Compare commits

..
Author SHA1 Message Date
Abimael MartellandClaude Opus 4.6 8c5fed2bdd extractPagesMarkdown: return classification metadata (0.7.0)
Combine per-page markdown extraction with layout classification into a
single parse. extractPagesMarkdown now returns PagesExtractionResult with
pages_with_tables, pages_with_columns, pages_needing_ocr, and is_complex
alongside the per-page markdown — eliminating redundant PDF parses for
callers that need both.

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
2026-04-14 13:06:42 -07:00
79 changed files with 920 additions and 21250 deletions
+31 -5
View File
@@ -20,7 +20,17 @@ jobs:
uses: dtolnay/rust-toolchain@stable
- name: Cache cargo
uses: Swatinem/rust-cache@v2
uses: actions/cache@v4
with:
path: |
~/.cargo/bin/
~/.cargo/registry/index/
~/.cargo/registry/cache/
~/.cargo/git/db/
target/
key: ${{ runner.os }}-cargo-${{ hashFiles('**/Cargo.lock') }}
restore-keys: |
${{ runner.os }}-cargo-
- name: Run tests
run: cargo test --verbose
@@ -51,9 +61,17 @@ jobs:
components: clippy
- name: Cache cargo
uses: Swatinem/rust-cache@v2
uses: actions/cache@v4
with:
key: clippy
path: |
~/.cargo/bin/
~/.cargo/registry/index/
~/.cargo/registry/cache/
~/.cargo/git/db/
target/
key: ${{ runner.os }}-cargo-clippy-${{ hashFiles('**/Cargo.lock') }}
restore-keys: |
${{ runner.os }}-cargo-clippy-
- name: Run clippy
run: cargo clippy -- -D warnings
@@ -71,9 +89,17 @@ jobs:
uses: dtolnay/rust-toolchain@stable
- name: Cache cargo
uses: Swatinem/rust-cache@v2
uses: actions/cache@v4
with:
key: build
path: |
~/.cargo/bin/
~/.cargo/registry/index/
~/.cargo/registry/cache/
~/.cargo/git/db/
target/
key: ${{ runner.os }}-cargo-build-${{ hashFiles('**/Cargo.lock') }}
restore-keys: |
${{ runner.os }}-cargo-build-
- name: Build
run: cargo build --release --verbose
-36
View File
@@ -1,36 +0,0 @@
name: Deploy landing page
on:
push:
branches: [main]
paths: ['site/**', '.github/workflows/pages.yml']
workflow_dispatch:
permissions:
contents: read
pages: write
id-token: write
# Allow one concurrent deployment; don't cancel an in-progress production deploy.
concurrency:
group: pages
cancel-in-progress: false
jobs:
deploy:
name: Build & deploy to GitHub Pages
runs-on: ubuntu-latest
environment:
name: github-pages
url: ${{ steps.deploy.outputs.page_url }}
steps:
- uses: actions/checkout@v4
- name: Upload site artifact
uses: actions/upload-pages-artifact@v3
with:
path: site
- name: Deploy to GitHub Pages
id: deploy
uses: actions/deploy-pages@v4
-87
View File
@@ -1,87 +0,0 @@
name: Publish Rust crate
on:
push:
branches: [main]
paths: ['Cargo.toml']
permissions:
contents: read
env:
CARGO_TERM_COLOR: always
jobs:
check-version:
name: Check version change
runs-on: ubuntu-latest
outputs:
changed: ${{ steps.check.outputs.changed }}
published: ${{ steps.check.outputs.published }}
version: ${{ steps.check.outputs.version }}
steps:
- uses: actions/checkout@v4
with:
fetch-depth: 2
- name: Check if version changed
id: check
run: |
NEW_VERSION=$(python3 -c 'import pathlib, tomllib; print(tomllib.loads(pathlib.Path("Cargo.toml").read_text())["package"]["version"])')
OLD_VERSION=$(git show HEAD~1:Cargo.toml | python3 -c 'import sys, tomllib; print(tomllib.loads(sys.stdin.read())["package"]["version"])')
echo "old=$OLD_VERSION new=$NEW_VERSION"
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
if [ "$NEW_VERSION" = "$OLD_VERSION" ]; then
echo "changed=false" >> "$GITHUB_OUTPUT"
echo "published=false" >> "$GITHUB_OUTPUT"
exit 0
fi
echo "changed=true" >> "$GITHUB_OUTPUT"
HTTP_STATUS=$(curl --silent --show-error --output /tmp/crate-version.json --write-out "%{http_code}" \
-H "User-Agent: firecrawl/pdf-inspector publish workflow (https://github.com/firecrawl/pdf-inspector)" \
"https://crates.io/api/v1/crates/pdf-inspector/$NEW_VERSION")
case "$HTTP_STATUS" in
200)
echo "published=true" >> "$GITHUB_OUTPUT"
echo "pdf-inspector v$NEW_VERSION is already published"
;;
404)
echo "published=false" >> "$GITHUB_OUTPUT"
;;
*)
cat /tmp/crate-version.json
echo "Unexpected crates.io response: $HTTP_STATUS" >&2
exit 1
;;
esac
publish:
name: Publish to crates.io
needs: check-version
if: needs.check-version.outputs.changed == 'true' && needs.check-version.outputs.published == 'false'
runs-on: ubuntu-latest
environment: crates-io
permissions:
contents: read
id-token: write
steps:
- uses: actions/checkout@v4
- name: Install Rust
uses: dtolnay/rust-toolchain@stable
- name: Verify package
run: cargo publish --dry-run
- name: Authenticate with crates.io
id: auth
uses: rust-lang/crates-io-auth-action@v1
- name: Publish crate
run: cargo publish
env:
CARGO_REGISTRY_TOKEN: ${{ steps.auth.outputs.token }}
+2 -31
View File
@@ -2,41 +2,14 @@ name: Publish npm package
on:
push:
branches: [main]
paths: ['napi/package.json']
tags: ['v*']
permissions:
contents: read
id-token: write
jobs:
check-version:
name: Check version change
runs-on: ubuntu-latest
outputs:
changed: ${{ steps.check.outputs.changed }}
version: ${{ steps.check.outputs.version }}
steps:
- uses: actions/checkout@v4
with:
fetch-depth: 2
- name: Check if version changed
id: check
run: |
NEW_VERSION=$(node -p "require('./napi/package.json').version")
OLD_VERSION=$(git show HEAD~1:napi/package.json | node -p "JSON.parse(require('fs').readFileSync('/dev/stdin','utf8')).version")
echo "old=$OLD_VERSION new=$NEW_VERSION"
if [ "$NEW_VERSION" != "$OLD_VERSION" ]; then
echo "changed=true" >> "$GITHUB_OUTPUT"
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
else
echo "changed=false" >> "$GITHUB_OUTPUT"
fi
build:
needs: check-version
if: needs.check-version.outputs.changed == 'true'
name: Build ${{ matrix.target }}
runs-on: ${{ matrix.os }}
strategy:
@@ -46,8 +19,6 @@ jobs:
target: x86_64-unknown-linux-gnu
- os: macos-14
target: aarch64-apple-darwin
- os: windows-latest
target: x86_64-pc-windows-msvc
steps:
- uses: actions/checkout@v4
@@ -97,7 +68,7 @@ jobs:
publish:
name: Publish to npm
needs: [check-version, build]
needs: build
runs-on: ubuntu-latest
permissions:
contents: read
-5
View File
@@ -35,8 +35,3 @@ scripts/
# Test output
test_output/
# Python
__pycache__/
*.pyc
.pytest_cache/
-79
View File
@@ -1,79 +0,0 @@
# pdf-inspector
Fast PDF text extraction to structured Markdown. CLI binary: `pdf2md`. Detection binary: `detect-pdf`.
## Build & Test
```bash
cargo fmt # format
cargo clippy -- -D warnings # lint (enforced, zero warnings)
cargo test # unit + integration tests (267+ unit, 73+ integration)
cargo build --release # release binary for benchmarks
```
All three must pass before committing.
## Binaries
- `pdf2md` — extract PDF → Markdown. Supports `--json` for structured output.
- `detect-pdf` — classify PDF type (TextBased/Scanned/Mixed/ImageBased). Supports `--analyze --json`.
## Architecture
```
src/
lib.rs public API, process_pdf_with_options, encoding issue detection
detector.rs PDF type classification, tiled-scan detection, page sampling
types.rs TextItem, TextLine, PdfRect, PdfLine
tounicode.rs CMap/ToUnicode parsing, CID decoding
text_utils.rs CJK/RTL handling, Otsu threshold, ligature expansion, NFKC
extractor/
mod.rs top-level extraction orchestrator
content_stream.rs PDF operator state machine (Tj/TJ/Td/Tm/q/Q)
fonts.rs font width/encoding, CMapDecisionCache, TrueType cmap fallback
layout.rs column detection (histogram), newspaper/tabular classification,
spanning-line pre-masking, sidebar detection
tables/
detect_rects.rs rect-based table detection (union-find clustering)
detect_heuristic.rs heuristic table detection (gap-histogram, body-font tables)
detect_lines.rs line-based table detection (H/V line grids)
grid.rs column/row boundaries, cell assignment
format.rs table→Markdown formatting, continuation row merging
markdown/
convert.rs core line→Markdown loop, struct-tree role support
analysis.rs font stats, heading tiers, paragraph thresholds
classify.rs line classification (header, list, code, caption)
preprocess.rs drop cap merging, heading line merging
postprocess.rs dot leaders, hyphenation, page numbers, URL formatting
```
## Key design decisions
- **Primary audience is AI agents.** Output optimized for token efficiency and semantic quality, not visual formatting. No cosmetic padding.
- **Three table detection strategies** run in priority order: rect-based → line-based → heuristic. First valid result wins.
- **Column detection** uses horizontal projection histograms with valley detection. Multi-item spanning lines (titles, headers) are pre-masked using column-aware thresholds before column assignment.
- **Newspaper vs tabular** classification determines reading order: newspaper reads columns sequentially, tabular Y-interleaves them.
- **Tiled-scan detection** catches scanned PDFs with JBIG2/strip images where no single tile exceeds the template threshold but aggregate area does (≥2M pixels).
- **Garbage text upgrade** reclassifies Mixed PDFs as Scanned when extracted text is <50% alphanumeric.
- **Tagged PDF support** uses structure tree roles (H1-H6, P, L, Code, BlockQuote) when available, falling back to font-size heuristics.
## Testing
- **Unit tests**: inline `#[cfg(test)] mod tests` in each module with synthetic data.
- **Integration tests**: `tests/integration_tests.rs` with fixture PDFs in `tests/fixtures/`.
- **Regression suite**: sibling repo `pdf-evals` with 179+ snapshot PDFs. Run `cargo build --release` then `bench.py test` in that repo before committing.
## Debugging
```bash
RUST_LOG=pdf_inspector::extractor::layout=debug cargo run --bin pdf2md -- file.pdf
RUST_LOG=pdf_inspector::tables=debug cargo run --bin pdf2md -- file.pdf
RUST_LOG=pdf_inspector::detector=debug cargo run --release --bin detect-pdf -- file.pdf
```
## Conventions
- Clippy: use `is_some_and(...)` not `map_or(false, ...)`
- lopdf quirk: `ParseError` is private — match by string for `InvalidFileHeader`
- Column limit for tables: 25 (wide statistical tables)
- `propagate_merged_cells` skipped for >10 columns (spanning rects = background fills)
+1 -2
View File
@@ -61,8 +61,7 @@ src/
- **Unit tests**: inline `#[cfg(test)] mod tests` in each module with synthetic data.
- **Integration tests**: `tests/integration_tests.rs` with fixture PDFs in `tests/fixtures/`.
- **Regression suite**: sibling repo `pdf-evals` with 187+ snapshot PDFs. Run `cargo build --release` then `bench.py test` in that repo before committing.
- **Semantic quality**: run `bench.py score` in `pdf-evals` for the semantic verdict (TEDS + MHS + reading order + char/word + list preservation, composited). Character-level diff alone misclassifies structural improvements (e.g., column-detection rewrites) as regressions — `score` is the tie-breaker. See `pdf-evals/CLAUDE.md` "Semantic scoring".
- **Regression suite**: sibling repo `pdf-evals` with 179+ snapshot PDFs. Run `cargo build --release` then `bench.py test` in that repo before committing.
## Debugging
+2 -2
View File
@@ -1,6 +1,6 @@
[package]
name = "pdf-inspector"
version = "0.1.4"
version = "0.1.0"
edition = "2021"
autobins = false
authors = ["Firecrawl Team"]
@@ -17,7 +17,7 @@ crate-type = ["lib", "cdylib"]
pyo3 = { version = "0.25", features = ["extension-module"], optional = true }
# PDF parsing
lopdf = { version = "0.41.0", features = ["rayon"] }
lopdf = { git = "https://github.com/J-F-Liu/lopdf", rev = "7a05512d831415b1f2b1ce522391d6beab8a1284", features = ["rayon"] }
# Error handling
thiserror = "2.0"
-21
View File
@@ -1,21 +0,0 @@
MIT License
Copyright (c) 2026 Firecrawl
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
SOFTWARE.
+16 -36
View File
@@ -1,9 +1,5 @@
# pdf-inspector
[![Crates.io](https://img.shields.io/crates/v/pdf-inspector.svg)](https://crates.io/crates/pdf-inspector)
[![npm](https://img.shields.io/npm/v/@firecrawl/pdf-inspector.svg)](https://www.npmjs.com/package/@firecrawl/pdf-inspector)
[![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Includes bindings for [Python](docs/python.md) and [Node.js](napi/README.md).
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
@@ -26,16 +22,16 @@ Evaluated on the [opendataloader-bench](https://github.com/opendataloader-projec
| Engine | Overall | Reading Order (NID) | Tables (TEDS) | Headings (MHS) | Speed (200 docs) |
|---|---|---|---|---|---|
| pdf-inspector | 0.83 | 0.88 | 0.66 | 0.74 | 4s |
| pdf-inspector | 0.78 | 0.87 | 0.59 | 0.57 | 4s |
| opendataloader | 0.84 | 0.91 | 0.49 | 0.74 | 11s |
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
| markitdown | 0.58 | 0.88 | 0.00 | 0.00 | 8s |
For context, engines that use OCR/ML (docling, marker, mineru) score 0.83-0.88 overall but take 2-180 minutes on the same corpus — pdf-inspector reaches the low end of that range without any OCR, in 4 seconds.
For context, engines that use OCR/ML (docling, marker, mineru) score 0.83-0.88 overall but take 2-180 minutes on the same corpus.
**Where we do well:** Speed (fastest of all engines), the best table detection of any engine shown, and heading detection now on par with opendataloader. Overall lands within 0.01 of opendataloader at roughly 2.5× the speed.
**Where we do well:** Speed (fastest of all engines), reading order, table detection vs other direct-text tools.
**Where we lag:** Reading order still trails opendataloader slightly, and table structure trails OCR-based engines that can see visual layout.
**Where we lag:** Heading detection trails opendataloader — many PDFs use bold text at body font size for headings, or headings that are only slightly larger than body text. Table detection trails OCR-based engines that can see visual table structure.
## Quick start
@@ -59,12 +55,12 @@ print(result.markdown) # Markdown string or None
### Node.js
```bash
npm install @firecrawl/pdf-inspector
npm install @firecrawl/pdf-inspector-js
```
```javascript
import { readFileSync } from 'fs';
import { processPdf, classifyPdf } from '@firecrawl/pdf-inspector';
import { processPdf, classifyPdf } from '@firecrawl/pdf-inspector-js';
const result = processPdf(readFileSync('document.pdf'));
console.log(result.pdfType); // "TextBased", "Scanned", "ImageBased", "Mixed"
@@ -75,17 +71,9 @@ console.log(result.markdown); // Markdown string or null
### Rust
Install from [crates.io](https://crates.io/crates/pdf-inspector):
```bash
cargo add pdf-inspector
```
Or add it manually:
```toml
[dependencies]
pdf-inspector = "0.1"
pdf-inspector = { git = "https://github.com/firecrawl/pdf-inspector" }
```
```rust
@@ -103,37 +91,29 @@ if let Some(markdown) = &result.markdown {
### CLI
```bash
# Install the CLI tools
cargo install pdf-inspector
# Convert PDF to Markdown
pdf2md document.pdf
cargo run --bin pdf2md -- document.pdf
# JSON output (for piping)
pdf2md document.pdf --json
# Positioned TextItem JSON, including is_underline metadata
pdf2md document.pdf --items-json
cargo run --bin pdf2md -- document.pdf --json
# Raw markdown only (no headers)
pdf2md document.pdf --raw
cargo run --bin pdf2md -- document.pdf --raw
# Insert page break markers (<!-- Page N -->)
pdf2md document.pdf --pages
cargo run --bin pdf2md -- document.pdf --pages
# Process only specific pages
pdf2md document.pdf --select-pages 1,3,5-10
cargo run --bin pdf2md -- document.pdf --select-pages 1,3,5-10
# Detection only (no extraction)
detect-pdf document.pdf
detect-pdf document.pdf --json
cargo run --bin detect-pdf -- document.pdf
cargo run --bin detect-pdf -- document.pdf --json
# Detection + layout analysis (tables, columns)
detect-pdf document.pdf --analyze --json
cargo run --bin detect-pdf -- document.pdf --analyze --json
```
From a source checkout, use `cargo run --bin pdf2md -- document.pdf` or `cargo run --bin detect-pdf -- document.pdf` instead.
## Architecture
```
@@ -243,4 +223,4 @@ See [docs/debugging.md](docs/debugging.md) for `RUST_LOG` environment variable u
## License
[MIT](LICENSE)
MIT
-33
View File
@@ -1,33 +0,0 @@
# Security Policy
## Reporting a Vulnerability
If you believe you've found a security vulnerability in pdf-inspector, please
report it privately so we can fix it before public disclosure.
**Preferred:** Email **help@firecrawl.dev** with:
- A description of the issue and its impact
- Steps to reproduce (a minimal PDF or input that triggers the bug is ideal)
- The version or commit hash of pdf-inspector you tested against
**Alternative:** Use GitHub's private vulnerability reporting under the
[Security tab](https://github.com/firecrawl/pdf-inspector/security/advisories/new).
We'll acknowledge your report in a timely manner and keep you updated on
remediation progress. Please do not open a public GitHub issue for security
bugs.
## Scope
In scope:
- Memory-safety issues (panics, OOB reads, UB) reachable from a crafted PDF
- Denial-of-service vectors (unbounded allocation, infinite loops) on
reasonably-sized inputs
- Bugs in the `pdf2md` / `detect-pdf` binaries or the `pdf-inspector` crate
that affect downstream consumers
Out of scope:
- Bugs in upstream dependencies (`lopdf`, etc.) — please report those upstream
- Extraction quality issues (wrong text, missing tables) — open a regular
GitHub issue instead
-21
View File
@@ -1,21 +0,0 @@
# Publishing
The Rust crate is published to [crates.io](https://crates.io/crates/pdf-inspector) with trusted publishing from GitHub Actions. The first release was published manually; future releases publish from `.github/workflows/publish-crate.yml` when a `Cargo.toml` version change lands on `main`.
## crates.io Trusted Publisher
Configure the trusted publisher for the `pdf-inspector` crate with:
- Repository: `firecrawl/pdf-inspector`
- Workflow: `publish-crate.yml`
- Environment: `crates-io`
The workflow uses `rust-lang/crates-io-auth-action@v1` to exchange GitHub's OIDC token for a short-lived crates.io token, then passes it to `cargo publish`.
## Release Steps
1. Update `version` in `Cargo.toml`.
2. Merge the version bump to `main`.
3. The publish workflow compares the new `Cargo.toml` version with `HEAD~1`, runs `cargo publish --dry-run`, then publishes if that version is not already on crates.io.
If `Cargo.toml` changes without a package version bump, the workflow exits without publishing.
-14
View File
@@ -42,14 +42,6 @@ text = pdf_inspector.extract_text("document.pdf")
items = pdf_inspector.extract_text_with_positions("document.pdf")
for item in items[:5]:
print(f"'{item.text}' at ({item.x:.0f}, {item.y:.0f}) size={item.font_size}")
# Per-page markdown (one Markdown string per page, plus layout metadata)
result = pdf_inspector.extract_pages_markdown("document.pdf")
for page in result.pages:
print(f"Page {page.page}: {len(page.markdown)} chars, needs_ocr={page.needs_ocr}")
# Restrict to specific 0-indexed pages (preserves caller order)
result = pdf_inspector.extract_pages_markdown("document.pdf", pages=[0, 2])
```
## API reference
@@ -68,8 +60,6 @@ result = pdf_inspector.extract_pages_markdown("document.pdf", pages=[0, 2])
| `extract_text_with_positions_bytes(data, pages=None)` | Text with positions from bytes |
| `extract_text_in_regions(path, page_regions)` | Extract text in bounding-box regions |
| `extract_text_in_regions_bytes(data, page_regions)` | Region extraction from bytes |
| `extract_pages_markdown(path, pages=None)` | Per-page Markdown + layout metadata (all pages by default) |
| `extract_pages_markdown_bytes(data, pages=None)` | Per-page Markdown from bytes |
## Types
@@ -82,7 +72,3 @@ result = pdf_inspector.extract_pages_markdown("document.pdf", pages=[0, 2])
**`RegionText` fields:** `text`, `needs_ocr`
**`PageRegionTexts` fields:** `page` (0-indexed), `regions` (list of RegionText)
**`PageMarkdown` fields:** `page` (0-indexed), `markdown`, `needs_ocr`
**`PagesExtractionResult` fields:** `pages` (list of PageMarkdown), `pages_with_tables` (1-indexed), `pages_with_columns` (1-indexed), `pages_needing_ocr` (1-indexed), `is_complex`
-25
View File
@@ -79,27 +79,6 @@ let bytes = std::fs::read("document.pdf")?;
let result = process_pdf_mem(&bytes)?;
```
Extract per-page Markdown (one string per page, plus document-wide layout
metadata):
```rust
use pdf_inspector::extract_pages_markdown;
// Pass `None` for every page in document order, or a slice of 0-indexed
// pages to restrict the output (caller-supplied order is preserved).
let result = extract_pages_markdown("document.pdf", None)?;
for page in &result.pages {
if page.needs_ocr {
// Route this page to OCR
} else {
println!("Page {}: {}", page.page, page.markdown);
}
}
println!("Complex layout? {}", result.is_complex);
```
## Processing modes
| Mode | What it does | Returns |
@@ -123,8 +102,6 @@ println!("Complex layout? {}", result.is_complex);
| `to_markdown(text, options)` | Convert plain text to Markdown |
| `to_markdown_from_items(items, options)` | Markdown from pre-extracted `TextItem`s |
| `to_markdown_from_items_with_rects(items, options, rects)` | Markdown with rectangle-based table detection |
| `extract_pages_markdown(path, pages)` | Per-page Markdown + layout metadata (file) |
| `extract_pages_markdown_mem(bytes, pages)` | Per-page Markdown from bytes |
Low-level detection functions are also available via the `detector` module (`detect_pdf_type`, `detect_pdf_type_with_config`, etc.) for callers who need `PdfTypeResult` instead of `PdfProcessResult`.
@@ -142,6 +119,4 @@ Low-level detection functions are also available via the `detector` module (`det
| `LayoutComplexity` | Layout analysis: is_complex, pages_with_tables, pages_with_columns |
| `TextItem` | Text with position, font info, and page number |
| `MarkdownOptions` | Configuration for Markdown formatting (page numbers, etc.) |
| `PageMarkdown` | Per-page result: page (0-indexed), markdown, needs_ocr |
| `PagesExtractionResult` | Per-page output + 1-indexed pages_with_tables / pages_with_columns / pages_needing_ocr, is_complex |
| `PdfError` | `Io`, `Parse`, `Encrypted`, `InvalidStructure`, `NotAPdf` |
+4 -5
View File
@@ -672,9 +672,8 @@ checksum = "5e5032e24019045c762d3c0f28f5b6b8bbf38563a65908389bf7978758920897"
[[package]]
name = "lopdf"
version = "0.41.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "67513274c50a2b51e5f75d9e682fcf4ab064a8a9c9ae2c3c59309084882bb24d"
version = "0.40.0"
source = "git+https://github.com/J-F-Liu/lopdf?rev=7a05512d831415b1f2b1ce522391d6beab8a1284#7a05512d831415b1f2b1ce522391d6beab8a1284"
dependencies = [
"aes",
"bitflags",
@@ -830,7 +829,7 @@ checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe"
[[package]]
name = "pdf-inspector"
version = "0.1.4"
version = "0.1.0"
dependencies = [
"env_logger",
"log",
@@ -845,7 +844,7 @@ dependencies = [
[[package]]
name = "pdf-inspector-napi"
version = "0.2.2"
version = "0.2.0"
dependencies = [
"napi",
"napi-build",
+1 -1
View File
@@ -1,6 +1,6 @@
[package]
name = "pdf-inspector-napi"
version = "0.2.2"
version = "0.2.0"
edition = "2021"
[lib]
+6 -7
View File
@@ -1,4 +1,4 @@
# PDF Inspector
# firecrawl-pdf-inspector
Fast PDF classification and region-based text extraction for Node.js/Bun. Native Rust performance via [napi-rs](https://napi.rs).
@@ -7,9 +7,9 @@ Built by [Firecrawl](https://firecrawl.dev) for hybrid OCR pipelines — extract
## Install
```bash
npm install @firecrawl/pdf-inspector
npm install firecrawl-pdf-inspector
# or
bun add @firecrawl/pdf-inspector
bun add firecrawl-pdf-inspector
```
Prebuilt binaries included for **linux-x64** and **macOS ARM64**. No Rust toolchain needed.
@@ -21,7 +21,7 @@ Prebuilt binaries included for **linux-x64** and **macOS ARM64**. No Rust toolch
Classify a PDF as TextBased, Scanned, Mixed, or ImageBased (~10-50ms). Returns which pages need OCR.
```typescript
import { classifyPdf } from '@firecrawl/pdf-inspector'
import { classifyPdf } from 'firecrawl-pdf-inspector'
import { readFileSync } from 'fs'
const pdf = readFileSync('document.pdf')
@@ -37,10 +37,10 @@ console.log(result.confidence) // 0.875
Extract text within bounding-box regions from a PDF. Designed for hybrid OCR pipelines where a layout model detects regions in rendered page images, and this function extracts text from the PDF structure for text-based pages — skipping GPU OCR.
Each region result includes a `needsOcr` flag that signals unreliable extraction (empty text, GID-encoded fonts, garbage text, encoding issues). When the cause is a suspected garbled text layer, `ocrReason` is set to `"suspected_garbled_text"`.
Each region result includes a `needsOcr` flag that signals unreliable extraction (empty text, GID-encoded fonts, garbage text, encoding issues).
```typescript
import { extractTextInRegions } from '@firecrawl/pdf-inspector'
import { extractTextInRegions } from 'firecrawl-pdf-inspector'
const result = extractTextInRegions(pdf, [
{
@@ -84,7 +84,6 @@ interface PageRegionTexts {
interface RegionText {
text: string
needsOcr: boolean // true when text is unreliable
ocrReason?: string // "suspected_garbled_text" when known
}
```
-131
View File
@@ -1,131 +0,0 @@
#!/usr/bin/env node
import { readFileSync, writeFileSync } from "fs";
import { createRequire } from "module";
const require = createRequire(import.meta.url);
const { version } = require("../package.json");
const HELP = `pdf-inspector v${version} — Fast PDF text extraction to Markdown
Usage:
pdf-inspector <file> Extract markdown (default)
pdf-inspector detect <file> Classify PDF type
Options:
--json Output as JSON
--pages <pages> Comma-separated page numbers (e.g. 1,3,5)
-o, --output <file> Write output to file instead of stdout
-h, --help Show this help
-v, --version Show version
Examples:
pdf-inspector document.pdf
pdf-inspector document.pdf --json
pdf-inspector document.pdf --pages 1,2,3
pdf-inspector detect document.pdf --json
cat document.pdf | pdf-inspector -`;
function die(msg) {
process.stderr.write(`error: ${msg}\n`);
process.exit(1);
}
function parseArgs(argv) {
const opts = { json: false, pages: null, output: null, file: null, command: "extract" };
let i = 0;
// Check for subcommand
if (argv[0] === "detect") {
opts.command = "detect";
i = 1;
}
while (i < argv.length) {
const arg = argv[i];
if (arg === "-h" || arg === "--help") {
process.stdout.write(HELP + "\n");
process.exit(0);
} else if (arg === "-v" || arg === "--version") {
process.stdout.write(`${version}\n`);
process.exit(0);
} else if (arg === "--json") {
opts.json = true;
} else if (arg === "--pages") {
i++;
if (!argv[i]) die("--pages requires a value (e.g. 1,3,5)");
opts.pages = argv[i].split(",").map((p) => {
const n = parseInt(p.trim(), 10);
if (Number.isNaN(n) || n < 1) die(`invalid page number: ${p}`);
return n;
});
} else if (arg === "-o" || arg === "--output") {
i++;
if (!argv[i]) die("-o requires a filename");
opts.output = argv[i];
} else if (arg === "-" || !arg.startsWith("-")) {
if (opts.file) die(`unexpected argument: ${arg}`);
opts.file = arg;
} else {
die(`unknown option: ${arg}`);
}
i++;
}
return opts;
}
function readInput(file) {
if (file === "-") {
return readFileSync(0); // stdin fd
}
try {
return readFileSync(file);
} catch (err) {
if (err.code === "ENOENT") die(`file not found: ${file}`);
die(err.message);
}
}
function output(text, outputPath) {
if (outputPath) {
writeFileSync(outputPath, text);
} else {
process.stdout.write(text);
}
}
// ---- main ----
const opts = parseArgs(process.argv.slice(2));
if (!opts.file) {
// Check if stdin is piped
if (process.stdin.isTTY !== false) {
process.stderr.write(HELP + "\n");
process.exit(1);
}
opts.file = "-";
}
const { processPdf, classifyPdf } = await import("../index.js");
const buffer = readInput(opts.file);
if (opts.command === "detect") {
const result = classifyPdf(buffer);
if (opts.json) {
output(JSON.stringify(result, null, 2) + "\n", opts.output);
} else {
const ocr = result.pagesNeedingOcr.length > 0
? `, ${result.pagesNeedingOcr.length} pages need OCR`
: "";
output(`${result.pdfType} (${result.pageCount} pages, confidence: ${result.confidence.toFixed(2)}${ocr})\n`, opts.output);
}
} else {
const result = processPdf(buffer, opts.pages ?? undefined);
if (opts.json) {
output(JSON.stringify(result, null, 2) + "\n", opts.output);
} else {
output((result.markdown ?? "") + "\n", opts.output);
}
}
+3 -8
View File
@@ -1,12 +1,9 @@
{
"name": "@firecrawl/pdf-inspector",
"version": "1.10.1",
"name": "firecrawl-pdf-inspector",
"version": "0.7.0",
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
"main": "index.js",
"types": "index.d.ts",
"bin": {
"pdf-inspector": "bin/pdf-inspector.mjs"
},
"license": "MIT",
"keywords": [
"pdf",
@@ -23,7 +20,6 @@
"index.js",
"index.d.ts",
"*.node",
"bin/",
"README.md"
],
"repository": {
@@ -38,8 +34,7 @@
"binaryName": "pdf-inspector",
"targets": [
"x86_64-unknown-linux-gnu",
"aarch64-apple-darwin",
"x86_64-pc-windows-msvc"
"aarch64-apple-darwin"
],
"package": {
"name": "@firecrawl/pdf-inspector-js"
-29
View File
@@ -1,29 +0,0 @@
import { readFileSync } from "node:fs";
import { createRequire } from "node:module";
const require = createRequire(import.meta.url);
const { detectVectorGridInRegion } = require("./index.js");
const pdfPath =
process.argv[2] ?? "/tmp/pdf_inspector_indent_fixtures/cis_edge_benchmark.pdf";
const pdf = readFileSync(pdfPath);
const dpi = Number(process.argv[3] ?? 200);
const crops = [
{ pageIdx: 29, box: [0, 0, 612, 792], label: "page30-full" },
{ pageIdx: 16, box: [0, 0, 612, 792], label: "page17-full" },
{ pageIdx: 23, box: [0, 0, 612, 792], label: "page24-full" },
];
for (const { pageIdx, box, label } of crops) {
const result = detectVectorGridInRegion(pdf, pageIdx, box, dpi);
if (!result) {
console.log(`${label}: null`);
continue;
}
const rows = result.structureTokens.filter((token) => token === "<tr>").length;
const cols = rows > 0 ? result.cellBboxes.length / rows : 0;
console.log(
`${label}: cells=${result.cellBboxes.length} rows=${rows} cols=${cols}`,
);
}
+4 -284
View File
@@ -40,8 +40,6 @@ pub struct PdfResult {
pub processing_time_ms: u32,
/// 1-indexed page numbers that need OCR.
pub pages_needing_ocr: Vec<u32>,
/// Machine-readable OCR reasons by 1-indexed page.
pub ocr_reasons_by_page: Vec<PageOcrReasons>,
pub title: Option<String>,
pub confidence: f64,
pub is_complex_layout: bool,
@@ -50,13 +48,6 @@ pub struct PdfResult {
pub has_encoding_issues: bool,
}
/// OCR reasons for a single 1-indexed page.
#[napi(object)]
pub struct PageOcrReasons {
pub page: u32,
pub reasons: Vec<String>,
}
/// Lightweight PDF classification result.
#[napi(object)]
pub struct PdfClassification {
@@ -80,12 +71,6 @@ pub struct TextItem {
pub page: u32,
pub is_bold: bool,
pub is_italic: bool,
/// Underline detected geometrically (drawn rule/thin rect under the
/// baseline) — PDFs carry no underline font flag.
pub is_underline: bool,
/// Strikeout detected geometrically (rule crossing the glyphs at mid
/// x-height).
pub is_strikeout: bool,
pub item_type: ItemType,
/// URL for link items, `None` for other types.
pub link_url: Option<String>,
@@ -105,8 +90,6 @@ pub struct RegionText {
pub text: String,
/// `true` when the text should not be trusted (empty, GID fonts, garbage, encoding issues).
pub needs_ocr: bool,
/// Machine-readable OCR reason when the cause is known.
pub ocr_reason: Option<String>,
}
/// Extracted text for one page's regions.
@@ -116,13 +99,6 @@ pub struct PageRegionTexts {
pub regions: Vec<RegionText>,
}
/// Vector-grid detection result compatible with `extractTablesWithStructure*`.
#[napi(object)]
pub struct VectorGridDetectionJs {
pub structure_tokens: Vec<String>,
pub cell_bboxes: Vec<Vec<f64>>,
}
// ---------------------------------------------------------------------------
// Helpers
// ---------------------------------------------------------------------------
@@ -143,7 +119,6 @@ fn to_napi_result(r: pdf_inspector::PdfProcessResult) -> PdfResult {
page_count: r.page_count,
processing_time_ms: r.processing_time_ms as u32,
pages_needing_ocr: r.pages_needing_ocr,
ocr_reasons_by_page: to_napi_page_ocr_reasons(r.ocr_reasons_by_page),
title: r.title,
confidence: r.confidence as f64,
is_complex_layout: r.layout.is_complex,
@@ -153,18 +128,6 @@ fn to_napi_result(r: pdf_inspector::PdfProcessResult) -> PdfResult {
}
}
fn to_napi_page_ocr_reasons(
reasons: Vec<pdf_inspector::PageOcrReasons>,
) -> Vec<PageOcrReasons> {
reasons
.into_iter()
.map(|reason| PageOcrReasons {
page: reason.page,
reasons: reason.reasons,
})
.collect()
}
fn convert_item_type(t: &pdf_inspector::types::ItemType) -> (ItemType, Option<String>) {
match t {
pdf_inspector::types::ItemType::Text => (ItemType::Text, None),
@@ -296,8 +259,6 @@ pub fn extract_text_with_positions(
page: item.page,
is_bold: item.is_bold,
is_italic: item.is_italic,
is_underline: item.is_underline,
is_strikeout: item.is_strikeout,
item_type,
link_url,
}
@@ -356,236 +317,6 @@ pub fn extract_tables_in_regions(
})
}
/// Detect a vector ruled-line / rectangle grid inside one page region.
///
/// Returns TSR-compatible structure tokens plus crop-pixel cell bboxes, or
/// `null` when the region does not contain a valid vector grid.
///
/// `pageIdx` is 0-indexed. `regionPdfPtBbox` is `[x1,y1,x2,y2]` in PDF
/// points with top-left origin. `renderDpi` is the DPI of the crop image that
/// will consume the returned cell bboxes.
#[napi]
pub fn detect_vector_grid_in_region(
buffer: Buffer,
page_idx: u32,
region_pdf_pt_bbox: Vec<f64>,
render_dpi: f64,
) -> Result<Option<VectorGridDetectionJs>> {
let bytes: Vec<u8> = buffer.to_vec();
let region = if region_pdf_pt_bbox.len() == 4 {
[
region_pdf_pt_bbox[0] as f32,
region_pdf_pt_bbox[1] as f32,
region_pdf_pt_bbox[2] as f32,
region_pdf_pt_bbox[3] as f32,
]
} else {
[0.0, 0.0, 0.0, 0.0]
};
catch_panic("detect_vector_grid_in_region", move || {
let result = pdf_inspector::detect_vector_grid_in_region_mem(
&bytes,
page_idx,
region,
render_dpi as f32,
)
.map_err(|e| to_napi_err(e, "detect_vector_grid_in_region"))?;
Ok(result.map(|r| VectorGridDetectionJs {
structure_tokens: r.structure_tokens,
cell_bboxes: r
.cell_bboxes
.into_iter()
.map(|bbox| bbox.into_iter().map(|v| v as f64).collect())
.collect(),
}))
})
}
/// One cropped table region plus its raw structure-recovery output, for
/// `extractTablesWithStructure`.
///
/// `structureTokens` and `cellBboxes` are typically produced by an external
/// table-structure recognition model (e.g. SLANet on PaddleOCR) running on
/// a rendered crop of the page. pdf-inspector uses the structure to lay out
/// the cells and pulls the cell text from the native PDF — no OCR involved.
#[napi(object)]
pub struct TsrTableInputJs {
/// 0-indexed page number where the crop was taken from.
pub page: u32,
/// Crop bbox on the page, `[x1, y1, x2, y2]` in PDF points with
/// top-left origin.
pub crop_pdf_pt_bbox: Vec<f64>,
/// DPI the crop image was rendered at (e.g. `200.0`).
pub render_dpi: f64,
/// Raw structure tokens emitted by the TSR model, in document order.
pub structure_tokens: Vec<String>,
/// One bbox per cell (in document order). May be 4-element
/// `[x1,y1,x2,y2]` or 8-element 4-corner polygon, in crop image-pixel
/// space.
pub cell_bboxes: Vec<Vec<f64>>,
}
/// Extract markdown tables using externally-supplied structure recovery.
///
/// For each input, pairs structure tokens with cell bboxes (rowspan/colspan
/// aware), converts each cell bbox from crop image-pixels into page PDF
/// points, pulls the cell's text from the native PDF, and emits a markdown
/// pipe-table.
///
/// Returns one markdown string per input, in input order.
#[napi]
pub fn extract_tables_with_structure(
buffer: Buffer,
inputs: Vec<TsrTableInputJs>,
) -> Result<Vec<String>> {
let bytes: Vec<u8> = buffer.to_vec();
let parsed = parse_tsr_inputs(&inputs);
catch_panic("extract_tables_with_structure", move || {
pdf_inspector::extract_tables_with_structure_mem(&bytes, &parsed)
.map_err(|e| to_napi_err(e, "extract_tables_with_structure"))
})
}
/// One resolved cell from `extractTablesWithStructureCells`.
#[napi(object)]
pub struct StructuredCellJs {
/// 0-indexed grid row.
pub row: u32,
/// 0-indexed grid column.
pub col: u32,
/// 1 for a normal cell.
pub rowspan: u32,
/// 1 for a normal cell.
pub colspan: u32,
/// `true` when the cell is a `<th>` or sits inside `<thead>`.
pub is_header: bool,
/// Text extracted from the native PDF for this cell (may be empty).
pub text: String,
/// Axis-aligned bbox `[x1, y1, x2, y2]` in page PDF-points, top-left
/// origin. Useful for debug overlays or per-cell post-processing.
pub page_pt_bbox: Vec<f64>,
}
/// Extract structured cells using externally-supplied structure recovery.
///
/// Lower-level sibling of [`extractTablesWithStructure`]: instead of
/// rendering markdown, returns the resolved cells (row, col, rowspan,
/// colspan, isHeader, text, pagePtBbox) so callers can drive their own
/// rendering, debug overlays, or per-cell post-processing.
///
/// Returns one `Array<StructuredCellJs>` per input, in input order.
#[napi]
pub fn extract_tables_with_structure_cells(
buffer: Buffer,
inputs: Vec<TsrTableInputJs>,
) -> Result<Vec<Vec<StructuredCellJs>>> {
let bytes: Vec<u8> = buffer.to_vec();
let parsed = parse_tsr_inputs(&inputs);
catch_panic("extract_tables_with_structure_cells", move || {
let result = pdf_inspector::extract_tables_with_structure_cells_mem(&bytes, &parsed)
.map_err(|e| to_napi_err(e, "extract_tables_with_structure_cells"))?;
Ok(result
.into_iter()
.map(|cells| {
cells
.into_iter()
.map(|c| StructuredCellJs {
row: c.row as u32,
col: c.col as u32,
rowspan: c.rowspan as u32,
colspan: c.colspan as u32,
is_header: c.is_header,
text: c.text,
page_pt_bbox: c.page_pt_bbox.iter().map(|v| *v as f64).collect(),
})
.collect()
})
.collect())
})
}
/// One result from `extractTablesWithStructureAuto` — markdown plus a
/// diagnostic flag identifying which path produced it.
///
/// `fallbackReason` is `null` when the TSR-hybrid path produced the
/// markdown directly. When stage 1's quality check fires (the cells
/// look like a SLANet detection pathology — phantom rows or multi-row
/// content in a single cell), the auto path may expand the TSR cells
/// in-place or run the heuristic table extractor on the same region.
/// `fallbackReason` carries the diagnostic label (for example
/// `"multi_row_in_cell_expanded"` or `"phantom_empty_row"`).
#[napi(object)]
pub struct TableExtractionResultJs {
pub markdown: String,
pub fallback_reason: Option<String>,
}
/// Auto-fallback variant of [`extractTablesWithStructure`].
///
/// Runs the TSR-hybrid path, checks the resulting cells for known
/// SLANet detection pathologies, expands multi-row cells in-place when
/// possible, and otherwise falls back to the heuristic
/// `extractTablesInRegions` for inputs where the TSR path looks
/// compromised.
///
/// On clean inputs this returns identical markdown to
/// `extractTablesWithStructure`; on flagged inputs `fallbackReason` is
/// set to the recovery path that produced the result.
#[napi]
pub fn extract_tables_with_structure_auto(
buffer: Buffer,
inputs: Vec<TsrTableInputJs>,
) -> Result<Vec<TableExtractionResultJs>> {
let bytes: Vec<u8> = buffer.to_vec();
let parsed = parse_tsr_inputs(&inputs);
catch_panic("extract_tables_with_structure_auto", move || {
let result = pdf_inspector::extract_tables_with_structure_auto_mem(&bytes, &parsed)
.map_err(|e| to_napi_err(e, "extract_tables_with_structure_auto"))?;
Ok(result
.into_iter()
.map(|r| TableExtractionResultJs {
markdown: r.markdown,
fallback_reason: r.fallback_reason,
})
.collect())
})
}
fn parse_tsr_inputs(inputs: &[TsrTableInputJs]) -> Vec<pdf_inspector::TsrTableInput> {
inputs
.iter()
.map(|i| {
let crop = if i.crop_pdf_pt_bbox.len() == 4 {
[
i.crop_pdf_pt_bbox[0] as f32,
i.crop_pdf_pt_bbox[1] as f32,
i.crop_pdf_pt_bbox[2] as f32,
i.crop_pdf_pt_bbox[3] as f32,
]
} else {
[0.0, 0.0, 0.0, 0.0]
};
let cell_bboxes: Vec<Vec<f32>> = i
.cell_bboxes
.iter()
.map(|bb| bb.iter().map(|v| *v as f32).collect())
.collect();
pdf_inspector::TsrTableInput {
page: i.page,
crop_pdf_pt_bbox: crop,
render_dpi: i.render_dpi as f32,
structure_tokens: i.structure_tokens.clone(),
cell_bboxes,
}
})
.collect()
}
/// Per-page markdown extraction result.
#[napi(object)]
pub struct PageMarkdownResult {
@@ -595,8 +326,6 @@ pub struct PageMarkdownResult {
pub markdown: String,
/// `true` when text on this page is unreliable.
pub needs_ocr: bool,
/// Machine-readable OCR reason when the cause is known.
pub ocr_reason: Option<String>,
}
/// Combined per-page markdown extraction and layout classification result.
@@ -610,30 +339,24 @@ pub struct PagesExtractionResult {
pub pages_with_columns: Vec<u32>,
/// 1-indexed pages that need OCR (scanned/image-based).
pub pages_needing_ocr: Vec<u32>,
/// Machine-readable OCR reasons by 1-indexed page.
pub ocr_reasons_by_page: Vec<PageOcrReasons>,
/// True if any page has tables or columns.
pub is_complex: bool,
}
/// Extract formatted markdown for pages of a PDF, with layout classification
/// metadata.
/// Extract formatted markdown for specific pages of a PDF, with layout
/// classification metadata.
///
/// Returns per-page markdown and classification data (tables, columns,
/// OCR needs) from a single parse. Font statistics are computed from the
/// full document so header detection is consistent across pages.
///
/// Omit `pages` (or pass `undefined`) to return every page in document
/// order. Pass an array of 0-indexed page numbers to restrict output to
/// those pages, in caller-supplied order.
#[napi]
pub fn extract_pages_markdown(
buffer: Buffer,
pages: Option<Vec<u32>>,
pages: Vec<u32>,
) -> Result<PagesExtractionResult> {
let bytes: Vec<u8> = buffer.to_vec();
catch_panic("extract_pages_markdown", move || {
let result = pdf_inspector::extract_pages_markdown_mem(&bytes, pages.as_deref())
let result = pdf_inspector::extract_pages_markdown_mem(&bytes, &pages)
.map_err(|e| to_napi_err(e, "extract_pages_markdown"))?;
Ok(PagesExtractionResult {
pages: result
@@ -643,13 +366,11 @@ pub fn extract_pages_markdown(
page: r.page,
markdown: r.markdown,
needs_ocr: r.needs_ocr,
ocr_reason: r.ocr_reason,
})
.collect(),
pages_with_tables: result.pages_with_tables,
pages_with_columns: result.pages_with_columns,
pages_needing_ocr: result.pages_needing_ocr,
ocr_reasons_by_page: to_napi_page_ocr_reasons(result.ocr_reasons_by_page),
is_complex: result.is_complex,
})
})
@@ -686,7 +407,6 @@ fn to_page_region_texts(results: Vec<pdf_inspector::PageRegionResult>) -> Vec<Pa
.map(|r| RegionText {
text: r.text,
needs_ocr: r.needs_ocr,
ocr_reason: r.ocr_reason,
})
.collect(),
})
-35
View File
@@ -7,8 +7,6 @@ import {
extractText,
extractTextWithPositions,
extractTextInRegions,
detectVectorGridInRegion,
extractPagesMarkdown,
} from './index.js';
const fixture = readFileSync('../tests/fixtures/thermo-freon12.pdf');
@@ -91,39 +89,6 @@ assert.equal(typeof regionResults[0].regions[0].text, 'string');
assert.equal(typeof regionResults[0].regions[0].needsOcr, 'boolean');
console.log(' extractTextInRegions: OK');
// --- detectVectorGridInRegion ---
console.log('Testing detectVectorGridInRegion...');
const vectorGrid = detectVectorGridInRegion(fixture, 0, [0, 0, 600, 800], 72);
assert.ok(vectorGrid === null || typeof vectorGrid === 'object');
if (vectorGrid) {
assert.ok(Array.isArray(vectorGrid.structureTokens));
assert.ok(Array.isArray(vectorGrid.cellBboxes));
assert.ok(vectorGrid.cellBboxes.every(bbox => Array.isArray(bbox) && bbox.length === 4));
}
console.log(' detectVectorGridInRegion: OK');
// --- extractPagesMarkdown ---
console.log('Testing extractPagesMarkdown...');
// omit pages → every page in document order
const allPages = extractPagesMarkdown(fixture);
assert.equal(allPages.pages.length, 3);
assert.deepEqual(allPages.pages.map(p => p.page), [0, 1, 2]);
assert.ok(typeof allPages.pages[0].markdown === 'string');
assert.equal(typeof allPages.pages[0].needsOcr, 'boolean');
assert.ok(Array.isArray(allPages.pagesWithTables));
assert.ok(Array.isArray(allPages.pagesWithColumns));
assert.ok(Array.isArray(allPages.pagesNeedingOcr));
assert.equal(typeof allPages.isComplex, 'boolean');
console.log(' extractPagesMarkdown (no pages arg): OK');
// selected pages preserve caller order
const picked = extractPagesMarkdown(fixture, [2, 0]);
assert.equal(picked.pages.length, 2);
assert.equal(picked.pages[0].page, 2);
assert.equal(picked.pages[1].page, 0);
console.log(' extractPagesMarkdown with pages: OK');
// --- Error handling ---
console.log('Testing error handling...');
assert.throws(() => processPdf(Buffer.from('not a pdf')), /process_pdf/);
-52
View File
@@ -38,8 +38,6 @@ class TextItem:
page: int
is_bold: bool
is_italic: bool
is_underline: bool
is_strikeout: bool
item_type: str
class RegionText:
@@ -54,28 +52,6 @@ class PageRegionTexts:
"""0-indexed page number."""
regions: list[RegionText]
class PageMarkdown:
"""Per-page markdown extraction result."""
page: int
"""0-indexed page number."""
markdown: str
"""Formatted markdown for this page (empty string when needs_ocr is True)."""
needs_ocr: bool
"""True when text on this page is unreliable and OCR should be used instead."""
class PagesExtractionResult:
"""Per-page markdown output with document-wide layout classification."""
pages: list[PageMarkdown]
"""Per-page markdown results, in the order requested."""
pages_with_tables: list[int]
"""1-indexed pages where tables were detected."""
pages_with_columns: list[int]
"""1-indexed pages where multi-column layout was detected."""
pages_needing_ocr: list[int]
"""1-indexed pages that need OCR."""
is_complex: bool
"""True if any page has tables or multi-column layout."""
def process_pdf(path: str, pages: Optional[list[int]] = None) -> PdfResult:
"""Process a PDF: detect type, extract text, convert to Markdown."""
...
@@ -139,31 +115,3 @@ def extract_text_in_regions_bytes(
page_regions: List of (page_0indexed, [[x1, y1, x2, y2], ...]) tuples.
"""
...
def extract_pages_markdown(
path: str,
pages: Optional[list[int]] = None,
) -> PagesExtractionResult:
"""Extract formatted markdown for pages of a PDF, with layout classification.
Args:
path: Path to the PDF file.
pages: Optional list of 0-indexed pages. When ``None`` (default), every
page is returned in document order. Otherwise, output matches the
caller-supplied order.
Returns:
PagesExtractionResult with per-page markdown and document-wide layout
classification (tables, columns, OCR needs).
"""
...
def extract_pages_markdown_bytes(
data: bytes,
pages: Optional[list[int]] = None,
) -> PagesExtractionResult:
"""Extract formatted markdown for pages of a PDF from bytes.
See :func:`extract_pages_markdown` for details.
"""
...
+1 -3
View File
@@ -4,9 +4,7 @@ build-backend = "maturin"
[project]
name = "pdf-inspector"
# Version is sourced from Cargo.toml [package] version by maturin so the Python
# artifact always tracks the crate release instead of drifting on its own.
dynamic = ["version"]
version = "0.1.0"
description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection"
license = { text = "MIT" }
requires-python = ">=3.8"
-3
View File
@@ -1,3 +0,0 @@
<svg width="200" height="284" viewBox="0 0 200 284" fill="none" xmlns="http://www.w3.org/2000/svg">
<path d="M166.862 90.7716C155.812 94.0514 147.483 101.471 141.383 109.53C140.073 111.26 137.343 109.96 137.863 107.841C149.543 59.8136 134.113 19.896 86.0157 0.247269C83.5758 -0.752669 81.0359 1.43719 81.6759 3.99704C103.555 91.8416 11.5294 84.432 23.1588 184.016C23.3588 185.726 21.4389 186.896 20.039 185.896C15.6792 182.766 10.8095 176.236 7.46963 171.647C6.48968 170.297 4.36978 170.677 3.9198 172.287C1.25994 181.906 0 190.965 0 199.965C0 234.963 17.9891 265.771 45.2177 283.63C46.7777 284.65 48.7776 283.19 48.2476 281.4C46.8477 276.7 46.0577 271.74 45.9977 266.611C45.9977 263.461 46.1977 260.241 46.6877 257.241C47.8276 249.702 50.4475 242.522 54.8473 235.983C69.9365 213.334 100.185 191.455 95.3552 161.747C95.0453 159.867 97.2651 158.627 98.6651 159.917C119.974 179.386 124.194 205.575 120.694 229.063C120.394 231.103 122.954 232.193 124.244 230.593C127.504 226.513 131.483 222.933 135.813 220.244C136.893 219.574 138.333 220.084 138.743 221.284C141.153 228.293 144.733 234.873 148.113 241.452C152.152 249.362 154.302 258.391 153.962 267.951C153.792 272.6 153.022 277.1 151.732 281.38C151.182 283.19 153.162 284.7 154.752 283.66C182.001 265.801 200 234.993 200 199.975C200 187.806 197.87 175.876 193.84 164.697C185.391 141.248 163.952 123.64 169.372 93.0815C169.632 91.6216 168.282 90.3517 166.862 90.7716Z" fill="#FA5D19" style="fill:#FA5D19;fill:color(display-p3 0.9816 0.3634 0.0984);fill-opacity:1;"/>
</svg>

Before

Width:  |  Height:  |  Size: 1.5 KiB

-12
View File
@@ -1,12 +0,0 @@
<svg width="172" height="40" viewBox="0 0 172 40" fill="none" xmlns="http://www.w3.org/2000/svg">
<path d="M23.3606 12.8281C21.8137 13.2873 20.6476 14.3261 19.7936 15.4544C19.6102 15.6966 19.228 15.5146 19.3008 15.2178C20.936 8.49401 18.7759 2.90556 12.0422 0.154735C11.7006 0.0147436 11.345 0.321324 11.4346 0.679702C14.4977 12.9779 1.61412 11.9406 3.24224 25.8823C3.27024 26.1217 3.00145 26.2855 2.80546 26.1455C2.19509 25.7073 1.51332 24.7932 1.04575 24.1506C0.908555 23.9616 0.611769 24.0148 0.548773 24.2402C0.176391 25.5869 0 26.8553 0 28.1152C0 33.0149 2.51847 37.328 6.33048 39.8283C6.54887 39.9711 6.82886 39.7667 6.75466 39.5161C6.55867 38.8581 6.44808 38.1638 6.43968 37.4456C6.43968 37.0046 6.46768 36.5539 6.53627 36.1339C6.69587 35.0784 7.06265 34.0732 7.67862 33.1577C9.79111 29.9869 14.0259 26.9239 13.3497 22.7647C13.3063 22.5015 13.6171 22.328 13.8131 22.5085C16.7964 25.2342 17.3871 28.9005 16.8972 32.1889C16.8552 32.4745 17.2135 32.6271 17.3941 32.4031C17.8505 31.832 18.4077 31.3308 19.0138 30.9542C19.165 30.8604 19.3666 30.9318 19.424 31.0998C19.7614 32.0811 20.2626 33.0023 20.7358 33.9234C21.3013 35.0308 21.6023 36.2949 21.5547 37.6332C21.5309 38.2842 21.4231 38.9141 21.2425 39.5133C21.1655 39.7667 21.4427 39.9781 21.6653 39.8325C25.4801 37.3322 28 33.0191 28 28.1166C28 26.4129 27.7018 24.7428 27.1376 23.1777C25.9547 19.8949 22.9533 17.4297 23.712 13.1515C23.7484 12.9471 23.5594 12.7693 23.3606 12.8281Z" fill="#262626" style="fill:#262626;fill:color(display-p3 0.1500 0.1500 0.1500);fill-opacity:1;"/>
<path d="M41 34.0521V10.9618H55.7586V14.3264H44.7969V21.0226H53.8436V24.2882H44.7969V34.0521H41Z" fill="#262626" style="fill:#262626;fill:color(display-p3 0.1500 0.1500 0.1500);fill-opacity:1;"/>
<path d="M59.9569 14.7882C58.7352 14.7882 57.7777 13.8976 57.7777 12.6441C57.7777 11.3906 58.7352 10.5 59.9569 10.5C61.1785 10.5 62.136 11.3906 62.136 12.6441C62.136 13.8976 61.1785 14.7882 59.9569 14.7882ZM58.1409 34.0521V17.1632H61.7068V34.0521H58.1409Z" fill="#262626" style="fill:#262626;fill:color(display-p3 0.1500 0.1500 0.1500);fill-opacity:1;"/>
<path d="M73.5885 17.1632H74.3809V20.4948H72.796C69.6264 20.4948 68.6029 22.9687 68.6029 25.5747V34.0521H65.0371V17.1632H68.2067L68.6029 19.7031C69.4613 18.2847 70.815 17.1632 73.5885 17.1632Z" fill="#262626" style="fill:#262626;fill:color(display-p3 0.1500 0.1500 0.1500);fill-opacity:1;"/>
<path d="M83.632 34.25C78.3163 34.25 74.9816 30.8194 74.9816 25.6406C74.9816 20.4288 78.3163 16.9653 83.3019 16.9653C88.1884 16.9653 91.457 20.066 91.5561 25.0139C91.5561 25.4427 91.5231 25.9045 91.457 26.3663H78.7125V26.5972C78.8116 29.467 80.6275 31.3472 83.4339 31.3472C85.613 31.3472 87.1979 30.2587 87.6931 28.3785H91.2589C90.6646 31.7101 87.8252 34.25 83.632 34.25ZM78.8446 23.7604H87.8582C87.561 21.2535 85.8112 19.8351 83.3349 19.8351C81.0567 19.8351 79.1087 21.3524 78.8446 23.7604Z" fill="#262626" style="fill:#262626;fill:color(display-p3 0.1500 0.1500 0.1500);fill-opacity:1;"/>
<path d="M102.033 34.25C96.9151 34.25 93.6465 30.9184 93.6465 25.6406C93.6465 20.4288 97.0142 16.9653 102.132 16.9653C106.49 16.9653 109.197 19.3733 109.891 23.1997H106.16C105.698 21.2205 104.278 20 102.066 20C99.1933 20 97.3113 22.309 97.3113 25.6406C97.3113 28.9392 99.1933 31.2153 102.066 31.2153C104.245 31.2153 105.698 29.9618 106.127 28.0156H109.891C109.23 31.842 106.358 34.25 102.033 34.25Z" fill="#262626" style="fill:#262626;fill:color(display-p3 0.1500 0.1500 0.1500);fill-opacity:1;"/>
<path d="M121.006 17.1632H121.799V20.4948H120.214C117.044 20.4948 116.021 22.9687 116.021 25.5747V34.0521H112.455V17.1632H115.625L116.021 19.7031C116.879 18.2847 118.233 17.1632 121.006 17.1632Z" fill="#262626" style="fill:#262626;fill:color(display-p3 0.1500 0.1500 0.1500);fill-opacity:1;"/>
<path d="M130.614 16.9653C135.104 16.9653 137.679 19.1094 137.679 23.1007V34.0521H134.576L134.279 31.6441C133.123 33.1615 131.505 34.25 128.831 34.25C125.133 34.25 122.657 32.4358 122.657 29.3021C122.657 25.8385 125.166 23.8924 129.92 23.8924H134.147V22.8698C134.147 20.9896 132.793 19.8351 130.449 19.8351C128.336 19.8351 126.916 20.8247 126.652 22.309H123.152C123.515 19.0104 126.355 16.9653 130.614 16.9653ZM129.425 31.4792C132.397 31.4792 134.114 29.7309 134.147 27.125V26.5312H129.722C127.51 26.5312 126.289 27.3559 126.289 29.0712C126.289 30.4896 127.477 31.4792 129.425 31.4792Z" fill="#262626" style="fill:#262626;fill:color(display-p3 0.1500 0.1500 0.1500);fill-opacity:1;"/>
<path d="M144.653 34.0521L139.139 17.1632H142.903L146.766 30.0937L150.629 17.1632H153.897L157.595 30.0937L161.59 17.1632H165.222L159.609 34.0521H155.779L152.214 22.5729L148.516 34.0521H144.653Z" fill="#262626" style="fill:#262626;fill:color(display-p3 0.1500 0.1500 0.1500);fill-opacity:1;"/>
<path d="M166.934 34.0521V10.9618H170.5V34.0521H166.934Z" fill="#262626" style="fill:#262626;fill:color(display-p3 0.1500 0.1500 0.1500);fill-opacity:1;"/>
</svg>

Before

Width:  |  Height:  |  Size: 4.8 KiB

-428
View File
@@ -1,428 +0,0 @@
<!DOCTYPE html>
<html lang="en">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<title>pdf-inspector — PDF classification &amp; text extraction, no OCR</title>
<meta name="description" content="Fast Rust library that classifies PDFs (text-based vs scanned) and extracts clean Markdown — no OCR, no ML models. Bindings for Rust, Python, and Node.js.">
<meta property="og:title" content="pdf-inspector">
<meta property="og:description" content="Classify PDFs and extract clean Markdown in milliseconds. No OCR. No ML. Pure Rust.">
<meta property="og:type" content="website">
<link rel="icon" href="data:image/svg+xml,%3Csvg xmlns='http://www.w3.org/2000/svg' viewBox='0 0 100 100'%3E%3Ctext y='.9em' font-size='90'%3E%F0%9F%93%84%3C/text%3E%3C/svg%3E">
<link rel="preconnect" href="https://fonts.googleapis.com">
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
<link href="https://fonts.googleapis.com/css2?family=Bricolage+Grotesque:opsz,wght@12..96,400;12..96,600;12..96,800&family=Hanken+Grotesk:wght@400;500;600&family=JetBrains+Mono:wght@400;500;700&display=swap" rel="stylesheet">
<style>
:root {
--paper: #f4efe4;
--paper-2: #eee7d8;
--ink: #1b1712;
--ink-soft: #4a433a;
--muted: #8b8375;
--line: #d9cfbb;
--accent: #dd3f22;
--accent-deep: #b32d15;
--card: #faf6ec;
--display: "Bricolage Grotesque", serif;
--body: "Hanken Grotesk", sans-serif;
--mono: "JetBrains Mono", monospace;
}
* { box-sizing: border-box; margin: 0; padding: 0; }
html { scroll-behavior: smooth; }
body {
background: var(--paper);
color: var(--ink);
font-family: var(--body);
font-size: 17px;
line-height: 1.6;
-webkit-font-smoothing: antialiased;
overflow-x: hidden;
background-image:
radial-gradient(circle at 1px 1px, rgba(27,23,18,0.05) 1px, transparent 0);
background-size: 22px 22px;
}
::selection { background: var(--accent); color: var(--paper); }
a { color: inherit; text-decoration: none; }
.wrap { max-width: 1120px; margin: 0 auto; padding: 0 28px; }
/* ── nav ── */
nav {
position: sticky; top: 0; z-index: 50;
background: rgba(244,239,228,0.82);
backdrop-filter: blur(10px);
border-bottom: 1px solid var(--line);
}
.nav-in { display: flex; align-items: center; gap: 22px; height: 60px; }
.brand { font-family: var(--mono); font-weight: 700; font-size: 15px; letter-spacing: -0.02em; display: flex; align-items: center; gap: 9px; }
.brand .dot { width: 9px; height: 9px; background: var(--accent); border-radius: 50%; box-shadow: 0 0 0 3px rgba(221,63,34,0.18); }
.nav-links { margin-left: auto; display: flex; gap: 24px; align-items: center; font-size: 14.5px; font-weight: 500; }
.nav-links a { color: var(--ink-soft); transition: color .15s; }
.nav-links a:hover { color: var(--accent); }
.nav-gh { border: 1px solid var(--ink); border-radius: 999px; padding: 6px 15px; color: var(--ink) !important; transition: all .15s; }
.nav-gh:hover { background: var(--ink); color: var(--paper) !important; }
@media (max-width: 680px) { .nav-links .hide-sm { display: none; } }
/* ── hero ── */
header { padding: 74px 0 40px; position: relative; }
.eyebrow { font-family: var(--mono); font-size: 12.5px; letter-spacing: 0.16em; text-transform: uppercase; color: var(--accent-deep); margin-bottom: 22px; }
h1 {
font-family: var(--display);
font-weight: 800;
font-size: clamp(2.9rem, 8vw, 6.1rem);
line-height: 0.96;
letter-spacing: -0.035em;
max-width: 15ch;
}
h1 .em { color: var(--accent); font-style: normal; position: relative; }
h1 .strike { position: relative; white-space: nowrap; }
h1 .strike::after { content: ""; position: absolute; left: -2%; right: -2%; top: 54%; height: 0.09em; background: var(--accent); transform: rotate(-3deg); }
.lede { margin-top: 28px; font-size: clamp(1.05rem, 2.2vw, 1.32rem); color: var(--ink-soft); max-width: 46ch; line-height: 1.5; }
.lede b { color: var(--ink); font-weight: 600; }
.hero-grid { display: grid; grid-template-columns: 1.15fr 0.85fr; gap: 48px; align-items: end; }
@media (max-width: 880px) { .hero-grid { grid-template-columns: 1fr; gap: 40px; } }
/* readout card */
.readout {
background: var(--ink); color: var(--paper);
border-radius: 14px; padding: 22px 22px 20px;
font-family: var(--mono); font-size: 13px;
box-shadow: 14px 14px 0 rgba(27,23,18,0.09);
position: relative;
}
.readout .rlabel { color: #b8ad98; font-size: 11px; letter-spacing: 0.14em; text-transform: uppercase; margin-bottom: 16px; display: flex; justify-content: space-between; }
.readout .rrow { display: flex; justify-content: space-between; align-items: center; padding: 9px 0; border-top: 1px solid rgba(255,255,255,0.09); }
.readout .rrow:first-of-type { border-top: none; }
.readout .k { color: #cfc6b3; }
.readout .v { font-weight: 700; }
.readout .v.hot { color: var(--accent); }
.bar { height: 6px; background: rgba(255,255,255,0.1); border-radius: 3px; overflow: hidden; margin-top: 3px; width: 96px; }
.bar > i { display: block; height: 100%; background: var(--accent); border-radius: 3px; }
/* ── install row ── */
.installs { display: grid; grid-template-columns: repeat(3,1fr); gap: 14px; margin-top: 54px; }
@media (max-width: 720px) { .installs { grid-template-columns: 1fr; } }
.inst {
background: var(--card); border: 1px solid var(--line); border-radius: 11px;
padding: 15px 17px; transition: transform .16s, border-color .16s, box-shadow .16s;
cursor: pointer;
}
.inst:hover { transform: translateY(-3px); border-color: var(--accent); box-shadow: 0 8px 22px rgba(27,23,18,0.07); }
.inst .reg { font-family: var(--mono); font-size: 11px; letter-spacing: 0.12em; text-transform: uppercase; color: var(--muted); margin-bottom: 8px; display: flex; justify-content: space-between; }
.inst code { font-family: var(--mono); font-size: 14px; color: var(--ink); font-weight: 500; }
.inst .arrow { color: var(--accent); opacity: 0; transition: opacity .16s; }
.inst:hover .arrow { opacity: 1; }
/* ── section scaffold ── */
section { padding: 66px 0; border-top: 1px solid var(--line); }
.sec-head { display: flex; align-items: baseline; gap: 16px; margin-bottom: 40px; flex-wrap: wrap; }
.sec-num { font-family: var(--mono); font-size: 13px; color: var(--accent); font-weight: 700; }
.sec-title { font-family: var(--display); font-weight: 600; font-size: clamp(1.7rem, 4vw, 2.6rem); letter-spacing: -0.025em; }
.sec-sub { color: var(--ink-soft); max-width: 52ch; font-size: 1.02rem; }
/* ── features ── */
.feat-grid { display: grid; grid-template-columns: repeat(4,1fr); gap: 1px; background: var(--line); border: 1px solid var(--line); border-radius: 14px; overflow: hidden; }
@media (max-width: 900px) { .feat-grid { grid-template-columns: repeat(2,1fr); } }
@media (max-width: 560px) { .feat-grid { grid-template-columns: 1fr; } }
.feat { background: var(--card); padding: 24px 22px; transition: background .16s; }
.feat:hover { background: #fff; }
.feat .fn { font-family: var(--mono); font-size: 12px; color: var(--accent); font-weight: 700; }
.feat h3 { font-family: var(--display); font-weight: 600; font-size: 1.16rem; margin: 12px 0 8px; letter-spacing: -0.01em; }
.feat p { font-size: 14.5px; color: var(--ink-soft); line-height: 1.5; }
/* ── benchmark ── */
.bench {
border: 1px solid var(--line); border-radius: 14px; overflow: hidden;
background: var(--card);
}
table { width: 100%; border-collapse: collapse; font-size: 15px; }
thead th { font-family: var(--mono); font-size: 11px; letter-spacing: 0.08em; text-transform: uppercase; color: var(--muted); text-align: right; padding: 15px 18px; background: var(--paper-2); border-bottom: 1px solid var(--line); font-weight: 500; }
thead th:first-child { text-align: left; }
tbody td { padding: 14px 18px; text-align: right; font-family: var(--mono); border-bottom: 1px solid var(--line); }
tbody td:first-child { text-align: left; font-family: var(--body); font-weight: 500; }
tbody tr:last-child td { border-bottom: none; }
tbody tr.us { background: rgba(221,63,34,0.06); }
tbody tr.us td:first-child { color: var(--accent-deep); font-weight: 700; }
tbody tr.us td:first-child::before { content: "▸ "; color: var(--accent); }
.bench-foot { padding: 15px 18px; font-size: 13.5px; color: var(--ink-soft); background: var(--paper-2); border-top: 1px solid var(--line); }
.bench-wrap { overflow-x: auto; }
.callouts { display: grid; grid-template-columns: 1fr 1fr; gap: 16px; margin-top: 22px; }
@media (max-width: 640px) { .callouts { grid-template-columns: 1fr; } }
.callout { border-left: 3px solid var(--accent); padding: 4px 0 4px 16px; }
.callout .ct { font-family: var(--mono); font-size: 11px; letter-spacing: 0.1em; text-transform: uppercase; color: var(--accent-deep); margin-bottom: 5px; }
.callout p { font-size: 14.5px; color: var(--ink-soft); }
/* ── quickstart tabs ── */
.tabs input { position: absolute; opacity: 0; pointer-events: none; }
.tablist { display: flex; gap: 6px; margin-bottom: 0; }
.tablist label {
font-family: var(--mono); font-size: 13px; font-weight: 500;
padding: 10px 18px; cursor: pointer; color: var(--muted);
border: 1px solid var(--line); border-bottom: none;
border-radius: 9px 9px 0 0; background: var(--paper-2); transition: all .15s;
}
.tablist label:hover { color: var(--ink); }
.panel { display: none; }
.code {
background: var(--ink); border-radius: 0 12px 12px 12px;
padding: 22px 24px; overflow-x: auto;
font-family: var(--mono); font-size: 13.5px; line-height: 1.7;
color: #e9e2d3;
box-shadow: 12px 12px 0 rgba(27,23,18,0.07);
}
.code .cm { color: #8a8069; }
.code .kw { color: #ff9f7a; }
.code .st { color: #cbb78a; }
.code .fn { color: #f4efe4; font-weight: 700; }
#t-rust:checked ~ .tablist label[for=t-rust],
#t-py:checked ~ .tablist label[for=t-py],
#t-node:checked ~ .tablist label[for=t-node],
#t-cli:checked ~ .tablist label[for=t-cli] {
background: var(--ink); color: var(--paper); border-color: var(--ink);
}
#t-rust:checked ~ .panels #p-rust,
#t-py:checked ~ .panels #p-py,
#t-node:checked ~ .panels #p-node,
#t-cli:checked ~ .panels #p-cli { display: block; }
.code a.ref { color: #ff9f7a; border-bottom: 1px dotted #ff9f7a; }
/* ── closing split (OSS vs hosted) ── */
.split { display: grid; grid-template-columns: 1fr 1fr; gap: 18px; }
@media (max-width: 780px) { .split { grid-template-columns: 1fr; } }
.path { border: 1px solid var(--line); border-radius: 16px; padding: 32px 30px; background: var(--card); display: flex; flex-direction: column; }
.path .ptag { font-family: var(--mono); font-size: 11px; letter-spacing: 0.12em; text-transform: uppercase; color: var(--muted); margin-bottom: 15px; }
.path h3 { font-family: var(--display); font-weight: 600; font-size: 1.5rem; letter-spacing: -0.02em; line-height: 1.05; margin-bottom: 12px; }
.path p { color: var(--ink-soft); font-size: 15px; line-height: 1.5; flex: 1; margin-bottom: 24px; }
.pbtns { display: flex; gap: 12px; flex-wrap: wrap; }
.path-pro { background: var(--ink); border-color: var(--ink); box-shadow: 14px 14px 0 rgba(27,23,18,0.09); }
.path-pro .ptag { color: #b8ad98; }
.path-pro .ptag b { color: var(--accent); font-weight: 700; }
.path-pro h3 { color: var(--paper); }
.path-pro p { color: #cfc6b3; }
.btn { font-family: var(--mono); font-size: 14px; font-weight: 500; padding: 13px 24px; border-radius: 999px; transition: all .15s; border: 1px solid var(--ink); }
.btn-p { background: var(--accent); border-color: var(--accent); color: var(--paper); }
.btn-p:hover { background: var(--accent-deep); border-color: var(--accent-deep); }
.btn-s:hover { background: var(--ink); color: var(--paper); }
.btn-pro { background: var(--accent); border-color: var(--accent); color: var(--paper); }
.btn-pro:hover { background: var(--accent-deep); border-color: var(--accent-deep); }
.btn-ghost { color: var(--paper); border-color: rgba(255,255,255,0.3); }
.btn-ghost:hover { border-color: var(--paper); background: rgba(255,255,255,0.08); }
.fc-mark { height: 34px; width: auto; display: block; margin-bottom: 20px; }
.fc-wordmark { height: 15px; width: auto; vertical-align: -2px; transition: opacity .15s; }
.fc-wordmark:hover { opacity: 0.65; }
footer { border-top: 1px solid var(--line); padding: 34px 0; font-size: 14px; color: var(--muted); }
.foot-in { display: flex; justify-content: space-between; align-items: center; gap: 18px; flex-wrap: wrap; }
.foot-in a { color: var(--ink-soft); }
.foot-in a:hover { color: var(--accent); }
.foot-links { display: flex; gap: 20px; font-family: var(--mono); font-size: 13px; }
/* ── load animation ── */
.reveal { opacity: 0; transform: translateY(14px); animation: rise .7s cubic-bezier(.2,.7,.3,1) forwards; }
@keyframes rise { to { opacity: 1; transform: none; } }
.d1 { animation-delay: .05s; } .d2 { animation-delay: .15s; } .d3 { animation-delay: .25s; }
.d4 { animation-delay: .35s; } .d5 { animation-delay: .45s; } .d6 { animation-delay: .55s; }
@media (prefers-reduced-motion: reduce) { .reveal { animation: none; opacity: 1; transform: none; } }
</style>
</head>
<body>
<nav>
<div class="wrap nav-in">
<a href="#top" class="brand"><span class="dot"></span>pdf-inspector</a>
<div class="nav-links">
<a href="#features" class="hide-sm">Features</a>
<a href="#benchmark" class="hide-sm">Benchmark</a>
<a href="#start">Quick start</a>
<a class="nav-gh" href="https://github.com/firecrawl/pdf-inspector">GitHub ↗</a>
</div>
</div>
</nav>
<header id="top">
<div class="wrap hero-grid">
<div>
<div class="eyebrow reveal d1">Rust · Python · Node · CLI</div>
<h1 class="reveal d2">Classify PDFs. Extract Markdown. <span class="strike">No OCR.</span></h1>
<p class="lede reveal d3">A fast Rust library that tells text-based PDFs from scanned ones, then extracts position-aware text and clean Markdown — <b>locally, in milliseconds</b>. Skip the OCR bill for the ~54% of PDFs that never needed it.</p>
</div>
<div class="readout reveal d4" aria-hidden="true">
<div class="rlabel"><span>classify_pdf()</span><span>~12ms</span></div>
<div class="rrow"><span class="k">type</span><span class="v hot">TextBased</span></div>
<div class="rrow"><span class="k">confidence</span><span class="v">0.98</span></div>
<div class="rrow"><span class="k">needs_ocr</span><span class="v">false</span></div>
<div class="rrow"><span class="k">route</span><span class="v">local&nbsp;&nbsp;md</span></div>
<div class="rrow" style="border-top:1px solid rgba(255,255,255,.09);padding-top:13px">
<span class="k">signal</span>
<span class="bar"><i style="width:98%"></i></span>
</div>
</div>
</div>
<div class="wrap">
<div class="installs">
<a class="inst reveal d4" href="https://crates.io/crates/pdf-inspector">
<div class="reg"><span>crates.io</span><span class="arrow"></span></div>
<code>cargo add pdf-inspector</code>
</a>
<a class="inst reveal d5" href="https://pypi.org/project/pdf-inspector/">
<div class="reg"><span>PyPI</span><span class="arrow"></span></div>
<code>pip install pdf-inspector</code>
</a>
<a class="inst reveal d6" href="https://www.npmjs.com/package/@firecrawl/pdf-inspector">
<div class="reg"><span>npm</span><span class="arrow"></span></div>
<code>npm i @firecrawl/pdf-inspector</code>
</a>
</div>
</div>
</header>
<section id="features">
<div class="wrap">
<div class="sec-head">
<span class="sec-num">01</span>
<h2 class="sec-title">Built for routing, not just reading</h2>
</div>
<div class="feat-grid">
<div class="feat"><div class="fn">01</div><h3>Smart classification</h3><p>TextBased, Scanned, ImageBased, or Mixed in ~1050ms by sampling content streams. Returns a confidence score and per-page OCR routing.</p></div>
<div class="feat"><div class="fn">02</div><h3>Position-aware text</h3><p>Extraction with font info, X/Y coordinates, and automatic multi-column reading order.</p></div>
<div class="feat"><div class="fn">03</div><h3>Markdown conversion</h3><p>Headings, bullet/numbered lists, code blocks, tables, bold/italic, URL linking, and page breaks.</p></div>
<div class="feat"><div class="fn">04</div><h3>Table detection</h3><p>Rectangle-based detection from drawing ops plus heuristic alignment detection. Financial tables, footnotes, and cross-page continuations.</p></div>
<div class="feat"><div class="fn">05</div><h3>CID font support</h3><p>ToUnicode CMap decoding for Type0/Identity-H fonts, with UTF-16BE, UTF-8, and Latin-1 encodings.</p></div>
<div class="feat"><div class="fn">06</div><h3>Multi-column layout</h3><p>Newspaper-style column detection, sequential reading order, and right-to-left text support.</p></div>
<div class="feat"><div class="fn">07</div><h3>Encoding checks</h3><p>Flags broken font encodings automatically so callers can fall back to OCR only when it's actually needed.</p></div>
<div class="feat"><div class="fn">08</div><h3>Lightweight</h3><p>Pure Rust. No ML models, no external services. A single parse shared between detection and extraction.</p></div>
</div>
</div>
</section>
<section id="benchmark">
<div class="wrap">
<div class="sec-head">
<span class="sec-num">02</span>
<h2 class="sec-title">Fastest of the direct-text engines</h2>
<p class="sec-sub">Evaluated on the <a href="https://github.com/opendataloader-project/opendataloader-bench" style="color:var(--accent-deep);border-bottom:1px solid var(--line)">opendataloader-bench</a> corpus (200 PDFs). Direct text-extraction engines only — no OCR, no ML. Higher is better.</p>
</div>
<div class="bench">
<div class="bench-wrap">
<table>
<thead>
<tr><th>Engine</th><th>Overall</th><th>Reading order</th><th>Tables</th><th>Headings</th><th>200 docs</th></tr>
</thead>
<tbody>
<tr class="us"><td>pdf-inspector</td><td>0.83</td><td>0.88</td><td>0.66</td><td>0.74</td><td>4s</td></tr>
<tr><td>opendataloader</td><td>0.84</td><td>0.91</td><td>0.49</td><td>0.74</td><td>11s</td></tr>
<tr><td>pymupdf4llm</td><td>0.73</td><td>0.89</td><td>0.40</td><td>0.41</td><td>18s</td></tr>
<tr><td>markitdown</td><td>0.58</td><td>0.88</td><td>0.00</td><td>0.00</td><td>8s</td></tr>
</tbody>
</table>
</div>
<div class="bench-foot">OCR/ML engines (docling, marker, mineru) score 0.830.88 overall — but take 2180 minutes on the same corpus.</div>
</div>
<div class="callouts">
<div class="callout"><div class="ct">Where we win</div><p>Fastest engine measured, the best table detection of any engine here, and heading quality now on par with opendataloader — at ~2.5× its speed.</p></div>
<div class="callout"><div class="ct">Where we're working</div><p>Reading order still trails opendataloader slightly, and tables that need the visual structure only an OCR engine can see.</p></div>
</div>
</div>
</section>
<section id="start">
<div class="wrap">
<div class="sec-head">
<span class="sec-num">03</span>
<h2 class="sec-title">Three lines to Markdown</h2>
</div>
<div class="tabs">
<input type="radio" name="tab" id="t-rust" checked>
<input type="radio" name="tab" id="t-py">
<input type="radio" name="tab" id="t-node">
<input type="radio" name="tab" id="t-cli">
<div class="tablist">
<label for="t-rust">Rust</label>
<label for="t-py">Python</label>
<label for="t-node">Node.js</label>
<label for="t-cli">CLI</label>
</div>
<div class="panels">
<div class="panel" id="p-rust"><pre class="code"><span class="kw">use</span> pdf_inspector::process_pdf;
<span class="kw">let</span> result = <span class="fn">process_pdf</span>(<span class="st">"document.pdf"</span>)?;
<span class="fn">println!</span>(<span class="st">"Type: {:?}"</span>, result.pdf_type);
<span class="kw">if let</span> <span class="kw">Some</span>(markdown) = &amp;result.markdown {
<span class="fn">println!</span>(<span class="st">"{}"</span>, markdown);
}
<span class="cm">// full reference → </span><a class="ref" href="https://github.com/firecrawl/pdf-inspector/blob/main/docs/rust-api.md">docs/rust-api.md</a></pre></div>
<div class="panel" id="p-py"><pre class="code"><span class="kw">import</span> pdf_inspector
result = pdf_inspector.<span class="fn">process_pdf</span>(<span class="st">"document.pdf"</span>)
<span class="fn">print</span>(result.pdf_type) <span class="cm"># "text_based" | "scanned" | "image_based" | "mixed"</span>
<span class="fn">print</span>(result.markdown) <span class="cm"># Markdown string or None</span>
<span class="cm"># full reference → </span><a class="ref" href="https://github.com/firecrawl/pdf-inspector/blob/main/docs/python.md">docs/python.md</a></pre></div>
<div class="panel" id="p-node"><pre class="code"><span class="kw">import</span> { readFileSync } <span class="kw">from</span> <span class="st">'fs'</span>;
<span class="kw">import</span> { processPdf } <span class="kw">from</span> <span class="st">'@firecrawl/pdf-inspector'</span>;
<span class="kw">const</span> result = <span class="fn">processPdf</span>(<span class="fn">readFileSync</span>(<span class="st">'document.pdf'</span>));
console.<span class="fn">log</span>(result.pdfType); <span class="cm">// "TextBased" | "Scanned" | ...</span>
console.<span class="fn">log</span>(result.markdown); <span class="cm">// Markdown string or null</span>
<span class="cm">// full reference → </span><a class="ref" href="https://github.com/firecrawl/pdf-inspector/blob/main/napi/README.md">napi/README.md</a></pre></div>
<div class="panel" id="p-cli"><pre class="code"><span class="cm"># install the CLI tools</span>
cargo <span class="fn">install</span> pdf-inspector
<span class="cm"># convert a PDF to Markdown</span>
<span class="fn">pdf2md</span> document.pdf
<span class="cm"># classify only — is it scanned?</span>
<span class="fn">detect-pdf</span> document.pdf --analyze --json
<span class="cm"># structured output for pipelines</span>
<span class="fn">pdf2md</span> document.pdf --json</pre></div>
</div>
</div>
</div>
</section>
<section class="close">
<div class="wrap">
<div class="sec-head">
<span class="sec-num">04</span>
<h2 class="sec-title">Two ways to parse</h2>
<p class="sec-sub">Run the classifier locally for text-based PDFs; hand the scanned, OCR, and at-scale work to Firecrawl.</p>
</div>
<div class="split">
<div class="path">
<div class="ptag">Open source · runs local</div>
<h3>Use pdf-inspector yourself</h3>
<p>Pure-Rust library and CLI. Classify and extract text-based PDFs on your own machine in milliseconds — no external calls, no OCR bill, MIT licensed.</p>
<div class="pbtns">
<a class="btn btn-p" href="https://github.com/firecrawl/pdf-inspector">Get started</a>
<a class="btn btn-s" href="https://crates.io/crates/pdf-inspector">crates.io</a>
</div>
</div>
<div class="path path-pro">
<img class="fc-mark" src="assets/firecrawl-mark.svg" alt="Firecrawl" width="24" height="34">
<div class="ptag"><b>Firecrawl Parse</b> · hosted API</div>
<h3>Or let Firecrawl handle the hard ones</h3>
<p>Scanned documents, OCR, DOCX / XLSX / HTML, and parsing at scale — clean, LLM-ready Markdown from one API call. The downstream route for everything local parsing can't reach.</p>
<div class="pbtns">
<a class="btn btn-pro" href="https://docs.firecrawl.dev/api-reference/endpoint/parse">Firecrawl Parse ↗</a>
<a class="btn btn-ghost" href="https://firecrawl.dev">firecrawl.dev</a>
</div>
</div>
</div>
</div>
</section>
<footer>
<div class="wrap foot-in">
<div style="display:flex;align-items:center;gap:7px">Built by <a href="https://firecrawl.dev"><img class="fc-wordmark" src="assets/firecrawl-wordmark.svg" alt="Firecrawl"></a> · MIT licensed</div>
<div class="foot-links">
<a href="https://github.com/firecrawl/pdf-inspector">GitHub</a>
<a href="https://crates.io/crates/pdf-inspector">crates.io</a>
<a href="https://pypi.org/project/pdf-inspector/">PyPI</a>
<a href="https://www.npmjs.com/package/@firecrawl/pdf-inspector">npm</a>
</div>
</div>
</footer>
</body>
</html>
+11 -33
View File
@@ -1,12 +1,8 @@
//! CLI tool for detecting PDF type (text-based vs scanned)
use pdf_inspector::{
detect_pdf_type, detector::estimate_page_count_from_bytes, process_pdf_with_options,
PdfOptions, PdfType, ProcessMode,
};
use pdf_inspector::{detect_pdf_type, process_pdf_with_options, PdfOptions, PdfType, ProcessMode};
use std::env;
use std::fmt::Write;
use std::fs;
use std::process;
use std::time::Instant;
@@ -68,32 +64,6 @@ fn pdf_type_str(pdf_type: &PdfType) -> &'static str {
}
}
fn page_count_hint(pdf_path: &str) -> Option<u32> {
fs::read(pdf_path)
.ok()
.map(|bytes| estimate_page_count_from_bytes(&bytes))
.filter(|&count| count > 0)
}
fn print_error(e: &pdf_inspector::PdfError, pdf_path: &str, json_output: bool) {
if json_output {
if let Some(count) = page_count_hint(pdf_path) {
println!(
r#"{{"error":"{}","page_count_hint":{}}}"#,
json_escape(&e.to_string()),
count
);
} else {
println!(r#"{{"error":"{}"}}"#, json_escape(&e.to_string()));
}
} else {
eprintln!("Error: {}", e);
if let Some(count) = page_count_hint(pdf_path) {
eprintln!("Page count hint: {}", count);
}
}
}
fn run_analyze(pdf_path: &str, json_output: bool, start: Instant) {
match process_pdf_with_options(pdf_path, PdfOptions::new().mode(ProcessMode::Analyze)) {
Ok(result) => {
@@ -165,7 +135,11 @@ fn run_analyze(pdf_path: &str, json_output: bool, start: Instant) {
}
}
Err(e) => {
print_error(&e, pdf_path, json_output);
if json_output {
println!(r#"{{"error":"{}"}}"#, e);
} else {
eprintln!("Error: {}", e);
}
process::exit(1);
}
}
@@ -262,7 +236,11 @@ fn run_detect_only(pdf_path: &str, json_output: bool, start: Instant) {
}
}
Err(e) => {
print_error(&e, pdf_path, json_output);
if json_output {
println!(r#"{{"error":"{}"}}"#, e);
} else {
eprintln!("Error: {}", e);
}
process::exit(1);
}
}
+3 -141
View File
@@ -1,10 +1,6 @@
//! CLI tool for PDF to Markdown conversion
use pdf_inspector::extractor::ItemType;
use pdf_inspector::{
extract_text_with_positions_pages, process_pdf_with_options, LayoutComplexity, PdfOptions,
PdfType, ProcessMode, TextItem,
};
use pdf_inspector::{process_pdf_with_options, LayoutComplexity, PdfOptions, PdfType, ProcessMode};
use std::collections::HashSet;
use std::env;
use std::fmt::Write;
@@ -35,110 +31,6 @@ fn json_escape(s: &str) -> String {
out
}
fn format_ocr_reasons_by_page(reasons: &[pdf_inspector::PageOcrReasons]) -> String {
reasons
.iter()
.map(|entry| {
let reasons_json = entry
.reasons
.iter()
.map(|reason| format!(r#""{}""#, json_escape(reason)))
.collect::<Vec<_>>()
.join(",");
format!(r#"{{"page":{},"reasons":[{}]}}"#, entry.page, reasons_json)
})
.collect::<Vec<_>>()
.join(",")
}
fn item_type_label(item_type: &ItemType) -> &'static str {
match item_type {
ItemType::Text => "text",
ItemType::Image => "image",
ItemType::Link(_) => "link",
ItemType::FormField => "form_field",
}
}
fn format_items_json(items: &[TextItem]) -> String {
let underlined_count = items.iter().filter(|item| item.is_underline).count();
let items_json = items
.iter()
.map(|item| {
let mcid = item
.mcid
.map(|value| value.to_string())
.unwrap_or_else(|| "null".to_string());
let link_url = match &item.item_type {
ItemType::Link(url) => format!(r#","url":"{}""#, json_escape(url)),
_ => String::new(),
};
format!(
r#"{{"text":"{}","page":{},"x":{:.2},"y":{:.2},"width":{:.2},"height":{:.2},"font":"{}","font_size":{:.2},"is_bold":{},"is_italic":{},"is_underline":{},"is_strikeout":{},"item_type":"{}","mcid":{}{}}}"#,
json_escape(&item.text),
item.page,
item.x,
item.y,
item.width,
item.height,
json_escape(&item.font),
item.font_size,
item.is_bold,
item.is_italic,
item.is_underline,
item.is_strikeout,
item_type_label(&item.item_type),
mcid,
link_url,
)
})
.collect::<Vec<_>>()
.join(",");
format!(
r#"{{"total_items":{},"underlined_count":{},"items":[{}]}}"#,
items.len(),
underlined_count,
items_json
)
}
#[cfg(test)]
mod tests {
use super::format_items_json;
use pdf_inspector::extractor::ItemType;
use pdf_inspector::TextItem;
#[test]
fn items_json_includes_position_and_underline_metadata() {
let items = vec![TextItem {
text: "A \"quoted\" item".to_string(),
x: 12.345,
y: 67.891,
width: 23.456,
height: 9.876,
font: "F1".to_string(),
font_size: 10.0,
page: 2,
is_bold: false,
is_italic: true,
is_underline: true,
is_strikeout: true,
item_type: ItemType::Text,
mcid: Some(7),
}];
let json = format_items_json(&items);
assert!(json.contains(r#""text":"A \"quoted\" item""#));
assert!(json.contains(r#""page":2"#));
assert!(json.contains(r#""x":12.35"#));
assert!(json.contains(r#""is_underline":true"#));
assert!(json.contains(r#""item_type":"text""#));
assert!(json.contains(r#""mcid":7"#));
}
}
/// Parse a page specification like "1,3,5-10,20" into a HashSet of page numbers.
fn parse_page_spec(spec: &str) -> Result<HashSet<u32>, String> {
let mut pages = HashSet::new();
@@ -196,7 +88,6 @@ fn main() {
if args.len() < 2 {
eprintln!("Usage: {} <pdf_file> [output_file]", args[0]);
eprintln!(" {} <pdf_file> --json", args[0]);
eprintln!(" {} <pdf_file> --items-json", args[0]);
eprintln!(" {} <pdf_file> --raw", args[0]);
eprintln!();
eprintln!("Converts PDF to Markdown with smart type detection.");
@@ -204,11 +95,9 @@ fn main() {
eprintln!();
eprintln!("Options:");
eprintln!(" --json Output result as JSON");
eprintln!(" --items-json Output positioned TextItem JSON");
eprintln!(" --raw Output only markdown (no headers)");
eprintln!(" --pages Insert page break markers (<!-- Page N -->)");
eprintln!(" --select-pages N Only process specified pages (e.g. 1,3,5-10)");
eprintln!(" --password PW Password for an encrypted PDF");
eprintln!(" --detect-only Only detect PDF type (no extraction)");
eprintln!(" --analyze Detect + extract + layout analysis (no markdown)");
process::exit(1);
@@ -216,22 +105,11 @@ fn main() {
let pdf_path = &args[1];
let json_output = args.iter().any(|a| a == "--json");
let items_json_output = args.iter().any(|a| a == "--items-json");
let raw_output = args.iter().any(|a| a == "--raw");
let page_numbers = args.iter().any(|a| a == "--pages");
let detect_only = args.iter().any(|a| a == "--detect-only");
let analyze = args.iter().any(|a| a == "--analyze");
// Parse --password value
let password = args.iter().position(|a| a == "--password").map(|i| {
args.get(i + 1)
.unwrap_or_else(|| {
eprintln!("Error: --password requires a value");
process::exit(1);
})
.clone()
});
// Parse --select-pages value
let page_filter = args
.iter()
@@ -251,17 +129,6 @@ fn main() {
})
});
if items_json_output {
match extract_text_with_positions_pages(pdf_path, page_filter.as_ref()) {
Ok(items) => println!("{}", format_items_json(&items)),
Err(e) => {
println!(r#"{{"error":"{}"}}"#, json_escape(&e.to_string()));
process::exit(1);
}
}
return;
}
let output_file = args
.get(2)
.filter(|a| !a.starts_with("--"))
@@ -280,7 +147,6 @@ fn main() {
if let Some(pages) = page_filter {
options.page_filter = Some(pages);
}
options.password = password;
match process_pdf_with_options(pdf_path, options) {
Ok(result) => {
@@ -311,14 +177,12 @@ fn main() {
.iter()
.map(|p| p.to_string())
.collect();
let ocr_reasons = format_ocr_reasons_by_page(&result.ocr_reasons_by_page);
println!(
r#"{{"pdf_type":"{}","page_count":{},"processing_time_ms":{},"pages_needing_ocr":[{}],"ocr_reasons_by_page":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"has_encoding_issues":{}}}"#,
r#"{{"pdf_type":"{}","page_count":{},"processing_time_ms":{},"pages_needing_ocr":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"has_encoding_issues":{}}}"#,
pdf_type_str,
result.page_count,
result.processing_time_ms,
ocr_pages.join(","),
ocr_reasons,
result.layout.is_complex,
table_pages.join(","),
col_pages.join(","),
@@ -359,9 +223,8 @@ fn main() {
.iter()
.map(|p| p.to_string())
.collect();
let ocr_reasons = format_ocr_reasons_by_page(&result.ocr_reasons_by_page);
println!(
r#"{{"pdf_type":"{}","page_count":{},"has_text":{},"processing_time_ms":{},"markdown_length":{},"pages_needing_ocr":[{}],"ocr_reasons_by_page":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"has_encoding_issues":{},"markdown":"{}"}}"#,
r#"{{"pdf_type":"{}","page_count":{},"has_text":{},"processing_time_ms":{},"markdown_length":{},"pages_needing_ocr":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"has_encoding_issues":{},"markdown":"{}"}}"#,
match result.pdf_type {
PdfType::TextBased => "text_based",
PdfType::Scanned => "scanned",
@@ -373,7 +236,6 @@ fn main() {
result.processing_time_ms,
result.markdown.as_ref().map(|m| m.len()).unwrap_or(0),
ocr_pages.join(","),
ocr_reasons,
result.layout.is_complex,
table_pages.join(","),
col_pages.join(","),
+114 -2011
View File
File diff suppressed because it is too large Load Diff
+41 -624
View File
@@ -14,13 +14,11 @@ use lopdf::{Document, Encoding, Object, ObjectId};
use std::collections::HashMap;
use super::fonts::{
build_font_encodings, build_font_widths, compute_string_width_ts, descriptor_style_flags,
extract_text_from_operand, get_font_file2_obj_num, get_operand_bytes, CMapDecisionCache,
FontStyleCache,
build_font_encodings, build_font_widths, compute_string_width_ts, extract_text_from_operand,
get_font_file2_obj_num, get_operand_bytes, CMapDecisionCache,
};
use super::underline::UnderlineLine;
use super::xobjects::{extract_form_xobject_text, get_page_xobjects, XObjectType};
use super::{get_number, image_bbox_from_ctm, multiply_matrices};
use super::{get_number, multiply_matrices};
/// Strip PDF comments (% to end of line) from content stream bytes.
///
@@ -76,56 +74,6 @@ fn strip_pdf_comments(data: &[u8]) -> Vec<u8> {
result
}
fn transform_path_point(x: f32, y: f32, ctm: &[f32; 6]) -> (f32, f32) {
(
x * ctm[0] + y * ctm[2] + ctm[4],
x * ctm[1] + y * ctm[3] + ctm[5],
)
}
fn transformed_stroke_width(
line_width: f32,
ctm: &[f32; 6],
x1: f32,
y1: f32,
x2: f32,
y2: f32,
) -> f32 {
let user_width = line_width.abs();
let dx = x2 - x1;
let dy = y2 - y1;
let len = (dx * dx + dy * dy).sqrt();
if len <= f32::EPSILON {
return user_width;
}
// PDF stroke width scales perpendicular to the path direction.
let nx = -dy / len;
let ny = dx / len;
let ndx = nx * ctm[0] + ny * ctm[2];
let ndy = nx * ctm[1] + ny * ctm[3];
user_width * (ndx * ndx + ndy * ndy).sqrt()
}
/// Text rise (Ts) displaces the glyph origin by (0, rise) in unscaled text
/// space — per the rendering-matrix definition it sits left of Tm, so the
/// offset maps through the text matrix's y column. Rise never contributes
/// to the advance, so callers apply it only to the rendering position and
/// keep advancing the unshifted text matrix.
fn rise_adjusted(tm: &[f32; 6], rise: f32) -> [f32; 6] {
if rise == 0.0 {
return *tm;
}
[
tm[0],
tm[1],
tm[2],
tm[3],
tm[4] + rise * tm[2],
tm[5] + rise * tm[3],
]
}
/// Returns `(page_extraction, has_gid_fonts)` where `has_gid_fonts` indicates
/// the page uses fonts with unresolvable gid-encoded glyphs.
pub(crate) fn extract_page_text_items(
@@ -134,7 +82,6 @@ pub(crate) fn extract_page_text_items(
page_num: u32,
font_cmaps: &FontCMaps,
include_invisible: bool,
style_cache: &mut FontStyleCache,
) -> Result<(PageExtraction, bool, bool), PdfError> {
use lopdf::content::Content;
@@ -142,7 +89,6 @@ pub(crate) fn extract_page_text_items(
let mut rects: Vec<PdfRect> = Vec::new();
let mut clip_rects: Vec<PdfRect> = Vec::new();
let mut lines: Vec<PdfLine> = Vec::new();
let mut underline_lines: Vec<UnderlineLine> = Vec::new();
// Path construction state for m/l/h → S/s line extraction
let mut path_subpath_start: Option<(f32, f32)> = None;
@@ -151,12 +97,6 @@ pub(crate) fn extract_page_text_items(
// Completed subpaths (each a vec of line segments) for f/f* rect extraction
let mut pending_subpaths: Vec<Vec<(f32, f32, f32, f32)>> = Vec::new();
let mut fill_rects: Vec<PdfRect> = Vec::new();
// `re` rects awaiting a paint operator. Underline detection must only
// see painted rects: a `re W n` clip path or `re n` no-op draws nothing
// on the page, so treating every `re` as ink would underline text that
// merely sits near an invisible clip boundary.
let mut pending_re_rects: Vec<PdfRect> = Vec::new();
let mut painted_rects: Vec<PdfRect> = Vec::new();
// Get fonts for encoding
let fonts = doc.get_page_fonts(page_id).unwrap_or_default();
@@ -174,8 +114,6 @@ pub(crate) fn extract_page_text_items(
std::collections::HashMap::new();
let mut inline_cmaps: std::collections::HashMap<String, crate::tounicode::CMapEntry> =
std::collections::HashMap::new();
let mut font_style_flags: std::collections::HashMap<String, (bool, bool)> =
std::collections::HashMap::new();
for (font_name, font_dict) in &fonts {
let resource_name = String::from_utf8_lossy(font_name).to_string();
if let Ok(base_font) = font_dict.get(b"BaseFont") {
@@ -184,12 +122,6 @@ pub(crate) fn extract_page_text_items(
font_base_names.insert(resource_name.clone(), base_name);
}
}
// Descriptor style flags rescue subset fonts whose BaseFont names
// are opaque tags the name heuristics can't read.
let style = descriptor_style_flags(doc, font_dict, style_cache);
if style != (false, false) {
font_style_flags.insert(resource_name.clone(), style);
}
// Track ToUnicode object reference, with FontFile2 fallback for Identity-H/V.
// Also handle inline ToUnicode streams.
match font_dict.get(b"ToUnicode") {
@@ -256,20 +188,7 @@ pub(crate) fn extract_page_text_items(
// Graphics state tracking
let mut ctm = [1.0f32, 0.0, 0.0, 1.0, 0.0, 0.0]; // Current Transformation Matrix
let mut text_rendering_mode: i32 = 0; // 0=fill, 1=stroke, 2=fill+stroke, 3=invisible
let mut line_width: f32 = 1.0;
#[derive(Clone)]
struct SavedGraphicsState {
ctm: [f32; 6],
text_rendering_mode: i32,
line_width: f32,
char_spacing: f32,
word_spacing: f32,
text_rise: f32,
text_leading: f32,
current_font: String,
current_font_size: f32,
}
let mut gstate_stack: Vec<SavedGraphicsState> = Vec::new();
let mut gstate_stack: Vec<([f32; 6], i32, f32, f32)> = Vec::new();
// Text state tracking
let mut current_font = String::new();
@@ -277,7 +196,6 @@ pub(crate) fn extract_page_text_items(
let mut text_leading: f32 = 0.0; // TL parameter (in text-space units)
let mut char_spacing: f32 = 0.0; // Tc parameter (extra spacing per character, unscaled)
let mut word_spacing: f32 = 0.0; // Tw parameter (extra spacing per space char, unscaled)
let mut text_rise: f32 = 0.0; // Ts parameter (baseline shift for super/subscripts, unscaled)
let mut text_matrix = [1.0f32, 0.0, 0.0, 1.0, 0.0, 0.0];
let mut line_matrix = [1.0f32, 0.0, 0.0, 1.0, 0.0, 0.0];
let mut in_text_block = false;
@@ -298,11 +216,6 @@ pub(crate) fn extract_page_text_items(
let mut marked_content_stack: Vec<MarkedContentEntry> = Vec::new();
let mut suppress_glyph_extraction = false;
let mut actual_text_start_tm: Option<[f32; 6]> = None; // text matrix at BDC entry
let mut actual_text_glyph_tm: Option<[f32; 6]> = None; // text matrix at first glyph inside BDC
// Text rise in effect at each captured matrix — the item must render at
// the rise of its GLYPHS, not whatever rise is set by EMC time.
let mut actual_text_start_rise: f32 = 0.0;
let mut actual_text_glyph_rise: Option<f32> = None;
/// Get the innermost MCID from the marked content stack.
fn current_mcid(stack: &[MarkedContentEntry]) -> Option<i64> {
stack.iter().rev().find_map(|e| e.mcid)
@@ -313,30 +226,15 @@ pub(crate) fn extract_page_text_items(
match op.operator.as_str() {
"q" => {
// Save graphics state
gstate_stack.push(SavedGraphicsState {
ctm,
text_rendering_mode,
line_width,
char_spacing,
word_spacing,
text_rise,
text_leading,
current_font: current_font.clone(),
current_font_size,
});
gstate_stack.push((ctm, text_rendering_mode, char_spacing, word_spacing));
}
"Q" => {
// Restore graphics state
if let Some(saved) = gstate_stack.pop() {
ctm = saved.ctm;
text_rendering_mode = saved.text_rendering_mode;
line_width = saved.line_width;
char_spacing = saved.char_spacing;
word_spacing = saved.word_spacing;
text_rise = saved.text_rise;
text_leading = saved.text_leading;
current_font = saved.current_font;
current_font_size = saved.current_font_size;
if let Some((saved_ctm, saved_tr, saved_tc, saved_tw)) = gstate_stack.pop() {
ctm = saved_ctm;
text_rendering_mode = saved_tr;
char_spacing = saved_tc;
word_spacing = saved_tw;
}
}
"cm" => {
@@ -353,11 +251,6 @@ pub(crate) fn extract_page_text_items(
ctm = multiply_matrices(&new_matrix, &ctm);
}
}
"w" => {
if let Some(width) = op.operands.first().and_then(get_number) {
line_width = width;
}
}
"BT" => {
// Begin text block
in_text_block = true;
@@ -406,12 +299,6 @@ pub(crate) fn extract_page_text_items(
word_spacing = tw;
}
}
"Ts" => {
// Set text rise (baseline shift for superscripts/subscripts)
if let Some(ts) = op.operands.first().and_then(get_number) {
text_rise = ts;
}
}
"Td" | "TD" => {
// Move text position: TLM = T(tx,ty) × TLM; Tm = TLM
// tx,ty are in text space — must be scaled by the text line matrix
@@ -462,16 +349,8 @@ pub(crate) fn extract_page_text_items(
)
})
});
// ActualText: suppress glyph extraction, just advance text matrix.
// Capture the FIRST glyph's text matrix as the rendering position
// for the ActualText item. Td ops between BDC and the first Tj
// may have moved the position to the correct line — the BDC-entry
// position (actual_text_start_tm) can be on the previous line.
// ActualText: suppress glyph extraction, just advance text matrix
if suppress_glyph_extraction {
if actual_text_glyph_tm.is_none() {
actual_text_glyph_tm = Some(text_matrix);
actual_text_glyph_rise = Some(text_rise);
}
if let Some(w_ts) = w_ts_opt {
text_matrix[4] += w_ts * text_matrix[0];
text_matrix[5] += w_ts * text_matrix[1];
@@ -498,10 +377,8 @@ pub(crate) fn extract_page_text_items(
&font_encodings,
&encoding_cache,
&mut cmap_decisions,
&font_widths,
) {
let combined =
multiply_matrices(&rise_adjusted(&text_matrix, text_rise), &ctm);
let combined = multiply_matrices(&text_matrix, &ctm);
let rendered_size = effective_font_size(current_font_size, &combined);
let (x, y) = (combined[4], combined[5]);
if combined[0].abs() >= combined[1].abs() {
@@ -523,10 +400,6 @@ pub(crate) fn extract_page_text_items(
.get(&current_font)
.map(|s| s.as_str())
.unwrap_or(&current_font);
let (desc_italic, desc_bold) = font_style_flags
.get(&current_font)
.copied()
.unwrap_or((false, false));
items.push(TextItem {
text: expand_ligatures(&text),
x,
@@ -536,10 +409,8 @@ pub(crate) fn extract_page_text_items(
font: current_font.clone(),
font_size: rendered_size,
page: page_num,
is_bold: is_bold_font(base_font) || desc_bold,
is_italic: is_italic_font(base_font) || desc_italic,
is_underline: false,
is_strikeout: false,
is_bold: is_bold_font(base_font),
is_italic: is_italic_font(base_font),
item_type: ItemType::Text,
mcid: current_mcid(&marked_content_stack),
});
@@ -554,11 +425,6 @@ pub(crate) fn extract_page_text_items(
let font_info = font_widths.get(&current_font);
let is_invisible = (text_rendering_mode == 3 && !include_invisible)
|| suppress_glyph_extraction;
// Capture first-glyph position for ActualText
if suppress_glyph_extraction && actual_text_glyph_tm.is_none() {
actual_text_glyph_tm = Some(text_matrix);
actual_text_glyph_rise = Some(text_rise);
}
// Compute space threshold based on font metrics when available
let space_threshold = if let Some(font_info) = font_info {
@@ -655,7 +521,6 @@ pub(crate) fn extract_page_text_items(
&font_encodings,
&encoding_cache,
&mut cmap_decisions,
&font_widths,
) {
current_text.push_str(&text);
}
@@ -678,10 +543,6 @@ pub(crate) fn extract_page_text_items(
.get(&current_font)
.map(|s| s.as_str())
.unwrap_or(&current_font);
let (desc_italic, desc_bold) = font_style_flags
.get(&current_font)
.copied()
.unwrap_or((false, false));
let scale_x = text_matrix[0] * ctm[0] + text_matrix[1] * ctm[2];
for (text, start_w, end_w) in &sub_items {
let offset_tm = [
@@ -692,8 +553,7 @@ pub(crate) fn extract_page_text_items(
text_matrix[4] + start_w * text_matrix[0],
text_matrix[5] + start_w * text_matrix[1],
];
let combined =
multiply_matrices(&rise_adjusted(&offset_tm, text_rise), &ctm);
let combined = multiply_matrices(&offset_tm, &ctm);
let (x, y) = (combined[4], combined[5]);
let width = if font_info.is_some() {
((end_w - start_w) * scale_x).abs()
@@ -709,10 +569,8 @@ pub(crate) fn extract_page_text_items(
font: current_font.clone(),
font_size: rendered_size,
page: page_num,
is_bold: is_bold_font(base_font) || desc_bold,
is_italic: is_italic_font(base_font) || desc_italic,
is_underline: false,
is_strikeout: false,
is_bold: is_bold_font(base_font),
is_italic: is_italic_font(base_font),
item_type: ItemType::Text,
mcid: current_mcid(&marked_content_stack),
});
@@ -736,26 +594,6 @@ pub(crate) fn extract_page_text_items(
line_matrix[4] += (-tl) * line_matrix[2];
line_matrix[5] += (-tl) * line_matrix[3];
text_matrix = line_matrix;
// Capture first-glyph position for ActualText AFTER the
// line move — the BDC-entry matrix is on the previous line.
if suppress_glyph_extraction && actual_text_glyph_tm.is_none() {
actual_text_glyph_tm = Some(text_matrix);
actual_text_glyph_rise = Some(text_rise);
}
// Advance width, as for Tj — without it the item stays
// zero-width and geometric underline/strikeout detection
// rejects it (`is_underline_candidate` needs width > 0).
let w_ts_opt = font_widths.get(&current_font).and_then(|fi| {
op.operands.first().and_then(get_operand_bytes).map(|raw| {
compute_string_width_ts(
raw,
fi,
current_font_size,
char_spacing,
word_spacing,
)
})
});
if !((text_rendering_mode == 3 && !include_invisible)
|| suppress_glyph_extraction
|| op.operands.is_empty())
@@ -770,11 +608,9 @@ pub(crate) fn extract_page_text_items(
&font_encodings,
&encoding_cache,
&mut cmap_decisions,
&font_widths,
) {
if !text.trim().is_empty() {
let combined =
multiply_matrices(&rise_adjusted(&text_matrix, text_rise), &ctm);
let combined = multiply_matrices(&text_matrix, &ctm);
if combined[0].abs() >= combined[1].abs() {
rotation_votes.horizontal += 1;
} else {
@@ -782,45 +618,27 @@ pub(crate) fn extract_page_text_items(
}
let rendered_size = effective_font_size(current_font_size, &combined);
let (x, y) = (combined[4], combined[5]);
let width = w_ts_opt
.map(|w_ts| {
(w_ts * (text_matrix[0] * ctm[0] + text_matrix[1] * ctm[2]))
.abs()
})
.unwrap_or(0.0);
let base_font = font_base_names
.get(&current_font)
.map(|s| s.as_str())
.unwrap_or(&current_font);
let (desc_italic, desc_bold) = font_style_flags
.get(&current_font)
.copied()
.unwrap_or((false, false));
items.push(TextItem {
text: expand_ligatures(&text),
x,
y,
width,
width: 0.0,
height: rendered_size,
font: current_font.clone(),
font_size: rendered_size,
page: page_num,
is_bold: is_bold_font(base_font) || desc_bold,
is_italic: is_italic_font(base_font) || desc_italic,
is_underline: false,
is_strikeout: false,
is_bold: is_bold_font(base_font),
is_italic: is_italic_font(base_font),
item_type: ItemType::Text,
mcid: current_mcid(&marked_content_stack),
});
}
}
}
// Advance regardless of visibility so later show-text
// operators on the same line stay positioned (as for Tj).
if let Some(w_ts) = w_ts_opt {
text_matrix[4] += w_ts * text_matrix[0];
text_matrix[5] += w_ts * text_matrix[1];
}
}
"Do" => {
// XObject invocation - could be an image or form
@@ -831,31 +649,7 @@ pub(crate) fn extract_page_text_items(
if let Some(xobj_type) = xobjects.get(&xobj_name) {
match xobj_type {
XObjectType::Image => {
// Emit a positional placeholder for the image
// so downstream consumers (layout-aware
// pipelines, figure-OCR routers) can locate
// raster figures without parsing the PDF
// again. The text field carries the
// XObject resource name in the legacy
// `[Image: Im0]` format that the markdown
// emitter already recognizes.
let (x, y, width, height) = image_bbox_from_ctm(&ctm);
items.push(TextItem {
text: format!("[Image: {}]", xobj_name),
x,
y,
width,
height,
font: String::new(),
font_size: 0.0,
page: page_num,
is_bold: false,
is_italic: false,
is_underline: false,
is_strikeout: false,
item_type: ItemType::Image,
mcid: current_mcid(&marked_content_stack),
});
// Skip images — text extraction only
}
XObjectType::Form(form_id) => {
// Extract text from Form XObject
@@ -866,7 +660,6 @@ pub(crate) fn extract_page_text_items(
font_cmaps,
&ctm,
&mut cmap_decisions,
style_cache,
);
items.extend(form_items);
}
@@ -907,9 +700,6 @@ pub(crate) fn extract_page_text_items(
if actual_text.is_some() {
suppress_glyph_extraction = true;
actual_text_start_tm = Some(text_matrix);
actual_text_start_rise = text_rise;
actual_text_glyph_tm = None; // reset — will be captured at first Tj/TJ
actual_text_glyph_rise = None;
}
marked_content_stack.push(MarkedContentEntry { actual_text, mcid });
}
@@ -917,16 +707,9 @@ pub(crate) fn extract_page_text_items(
// End Marked Content — emit ActualText item with correct width
if let Some(entry) = marked_content_stack.pop() {
if let Some(at) = entry.actual_text {
// Use the first-glyph position (if available) instead of the
// BDC-entry position. Td operators between BDC and the first
// Tj may have moved the text position to the correct line —
// the BDC-entry position can be on the previous line.
let glyph_tm = actual_text_glyph_tm.take();
let glyph_rise = actual_text_glyph_rise.take();
let entry_tm = actual_text_start_tm.take();
if let Some(start_tm) = glyph_tm.or(entry_tm) {
let rise = glyph_rise.unwrap_or(actual_text_start_rise);
let combined = multiply_matrices(&rise_adjusted(&start_tm, rise), &ctm);
// Compute width from text matrix advancement during BDC..EMC
if let Some(start_tm) = actual_text_start_tm.take() {
let combined = multiply_matrices(&start_tm, &ctm);
if combined[0].abs() >= combined[1].abs() {
rotation_votes.horizontal += 1;
} else {
@@ -943,10 +726,6 @@ pub(crate) fn extract_page_text_items(
.get(&current_font)
.map(|s| s.as_str())
.unwrap_or(&current_font);
let (desc_italic, desc_bold) = font_style_flags
.get(&current_font)
.copied()
.unwrap_or((false, false));
items.push(TextItem {
text: expand_ligatures(&at),
x,
@@ -956,10 +735,8 @@ pub(crate) fn extract_page_text_items(
font: current_font.clone(),
font_size: rendered_size,
page: page_num,
is_bold: is_bold_font(base_font) || desc_bold,
is_italic: is_italic_font(base_font) || desc_italic,
is_underline: false,
is_strikeout: false,
is_bold: is_bold_font(base_font),
is_italic: is_italic_font(base_font),
item_type: ItemType::Text,
mcid: entry
.mcid
@@ -984,19 +761,13 @@ pub(crate) fn extract_page_text_items(
let y_dev = rx * ctm[1] + ry * ctm[3] + ctm[5];
let w_dev = rw * ctm[0];
let h_dev = rh * ctm[3];
let rect = PdfRect {
rects.push(PdfRect {
x: x_dev,
y: y_dev,
width: w_dev,
height: h_dev,
page: page_num,
};
// Underline detection must only see rects that are
// actually painted — a `re` used purely as a clip path
// (`re W n`) or discarded (`re n`) draws nothing. Hold
// the rect as pending until a paint operator confirms it.
pending_re_rects.push(rect.clone());
rects.push(rect);
});
}
}
// ── Path construction operators ──────────────────────
@@ -1046,8 +817,10 @@ pub(crate) fn extract_page_text_items(
}
}
for (x1, y1, x2, y2) in pending_lines.drain(..) {
let (x1d, y1d) = transform_path_point(x1, y1, &ctm);
let (x2d, y2d) = transform_path_point(x2, y2, &ctm);
let x1d = x1 * ctm[0] + y1 * ctm[2] + ctm[4];
let y1d = x1 * ctm[1] + y1 * ctm[3] + ctm[5];
let x2d = x2 * ctm[0] + y2 * ctm[2] + ctm[4];
let y2d = x2 * ctm[1] + y2 * ctm[3] + ctm[5];
lines.push(PdfLine {
x1: x1d,
y1: y1d,
@@ -1055,16 +828,7 @@ pub(crate) fn extract_page_text_items(
y2: y2d,
page: page_num,
});
underline_lines.push(UnderlineLine {
x1: x1d,
y1: y1d,
x2: x2d,
y2: y2d,
stroke_width: transformed_stroke_width(line_width, &ctm, x1, y1, x2, y2),
page: page_num,
});
}
painted_rects.append(&mut pending_re_rects);
pending_subpaths.clear();
path_subpath_start = None;
path_current = None;
@@ -1080,8 +844,10 @@ pub(crate) fn extract_page_text_items(
}
}
for (x1, y1, x2, y2) in pending_lines.drain(..) {
let (x1d, y1d) = transform_path_point(x1, y1, &ctm);
let (x2d, y2d) = transform_path_point(x2, y2, &ctm);
let x1d = x1 * ctm[0] + y1 * ctm[2] + ctm[4];
let y1d = x1 * ctm[1] + y1 * ctm[3] + ctm[5];
let x2d = x2 * ctm[0] + y2 * ctm[2] + ctm[4];
let y2d = x2 * ctm[1] + y2 * ctm[3] + ctm[5];
lines.push(PdfLine {
x1: x1d,
y1: y1d,
@@ -1089,16 +855,7 @@ pub(crate) fn extract_page_text_items(
y2: y2d,
page: page_num,
});
underline_lines.push(UnderlineLine {
x1: x1d,
y1: y1d,
x2: x2d,
y2: y2d,
stroke_width: transformed_stroke_width(line_width, &ctm, x1, y1, x2, y2),
page: page_num,
});
}
painted_rects.append(&mut pending_re_rects);
pending_subpaths.clear();
path_subpath_start = None;
path_current = None;
@@ -1156,7 +913,6 @@ pub(crate) fn extract_page_text_items(
}
}
}
painted_rects.append(&mut pending_re_rects);
pending_lines.clear();
path_subpath_start = None;
path_current = None;
@@ -1222,10 +978,7 @@ pub(crate) fn extract_page_text_items(
// Do NOT clear pending_lines — the following `n` does that
}
"n" => {
// end path (no-op): discard — including any `re` rects that
// were only ever part of a clip path (`re W n`), which draw
// no ink and must not feed underline detection.
pending_re_rects.clear();
// end path (no-op): discard
pending_lines.clear();
pending_subpaths.clear();
path_subpath_start = None;
@@ -1235,29 +988,15 @@ pub(crate) fn extract_page_text_items(
}
}
// Underline detection reads only painted ink: `re` rects confirmed by
// a paint operator plus filled-subpath rects — never clip-only rects,
// which draw nothing.
let mut underline_rects = painted_rects;
underline_rects.extend(fill_rects.iter().cloned());
// Only use clip/fill rects when no `re` rects exist on this page.
// Clip rects take priority over fill rects, but first we deduplicate
// them: some PDFs wrap every text block in a full-page W* clip path,
// producing thousands of identical rects that yield a degenerate grid.
// After dedup, if too few unique clip rects remain we fall through to
// fill rects (explicitly drawn visible rectangles).
//
// When fill rects substantially outnumber clip rects, the clips are
// typically section-level wrappers and the fills are the actual table
// cell backgrounds (e.g. shaded-header tables drawn with `m`/`l`/`h`/`f*`
// sequences). In that case, prefer fills.
if rects.is_empty() {
dedup_rects(&mut clip_rects);
let prefer_fills = !fill_rects.is_empty() && fill_rects.len() >= clip_rects.len() * 3;
if prefer_fills {
rects = fill_rects;
} else if clip_rects.len() >= 4 {
if clip_rects.len() >= 4 {
rects = clip_rects;
} else if !fill_rects.is_empty() {
rects = fill_rects;
@@ -1270,17 +1009,8 @@ pub(crate) fn extract_page_text_items(
// Some PDFs embed landscape content in portrait pages using a rotated text
// matrix (e.g. [0, b, -b, 0, tx, ty] for 90° CCW). The layout engine
// assumes x=horizontal, y=vertical — so we swap coordinates to match.
let (mut items, rects, lines, coords_rotated) =
let (items, rects, lines, coords_rotated) =
correct_rotated_page(items, rects, lines, &rotation_votes);
if coords_rotated {
rotate_underline_graphics(&mut underline_rects, &mut underline_lines);
}
super::underline::mark_underlined_items(
&mut items,
&underline_rects,
&underline_lines,
page_num,
);
let items = super::merge_text_items(items);
let items = super::merge_subscript_items(items);
@@ -1366,27 +1096,6 @@ fn correct_rotated_page(
(items, rects, lines, true)
}
fn rotate_underline_graphics(rects: &mut [PdfRect], lines: &mut [UnderlineLine]) {
for rect in rects {
let new_x = rect.y;
let new_y = -(rect.x + rect.width.abs());
rect.x = new_x;
rect.y = new_y;
std::mem::swap(&mut rect.width, &mut rect.height);
}
for line in lines {
let new_x1 = line.y1;
let new_y1 = -line.x1;
let new_x2 = line.y2;
let new_y2 = -line.x2;
line.x1 = new_x1;
line.y1 = new_y1;
line.x2 = new_x2;
line.y2 = new_y2;
}
}
/// Remove near-duplicate rects (same coordinates within 0.5 pt tolerance).
/// Some PDFs emit a full-page clip path for every text block, producing
/// thousands of identical rects. After dedup these collapse to one rect,
@@ -1436,64 +1145,6 @@ mod tests {
}
}
fn simple_doc_with_content(content: &[u8]) -> (lopdf::Document, lopdf::ObjectId) {
use lopdf::{dictionary, Object, Stream};
let mut doc = lopdf::Document::new();
let widths: Vec<Object> = (0..=255).map(|_| 600.into()).collect();
let font_id = doc.add_object(dictionary! {
"Type" => "Font",
"Subtype" => "Type1",
"BaseFont" => "Helvetica",
"FirstChar" => 0,
"LastChar" => 255,
"Widths" => Object::Array(widths),
});
let content_id = doc.add_object(Object::Stream(Stream::new(
dictionary! {},
content.to_vec(),
)));
let page_id = doc.add_object(dictionary! {
"Type" => "Page",
"Contents" => Object::Reference(content_id),
"Resources" => dictionary! {
"Font" => dictionary! {
"F1" => Object::Reference(font_id),
},
},
"MediaBox" => vec![0.into(), 0.into(), 612.into(), 792.into()],
});
let pages_id = doc.add_object(dictionary! {
"Type" => "Pages",
"Count" => Object::Integer(1),
"Kids" => vec![Object::Reference(page_id)],
});
let catalog_id = doc.add_object(dictionary! {
"Type" => "Catalog",
"Pages" => Object::Reference(pages_id),
});
doc.trailer.set("Root", Object::Reference(catalog_id));
(doc, page_id)
}
fn extract_simple_items(content: &[u8]) -> Vec<TextItem> {
use crate::tounicode::FontCMaps;
let (doc, page_id) = simple_doc_with_content(content);
let font_cmaps = FontCMaps::from_doc(&doc);
let ((items, _, _), _, _) = extract_page_text_items(
&doc,
page_id,
1,
&font_cmaps,
false,
&mut FontStyleCache::new(),
)
.unwrap();
items
}
#[test]
fn test_dedup_rects_identical() {
let mut rects = vec![rect(0.0, 0.0, 612.0, 792.0, 1); 3759];
@@ -1543,140 +1194,6 @@ mod tests {
assert_eq!(single.len(), 1);
}
#[test]
fn thick_stroked_rule_does_not_mark_underline() {
let content = b"BT /F1 12 Tf 1 0 0 1 100 500 Tm (THICK) Tj ET
4 w
100 498 m 170 498 l S
BT /F1 12 Tf 1 0 0 1 100 480 Tm (THIN) Tj ET
1 w
100 478 m 160 478 l S";
let items = extract_simple_items(content);
let thick = items.iter().find(|item| item.text == "THICK").unwrap();
let thin = items.iter().find(|item| item.text == "THIN").unwrap();
assert!(!thick.is_underline);
assert!(thin.is_underline);
}
#[test]
fn rotated_page_underline_is_detected_after_coordinate_correction() {
let content = b"BT /F1 12 Tf 0 1 -1 0 200 100 Tm (HELLO) Tj ET
BT /F1 12 Tf 0 1 -1 0 240 100 Tm (WORLD) Tj ET
1 w
202 100 m 202 170 l S";
let items = extract_simple_items(content);
let hello = items.iter().find(|item| item.text == "HELLO").unwrap();
let world = items.iter().find(|item| item.text == "WORLD").unwrap();
assert!(hello.is_underline);
assert!(!world.is_underline);
}
#[test]
fn quote_operator_text_carries_advance_width() {
// `'` (move-to-next-line-and-show-text) must retain the string's
// advance width like Tj — zero-width items are invisible to
// geometric underline/strikeout detection.
let content = b"BT /F1 12 Tf 12 TL 1 0 0 1 100 512 Tm (first) Tj (struck) ' ET
1 w
99 503 m 145 503 l S";
let items = extract_simple_items(content);
let struck = items.iter().find(|item| item.text == "struck").unwrap();
// 6 glyphs x 600/1000 x 12pt = 43.2pt, drawn one leading below Tm.
assert!((struck.width - 43.2).abs() < 0.1);
assert!((struck.y - 500.0).abs() < 0.1);
assert!(struck.is_strikeout);
assert!(!struck.is_underline);
}
#[test]
fn quote_operator_advances_text_matrix() {
// Text shown after `'` on the same line must start past the shown
// string: "CD" lands at x=114.4 (2 glyphs x 600/1000 x 12pt after
// x=100), flush against "AB", so the merge pass joins them. Without
// the advance "CD" overlaps "AB" at x=100 and the items stay apart.
let content = b"BT /F1 12 Tf 12 TL 1 0 0 1 100 512 Tm (AB) ' (CD) Tj ET";
let items = extract_simple_items(content);
let merged = items.iter().find(|item| item.text == "ABCD").unwrap();
assert!((merged.x - 100.0).abs() < 0.1);
assert!((merged.width - 28.8).abs() < 0.1);
assert!((merged.y - 500.0).abs() < 0.1);
}
#[test]
fn text_rise_shifts_item_baseline() {
// Ts displaces the glyph origin vertically without touching the
// advance; the next run at rise 0 must return to the original
// baseline and follow the raised run horizontally.
let content =
b"BT /F1 12 Tf 1 0 0 1 100 500 Tm (base) Tj 5 Ts (super) Tj 0 Ts (after) Tj ET";
let items = extract_simple_items(content);
let base = items.iter().find(|item| item.text == "base").unwrap();
let raised = items.iter().find(|item| item.text == "super").unwrap();
let after = items.iter().find(|item| item.text == "after").unwrap();
assert!((base.y - 500.0).abs() < 0.1);
assert!((raised.y - 505.0).abs() < 0.1);
assert!((after.y - 500.0).abs() < 0.1);
assert!(after.x > raised.x);
}
#[test]
fn actual_text_item_uses_glyph_rise() {
// The ActualText replacement item must render at the rise in
// effect when its glyphs were drawn — not the unshifted BDC
// baseline, and not whatever rise is set by EMC time.
let content = b"BT /F1 12 Tf 1 0 0 1 100 500 Tm \
/Span <</ActualText (super) >> BDC 5 Ts (sup) Tj 0 Ts EMC (after) Tj ET";
let items = extract_simple_items(content);
let sup = items.iter().find(|item| item.text == "super").unwrap();
let after = items.iter().find(|item| item.text == "after").unwrap();
assert!((sup.y - 505.0).abs() < 0.1);
assert!((after.y - 500.0).abs() < 0.1);
}
#[test]
fn actual_text_shown_with_quote_op_uses_moved_risen_baseline() {
// When the tagged span's show op is `'`, the glyph position is
// only known AFTER its line move — falling back to the BDC-entry
// matrix would place the item on the previous line, unrisen.
let content = b"BT /F1 12 Tf 14 TL 1 0 0 1 100 500 Tm \
/Span <</ActualText (replaced) >> BDC 3 Ts (raw) ' 0 Ts EMC ET";
let items = extract_simple_items(content);
let item = items.iter().find(|item| item.text == "replaced").unwrap();
// Line move: 500 - 14 = 486; rise: +3 -> 489.
assert!((item.y - 489.0).abs() < 0.1);
assert!(item.width > 0.0);
}
#[test]
fn strikeout_detected_on_risen_text() {
// The rule crosses the glyphs at their risen position; without the
// rise in item.y the strike window sits 4pt too low and misses.
let content = b"BT /F1 12 Tf 1 0 0 1 100 500 Tm 4 Ts (struck) Tj ET
1 w
99 507 m 145 507 l S";
let items = extract_simple_items(content);
let struck = items.iter().find(|item| item.text == "struck").unwrap();
assert!((struck.y - 504.0).abs() < 0.1);
assert!(struck.is_strikeout);
assert!(!struck.is_underline);
}
#[test]
fn test_skip_excessive_operations() {
use crate::tounicode::FontCMaps;
@@ -1711,113 +1228,13 @@ BT /F1 12 Tf 0 1 -1 0 240 100 Tm (WORLD) Tj ET
doc.add_object(catalog);
let font_cmaps = FontCMaps::from_doc(&doc);
let result = extract_page_text_items(
&doc,
page_id,
1,
&font_cmaps,
false,
&mut FontStyleCache::new(),
)
.unwrap();
let result = extract_page_text_items(&doc, page_id, 1, &font_cmaps, false).unwrap();
let ((items, rects, lines), _has_gid, _coords_rotated) = result;
assert!(items.is_empty());
assert!(rects.is_empty());
assert!(lines.is_empty());
}
#[test]
fn test_q_restores_current_font_for_text_decoding() {
use crate::tounicode::FontCMaps;
use lopdf::{dictionary, Object, Stream};
fn cmap_stream(dst_hex: &str) -> Stream {
let cmap = format!(
r#"/CIDInit /ProcSet findresource begin
12 dict begin
begincmap
/CIDSystemInfo << /Registry (Adobe) /Ordering (UCS) /Supplement 0 >> def
/CMapName /Test-UCS def
/CMapType 2 def
1 begincodespacerange
<00> <FF>
endcodespacerange
1 beginbfchar
<41> <{dst_hex}>
endbfchar
endcmap
CMapName currentdict /CMap defineresource pop
end
end"#
);
Stream::new(dictionary! {}, cmap.into_bytes())
}
let mut doc = lopdf::Document::new();
let f1_cmap = doc.add_object(Object::Stream(cmap_stream("0058"))); // X
let f2_cmap = doc.add_object(Object::Stream(cmap_stream("0059"))); // Y
let f1 = doc.add_object(dictionary! {
"Type" => "Font",
"Subtype" => "Type1",
"BaseFont" => "Helvetica",
"ToUnicode" => Object::Reference(f1_cmap),
});
let f2 = doc.add_object(dictionary! {
"Type" => "Font",
"Subtype" => "Type1",
"BaseFont" => "Helvetica",
"ToUnicode" => Object::Reference(f2_cmap),
});
let content = b"BT /F1 12 Tf 10 700 Tm <41> Tj ET
q
BT /F2 12 Tf 20 700 Tm <41> Tj ET
Q
BT 30 700 Tm <41> Tj ET";
let content_id = doc.add_object(Object::Stream(Stream::new(
dictionary! {},
content.to_vec(),
)));
let page_id = doc.add_object(dictionary! {
"Type" => "Page",
"Contents" => Object::Reference(content_id),
"Resources" => dictionary! {
"Font" => dictionary! {
"F1" => Object::Reference(f1),
"F2" => Object::Reference(f2),
},
},
"MediaBox" => vec![0.into(), 0.into(), 612.into(), 792.into()],
});
let pages_id = doc.add_object(dictionary! {
"Type" => "Pages",
"Count" => Object::Integer(1),
"Kids" => vec![Object::Reference(page_id)],
});
let catalog_id = doc.add_object(dictionary! {
"Type" => "Catalog",
"Pages" => Object::Reference(pages_id),
});
doc.trailer.set("Root", Object::Reference(catalog_id));
let font_cmaps = FontCMaps::from_doc(&doc);
let ((items, _, _), _, _) = extract_page_text_items(
&doc,
page_id,
1,
&font_cmaps,
false,
&mut FontStyleCache::new(),
)
.unwrap();
let text = items
.iter()
.map(|item| item.text.as_str())
.collect::<String>();
assert_eq!(text, "XYX");
}
#[test]
fn test_strip_pdf_comments() {
// Basic comment stripping
+16 -699
View File
@@ -4,7 +4,7 @@ use crate::glyph_names::glyph_to_char;
use crate::tounicode::FontCMaps;
use crate::types::{FontEncodingMap, FontWidthInfo, PageFontEncodings, PageFontWidths};
use log::debug;
use lopdf::{Document, Encoding, Object, ObjectId};
use lopdf::{Document, Encoding, Object};
use std::collections::HashMap;
#[derive(Debug, Copy, Clone, PartialEq, Eq)]
@@ -526,11 +526,6 @@ pub(crate) fn parse_font_encoding(
font_dict: &lopdf::Dictionary,
) -> Option<EncodingResult> {
let encoding_obj = font_dict.get(b"Encoding").ok()?;
let base_font_name = font_dict
.get(b"BaseFont")
.ok()
.and_then(|o| o.as_name().ok())
.map(|n| String::from_utf8_lossy(n).to_string());
// Encoding can be a name or a dictionary
match encoding_obj {
@@ -543,14 +538,12 @@ pub(crate) fn parse_font_encoding(
Object::Reference(obj_ref) => {
// Reference to encoding dictionary
if let Ok(enc_dict) = doc.get_dictionary(*obj_ref) {
parse_encoding_dictionary(doc, enc_dict, base_font_name.as_deref())
parse_encoding_dictionary(doc, enc_dict)
} else {
None
}
}
Object::Dictionary(enc_dict) => {
parse_encoding_dictionary(doc, enc_dict, base_font_name.as_deref())
}
Object::Dictionary(enc_dict) => parse_encoding_dictionary(doc, enc_dict),
_ => None,
}
}
@@ -569,7 +562,6 @@ pub(crate) struct EncodingResult {
pub(crate) fn parse_encoding_dictionary(
doc: &Document,
enc_dict: &lopdf::Dictionary,
base_font_name: Option<&str>,
) -> Option<EncodingResult> {
let differences = enc_dict.get(b"Differences").ok()?;
@@ -599,9 +591,11 @@ pub(crate) fn parse_encoding_dictionary(
Object::Name(name) => {
// Map current code to glyph name -> Unicode
let glyph_name = String::from_utf8_lossy(&name).to_string();
let mapped_char = glyph_to_char(&glyph_name)
.or_else(|| private_glyph_to_char(&glyph_name, base_font_name));
if mapped_char.is_some_and(is_ligature_char) {
if glyph_name == "fi"
|| glyph_name == "fl"
|| glyph_name == "ffi"
|| glyph_name == "ffl"
{
debug!(
" Differences: code=0x{:02X} glyph={:?} (ligature)",
current_code, glyph_name
@@ -616,7 +610,7 @@ pub(crate) fn parse_encoding_dictionary(
{
gid_glyph_count += 1;
}
if let Some(ch) = mapped_char {
if let Some(ch) = glyph_to_char(&glyph_name) {
encoding_map.insert(current_code, ch);
} else {
debug!(
@@ -651,31 +645,6 @@ pub(crate) fn parse_encoding_dictionary(
})
}
fn private_glyph_to_char(glyph_name: &str, base_font_name: Option<&str>) -> Option<char> {
let base_font_name = strip_subset_prefix(base_font_name?);
// Aptos CFF subsets from Office PDFs can expose the ff ligature as /g431
// without a ToUnicode map. Keep this font-scoped because /gNNN names are private.
if base_font_name.eq_ignore_ascii_case("Aptos") && glyph_name == "g431" {
Some('\u{FB00}')
} else {
None
}
}
fn strip_subset_prefix(font_name: &str) -> &str {
font_name
.split_once('+')
.map_or(font_name, |(_, stripped)| stripped)
}
fn is_ligature_char(ch: char) -> bool {
matches!(
ch,
'\u{FB00}' | '\u{FB01}' | '\u{FB02}' | '\u{FB03}' | '\u{FB04}'
)
}
/// Get the CMap lookup key for an Identity-H/V CID font without ToUnicode.
/// Returns the object number used by `collect_cmaps_from_fonts` to store the CMap:
/// - FontFile2 or FontFile3 obj_num (for embedded font cmap)
@@ -739,171 +708,6 @@ pub(crate) fn get_font_file2_obj_num(doc: &Document, font_dict: &lopdf::Dictiona
.map(|r| r.0)
}
/// Document-scoped memo of embedded-font style flags, keyed by the
/// FontFile2/FontFile3 stream's object id. The same font program is
/// referenced from every page that uses the font, and decompressing +
/// parsing it dominates `descriptor_style_flags` — without the memo that
/// cost repeats per page whenever the descriptor leaves a flag unset
/// (the common case: regular fonts report neither italic nor bold).
#[derive(Debug, Default)]
pub(crate) struct FontStyleCache {
by_font_file: HashMap<ObjectId, (bool, bool)>,
}
impl FontStyleCache {
pub(crate) fn new() -> Self {
Self::default()
}
}
/// Style flags from the FontDescriptor, which survive subset fonts whose
/// BaseFont names are opaque tags ("Tc1", "ABCDEF+F1") that defeat the
/// name-based bold/italic heuristics.
///
/// Italic: `ItalicAngle` beyond a few degrees, or Flags bit 7 (Italic,
/// value 64). Bold: Flags bit 19 (ForceBold, value 1<<18). The small
/// ItalicAngle threshold skips fonts that declare a token slant.
pub(crate) fn descriptor_style_flags(
doc: &Document,
font_dict: &lopdf::Dictionary,
style_cache: &mut FontStyleCache,
) -> (bool, bool) {
let descriptor = font_dict
.get(b"FontDescriptor")
.ok()
.and_then(|obj| resolve_dict(doc, obj))
.or_else(|| {
// Type0 fonts hang the descriptor off DescendantFonts[0].
let desc_fonts = font_dict.get(b"DescendantFonts").ok()?;
let desc_fonts = resolve_array(doc, desc_fonts)?;
let cid_font_dict = resolve_dict(doc, desc_fonts.first()?)?;
resolve_dict(doc, cid_font_dict.get(b"FontDescriptor").ok()?)
});
let Some(descriptor) = descriptor else {
return (false, false);
};
let italic_angle = descriptor
.get(b"ItalicAngle")
.ok()
.and_then(|obj| match obj {
Object::Integer(i) => Some(*i as f32),
Object::Real(r) => Some(*r),
_ => None,
})
.unwrap_or(0.0);
let flags = descriptor
.get(b"Flags")
.ok()
.and_then(|obj| obj.as_i64().ok())
.unwrap_or(0);
let mut italic = italic_angle.abs() >= 4.0 || flags & (1 << 6) != 0;
let mut bold = flags & (1 << 18) != 0;
// Descriptors lie: subset generators write ItalicAngle 0 for genuinely
// italic faces. The embedded font file keeps the truth — OS/2
// fsSelection (via `Face::is_italic`) and the post table's italicAngle.
if !italic || !bold {
if let Some(ff_ref) = font_file_ref(descriptor) {
let (emb_italic, emb_bold) = *style_cache
.by_font_file
.entry(ff_ref)
.or_insert_with(|| embedded_style_flags(doc, ff_ref));
italic = italic || emb_italic;
bold = bold || emb_bold;
}
}
(italic, bold)
}
/// Style flags parsed from an embedded font program stream.
fn embedded_style_flags(doc: &Document, ff_ref: ObjectId) -> (bool, bool) {
let Some(data) = font_file_data(doc, ff_ref) else {
return (false, false);
};
if let Ok(face) = ttf_parser::Face::parse(&data, 0) {
(
face.is_italic() || face.italic_angle().abs() >= 4.0,
face.is_bold(),
)
} else if let Some(name) = cff_font_name(&data) {
// FontFile3 is bare CFF (no sfnt container) — ttf_parser
// can't open it, but the CFF Name INDEX keeps the real
// PostScript name ("XXXXXX+Amplitude-LightItalic") even
// when the descriptor was rewritten to claim upright.
(
crate::text_utils::is_italic_font(&name),
crate::text_utils::is_bold_font(&name),
)
} else {
(false, false)
}
}
/// First PostScript name from a bare CFF font's Name INDEX (CFF spec §7).
fn cff_font_name(data: &[u8]) -> Option<String> {
// Header: major(1) minor(1) hdrSize(1) offSize(1); major must be 1.
if data.len() < 4 || data[0] != 1 {
return None;
}
let hdr_size = data[2] as usize;
// Name INDEX: count(u16) offSize(u8) offsets[count+1] data
let count = u16::from_be_bytes([*data.get(hdr_size)?, *data.get(hdr_size + 1)?]) as usize;
if count == 0 {
return None;
}
let off_size = *data.get(hdr_size + 2)? as usize;
if !(1..=4).contains(&off_size) {
return None;
}
let read_offset = |idx: usize| -> Option<usize> {
let at = hdr_size + 3 + idx * off_size;
let bytes = data.get(at..at + off_size)?;
let mut v = 0usize;
for b in bytes {
v = (v << 8) | *b as usize;
}
Some(v)
};
let start = read_offset(0)?;
let end = read_offset(1)?;
if start == 0 || end < start {
return None;
}
// Offsets are 1-based from the byte before the object data.
let objects_base = hdr_size + 3 + (count + 1) * off_size - 1;
let name = data.get(objects_base + start..objects_base + end)?;
Some(String::from_utf8_lossy(name).to_string())
}
/// FontFile2/FontFile3 stream reference from a FontDescriptor.
fn font_file_ref(descriptor: &lopdf::Dictionary) -> Option<ObjectId> {
descriptor
.get(b"FontFile2")
.ok()
.and_then(|o| o.as_reference().ok())
.or_else(|| {
descriptor
.get(b"FontFile3")
.ok()
.and_then(|o| o.as_reference().ok())
})
}
/// Decompressed embedded font program bytes.
fn font_file_data(doc: &Document, ff_ref: ObjectId) -> Option<Vec<u8>> {
let stream = doc
.get_object(ff_ref)
.and_then(lopdf::Object::as_stream)
.ok()?;
Some(
stream
.decompressed_content()
.unwrap_or_else(|_| stream.content.clone()),
)
}
/// Decode text from a PDF string operand using font CMaps, encodings, and fallbacks.
#[allow(clippy::too_many_arguments)]
pub(crate) fn extract_text_from_operand(
@@ -916,13 +720,7 @@ pub(crate) fn extract_text_from_operand(
font_encodings: &PageFontEncodings,
encoding_cache: &HashMap<String, Encoding<'_>>,
cmap_decisions: &mut CMapDecisionCache,
font_widths: &PageFontWidths,
) -> Option<String> {
let is_type0_cid_font = font_widths
.get(current_font)
.is_some_and(|info| info.is_cid);
let use_cp1252_fallback =
should_use_cp1252_single_byte_fallback(base_font_name, is_type0_cid_font);
let result = (|| -> Option<String> {
if let Object::String(bytes, _) = obj {
let mut decode_with_entry = |entry: &crate::tounicode::CMapEntry| -> Option<String> {
@@ -953,12 +751,9 @@ pub(crate) fn extract_text_from_operand(
return Some(ch.to_string());
}
}
// 4. Printable single-byte fallback
// 4. Printable ASCII/Latin-1 fallback
if b >= 0x20 {
return Some(
decode_single_byte_fallback_char(b, use_cp1252_fallback)
.to_string(),
);
return Some((b as char).to_string());
}
None
})
@@ -1059,13 +854,6 @@ pub(crate) fn extract_text_from_operand(
// unmapped. Don't fall through to text-interpretation fallbacks
// (Latin-1, UTF-16, etc.) which would misinterpret CID bytes as
// character codes (e.g. CID 0x01A9 → Latin-1 "©").
if is_type0_cid_font && bytes.iter().any(|&b| b > 0x7F) {
// 2-byte CIDs (Identity-H) are by far the common case; for
// an odd byte count we still emit at least one marker so
// detection downstream fires.
let cid_count = (bytes.len() / 2).max(1);
return Some("\u{FFFD}".repeat(cid_count));
}
// Try our custom encoding map from Differences arrays.
// The Differences array overrides specific codes in a base encoding (typically
@@ -1081,9 +869,8 @@ pub(crate) fn extract_text_from_operand(
Some(ch)
} else if b >= 0x20 {
// Base encoding fallback for printable bytes.
// Most PDFs with simple fonts use WinAnsi/PDFDocEncoding
// semantics, not ISO-8859-1 C1 controls.
Some(decode_single_byte_fallback_char(b, use_cp1252_fallback))
// For codes 0x20-0x7E this matches all standard PDF encodings.
Some(b as char)
} else {
None // Skip unmapped control characters
}
@@ -1139,7 +926,6 @@ pub(crate) fn extract_text_from_operand(
// Try to decode using cached font encoding from lopdf
if let Some(encoding) = encoding_cache.get(current_font) {
if let Ok(text) = Document::decode_text(encoding, bytes) {
let text = normalize_cp1252_controls(text, use_cp1252_fallback);
if text.contains('\u{FFFD}') {
debug!(
"decode_text produced replacement for font={} bytes_len={}",
@@ -1176,119 +962,13 @@ pub(crate) fn extract_text_from_operand(
return Some(symbol_text);
}
// Non-CID (Type1 / TrueType / Type3) fonts use single-byte
// encodings. In practice the fallback should follow WinAnsi for
// 0x80..=0x9F so bytes like 0x92 become smart punctuation instead
// of C1 controls that look like CID mojibake.
Some(decode_single_byte_fallback(bytes, use_cp1252_fallback))
// Latin-1 fallback
Some(bytes.iter().map(|&b| b as char).collect())
} else {
None
}
})();
result.map(|text| {
let text = clean_symbol_pua(text);
normalize_cp1252_controls(text, use_cp1252_fallback)
})
}
fn decode_single_byte_fallback(bytes: &[u8], use_cp1252_fallback: bool) -> String {
bytes
.iter()
.map(|&b| decode_single_byte_fallback_char(b, use_cp1252_fallback))
.collect()
}
fn decode_single_byte_fallback_char(byte: u8, use_cp1252_fallback: bool) -> char {
if !use_cp1252_fallback {
return byte as char;
}
match byte {
0x80 => '\u{20AC}',
0x82 => '\u{201A}',
0x83 => '\u{0192}',
0x84 => '\u{201E}',
0x85 => '\u{2026}',
0x86 => '\u{2020}',
0x87 => '\u{2021}',
0x88 => '\u{02C6}',
0x89 => '\u{2030}',
0x8A => '\u{0160}',
0x8B => '\u{2039}',
0x8C => '\u{0152}',
0x8E => '\u{017D}',
0x91 => '\u{2018}',
0x92 => '\u{2019}',
0x93 => '\u{201C}',
0x94 => '\u{201D}',
0x95 => '\u{2022}',
0x96 => '\u{2013}',
0x97 => '\u{2014}',
0x98 => '\u{02DC}',
0x99 => '\u{2122}',
0x9A => '\u{0161}',
0x9B => '\u{203A}',
0x9C => '\u{0153}',
0x9E => '\u{017E}',
0x9F => '\u{0178}',
_ => byte as char,
}
}
fn normalize_cp1252_controls(text: String, use_cp1252_fallback: bool) -> String {
if !use_cp1252_fallback {
return text;
}
if !text
.chars()
.any(|ch| ('\u{0080}'..='\u{009F}').contains(&ch))
{
return text;
}
text.chars()
.map(|ch| {
if ('\u{0080}'..='\u{009F}').contains(&ch) {
decode_single_byte_fallback_char(ch as u8, true)
} else {
ch
}
})
.collect()
}
fn should_use_cp1252_single_byte_fallback(
base_font_name: Option<&str>,
is_type0_cid_font: bool,
) -> bool {
if is_type0_cid_font {
return false;
}
let Some(base_font_name) = base_font_name else {
return true;
};
let font_name = base_font_name
.rsplit_once('+')
.map_or(base_font_name, |(_, stripped)| stripped)
.to_ascii_lowercase();
// TeX/Computer Modern and math/symbol fonts often place ligatures or
// symbols in the C1 byte range. Treating those bytes as Windows-1252 makes
// words like "deficiente" become "de…ciente" and "fluid" become "‡uid".
let non_cp1252_prefixes = [
"cmr", "cmb", "cmmi", "cmsy", "cmex", "cmtt", "cmss", "cmti", "ecrm", "ecbx", "ecti",
"tcrm", "tctt", "msam", "msbm", "ttdc",
];
if non_cp1252_prefixes
.iter()
.any(|prefix| font_name.starts_with(prefix))
{
return false;
}
let non_cp1252_names = ["math", "symbol", "dingbat", "emoji"];
!non_cp1252_names.iter().any(|name| font_name.contains(name))
result.map(clean_symbol_pua)
}
/// Replace PUA characters in the F000-F0FF range with standard Unicode equivalents.
@@ -1410,7 +1090,6 @@ fn score_text(text: &str) -> i32 {
#[cfg(test)]
mod tests {
use super::*;
use lopdf::dictionary;
fn make_font_info(widths: &[(u16, u16)], default_width: u16, is_cid: bool) -> FontWidthInfo {
FontWidthInfo {
@@ -1427,172 +1106,6 @@ mod tests {
}
}
fn doc_with_descriptor(descriptor: lopdf::Dictionary) -> (Document, lopdf::Dictionary) {
let mut doc = Document::with_version("1.4");
let desc_id = doc.add_object(descriptor);
let font_dict = dictionary! {
"Type" => "Font",
"Subtype" => "TrueType",
"BaseFont" => "Tc1",
"FontDescriptor" => desc_id,
};
(doc, font_dict)
}
#[test]
fn descriptor_italic_angle_sets_italic() {
// Subset font with an opaque BaseFont name ("Tc1") — the name
// heuristic sees nothing, the descriptor carries the truth.
let (doc, font_dict) = doc_with_descriptor(dictionary! {
"Type" => "FontDescriptor",
"FontName" => "Tc1",
"ItalicAngle" => -12,
"Flags" => 32,
});
assert_eq!(
descriptor_style_flags(&doc, &font_dict, &mut FontStyleCache::new()),
(true, false)
);
}
#[test]
fn descriptor_italic_flag_bit_sets_italic() {
let (doc, font_dict) = doc_with_descriptor(dictionary! {
"Type" => "FontDescriptor",
"FontName" => "Tc1",
"ItalicAngle" => 0,
"Flags" => 64, // bit 7: Italic
});
assert_eq!(
descriptor_style_flags(&doc, &font_dict, &mut FontStyleCache::new()),
(true, false)
);
}
#[test]
fn descriptor_force_bold_flag_sets_bold() {
let (doc, font_dict) = doc_with_descriptor(dictionary! {
"Type" => "FontDescriptor",
"FontName" => "Tc1",
"ItalicAngle" => 0,
"Flags" => 1 << 18, // ForceBold
});
assert_eq!(
descriptor_style_flags(&doc, &font_dict, &mut FontStyleCache::new()),
(false, true)
);
}
#[test]
fn tiny_italic_angle_is_not_italic() {
// A token 1-degree slant is optical correction, not italic.
let (doc, font_dict) = doc_with_descriptor(dictionary! {
"Type" => "FontDescriptor",
"FontName" => "Tc1",
"ItalicAngle" => lopdf::Object::Real(-1.0),
"Flags" => 32,
});
assert_eq!(
descriptor_style_flags(&doc, &font_dict, &mut FontStyleCache::new()),
(false, false)
);
}
#[test]
fn missing_descriptor_yields_no_flags() {
let doc = Document::with_version("1.4");
let font_dict = dictionary! { "Type" => "Font", "BaseFont" => "Tc1" };
assert_eq!(
descriptor_style_flags(&doc, &font_dict, &mut FontStyleCache::new()),
(false, false)
);
}
#[test]
fn type0_descendant_descriptor_is_resolved() {
let mut doc = Document::with_version("1.4");
let desc_id = doc.add_object(dictionary! {
"Type" => "FontDescriptor",
"FontName" => "ABCDEF+F1",
"ItalicAngle" => -15,
});
let cid_id = doc.add_object(dictionary! {
"Type" => "Font",
"Subtype" => "CIDFontType2",
"FontDescriptor" => desc_id,
});
let font_dict = dictionary! {
"Type" => "Font",
"Subtype" => "Type0",
"BaseFont" => "ABCDEF+F1",
"DescendantFonts" => vec![lopdf::Object::Reference(cid_id)],
};
assert_eq!(
descriptor_style_flags(&doc, &font_dict, &mut FontStyleCache::new()),
(true, false)
);
}
/// Bare CFF: header + Name INDEX only — enough for `cff_font_name`.
fn bare_cff_with_name(name: &str) -> Vec<u8> {
let mut data = vec![1, 0, 4, 1]; // major, minor, hdrSize, offSize
data.extend_from_slice(&1u16.to_be_bytes()); // Name INDEX count
data.push(1); // offSize
data.push(1); // offset of first name
data.push(1 + name.len() as u8); // offset past last name
data.extend_from_slice(name.as_bytes());
data
}
#[test]
fn embedded_font_style_is_cached_by_font_file_object() {
use lopdf::{Object, Stream};
let mut doc = Document::with_version("1.4");
let ff_id = doc.add_object(Object::Stream(Stream::new(
dictionary! {},
bare_cff_with_name("ABCDEF+Test-BoldItalic"),
)));
let desc_id = doc.add_object(dictionary! {
"Type" => "FontDescriptor",
"FontName" => "ABCDEF+Test-BoldItalic",
"ItalicAngle" => 0,
"Flags" => 32,
"FontFile3" => ff_id,
});
let font_dict = dictionary! {
"Type" => "Font",
"Subtype" => "Type1",
"BaseFont" => "Tc1",
"FontDescriptor" => desc_id,
};
let mut cache = FontStyleCache::new();
assert_eq!(
descriptor_style_flags(&doc, &font_dict, &mut cache),
(true, true)
);
assert_eq!(cache.by_font_file.len(), 1);
// Replace the font program with garbage: a repeat call must serve
// the memo instead of re-reading the stream — repeated per-page
// decompression is exactly what the cache exists to avoid.
doc.objects.insert(
ff_id,
Object::Stream(Stream::new(dictionary! {}, vec![0u8; 4])),
);
assert_eq!(
descriptor_style_flags(&doc, &font_dict, &mut cache),
(true, true)
);
// A cold cache parses the (now garbage) stream, proving the warm
// call above answered from the memo.
assert_eq!(
descriptor_style_flags(&doc, &font_dict, &mut FontStyleCache::new()),
(false, false)
);
}
#[test]
fn compute_string_width_ts_no_tc_tw() {
// Without Tc/Tw (both 0), width = glyph widths only
@@ -1700,200 +1213,4 @@ mod tests {
let bad = "###!!!@@@$$$";
assert!(score_text(good) > score_text(bad));
}
fn doc_with_private_differences() -> (Document, lopdf::ObjectId) {
let mut doc = Document::with_version("1.7");
let encoding_id = doc.add_object(dictionary! {
"Differences" => Object::Array(vec![
Object::Integer(0x88),
Object::Name(b"g431".to_vec()),
Object::Name(b"fi".to_vec()),
Object::Integer(0xAD),
Object::Name(b"fl".to_vec()),
]),
});
(doc, encoding_id)
}
#[test]
fn aptos_private_g431_maps_to_ff_ligature() {
let (doc, encoding_id) = doc_with_private_differences();
let font_dict = dictionary! {
"BaseFont" => Object::Name(b"NJEQOD+Aptos".to_vec()),
"Encoding" => Object::Reference(encoding_id),
};
let result = parse_font_encoding(&doc, &font_dict).expect("encoding should parse");
assert_eq!(result.map.get(&0x88u8), Some(&'\u{FB00}'));
assert_eq!(result.map.get(&0x89u8), Some(&'\u{FB01}'));
assert_eq!(result.map.get(&0xADu8), Some(&'\u{FB02}'));
}
#[test]
fn private_g431_does_not_map_for_unrelated_fonts() {
let (doc, encoding_id) = doc_with_private_differences();
let font_dict = dictionary! {
"BaseFont" => Object::Name(b"ABCDEF+OtherFont".to_vec()),
"Encoding" => Object::Reference(encoding_id),
};
let result = parse_font_encoding(&doc, &font_dict).expect("encoding should parse");
assert!(!result.map.contains_key(&0x88u8));
assert_eq!(result.map.get(&0x89u8), Some(&'\u{FB01}'));
assert_eq!(result.map.get(&0xADu8), Some(&'\u{FB02}'));
}
#[test]
fn cid_font_with_unparseable_cmap_does_not_emit_latin1_mojibake() {
// Type0/CID font (font_widths reports `is_cid=true`) where the
// ToUnicode CMap couldn't be parsed (FontCMaps doesn't have the
// obj_num). Bytes are a 2-byte CID stream containing high bytes
// that aren't valid UTF-8 — exactly the case in the production
// samples (Identity-H text where the ToUnicode CMap was missing
// or malformed, scrape_id 019de78c-..., e.g. "Í Ù Z)¿").
//
// Without the guard, the function falls through to the byte-by-byte
// Latin-1 fallback and produces "ÍÙ" (U+00CD U+00D9). The correct
// behavior is to emit U+FFFD per CID so downstream
// `detect_encoding_issues` flags the page for OCR.
let bytes = vec![0xCD_u8, 0xD9, 0xCD, 0xD9];
let obj = Object::String(bytes, lopdf::StringFormat::Hexadecimal);
let font_cmaps = FontCMaps::default();
let mut font_tounicode_refs: HashMap<String, u32> = HashMap::new();
font_tounicode_refs.insert("F0".to_string(), 999);
let inline_cmaps = HashMap::new();
let font_encodings: PageFontEncodings = HashMap::new();
let encoding_cache: HashMap<String, Encoding<'_>> = HashMap::new();
let mut decisions = CMapDecisionCache::new();
let mut font_widths: PageFontWidths = HashMap::new();
font_widths.insert("F0".to_string(), make_font_info(&[], 1000, true));
let result = extract_text_from_operand(
&obj,
"F0",
None,
&font_cmaps,
&font_tounicode_refs,
&inline_cmaps,
&font_encodings,
&encoding_cache,
&mut decisions,
&font_widths,
);
let text = result.expect("CID font fallback should still emit a marker");
assert!(
!text.contains('\u{00CD}') && !text.contains('\u{00D9}'),
"CID font with unparseable CMap leaked Latin-1 mojibake: {text:?}"
);
assert!(
text.contains('\u{FFFD}'),
"CID font with unparseable CMap should emit U+FFFD so detect_encoding_issues fires: {text:?}"
);
}
#[test]
fn simple_font_single_byte_fallback_passes_high_bytes_through() {
// A Type1/TrueType simple font (is_cid=false) with a `/ToUnicode`
// reference but no usable CMap and no `/Differences` map.
// Per-byte fallback is the canonical interpretation here — these
// bytes are character codes, not CIDs. The CID guard must NOT strip
// them. Reproduces the false positive that an earlier version of the
// guard introduced for fonts in PDFs like pdf-evals/Navigating-
// Artificial-Intelligence-..., where bytes like 0xB6 are legitimate
// single-byte character codes.
let bytes = vec![0x24_u8, 0x47, 0xB6, 0x56]; // "$G¶V"
let obj = Object::String(bytes, lopdf::StringFormat::Hexadecimal);
let font_cmaps = FontCMaps::default();
let mut font_tounicode_refs: HashMap<String, u32> = HashMap::new();
font_tounicode_refs.insert("F1".to_string(), 999);
let inline_cmaps = HashMap::new();
let font_encodings: PageFontEncodings = HashMap::new();
let encoding_cache: HashMap<String, Encoding<'_>> = HashMap::new();
let mut decisions = CMapDecisionCache::new();
let mut font_widths: PageFontWidths = HashMap::new();
font_widths.insert("F1".to_string(), make_font_info(&[], 1000, false));
let text = extract_text_from_operand(
&obj,
"F1",
None,
&font_cmaps,
&font_tounicode_refs,
&inline_cmaps,
&font_encodings,
&encoding_cache,
&mut decisions,
&font_widths,
)
.expect("simple font should round-trip Latin-1 bytes");
assert_eq!(text, "$G\u{00B6}V");
assert!(
!text.contains('\u{FFFD}'),
"simple font fallback must not stamp FFFD over legitimate bytes: {text:?}"
);
}
#[test]
fn simple_font_single_byte_fallback_maps_cp1252_punctuation() {
let bytes = vec![b'l', 0x92_u8, b'a', b'c', b'a', b'd'];
let obj = Object::String(bytes, lopdf::StringFormat::Hexadecimal);
let font_cmaps = FontCMaps::default();
let font_tounicode_refs: HashMap<String, u32> = HashMap::new();
let inline_cmaps = HashMap::new();
let font_encodings: PageFontEncodings = HashMap::new();
let encoding_cache: HashMap<String, Encoding<'_>> = HashMap::new();
let mut decisions = CMapDecisionCache::new();
let font_widths: PageFontWidths = HashMap::new();
let text = extract_text_from_operand(
&obj,
"F1",
None,
&font_cmaps,
&font_tounicode_refs,
&inline_cmaps,
&font_encodings,
&encoding_cache,
&mut decisions,
&font_widths,
)
.expect("simple font should decode CP1252 punctuation");
assert_eq!(text, "lacad");
}
#[test]
fn cached_encoding_decode_normalizes_cp1252_controls() {
let text = normalize_cp1252_controls("d\u{92}un \u{96} test".to_string(), true);
assert_eq!(text, "dun test");
}
#[test]
fn tex_font_decode_keeps_c1_ligature_bytes_unmodified() {
let text = normalize_cp1252_controls("de\u{85}ciente \u{87}uid".to_string(), false);
assert_eq!(text, "de\u{85}ciente \u{87}uid");
assert!(!should_use_cp1252_single_byte_fallback(
Some("TTdcr10"),
false
));
assert!(!should_use_cp1252_single_byte_fallback(
Some("cmr10"),
false
));
}
#[test]
fn winansi_text_font_uses_cp1252_fallback() {
assert!(should_use_cp1252_single_byte_fallback(
Some("BJPQNQ+Times-Roman"),
false
));
}
}
+7 -108
View File
@@ -29,12 +29,8 @@ pub(crate) fn detect_columns(
const MIN_ITEMS_PER_COLUMN: usize = 10;
const NOISE_FRACTION: f32 = 0.15;
// Get items for this page. Strip Image placeholders — an image's left edge
// would otherwise count toward the column projection profile.
let page_items: Vec<&TextItem> = items
.iter()
.filter(|i| i.page == page && crate::extractor::is_text_layout_item(i))
.collect();
// Get items for this page
let page_items: Vec<&TextItem> = items.iter().filter(|i| i.page == page).collect();
if page_items.is_empty() {
return vec![];
@@ -630,33 +626,6 @@ fn find_relative_valleys(
valleys
}
/// Detect whether a side of a gutter consists predominantly of list-marker
/// glyphs (•, ●, ○, ◦, ▪, ▫, ◆, ◇). A column of bullets on the left margin
/// creates a spurious histogram valley between the bullet and the content.
/// Treating it as a real column splits each list item's text across two
/// "columns," so we reject these candidates.
fn is_list_marker_column(items: &[&&TextItem]) -> bool {
const LIST_MARKERS: &[char] = &['•', '●', '○', '◦', '▪', '▫', '◆', '◇', '■', '□'];
if items.is_empty() {
return false;
}
let marker_count = items
.iter()
.filter(|i| {
let t = i.text.trim();
let mut chars = t.chars();
match (chars.next(), chars.next()) {
(Some(c), None) => LIST_MARKERS.contains(&c),
_ => false,
}
})
.count();
// Require ≥80% of items on this side to be standalone markers. A handful
// of non-marker items (stray page numbers, footnote refs) shouldn't
// defeat the check.
marker_count as f32 / items.len() as f32 >= 0.8
}
/// Validate valley candidates with vertical consistency checks and build column regions.
///
/// When `center_assign` is true, items are assigned to columns based on their
@@ -725,19 +694,6 @@ fn validate_and_build_columns(
continue;
}
// Reject valleys where the smaller side is just a column of list
// markers (bullets aligned at the left margin). This is a common
// pattern in PDFs where ● starts each list item: histogram detection
// sees the gap between bullet and content as a gutter.
let smaller_items: &[&&TextItem] = if left_items.len() <= right_items.len() {
&left_items
} else {
&right_items
};
if is_list_marker_column(smaller_items) {
continue;
}
// Check vertical overlap
if y_range > 0.0 {
let left_y_min = left_items.iter().map(|i| i.y).fold(f32::INFINITY, f32::min);
@@ -1234,7 +1190,11 @@ pub(crate) fn group_into_lines_with_thresholds(
ci,
item.x,
item.y,
super::trace_text_preview(&item.text, 60)
if item.text.len() > 60 {
&item.text[..60]
} else {
&item.text
}
);
}
}
@@ -1494,8 +1454,6 @@ mod tests {
page,
is_bold: false,
is_italic: false,
is_underline: false,
is_strikeout: false,
item_type: ItemType::Text,
mcid: None,
}
@@ -1625,8 +1583,6 @@ mod tests {
page,
is_bold: false,
is_italic: false,
is_underline: false,
is_strikeout: false,
item_type: ItemType::Text,
mcid: None,
});
@@ -1824,63 +1780,6 @@ mod tests {
);
}
#[test]
fn bullet_marker_column_not_detected_as_column() {
// Pattern: every line is `● <content>`, with ● at x=90 and content
// starting at x=104. Histogram detection sees a gutter between them
// and would split the page into a "bullet column" and "content column",
// scrambling every list item.
let mut items = Vec::new();
for i in 0..15 {
let y = 750.0 - i as f32 * 30.0;
items.push(make_item(1, 90.0, y, ""));
items.push(make_item(
1,
104.0,
y,
"FullContentLineTextHere________________",
));
}
// Pad with content to satisfy min item count for column detection.
for i in 0..15 {
let y = 300.0 - i as f32 * 14.0;
items.push(make_item(1, 72.0, y, "FootnoteText_____________________"));
}
let cols = detect_columns(&items, 1, false);
assert_eq!(
cols.len(),
1,
"Bullet markers aligned at left margin should not be treated as their own column"
);
}
#[test]
fn is_list_marker_column_detects_bullets() {
let items = vec![
make_item(1, 90.0, 100.0, ""),
make_item(1, 90.0, 114.0, ""),
make_item(1, 90.0, 128.0, ""),
make_item(1, 90.0, 142.0, ""),
];
let refs: Vec<&TextItem> = items.iter().collect();
let wrapped: Vec<&&TextItem> = refs.iter().collect();
assert!(is_list_marker_column(&wrapped));
}
#[test]
fn is_list_marker_column_rejects_prose() {
let items = vec![
make_item(1, 30.0, 100.0, "Regular prose line"),
make_item(1, 30.0, 114.0, "Another sentence"),
make_item(1, 30.0, 128.0, "Third line"),
make_item(1, 30.0, 142.0, "Fourth line"),
];
let refs: Vec<&TextItem> = items.iter().collect();
let wrapped: Vec<&&TextItem> = refs.iter().collect();
assert!(!is_list_marker_column(&wrapped));
}
#[test]
fn premask_narrow_line_not_masked() {
// Items that form a line spanning only ~40% of column width → not masked
-4
View File
@@ -78,8 +78,6 @@ pub fn extract_page_links(doc: &Document, page_id: ObjectId, page_num: u32) -> V
page: page_num,
is_bold: false,
is_italic: false,
is_underline: false,
is_strikeout: false,
item_type: ItemType::Link(url),
mcid: None,
});
@@ -318,8 +316,6 @@ pub(crate) fn walk_form_fields(
page: page_num,
is_bold: false,
is_italic: false,
is_underline: false,
is_strikeout: false,
item_type: ItemType::FormField,
mcid: None,
});
+51 -829
View File
File diff suppressed because it is too large Load Diff
-563
View File
@@ -1,563 +0,0 @@
//! Geometric underline detection.
//!
//! PDFs have no underline font flag — underlines are drawn as separate
//! graphics: stroked horizontal lines (`l`/`S` operators) or thin filled
//! rectangles (`re`/`f`). This pass correlates those graphics with text
//! items after extraction: an item is underlined when a horizontal
//! line/thin rect sits just below its baseline and covers most of its
//! horizontal extent.
//!
//! Repeated same-span rules are treated as table/form rulings rather than
//! underlines, which avoids marking every cell in ruled tables.
use std::collections::HashSet;
use crate::types::{ItemType, PdfRect, TextItem};
/// Max thickness (pt) for a stroked line / filled rect to count as an
/// underline rule rather than a border or decorative band.
const MAX_RULE_THICKNESS: f32 = 2.0;
/// Fraction of the item's width that the rule must cover horizontally.
const MIN_X_OVERLAP: f32 = 0.6;
/// Same-span rules repeated at this many y-levels are usually table/form
/// rulings, not semantic underlines.
const MIN_REPEATED_RULE_LEVELS: usize = 3;
/// Vertical tolerance for considering two rules to be on the same row edge.
const RULE_Y_DEDUP_EPS: f32 = 2.0;
/// Horizontal span similarity required when clustering repeated rulings.
const RULE_SPAN_OVERLAP_RATIO: f32 = 0.8;
const RULE_SPAN_WIDTH_RATIO: f32 = 1.5;
/// Multiple separated rule segments on one row are usually per-column table
/// header/body separators.
const MIN_SEGMENTED_ROW_RULES: usize = 3;
const MIN_SEGMENTED_ROW_GAPS: usize = 2;
const SEGMENTED_ROW_GAP_MIN: f32 = 12.0;
/// A single rule under several widely separated items is usually a table
/// header/body separator, not a sentence underline.
const MIN_TABULAR_RULE_ITEMS: usize = 3;
const MIN_TABULAR_RULE_GAPS: usize = 2;
const TABULAR_RULE_GAP_EM: f32 = 2.0;
#[derive(Clone)]
pub(crate) struct UnderlineLine {
pub(crate) x1: f32,
pub(crate) y1: f32,
pub(crate) x2: f32,
pub(crate) y2: f32,
pub(crate) stroke_width: f32,
pub(crate) page: u32,
}
/// A horizontal rule candidate in page coordinates (PDF y-up).
#[derive(Clone)]
struct Rule {
x1: f32,
x2: f32,
y: f32,
}
impl Rule {
fn width(&self) -> f32 {
self.x2 - self.x1
}
}
fn rules_from_graphics(rects: &[PdfRect], lines: &[UnderlineLine], page: u32) -> Vec<Rule> {
let mut rules: Vec<Rule> = Vec::new();
for l in lines {
if l.page != page {
continue;
}
// Horizontal stroked line (tolerate slight skew).
if l.stroke_width <= MAX_RULE_THICKNESS && (l.y1 - l.y2).abs() <= MAX_RULE_THICKNESS {
let (x1, x2) = if l.x1 <= l.x2 {
(l.x1, l.x2)
} else {
(l.x2, l.x1)
};
if x2 - x1 > 1.0 {
rules.push(Rule {
x1,
x2,
y: (l.y1 + l.y2) / 2.0,
});
}
}
}
for r in rects {
if r.page != page {
continue;
}
// Thin filled rect used as an underline rule. Extents are
// normalized first: `re` operands pass through the CTM, so
// width/height can be negative (flipped axes / negative scale) —
// without normalization negative-width rules are missed and
// negative-height bands sneak past the thickness check.
let (x1, x2) = if r.width >= 0.0 {
(r.x, r.x + r.width)
} else {
(r.x + r.width, r.x)
};
if r.height.abs() <= MAX_RULE_THICKNESS && x2 - x1 > 1.0 {
rules.push(Rule {
x1,
x2,
y: r.y + r.height / 2.0,
});
}
}
rules
}
fn discard_repeated_ruling_rules(rules: Vec<Rule>) -> Vec<Rule> {
if rules.len() < MIN_REPEATED_RULE_LEVELS {
return rules;
}
rules
.iter()
.filter(|rule| {
!is_repeated_ruling_rule(rule, &rules) && !is_segmented_row_ruling_rule(rule, &rules)
})
.cloned()
.collect()
}
fn is_repeated_ruling_rule(rule: &Rule, rules: &[Rule]) -> bool {
let mut y_levels: Vec<f32> = rules
.iter()
.filter(|other| has_similar_span(rule, other))
.map(|other| other.y)
.collect();
y_levels.sort_by(|a, b| a.total_cmp(b));
y_levels.dedup_by(|a, b| (*a - *b).abs() <= RULE_Y_DEDUP_EPS);
y_levels.len() >= MIN_REPEATED_RULE_LEVELS
}
fn is_segmented_row_ruling_rule(rule: &Rule, rules: &[Rule]) -> bool {
let mut row_rules: Vec<&Rule> = rules
.iter()
.filter(|other| (other.y - rule.y).abs() <= RULE_Y_DEDUP_EPS)
.collect();
if row_rules.len() < MIN_SEGMENTED_ROW_RULES {
return false;
}
row_rules.sort_by(|a, b| a.x1.total_cmp(&b.x1));
let large_gaps = row_rules
.windows(2)
.filter(|pair| pair[1].x1 - pair[0].x2 > SEGMENTED_ROW_GAP_MIN)
.count();
large_gaps >= MIN_SEGMENTED_ROW_GAPS
}
fn has_similar_span(a: &Rule, b: &Rule) -> bool {
let a_width = a.width();
let b_width = b.width();
if a_width <= 1.0 || b_width <= 1.0 {
return false;
}
let width_ratio = a_width.max(b_width) / a_width.min(b_width);
if width_ratio > RULE_SPAN_WIDTH_RATIO {
return false;
}
let overlap = a.x2.min(b.x2) - a.x1.max(b.x1);
overlap >= a_width.min(b_width) * RULE_SPAN_OVERLAP_RATIO
}
fn tabular_row_separator_rule_indices(rules: &[Rule], items: &[TextItem]) -> HashSet<usize> {
let mut tabular_rules = HashSet::new();
for (rule_idx, rule) in rules.iter().enumerate() {
let mut matched_items: Vec<&TextItem> = items
.iter()
.filter(|item| is_underline_candidate(item) && rule_matches_item(rule, item))
.collect();
if matched_items.len() < MIN_TABULAR_RULE_ITEMS {
continue;
}
matched_items.sort_by(|a, b| a.x.total_cmp(&b.x));
let large_gaps = matched_items
.windows(2)
.filter(|pair| {
let left = pair[0];
let right = pair[1];
let gap = right.x - (left.x + left.width);
let font_size = left.font_size.max(right.font_size).max(1.0);
gap > font_size * TABULAR_RULE_GAP_EM
})
.count();
if large_gaps >= MIN_TABULAR_RULE_GAPS {
tabular_rules.insert(rule_idx);
}
}
tabular_rules
}
fn is_underline_candidate(item: &TextItem) -> bool {
matches!(item.item_type, ItemType::Text) && !item.text.trim().is_empty() && item.width > 0.0
}
fn rule_matches_item(rule: &Rule, item: &TextItem) -> bool {
// Vertical window: underlines sit at or slightly below the baseline.
// Fonts draw them at roughly 5-15% of the em below; allow up to 35%
// (min 3pt) below and 1pt above for rounding.
let below = (item.font_size * 0.35).max(3.0);
let y_min = item.y - below;
let y_max = item.y + 1.0;
if rule.y < y_min || rule.y > y_max {
return false;
}
let ix1 = item.x;
let ix2 = item.x + item.width;
let min_overlap = item.width * MIN_X_OVERLAP;
let overlap = rule.x2.min(ix2) - rule.x1.max(ix1);
overlap >= min_overlap
}
/// Strikeout window: a rule crossing the glyphs. Strikethroughs sit at
/// roughly 20-35% of the em above the baseline (about half the x-height);
/// accept a band well inside the glyph body so baseline underlines and
/// overlines never qualify.
fn rule_strikes_item(rule: &Rule, item: &TextItem) -> bool {
let y_min = item.y + item.font_size * 0.12;
let y_max = item.y + item.font_size * 0.55;
if rule.y < y_min || rule.y > y_max {
return false;
}
let ix1 = item.x;
let ix2 = item.x + item.width;
let min_overlap = item.width * MIN_X_OVERLAP;
let overlap = rule.x2.min(ix2) - rule.x1.max(ix1);
overlap >= min_overlap
}
/// Mark `is_underline` on text items that have a horizontal rule just
/// below their baseline, and `is_strikeout` on items whose glyphs a rule
/// crosses at mid x-height. `items`, `rects`, and `lines` are a single
/// page's extraction output (all in PDF coordinates, y-up, where
/// `TextItem::y` is the text baseline).
pub(crate) fn mark_underlined_items(
items: &mut [TextItem],
rects: &[PdfRect],
lines: &[UnderlineLine],
page: u32,
) {
let rules = discard_repeated_ruling_rules(rules_from_graphics(rects, lines, page));
if rules.is_empty() {
return;
}
let tabular_rules = tabular_row_separator_rule_indices(&rules, items);
for item in items.iter_mut() {
if !is_underline_candidate(item) {
continue;
}
for (rule_idx, rule) in rules.iter().enumerate() {
if tabular_rules.contains(&rule_idx) {
continue;
}
if rule_matches_item(rule, item) {
item.is_underline = true;
}
if rule_strikes_item(rule, item) {
item.is_strikeout = true;
}
if item.is_underline && item.is_strikeout {
break;
}
}
}
}
#[cfg(test)]
mod tests {
use super::*;
use crate::types::ItemType;
fn item(text: &str, x: f32, y: f32, width: f32, font_size: f32) -> TextItem {
TextItem {
text: text.to_string(),
x,
y,
width,
height: font_size,
font: "F1".to_string(),
font_size,
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
is_strikeout: false,
item_type: ItemType::Text,
mcid: None,
}
}
fn hline(x1: f32, x2: f32, y: f32) -> UnderlineLine {
UnderlineLine {
x1,
y1: y,
x2,
y2: y,
stroke_width: 1.0,
page: 1,
}
}
fn thin_rect(x: f32, y: f32, width: f32) -> PdfRect {
PdfRect {
x,
y,
width,
height: 0.8,
page: 1,
}
}
#[test]
fn stroked_line_under_baseline_marks_underline() {
let mut items = vec![item("underlined", 100.0, 500.0, 60.0, 10.0)];
let lines = vec![hline(99.0, 161.0, 498.5)];
mark_underlined_items(&mut items, &[], &lines, 1);
assert!(items[0].is_underline);
}
#[test]
fn thin_filled_rect_under_baseline_marks_underline() {
let mut items = vec![item("underlined", 100.0, 500.0, 60.0, 10.0)];
let rects = vec![thin_rect(100.0, 497.8, 60.0)];
mark_underlined_items(&mut items, &rects, &[], 1);
assert!(items[0].is_underline);
}
#[test]
fn long_rule_under_multiple_items_marks_each() {
// One underline drawn under a whole sentence: every overlapped
// item gets the flag.
let mut items = vec![
item("first", 100.0, 500.0, 40.0, 10.0),
item("second", 145.0, 500.0, 50.0, 10.0),
];
let lines = vec![hline(98.0, 200.0, 498.0)];
mark_underlined_items(&mut items, &[], &lines, 1);
assert!(items[0].is_underline);
assert!(items[1].is_underline);
}
#[test]
fn line_far_below_baseline_is_not_an_underline() {
// A horizontal rule 30pt below (section divider) must not mark.
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
let lines = vec![hline(90.0, 300.0, 470.0)];
mark_underlined_items(&mut items, &[], &lines, 1);
assert!(!items[0].is_underline);
}
#[test]
fn thick_stroked_line_is_not_an_underline() {
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
let mut line = hline(99.0, 161.0, 498.5);
line.stroke_width = 4.0;
mark_underlined_items(&mut items, &[], &[line], 1);
assert!(!items[0].is_underline);
}
#[test]
fn mid_glyph_rule_marks_strikeout_not_underline() {
// Rule at ~30% of the em above the baseline crosses the glyphs.
let mut items = vec![item("struck out", 100.0, 500.0, 60.0, 10.0)];
let lines = vec![hline(99.0, 161.0, 503.0)];
mark_underlined_items(&mut items, &[], &lines, 1);
assert!(items[0].is_strikeout);
assert!(!items[0].is_underline);
}
#[test]
fn baseline_rule_marks_underline_not_strikeout() {
let mut items = vec![item("underlined", 100.0, 500.0, 60.0, 10.0)];
let lines = vec![hline(99.0, 161.0, 498.5)];
mark_underlined_items(&mut items, &[], &lines, 1);
assert!(items[0].is_underline);
assert!(!items[0].is_strikeout);
}
#[test]
fn overline_is_neither_underline_nor_strikeout() {
// Rule just above the cap height (overline / next line's rule).
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
let lines = vec![hline(99.0, 161.0, 507.0)];
mark_underlined_items(&mut items, &[], &lines, 1);
assert!(!items[0].is_underline);
assert!(!items[0].is_strikeout);
}
#[test]
fn thin_filled_rect_at_mid_glyph_marks_strikeout() {
let mut items = vec![item("struck out", 100.0, 500.0, 60.0, 10.0)];
let rects = vec![thin_rect(100.0, 502.6, 60.0)];
mark_underlined_items(&mut items, &rects, &[], 1);
assert!(items[0].is_strikeout);
assert!(!items[0].is_underline);
}
#[test]
fn line_above_baseline_is_not_an_underline() {
// Strikethrough / overline geometry must not mark.
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
let lines = vec![hline(90.0, 300.0, 505.0)];
mark_underlined_items(&mut items, &[], &lines, 1);
assert!(!items[0].is_underline);
}
#[test]
fn insufficient_horizontal_overlap_is_not_an_underline() {
// Rule under only a quarter of the item (e.g. neighboring cell
// border) must not mark.
let mut items = vec![item("wide text item", 100.0, 500.0, 100.0, 10.0)];
let lines = vec![hline(100.0, 125.0, 498.5)];
mark_underlined_items(&mut items, &[], &lines, 1);
assert!(!items[0].is_underline);
}
#[test]
fn negative_width_rect_is_normalized_and_marks_underline() {
// A CTM with negative x-scale (or negative `re` operands) produces
// rects whose width is negative; the rule extents must normalize.
let mut items = vec![item("underlined", 100.0, 500.0, 60.0, 10.0)];
let rects = vec![PdfRect {
x: 160.0,
y: 497.8,
width: -60.0,
height: 0.8,
page: 1,
}];
mark_underlined_items(&mut items, &rects, &[], 1);
assert!(items[0].is_underline);
}
#[test]
fn negative_height_band_is_not_an_underline() {
// A 14pt band expressed with negative height must not pass the
// thickness check via sign trickery.
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
let rects = vec![PdfRect {
x: 95.0,
y: 509.0,
width: 80.0,
height: -14.0,
page: 1,
}];
mark_underlined_items(&mut items, &rects, &[], 1);
assert!(!items[0].is_underline);
}
#[test]
fn thick_band_is_not_an_underline() {
// A highlight bar / filled cell background (tall rect) must not mark.
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
let rects = vec![PdfRect {
x: 95.0,
y: 495.0,
width: 80.0,
height: 14.0,
page: 1,
}];
mark_underlined_items(&mut items, &rects, &[], 1);
assert!(!items[0].is_underline);
}
#[test]
fn vertical_line_is_not_an_underline() {
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
let lines = vec![UnderlineLine {
x1: 120.0,
y1: 498.0,
x2: 120.0,
y2: 400.0,
stroke_width: 1.0,
page: 1,
}];
mark_underlined_items(&mut items, &[], &lines, 1);
assert!(!items[0].is_underline);
}
#[test]
fn other_pages_graphics_do_not_mark() {
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
let mut line = hline(99.0, 161.0, 498.5);
line.page = 2;
mark_underlined_items(&mut items, &[], &[line], 1);
assert!(!items[0].is_underline);
}
#[test]
fn repeated_table_row_rules_do_not_mark_cell_text() {
let mut items = vec![
item("A", 110.0, 500.0, 20.0, 10.0),
item("B", 110.0, 480.0, 20.0, 10.0),
item("C", 110.0, 460.0, 20.0, 10.0),
];
let lines = vec![
hline(100.0, 150.0, 498.0),
hline(100.0, 150.0, 478.0),
hline(100.0, 150.0, 458.0),
];
mark_underlined_items(&mut items, &[], &lines, 1);
assert!(items.iter().all(|item| !item.is_underline));
}
#[test]
fn row_separator_under_spaced_column_labels_is_not_an_underline() {
let mut items = vec![
item("Date", 100.0, 500.0, 25.0, 10.0),
item("Rate", 200.0, 500.0, 25.0, 10.0),
item("Yield", 300.0, 500.0, 30.0, 10.0),
];
let lines = vec![hline(90.0, 340.0, 498.0)];
mark_underlined_items(&mut items, &[], &lines, 1);
assert!(items.iter().all(|item| !item.is_underline));
}
#[test]
fn same_row_spaced_rule_segments_do_not_mark_column_labels() {
let mut items = vec![
item("Date", 100.0, 500.0, 25.0, 10.0),
item("Rate", 200.0, 500.0, 25.0, 10.0),
item("Yield", 300.0, 500.0, 30.0, 10.0),
];
let lines = vec![
hline(98.0, 128.0, 498.0),
hline(198.0, 228.0, 498.0),
hline(298.0, 333.0, 498.0),
];
mark_underlined_items(&mut items, &[], &lines, 1);
assert!(items.iter().all(|item| !item.is_underline));
}
}
+18 -69
View File
@@ -1,6 +1,5 @@
//! Form XObject and image XObject extraction.
use super::fonts::descriptor_style_flags;
use crate::text_utils::{effective_font_size, expand_ligatures, is_bold_font, is_italic_font};
use crate::tounicode::FontCMaps;
use crate::types::{ItemType, TextItem};
@@ -9,9 +8,9 @@ use std::collections::HashMap;
use super::fonts::{
build_font_encodings, build_font_widths, compute_string_width_ts, extract_text_from_operand,
get_font_file2_obj_num, get_operand_bytes, CMapDecisionCache, FontStyleCache,
get_font_file2_obj_num, get_operand_bytes, CMapDecisionCache,
};
use super::{get_number, image_bbox_from_ctm, multiply_matrices};
use super::{get_number, multiply_matrices};
const MAX_FORM_XOBJECT_DEPTH: u8 = 5;
@@ -115,7 +114,6 @@ pub(crate) fn extract_form_xobject_text(
font_cmaps: &FontCMaps,
parent_ctm: &[f32; 6],
cmap_decisions: &mut CMapDecisionCache,
style_cache: &mut FontStyleCache,
) -> Vec<TextItem> {
extract_form_xobject_text_inner(
doc,
@@ -124,12 +122,10 @@ pub(crate) fn extract_form_xobject_text(
font_cmaps,
parent_ctm,
cmap_decisions,
style_cache,
0,
)
}
#[allow(clippy::too_many_arguments)]
fn extract_form_xobject_text_inner(
doc: &Document,
form_id: ObjectId,
@@ -137,7 +133,6 @@ fn extract_form_xobject_text_inner(
font_cmaps: &FontCMaps,
parent_ctm: &[f32; 6],
cmap_decisions: &mut CMapDecisionCache,
style_cache: &mut FontStyleCache,
depth: u8,
) -> Vec<TextItem> {
use lopdf::content::Content;
@@ -172,7 +167,6 @@ fn extract_form_xobject_text_inner(
let mut font_tounicode_refs: HashMap<String, u32> = HashMap::new();
let mut inline_cmaps: HashMap<String, crate::tounicode::CMapEntry> = HashMap::new();
let mut font_style_flags: HashMap<String, (bool, bool)> = HashMap::new();
for (font_name, font_dict) in &form_fonts {
let resource_name = String::from_utf8_lossy(font_name).to_string();
if let Ok(base_font) = font_dict.get(b"BaseFont") {
@@ -181,10 +175,6 @@ fn extract_form_xobject_text_inner(
font_base_names.insert(resource_name.clone(), base_name);
}
}
let style = descriptor_style_flags(doc, font_dict, style_cache);
if style != (false, false) {
font_style_flags.insert(resource_name.clone(), style);
}
match font_dict.get(b"ToUnicode") {
Ok(tounicode) => {
if let Ok(obj_ref) = tounicode.as_reference() {
@@ -272,46 +262,19 @@ fn extract_form_xobject_text_inner(
if !op.operands.is_empty() {
if let Ok(name) = op.operands[0].as_name() {
let xobj_name = String::from_utf8_lossy(name).to_string();
match form_xobjects.get(&xobj_name) {
Some(XObjectType::Form(nested_id)) => {
if depth < MAX_FORM_XOBJECT_DEPTH {
let nested_items = extract_form_xobject_text_inner(
doc,
*nested_id,
page_num,
font_cmaps,
&ctm,
cmap_decisions,
style_cache,
depth + 1,
);
items.extend(nested_items);
}
if let Some(XObjectType::Form(nested_id)) = form_xobjects.get(&xobj_name) {
if depth < MAX_FORM_XOBJECT_DEPTH {
let nested_items = extract_form_xobject_text_inner(
doc,
*nested_id,
page_num,
font_cmaps,
&ctm,
cmap_decisions,
depth + 1,
);
items.extend(nested_items);
}
Some(XObjectType::Image) => {
// Mirror the top-level Image-XObject emission
// in content_stream.rs so figures embedded
// inside Form XObjects (common in print-to-PDF
// workflows) aren't silently dropped.
let (x, y, width, height) = image_bbox_from_ctm(&ctm);
items.push(TextItem {
text: format!("[Image: {}]", xobj_name),
x,
y,
width,
height,
font: String::new(),
font_size: 0.0,
page: page_num,
is_bold: false,
is_italic: false,
is_underline: false,
is_strikeout: false,
item_type: ItemType::Image,
mcid: None,
});
}
None => {}
}
}
}
@@ -410,7 +373,6 @@ fn extract_form_xobject_text_inner(
&font_encodings,
&encoding_cache,
cmap_decisions,
&font_widths,
) {
let combined = multiply_matrices(&text_matrix, &ctm);
let rendered_size = effective_font_size(current_font_size, &combined);
@@ -440,10 +402,6 @@ fn extract_form_xobject_text_inner(
.get(&current_font)
.map(|s| s.as_str())
.unwrap_or(&current_font);
let (desc_italic, desc_bold) = font_style_flags
.get(&current_font)
.copied()
.unwrap_or((false, false));
items.push(TextItem {
text: expand_ligatures(&text),
x,
@@ -453,10 +411,8 @@ fn extract_form_xobject_text_inner(
font: current_font.clone(),
font_size: rendered_size,
page: page_num,
is_bold: is_bold_font(base_font) || desc_bold,
is_italic: is_italic_font(base_font) || desc_italic,
is_underline: false,
is_strikeout: false,
is_bold: is_bold_font(base_font),
is_italic: is_italic_font(base_font),
item_type: ItemType::Text,
mcid: None,
});
@@ -561,7 +517,6 @@ fn extract_form_xobject_text_inner(
&font_encodings,
&encoding_cache,
cmap_decisions,
&font_widths,
) {
current_text.push_str(&text);
}
@@ -577,10 +532,6 @@ fn extract_form_xobject_text_inner(
.get(&current_font)
.map(|s| s.as_str())
.unwrap_or(&current_font);
let (desc_italic, desc_bold) = font_style_flags
.get(&current_font)
.copied()
.unwrap_or((false, false));
let scale_x = text_matrix[0] * ctm[0] + text_matrix[1] * ctm[2];
for (text, start_w, end_w) in &sub_items {
let offset_tm = [
@@ -607,10 +558,8 @@ fn extract_form_xobject_text_inner(
font: current_font.clone(),
font_size: rendered_size,
page: page_num,
is_bold: is_bold_font(base_font) || desc_bold,
is_italic: is_italic_font(base_font) || desc_italic,
is_underline: false,
is_strikeout: false,
is_bold: is_bold_font(base_font),
is_italic: is_italic_font(base_font),
item_type: ItemType::Text,
mcid: None,
});
+198 -4785
View File
File diff suppressed because it is too large Load Diff
-183
View File
@@ -130,123 +130,6 @@ pub(crate) fn has_dot_leaders(text: &str) -> bool {
dot_groups >= 2
}
/// Detect a table-of-contents entry: a line ending in a page number preceded by
/// a dot-leader group (e.g. "Measurement Lab worksheet ... 3"). `has_dot_leaders`
/// misses single-group leaders ("..."), but a trailing "<dots> <number>" is a
/// strong TOC signal on its own. Such lines must never be promoted to headings.
pub(crate) fn is_toc_entry_line(text: &str) -> bool {
let trimmed = text.trim_end();
let digits = trimmed
.chars()
.rev()
.take_while(|c| c.is_ascii_digit())
.count();
if digits == 0 || digits > 4 {
return false;
}
let before_number = trimmed[..trimmed.len() - digits].trim_end();
let dots = before_number
.chars()
.rev()
.take_while(|c| *c == '.')
.count();
dots >= 3
}
/// A heading that announces a table of contents ("Contents", "Table of
/// Contents"). Lines after it on the same page are ToC entries — section
/// titles that look exactly like headings but must not be promoted.
pub(crate) fn is_toc_marker_heading(text: &str) -> bool {
let t = text.trim().trim_end_matches(':').trim().to_lowercase();
matches!(t.as_str(), "contents" | "table of contents")
}
/// Lines that resemble headings structurally but are display-math fragments:
/// equations ending in an equation number ("S = kB ln W, (2)") or equation
/// lead-ins ("Rearranging Equation (8) gives:"). Both carry an "(N)" equation
/// reference — but a trailing "(N)" alone is not enough: real headings end
/// with parenthesized numbers too ("Nicaea (325)", appendix numbering), so
/// the suffix form additionally requires math evidence — an "=" in the line
/// or a comma immediately before the number, both present in every display
/// equation and absent from name-plus-number headings. A bare trailing colon
/// is NOT a fragment signal either: real headings frequently end with colons
/// ("Procedure:", "Steps for Using the Microscope:").
pub(crate) fn is_heading_fragment(text: &str) -> bool {
let t = text.trim_end();
// A lowercase-initial one-or-two-word "heading" is a mid-sentence
// fragment beside display math ("or inversely", "and therefore") —
// real headings that short start uppercase. Measured as spurious
// headings on academic docs (fire-pdf ENG-5029 / opendataloader MHS).
{
let words: Vec<&str> = t.split_whitespace().collect();
if words.len() <= 2 {
if let Some(first_alpha) = t.chars().find(|c| c.is_alphabetic()) {
if first_alpha.is_lowercase() {
return true;
}
}
}
}
fn is_equation_number(s: &str) -> bool {
s.strip_prefix('(')
.and_then(|r| r.strip_suffix(')'))
.is_some_and(|inner| {
!inner.is_empty() && inner.len() <= 3 && inner.chars().all(|c| c.is_ascii_digit())
})
}
// Equation-number suffix with math evidence: "S = kB ln W, (2)"
let mut rev = t.rsplit(' ');
let last = rev.next().unwrap_or("");
if is_equation_number(last) {
// Page-of-total running headers: "LIVSMEDELSVERKET PM 2 (10)"
if let Some(prev_word) = t.rsplit(' ').nth(1) {
if let (Ok(page), Some(total)) = (
prev_word.parse::<u32>(),
last.trim_start_matches('(')
.trim_end_matches(')')
.parse::<u32>()
.ok(),
) {
if page <= total {
return true;
}
}
}
let punct_before = rev
.next()
.is_some_and(|w| w.ends_with(',') || w.ends_with(':'));
let has_math_op = t.chars().any(|c| {
matches!(
c,
'=' | '<'
| '>'
| '≤'
| '≥'
| '≪'
| '≫'
| '≈'
| '≠'
| '±'
| '∑'
| '∫'
| '√'
| '∝'
)
});
if punct_before || has_math_op {
return true;
}
}
// Lead-in: ends with a colon AND references an equation number inline
if t.ends_with(':') && t.split_whitespace().any(is_equation_number) {
return true;
}
false
}
/// Compute the Y-gap threshold for paragraph break detection.
///
/// Instead of using a fixed multiple of base_size (which fails for double-spaced
@@ -437,69 +320,3 @@ pub(crate) fn detect_header_level(
Some(4)
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn toc_entry_with_single_dot_group() {
assert!(is_toc_entry_line("Measurement Lab worksheet ... 3"));
assert!(is_toc_entry_line("Results ........ 12"));
assert!(is_toc_entry_line("Appendix B...42"));
}
#[test]
fn non_toc_lines_pass() {
assert!(!is_toc_entry_line(
"6.2. Expectations for Re-Hiring Employees"
));
assert!(!is_toc_entry_line("What happened in 2020"));
assert!(!is_toc_entry_line("IMPLEMENTATION"));
// Ellipsis without a trailing page number
assert!(!is_toc_entry_line("and so it goes ..."));
// Long numbers are data, not page refs
assert!(!is_toc_entry_line("ISBN ... 97814"));
}
#[test]
fn toc_marker_headings() {
assert!(is_toc_marker_heading("Contents"));
assert!(is_toc_marker_heading("CONTENTS"));
assert!(is_toc_marker_heading("Table of Contents"));
assert!(is_toc_marker_heading("Table of contents:"));
assert!(!is_toc_marker_heading("Contents of the Shipment"));
assert!(!is_toc_marker_heading("Introduction"));
}
#[test]
fn heading_fragments() {
// Equation lead-ins: colon ending + inline equation reference
assert!(is_heading_fragment("or inversely"));
assert!(is_heading_fragment("and therefore"));
assert!(!is_heading_fragment("Introduction"));
assert!(!is_heading_fragment("iPhone Sales Strategy Overview")); // 4 words, exempt
assert!(is_heading_fragment("Rearranging Equation (8) gives:"));
// Display-equation neighbours ending in an equation number
assert!(is_heading_fragment("S = kB ln W, (2)"));
assert!(is_heading_fragment("E = mc2 (12)"));
assert!(is_heading_fragment("x + y = z, (3)"));
// Page-of-total running headers
assert!(is_heading_fragment("LIVSMEDELSVERKET PM 2 (10)"));
// Comparison-operator evidence and colon-before-number
assert!(is_heading_fragment(
"PLL\u{fe} PHH\u{226a} PLH\u{fe} PHL: (12)"
));
// Real headings pass — including name-plus-number and colon-ended ones
assert!(!is_heading_fragment("Nicaea (325)"));
assert!(!is_heading_fragment(
"\u{627}\u{644}\u{645}\u{644}\u{62d}\u{642} \u{631}\u{642}\u{645} (1)"
));
assert!(!is_heading_fragment("4. Entropy"));
assert!(!is_heading_fragment("Procedure:"));
assert!(!is_heading_fragment("Steps for Using the Microscope:"));
assert!(!is_heading_fragment("Changing objectives:"));
assert!(!is_heading_fragment("Sales by Region (2024)"));
assert!(!is_heading_fragment("Results (preliminary)"));
}
}
-73
View File
@@ -64,22 +64,6 @@ pub(crate) fn is_caption_line(text: &str) -> bool {
false
}
/// Check if text starts with an unambiguous bullet marker (●, •, ○, ◦).
///
/// Narrower than [`is_list_item`]: it excludes numbered/lettered patterns
/// like `1.` or `a)`, which legitimately appear as section headings in many
/// documents. Used by the heading classifier to reject bullet lines without
/// also demoting numbered headings.
pub(crate) fn starts_with_bullet_marker(text: &str) -> bool {
let trimmed = text.trim_start();
trimmed.starts_with("")
|| trimmed.starts_with("")
|| trimmed.starts_with("")
|| trimmed.starts_with("")
|| trimmed.starts_with("- ")
|| trimmed.starts_with("* ")
}
/// Check if text looks like a list item
pub(crate) fn is_list_item(text: &str) -> bool {
let trimmed = text.trim_start();
@@ -131,17 +115,6 @@ pub(crate) fn format_list_item(text: &str) -> String {
if let Some(rest) = trimmed.strip_prefix(*bullet) {
return format!("- {}", rest.trim_start());
}
// Bullet inside a leading style run (e.g. "**● Label:** rest" or
// "<u>● Label</u>"). The run wraps both the marker and the following
// label because both carry the style in the PDF. The marker must move
// outside the wrapper so markdown still sees a list item.
for wrapper in ["**", "*", "<u>"] {
if let Some(after_open) = trimmed.strip_prefix(wrapper) {
if let Some(rest) = after_open.strip_prefix(*bullet) {
return format!("- {}{}", wrapper, rest.trim_start());
}
}
}
}
if trimmed.starts_with("- ") || trimmed.starts_with("* ") {
@@ -225,49 +198,3 @@ pub(crate) fn is_monospace_font(font_name: &str) -> bool {
patterns.iter().any(|p| lower.contains(p))
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn format_list_item_plain_bullet() {
assert_eq!(format_list_item("● Item"), "- Item");
assert_eq!(format_list_item("• Item"), "- Item");
}
#[test]
fn format_list_item_bullet_inside_underline() {
// Fully-underlined bullet line: the marker must move outside the
// <u> wrapper so markdown still renders a list item.
assert_eq!(format_list_item("<u>● Item text</u>"), "- <u>Item text</u>");
}
#[test]
fn format_list_item_bullet_inside_bold() {
// PDF that uses bold font for both the marker and the label produces
// a single bold run like "**● Label:** rest"; the bullet must still
// be stripped and the bold wrapper preserved on the label.
assert_eq!(
format_list_item("**● Fraud: Willing cooperation;**"),
"- **Fraud: Willing cooperation;**"
);
assert_eq!(
format_list_item("**● Label:** rest of line"),
"- **Label:** rest of line"
);
assert_eq!(format_list_item("*● Italic:* rest"), "- *Italic:* rest");
}
#[test]
fn format_list_item_already_dash() {
assert_eq!(format_list_item("- existing"), "- existing");
}
#[test]
fn is_list_item_with_bullet_space() {
assert!(is_list_item("● Item"));
assert!(is_list_item("• Item"));
assert!(is_list_item("- Item"));
}
}
+15 -482
View File
@@ -7,12 +7,9 @@ use crate::types::TextLine;
use super::analysis::{
bold_heading_level, calculate_font_stats, compute_heading_tiers, compute_paragraph_threshold,
detect_header_level, font_size_rarity, has_dot_leaders, is_heading_fragment, is_toc_entry_line,
is_toc_marker_heading,
};
use super::classify::{
format_list_item, is_caption_line, is_list_item, is_monospace_font, starts_with_bullet_marker,
detect_header_level, font_size_rarity, has_dot_leaders,
};
use super::classify::{format_list_item, is_caption_line, is_list_item, is_monospace_font};
use super::postprocess::clean_markdown;
use super::preprocess::{merge_drop_caps, merge_heading_lines};
use super::MarkdownOptions;
@@ -141,11 +138,8 @@ fn find_isolated_lines(lines: &[TextLine], base_size: f32, para_threshold: f32)
}
}
for (&page, &(total, isolated)) in &page_line_counts {
// The ratio only means something on pages dense enough for a
// multi-column misfire; on sparse pages (covers, ToC pages with a
// lone title) one isolated line is 25%+ of the page and exactly the
// line isolation exists to find.
if total >= 10 && isolated as f32 / total as f32 > 0.25 {
if total > 0 && isolated as f32 / total as f32 > 0.25 {
// Too many isolated lines on this page — remove them all
set.retain(|&i| lines[i].page != page);
}
}
@@ -153,79 +147,6 @@ fn find_isolated_lines(lines: &[TextLine], base_size: f32, para_threshold: f32)
set
}
/// Pre-scan body-size all-bold runs that are too long to be headings.
///
/// Some academic PDFs use an all-bold abstract/summary paragraph immediately
/// after the author block. A line-local bold heading heuristic sees each
/// wrapped visual line as "standalone" once the first line is misclassified,
/// producing a stack of `##` headings. Multi-line body-size bold runs with a
/// paragraph-sized word count should stay paragraph text.
fn find_wrapped_bold_paragraph_lines(
lines: &[TextLine],
base_size: f32,
para_threshold: f32,
) -> HashSet<usize> {
let mut set = HashSet::new();
let mut i = 0usize;
while i < lines.len() {
if !is_body_size_all_bold_line(&lines[i], base_size) {
i += 1;
continue;
}
let start = i;
let mut end = i;
let mut word_count = lines[i].text().split_whitespace().count();
while end + 1 < lines.len()
&& is_body_size_all_bold_line(&lines[end + 1], base_size)
&& is_wrapped_same_style_line(&lines[end], &lines[end + 1], para_threshold)
{
end += 1;
word_count += lines[end].text().split_whitespace().count();
}
let line_count = end - start + 1;
if line_count >= 3 && word_count > 20 {
for idx in start..=end {
set.insert(idx);
}
}
i = end + 1;
}
set
}
fn is_body_size_all_bold_line(line: &TextLine, base_size: f32) -> bool {
let Some(first) = line.items.first() else {
return false;
};
first.font_size >= base_size * 0.95
&& first.font_size < base_size * 1.2
&& line
.items
.iter()
.all(|item| item.is_bold && (item.font_size - first.font_size).abs() < 0.5)
}
fn is_wrapped_same_style_line(prev: &TextLine, next: &TextLine, para_threshold: f32) -> bool {
if prev.page != next.page {
return false;
}
let y_gap = prev.y - next.y;
if !(y_gap > 0.0 && y_gap <= para_threshold) {
return false;
}
let prev_x = prev.items.first().map(|item| item.x).unwrap_or(0.0);
let next_x = next.items.first().map(|item| item.x).unwrap_or(0.0);
(prev_x - next_x).abs() <= 40.0
}
/// Resolve the dominant structure role for a text line by looking up its items' MCIDs.
///
/// Returns the first non-container role found (skipping Document/Part/Sect/Div/NonStruct/Span).
@@ -474,8 +395,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
// between paragraphs at body font size. Inspired by opendataloader's
// lookahead in HeadingProcessor (prevNode/nextNode context).
let isolated_lines = find_isolated_lines(&lines, base_size, para_threshold);
let wrapped_bold_paragraph_lines =
find_wrapped_bold_paragraph_lines(&lines, base_size, para_threshold);
// Detect struct heading levels that are overused (body text mistagged as headings)
let overused_heading_levels = detect_overused_struct_heading_levels(&lines, struct_roles);
@@ -489,8 +408,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
let mut last_list_x: Option<f32> = None;
let mut in_code_block = false;
let mut prev_had_dot_leaders = false;
let mut paragraph_in_wrapped_bold_run = false;
let mut toc_suppress_page: Option<u32> = None;
let mut inserted_tables: HashSet<(u32, usize)> = HashSet::new();
let mut inserted_images: HashSet<(u32, usize)> = HashSet::new();
@@ -556,7 +473,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
current_page = line.page;
prev_y = f32::MAX;
prev_x = 0.0;
paragraph_in_wrapped_bold_run = false;
if options.include_page_numbers {
output.push_str(&format!("<!-- Page {} -->\n\n", current_page));
@@ -571,7 +487,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
output.push('\n');
output.push_str(table_md);
@@ -589,7 +504,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
output.push('\n');
output.push_str(image_md);
@@ -611,18 +525,9 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
&& y_gap.abs() <= para_threshold
&& (prev_x - line_x).abs() > 50.0
&& prev_y < f32::MAX;
let line_all_bold = !line.items.is_empty() && line.items.iter().all(|item| item.is_bold);
let line_in_wrapped_bold_run = wrapped_bold_paragraph_lines.contains(&line_idx);
let is_bold_to_regular_break = in_paragraph
&& paragraph_in_wrapped_bold_run
&& !line_in_wrapped_bold_run
&& !line_all_bold
&& y_gap > base_size * 1.2
&& y_gap <= para_threshold;
if (is_para_break || is_band_switch || is_bold_to_regular_break) && in_paragraph {
if (is_para_break || is_band_switch) && in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
// Don't immediately end list on paragraph break
// Let the continuation check below decide if we're still in a list
@@ -630,11 +535,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
prev_x = line_x;
// Get text with optional bold/italic formatting
let text = line.text_with_formatting(
options.detect_bold,
options.detect_italic,
options.detect_underline,
);
let text = line.text_with_formatting(options.detect_bold, options.detect_italic);
let trimmed = text.trim();
// Also get plain text for pattern matching (list detection, captions, etc.)
@@ -669,7 +570,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
output.push_str(trimmed);
output.push_str("\n\n");
@@ -684,41 +584,9 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
.as_ref()
.and_then(struct_role_heading_level)
.filter(|level| !overused_heading_levels.contains(level));
// Protect wrapped list items: when inside a list, a visually-continuing
// line (same indent, line-wrap spacing) must not be reclassified as a
// heading by the font heuristic — PDFs often bold the lead phrase of a
// list item across multiple wrap lines, and an all-bold middle line
// would otherwise split one item into a heading + stray body text.
// We gate on the document's paragraph threshold so genuine section
// headings that follow a numbered paragraph (y_gap > para_threshold)
// remain detectable.
let looks_like_list_continuation = in_list
&& match (last_list_x, line.items.first().map(|i| i.x)) {
(Some(list_x), Some(curr_x)) => {
let x_ok = curr_x >= list_x - 5.0 && curr_x <= list_x + 50.0;
let y_ok = y_gap >= 0.0 && y_gap <= para_threshold;
x_ok && y_ok && !is_list_item(plain_trimmed)
}
_ => false,
};
// Lines explicitly tagged with a non-heading content role must never
// be promoted by the visual heuristic — a tagged list item, quote, or
// code line can look exactly like a heading (short, isolated).
let non_heading_role = struct_role
.as_ref()
.is_some_and(StructRole::is_non_heading_content);
let heuristic_heading = if options.detect_headers
&& !non_heading_role
&& !is_code_line
&& !looks_like_list_continuation
&& plain_trimmed.len() > 3
&& plain_trimmed.split_whitespace().count() <= 15
&& !starts_with_bullet_marker(plain_trimmed)
&& !is_toc_entry_line(plain_trimmed)
&& !is_heading_fragment(plain_trimmed)
&& toc_suppress_page != Some(line.page)
{
let line_font_size = line.items.first().map(|i| i.font_size).unwrap_or(base_size);
detect_header_level(line_font_size, base_size, &heading_tiers).or_else(|| {
@@ -734,9 +602,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
if !(1..=15).contains(&word_count) {
return None;
}
if wrapped_bold_paragraph_lines.contains(&line_idx) {
return None;
}
let rarity = font_size_rarity(line_font_size, &font_stats);
let all_bold = !line.items.is_empty() && line.items.iter().all(|i| i.is_bold);
let standalone = !in_paragraph;
@@ -754,11 +619,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
// paragraph continuity and minor font-size variation
// inflates rarity scores.
let has_strong_signal = all_bold || isolated || (rarity >= 0.97 && word_count <= 8);
// Single-word headings ("IMPLEMENTATION", "CONTENTS") are common;
// accept them only with the strongest signal combination.
let enough_words =
word_count >= 2 || (all_bold && isolated && plain_trimmed.len() >= 4);
if score >= 0.5 && standalone && enough_words && has_strong_signal {
if score >= 0.5 && standalone && word_count >= 2 && has_strong_signal {
Some(bold_heading_level(&heading_tiers))
} else {
None
@@ -772,39 +633,23 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
let prefix = "#".repeat(level);
// Plain text for headers (no redundant bold/italic inside `#`),
// but underline is preserved: `<u>` carries meaning `#` doesn't.
let heading_text = if options.detect_underline {
line.text_with_formatting(false, false, true)
} else {
plain_text.clone()
};
output.push_str(&format!("{} {}\n\n", prefix, heading_text.trim()));
if is_toc_marker_heading(plain_trimmed) {
toc_suppress_page = Some(line.page);
}
// Use plain text for headers to avoid redundant formatting
output.push_str(&format!("{} {}\n\n", prefix, plain_trimmed));
in_list = false;
continue;
}
// Structure-tree list item (LI only — LBody is a continuation, not a new item).
// Some tagged PDFs use a "flat" style where every wrapped line in a list item
// gets its own MCID tagged directly under LI. When we're already inside a list
// and the line has no visible bullet marker, treat it as a continuation (falls
// through to the continuation logic below) rather than a new list item.
// Structure-tree list item (LI only — LBody is a continuation, not a new item)
if struct_role
.as_ref()
.is_some_and(|r| matches!(r, StructRole::LI))
&& !is_list_item(plain_trimmed)
&& !in_list
{
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
output.push_str(&format!("- {}", trimmed));
output.push('\n');
@@ -818,7 +663,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
let formatted = format_list_item(trimmed);
output.push_str(&formatted);
@@ -865,7 +709,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
output.push_str(&format!("> {}\n", trimmed));
continue;
@@ -876,7 +719,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
if !in_code_block {
output.push_str("```\n");
@@ -897,11 +739,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
}
}
output.push_str(trimmed);
paragraph_in_wrapped_bold_run = if in_paragraph {
paragraph_in_wrapped_bold_run || line_in_wrapped_bold_run
} else {
line_in_wrapped_bold_run
};
in_paragraph = true;
prev_had_dot_leaders = cur_dot_leaders;
}
@@ -971,8 +808,6 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
let para_threshold = compute_paragraph_threshold(&lines, base_size);
let isolated_lines = find_isolated_lines(&lines, base_size, para_threshold);
let wrapped_bold_paragraph_lines =
find_wrapped_bold_paragraph_lines(&lines, base_size, para_threshold);
let mut output = String::new();
let mut current_page = 0u32;
@@ -981,8 +816,6 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
let mut in_paragraph = false;
let mut last_list_x: Option<f32> = None;
let mut prev_had_dot_leaders = false;
let mut paragraph_in_wrapped_bold_run = false;
let mut toc_suppress_page: Option<u32> = None;
for (line_idx, line) in lines.iter().enumerate() {
// Page break
@@ -999,7 +832,6 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
in_list = false;
last_list_x = None;
prev_had_dot_leaders = false;
paragraph_in_wrapped_bold_run = false;
if options.include_page_numbers {
output.push_str(&format!("<!-- Page {} -->\n\n", current_page));
@@ -1010,29 +842,16 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
// (newspaper columns emitted sequentially on the same page).
let y_gap = prev_y - line.y;
let is_para_break = y_gap.abs() > para_threshold;
let line_all_bold = !line.items.is_empty() && line.items.iter().all(|item| item.is_bold);
let line_in_wrapped_bold_run = wrapped_bold_paragraph_lines.contains(&line_idx);
let is_bold_to_regular_break = in_paragraph
&& paragraph_in_wrapped_bold_run
&& !line_in_wrapped_bold_run
&& !line_all_bold
&& y_gap > base_size * 1.2
&& y_gap <= para_threshold;
if (is_para_break || is_bold_to_regular_break) && in_paragraph {
if is_para_break && in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
// Don't immediately end list on paragraph break
// Let the continuation check below decide if we're still in a list
prev_y = line.y;
// Get text with optional bold/italic formatting
let text = line.text_with_formatting(
options.detect_bold,
options.detect_italic,
options.detect_underline,
);
let text = line.text_with_formatting(options.detect_bold, options.detect_italic);
let trimmed = text.trim();
// Also get plain text for pattern matching
@@ -1049,7 +868,6 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
output.push_str(trimmed);
output.push_str("\n\n");
@@ -1061,10 +879,6 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
if options.detect_headers
&& plain_trimmed.len() > 3
&& plain_trimmed.split_whitespace().count() <= 15
&& !is_toc_entry_line(plain_trimmed)
&& !is_heading_fragment(plain_trimmed)
&& toc_suppress_page != Some(line.page)
&& !(options.detect_code && line.items.iter().any(|i| is_monospace_font(&i.font)))
{
let line_font_size = line.items.first().map(|i| i.font_size).unwrap_or(base_size);
if let Some(header_level) =
@@ -1076,9 +890,6 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
if !(1..=15).contains(&word_count) {
return None;
}
if wrapped_bold_paragraph_lines.contains(&line_idx) {
return None;
}
let rarity = font_size_rarity(line_font_size, &font_stats);
let all_bold = !line.items.is_empty() && line.items.iter().all(|i| i.is_bold);
let standalone = !in_paragraph;
@@ -1087,9 +898,7 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
+ if all_bold { 0.3 } else { 0.0 }
+ if standalone { 0.2 } else { 0.0 }
+ if isolated { 0.3 } else { 0.0 };
let enough_words =
word_count >= 2 || (all_bold && isolated && plain_trimmed.len() >= 4);
if score >= 0.5 && standalone && enough_words {
if score >= 0.5 && standalone && word_count >= 2 {
return Some(bold_heading_level(&heading_tiers));
}
None
@@ -1098,19 +907,10 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
let prefix = "#".repeat(header_level);
// Plain text for headers, except underline (see above).
let heading_text = if options.detect_underline {
line.text_with_formatting(false, false, true)
} else {
plain_text.clone()
};
output.push_str(&format!("{} {}\n\n", prefix, heading_text.trim()));
if is_toc_marker_heading(plain_trimmed) {
toc_suppress_page = Some(line.page);
}
// Use plain text for headers to avoid redundant formatting
output.push_str(&format!("{} {}\n\n", prefix, plain_trimmed));
in_list = false;
continue;
}
@@ -1121,7 +921,6 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
let formatted = format_list_item(trimmed);
output.push_str(&formatted);
@@ -1166,7 +965,6 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
// Use plain text for code blocks
output.push_str(&format!("```\n{}\n```\n", plain_trimmed));
@@ -1184,11 +982,6 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
}
}
output.push_str(trimmed);
paragraph_in_wrapped_bold_run = if in_paragraph {
paragraph_in_wrapped_bold_run || line_in_wrapped_bold_run
} else {
line_in_wrapped_bold_run
};
in_paragraph = true;
prev_had_dot_leaders = cur_dot_leaders;
}
@@ -1221,8 +1014,6 @@ mod tests {
page,
is_bold: false,
is_italic: false,
is_underline: false,
is_strikeout: false,
item_type: crate::types::ItemType::Text,
mcid,
}
@@ -1239,42 +1030,6 @@ mod tests {
}
}
fn line_at(text: &str, page: u32, y: f32) -> TextLine {
let mut item = make_item(text, page, None);
item.y = y;
make_line(vec![item])
}
#[test]
fn isolated_lines_kept_on_sparse_pages() {
// A ToC page with a lone title and one entry far below: the density
// ratio is 50% but the page is too sparse for the multi-column
// misfire the guard targets — the title must stay isolated.
let lines = vec![
line_at("CONTENTS", 1, 700.0),
line_at("Chapter One 5", 1, 500.0),
];
let isolated = find_isolated_lines(&lines, 12.0, 20.0);
assert!(
isolated.contains(&0),
"sparse-page title must stay isolated"
);
}
#[test]
fn isolated_lines_wiped_on_dense_pages() {
// 12 short lines all with paragraph gaps — the multi-column misfire
// shape. The guard must clear them all.
let lines: Vec<TextLine> = (0..12)
.map(|i| line_at("Short column line", 1, 700.0 - i as f32 * 50.0))
.collect();
let isolated = find_isolated_lines(&lines, 12.0, 20.0);
assert!(
isolated.is_empty(),
"dense page of isolated lines must be wiped"
);
}
#[test]
fn test_struct_role_heading() {
let lines = vec![
@@ -1335,59 +1090,6 @@ mod tests {
);
}
#[test]
fn test_struct_role_li_flat_continuation_lines_merge() {
// Regression: some tagged PDFs put each wrapped visual line of a list
// item under its own MCID, all tagged directly as LI. Continuation
// lines (no bullet marker) must merge into the bulleted parent item,
// not each become their own list item.
let make = |text: &str, mcid: i64, x: f32, y: f32| {
let mut item = make_item(text, 1, Some(mcid));
item.x = x;
item.y = y;
item
};
let lines = vec![
make_line(vec![make("● First item that wraps onto", 0, 90.0, 322.0)]),
make_line(vec![make("a continuation line.", 1, 108.0, 306.0)]),
make_line(vec![make("● Second bullet also wraps", 2, 90.0, 290.0)]),
make_line(vec![make("to a second line here.", 3, 108.0, 274.0)]),
];
let mut page_roles = HashMap::new();
for mcid in 0..4 {
page_roles.insert(mcid, StructRole::LI);
}
let mut roles = HashMap::new();
roles.insert(1u32, page_roles);
let md = to_markdown_from_lines_with_tables_and_images(
lines,
MarkdownOptions::default(),
HashMap::new(),
HashMap::new(),
&std::collections::HashSet::new(),
Some(&roles),
);
assert!(
md.contains("- First item that wraps onto a continuation line."),
"continuation should merge into first bullet: {md}"
);
assert!(
md.contains("- Second bullet also wraps to a second line here."),
"continuation should merge into second bullet: {md}"
);
assert!(
!md.contains("- a continuation line."),
"continuation line should not get its own bullet: {md}"
);
assert!(
!md.contains("- to a second line here."),
"continuation line should not get its own bullet: {md}"
);
}
#[test]
fn test_struct_role_blockquote() {
let lines = vec![make_line(vec![make_item("Quoted text", 1, Some(0))])];
@@ -1573,109 +1275,6 @@ mod tests {
);
}
#[test]
fn test_wrapped_bold_abstract_is_not_split_into_headings() {
// Regression for arXiv 1107.1353: the opening abstract paragraph is
// entirely bold at body size. The first wrapped lines used to become
// separate H2 headings, and the following body paragraph was joined to
// the bold abstract because the paragraph gap is modest.
let make = |text: &str, y: f32, font_size: f32, bold: bool| {
let mut item = make_item(text, 1, None);
item.y = y;
item.font_size = font_size;
item.height = font_size;
item.is_bold = bold;
item
};
let lines = vec![
make_line(vec![make(
"Quantum Nature of Light Measured With a Single Detector",
747.7,
25.0,
true,
)]),
make_line(vec![make(
"Gesine A. Steudle1*, Stefan Schietinger1, David Höckel1",
651.1,
11.0,
false,
)]),
make_line(vec![make(
"Zwiller2, and Oliver Benson1",
638.5,
11.0,
false,
)]),
make_line(vec![make(
"The introduction of light quanta by Einstein in 1905 triggered strong efforts to",
607.5,
11.0,
true,
)]),
make_line(vec![make(
"demonstrate the quantum properties of light directly, without involving matter",
594.8,
11.0,
true,
)]),
make_line(vec![make(
"quantization. It however took more than seven decades for the quantum granularity",
582.2,
11.0,
true,
)]),
make_line(vec![make(
"of light to be observed in the fluorescence of single atoms. Single atoms emit",
569.5,
11.0,
true,
)]),
make_line(vec![make(
"photons one at a time, this is typically demonstrated with a Hanbury-Brown-Twiss",
556.9,
11.0,
true,
)]),
make_line(vec![make(
"Our work significantly simplifies a widely used photon-correlation technique.",
544.2,
11.0,
true,
)]),
make_line(vec![make(
"A photon is a single excitation of a mode of the electromagnetic field.",
528.7,
11.0,
false,
)]),
];
let md = to_markdown_from_lines_with_tables_and_images(
lines,
MarkdownOptions::default(),
HashMap::new(),
HashMap::new(),
&std::collections::HashSet::new(),
None,
);
assert!(
md.contains("# Quantum Nature of Light Measured With a Single Detector"),
"title should remain a heading: {md}"
);
assert!(
!md.contains("## The introduction")
&& !md.contains("## demonstrate")
&& !md.contains("## quantization"),
"bold abstract lines should not become headings: {md}"
);
assert!(
md.contains("technique.**\n\nA photon is a single excitation"),
"body paragraph should be separated from bold abstract: {md}"
);
}
#[test]
fn test_struct_role_code_multiline_accumulation() {
let mut line1 = make_item("fn main() {", 1, Some(0));
@@ -1793,70 +1392,4 @@ mod tests {
overused
);
}
#[test]
fn test_wrapped_bold_lead_in_list_item_not_heading() {
// Regression: numbered-list items whose bold "lead" phrase wraps onto
// a second line (e.g. definitions in system cards) must not have the
// wrapped line reclassified as a heading. The middle line is
// all_bold + standalone (in_paragraph=false while in_list), which
// previously tripped the rarity heuristic and emitted #### in the
// middle of the item, splitting the body into stray bullets.
let make = |text: &str, x: f32, y: f32, bold: bool| {
let mut item = make_item(text, 1, None);
item.x = x;
item.y = y;
item.is_bold = bold;
item
};
let lines = vec![
// "1. **bold lead phrase start**"
make_line(vec![
make("1. ", 72.0, 700.0, false),
make(
"Chemical and biological weapons threat model 1 (CB-1): Non-novel",
90.0,
700.0,
true,
),
]),
// wrapped continuation of the bold lead — all_bold, same indent
make_line(vec![make(
"chemical/biological weapons production capabilities: A model has CB-1",
90.0,
686.0,
true,
)]),
// body text of the same list item
make_line(vec![make(
"capabilities if it has the ability to significantly help individuals.",
90.0,
672.0,
false,
)]),
];
let md = to_markdown_from_lines_with_tables_and_images(
lines,
MarkdownOptions::default(),
HashMap::new(),
HashMap::new(),
&std::collections::HashSet::new(),
None,
);
assert!(
!md.contains("#### "),
"wrapped bold lead must not become a heading: {md}"
);
assert!(
md.lines().filter(|l| l.starts_with("- ")).count() == 0,
"continuation body must not become a stray bullet: {md}"
);
assert!(
md.contains("1. ") && md.contains("A model has CB-1"),
"numbered list item should remain intact: {md}"
);
}
}
+1 -15
View File
@@ -400,8 +400,6 @@ pub struct MarkdownOptions {
pub detect_bold: bool,
/// Detect and format italic text from font names
pub detect_italic: bool,
/// Emit `<u>` runs for text with a geometrically-detected underline
pub detect_underline: bool,
/// Include image placeholders in output
pub include_images: bool,
/// Include extracted hyperlinks
@@ -424,17 +422,7 @@ impl Default for MarkdownOptions {
fix_hyphenation: true,
detect_bold: true,
detect_italic: true,
detect_underline: true,
// `include_images: false` is intentional. The content-stream walker
// now emits `ItemType::Image` `TextItem`s for every Image XObject
// it encounters (see `extractor/content_stream.rs`). If we rendered
// those into markdown by default, every existing caller would
// suddenly see `![Image: Im0](image)` placeholders inserted
// throughout their output — a silent regression for anyone who
// upgrades. Image bboxes are still available via
// `extract_text_with_positions` for callers (e.g. layout-aware
// pipelines) that want to crop + caption figures themselves.
include_images: false,
include_images: true,
include_links: true,
include_page_numbers: false,
strip_headers_footers: true,
@@ -1220,8 +1208,6 @@ mod tests {
page,
is_bold: false,
is_italic: false,
is_underline: false,
is_strikeout: false,
item_type: crate::types::ItemType::Text,
mcid: None,
}
-84
View File
@@ -29,8 +29,6 @@ pub(crate) fn clean_markdown(mut text: String, options: &MarkdownOptions) -> Str
// text item, which combine with gap-based space insertion to produce
// double spaces ("Vice President" instead of "Vice President").
collapse_consecutive_spaces(&mut text);
remove_spaces_before_closing_brackets(&mut text);
remove_spaces_before_sentence_punctuation(&mut text);
// Remove excessive newlines (more than 2 in a row)
while text.contains("\n\n\n") {
@@ -73,46 +71,6 @@ fn collapse_consecutive_spaces(text: &mut String) {
*text = result;
}
/// Remove spaces before closing square brackets.
/// Unit markers and markdown links occasionally pick up a gap-inserted space
/// before `]` (e.g. `[kg/m3 ]`), which is cosmetic padding.
fn remove_spaces_before_closing_brackets(text: &mut String) {
let mut result = String::with_capacity(text.len());
for ch in text.chars() {
if ch == ']' && result.ends_with(' ') {
result.pop();
}
result.push(ch);
}
*text = result;
}
/// Remove a stray space before sentence punctuation ("word ." → "word.").
/// Style-boundary item splits (bold/italic/underline runs) can strand a
/// trailing period or comma in its own fragment, and several assembly paths
/// join fragments with spaces. Only fires when the punctuation ends the
/// token (followed by whitespace or end of text), so decimals ("3 .14" stays
/// untouched — no such input exists, but the guard is cheap) and dot leaders
/// (" ... ") are unaffected.
fn remove_spaces_before_sentence_punctuation(text: &mut String) {
let chars: Vec<char> = text.chars().collect();
let mut result = String::with_capacity(text.len());
for (i, &ch) in chars.iter().enumerate() {
if matches!(ch, '.' | ',' | ';') && result.ends_with(' ') {
let next = chars.get(i + 1);
// `|` counts as a token end so table cells get the same fix.
let token_ends = next.is_none_or(|c| c.is_whitespace() || *c == '|');
// Never touch runs of dots (ellipsis / dot leaders).
let in_dot_run = ch == '.' && next == Some(&'.');
if token_ends && !in_dot_run {
result.pop();
}
}
result.push(ch);
}
*text = result;
}
/// Collapse dot leaders (runs of 4+ dots) into " ... "
/// Common in tables of contents: "Introduction...............................1" -> "Introduction ... 1"
fn collapse_dot_leaders(text: &str) -> String {
@@ -384,48 +342,6 @@ mod tests {
assert!(result.contains("Chapter 2 ... 20"));
}
// --- remove_spaces_before_closing_brackets ---
#[test]
fn test_remove_spaces_before_closing_brackets() {
let mut input = "Density [kg/m3 ] and [linked text ](https://example.com)".to_string();
remove_spaces_before_closing_brackets(&mut input);
assert_eq!(
input,
"Density [kg/m3] and [linked text](https://example.com)"
);
}
// --- remove_spaces_before_sentence_punctuation ---
#[test]
fn strips_space_before_trailing_period() {
let mut t = "Foreign insurance companies . The provisions".to_string();
remove_spaces_before_sentence_punctuation(&mut t);
assert_eq!(t, "Foreign insurance companies. The provisions");
}
#[test]
fn strips_space_before_period_at_cell_boundary() {
let mut t = "|Applicability date .|This section|".to_string();
remove_spaces_before_sentence_punctuation(&mut t);
assert_eq!(t, "|Applicability date.|This section|");
}
#[test]
fn keeps_dot_leaders_and_ellipses() {
let mut t = "Introduction ... 1".to_string();
remove_spaces_before_sentence_punctuation(&mut t);
assert_eq!(t, "Introduction ... 1");
}
#[test]
fn keeps_mid_token_periods() {
let mut t = "version 3 .14 released".to_string();
remove_spaces_before_sentence_punctuation(&mut t);
assert_eq!(t, "version 3 .14 released");
}
// --- fix_hyphenation ---
#[test]
+1 -102
View File
@@ -3,7 +3,7 @@
use std::collections::{HashMap, HashSet};
use crate::structure_tree::StructRole;
use crate::types::{TextItem, TextLine};
use crate::types::TextLine;
use super::analysis::detect_header_level;
@@ -87,41 +87,6 @@ pub(crate) fn merge_heading_lines(
false
};
// Bold headings at body font size never reach a tier, so wrapped ones
// split into two output headings ("…of wood pellets and cost" /
// "structure in Japan"). Merge a fully-bold line into the previous
// fully-bold line when it reads as a wrap continuation: starts
// lowercase, tiny Y gap, and the previous line has no terminal
// punctuation. Kept deliberately narrow — bold list labels and bold
// sentences start with markers or capitals and are unaffected.
let should_merge = should_merge
|| if let Some(prev) = result.last() {
let all_bold = |l: &TextLine| {
!l.items.is_empty() && l.items.iter().all(|i: &TextItem| i.is_bold)
};
let prev_text = prev.text();
let prev_trim = prev_text.trim_end();
let curr_text = line.text();
let curr_trim = curr_text.trim();
let y_gap = prev.y - line.y;
// Both lines must be tier-less: a tiered/tagged bold heading
// followed by bold body text must not absorb it.
line_level.is_none()
&& effective_heading_level(prev, base_size, heading_tiers, struct_roles)
.is_none()
&& prev.page == line.page
&& all_bold(prev)
&& all_bold(&line)
&& y_gap > 0.0
&& y_gap < line_font * 1.6
&& curr_trim.chars().next().is_some_and(|c| c.is_lowercase())
&& !prev_trim.ends_with(['.', ':', ';', '!', '?'])
&& prev_trim.split_whitespace().count() + curr_trim.split_whitespace().count()
<= 20
} else {
false
};
if should_merge {
// Append this line's items to the previous line
let prev = result.last_mut().unwrap();
@@ -577,8 +542,6 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
is_strikeout: false,
item_type: ItemType::Text,
mcid,
}
@@ -720,68 +683,4 @@ mod tests {
.unwrap();
assert_eq!(first_header.page, 1, "first occurrence should be on page 1");
}
fn make_bold_line(text: &str, page: u32, y: f32) -> TextLine {
let mut item = make_item(text, 12.0, None);
item.is_bold = true;
TextLine {
items: vec![item],
y,
page,
adaptive_threshold: 0.10,
}
}
#[test]
fn merge_wrapped_bold_heading_lowercase_continuation() {
// Bold-at-body-size heading wrapped across two lines: the second line
// starts lowercase and must merge into the first.
let lines = vec![
make_bold_line(
"3. Perspective of supply and demand balance and cost",
1,
700.0,
),
make_bold_line("structure in Japan", 1, 686.0),
make_line("Body text paragraph follows here.", 12.0, 1, 660.0, None),
];
let result = merge_heading_lines(lines, 12.0, &[], None);
assert_eq!(result.len(), 2, "wrapped bold heading should merge");
assert!(result[0].text().contains("cost structure in Japan"));
}
#[test]
fn no_merge_for_bold_sentences_or_new_headings() {
// Second bold line starts with a capital — a new heading or label,
// not a wrap continuation.
let lines = vec![
make_bold_line("Replace", 1, 700.0),
make_bold_line("Trash", 1, 686.0),
];
let result = merge_heading_lines(lines, 12.0, &[], None);
assert_eq!(result.len(), 2, "distinct bold lines must not merge");
// Previous line ends a sentence — continuation must not merge.
let lines = vec![
make_bold_line("This is a bold sentence.", 1, 700.0),
make_bold_line("another bold line", 1, 686.0),
];
let result = merge_heading_lines(lines, 12.0, &[], None);
assert_eq!(result.len(), 2, "sentence-final bold line must not merge");
}
#[test]
fn tiered_bold_heading_does_not_absorb_bold_body() {
// Previous line is a tier-level bold heading (16pt vs 12pt body);
// a following lowercase bold body line must NOT merge into it.
let mut heading = make_bold_line("Section Title", 1, 700.0);
heading.items[0].font_size = 16.0;
heading.items[0].height = 16.0;
let lines = vec![
heading,
make_bold_line("emphasized body text continues here", 1, 686.0),
];
let result = merge_heading_lines(lines, 12.0, &[16.0], None);
assert_eq!(result.len(), 2, "tiered heading must not absorb bold body");
}
}
-176
View File
@@ -30,9 +30,6 @@ pub struct PyPdfResult {
/// 1-indexed page numbers that need OCR.
#[pyo3(get)]
pub pages_needing_ocr: Vec<u32>,
/// Machine-readable OCR reasons by 1-indexed page.
#[pyo3(get)]
pub ocr_reasons_by_page: Vec<PyPageOcrReasons>,
/// Title from PDF metadata.
#[pyo3(get)]
pub title: Option<String>,
@@ -63,28 +60,6 @@ impl PyPdfResult {
}
}
/// OCR reasons for a single 1-indexed page.
#[pyclass(name = "PageOcrReasons")]
#[derive(Clone)]
pub struct PyPageOcrReasons {
/// 1-indexed page number.
#[pyo3(get)]
pub page: u32,
/// Machine-readable OCR reason identifiers.
#[pyo3(get)]
pub reasons: Vec<String>,
}
#[pymethods]
impl PyPageOcrReasons {
fn __repr__(&self) -> String {
format!(
"PageOcrReasons(page={}, reasons={:?})",
self.page, self.reasons
)
}
}
// ---------------------------------------------------------------------------
// Classification wrapper (lightweight)
// ---------------------------------------------------------------------------
@@ -131,9 +106,6 @@ pub struct PyRegionText {
/// True when the text should not be trusted (empty, GID fonts, garbage, encoding issues).
#[pyo3(get)]
pub needs_ocr: bool,
/// Machine-readable OCR reason when the cause is known.
#[pyo3(get)]
pub ocr_reason: Option<String>,
}
#[pymethods]
@@ -174,73 +146,6 @@ impl PyPageRegionTexts {
// Text item wrapper
// ---------------------------------------------------------------------------
/// Per-page markdown extraction result.
#[pyclass(name = "PageMarkdown")]
#[derive(Clone)]
pub struct PyPageMarkdown {
/// 0-indexed page number.
#[pyo3(get)]
pub page: u32,
/// Formatted markdown for this page.
#[pyo3(get)]
pub markdown: String,
/// True when text on this page is unreliable (GID-encoded fonts,
/// encoding issues, garbage text, or empty extraction).
#[pyo3(get)]
pub needs_ocr: bool,
/// Machine-readable OCR reason when the cause is known.
#[pyo3(get)]
pub ocr_reason: Option<String>,
}
#[pymethods]
impl PyPageMarkdown {
fn __repr__(&self) -> String {
format!(
"PageMarkdown(page={}, markdown='{}', needs_ocr={})",
self.page,
self.markdown.chars().take(40).collect::<String>(),
self.needs_ocr
)
}
}
/// Combined per-page markdown extraction and layout classification result.
#[pyclass(name = "PagesExtractionResult")]
#[derive(Clone)]
pub struct PyPagesExtractionResult {
/// Per-page markdown results, in the order requested.
#[pyo3(get)]
pub pages: Vec<PyPageMarkdown>,
/// 1-indexed pages where tables were detected.
#[pyo3(get)]
pub pages_with_tables: Vec<u32>,
/// 1-indexed pages where multi-column layout was detected.
#[pyo3(get)]
pub pages_with_columns: Vec<u32>,
/// 1-indexed pages that need OCR (scanned/image-based or unreliable text).
#[pyo3(get)]
pub pages_needing_ocr: Vec<u32>,
/// Machine-readable OCR reasons by 1-indexed page.
#[pyo3(get)]
pub ocr_reasons_by_page: Vec<PyPageOcrReasons>,
/// True if any page has tables or columns.
#[pyo3(get)]
pub is_complex: bool,
}
#[pymethods]
impl PyPagesExtractionResult {
fn __repr__(&self) -> String {
format!(
"PagesExtractionResult(pages={}, pages_with_tables={:?}, is_complex={})",
self.pages.len(),
self.pages_with_tables,
self.is_complex
)
}
}
/// A positioned text item extracted from a PDF.
#[pyclass(name = "TextItem")]
#[derive(Clone)]
@@ -266,10 +171,6 @@ pub struct PyTextItem {
#[pyo3(get)]
pub is_italic: bool,
#[pyo3(get)]
pub is_underline: bool,
#[pyo3(get)]
pub is_strikeout: bool,
#[pyo3(get)]
pub item_type: String,
}
@@ -306,7 +207,6 @@ fn to_py_result(r: crate::PdfProcessResult) -> PyPdfResult {
page_count: r.page_count,
processing_time_ms: r.processing_time_ms,
pages_needing_ocr: r.pages_needing_ocr,
ocr_reasons_by_page: to_py_page_ocr_reasons(r.ocr_reasons_by_page),
title: r.title,
confidence: r.confidence,
is_complex_layout: r.layout.is_complex,
@@ -316,16 +216,6 @@ fn to_py_result(r: crate::PdfProcessResult) -> PyPdfResult {
}
}
fn to_py_page_ocr_reasons(reasons: Vec<crate::PageOcrReasons>) -> Vec<PyPageOcrReasons> {
reasons
.into_iter()
.map(|reason| PyPageOcrReasons {
page: reason.page,
reasons: reason.reasons,
})
.collect()
}
fn to_py_err(e: crate::PdfError) -> PyErr {
PyValueError::new_err(e.to_string())
}
@@ -353,8 +243,6 @@ fn convert_text_items(items: Vec<crate::TextItem>) -> Vec<PyTextItem> {
page: item.page,
is_bold: item.is_bold,
is_italic: item.is_italic,
is_underline: item.is_underline,
is_strikeout: item.is_strikeout,
item_type: item_type_str(&item.item_type),
})
.collect()
@@ -392,26 +280,6 @@ fn parse_page_regions(
.collect()
}
fn to_py_pages_result(r: crate::PagesExtractionResult) -> PyPagesExtractionResult {
PyPagesExtractionResult {
pages: r
.pages
.into_iter()
.map(|p| PyPageMarkdown {
page: p.page,
markdown: p.markdown,
needs_ocr: p.needs_ocr,
ocr_reason: p.ocr_reason,
})
.collect(),
pages_with_tables: r.pages_with_tables,
pages_with_columns: r.pages_with_columns,
pages_needing_ocr: r.pages_needing_ocr,
ocr_reasons_by_page: to_py_page_ocr_reasons(r.ocr_reasons_by_page),
is_complex: r.is_complex,
}
}
fn convert_region_results(results: Vec<crate::PageRegionResult>) -> Vec<PyPageRegionTexts> {
results
.into_iter()
@@ -423,7 +291,6 @@ fn convert_region_results(results: Vec<crate::PageRegionResult>) -> Vec<PyPageRe
.map(|r| PyRegionText {
text: r.text,
needs_ocr: r.needs_ocr,
ocr_reason: r.ocr_reason,
})
.collect(),
})
@@ -575,55 +442,14 @@ fn extract_text_in_regions_bytes(
Ok(convert_region_results(results))
}
/// Extract formatted markdown for pages of a PDF file, with layout
/// classification metadata.
///
/// Returns per-page markdown and classification data (tables, columns,
/// OCR needs) from a single parse. Font statistics are computed from the
/// full document so header detection is consistent across pages.
///
/// Args:
/// path: Path to the PDF file.
/// pages: Optional list of 0-indexed pages. When None (default), every
/// page is returned in document order. When provided, output
/// matches the caller-supplied order.
///
/// Returns:
/// PagesExtractionResult with per-page markdown and classification data.
#[pyfunction]
#[pyo3(signature = (path, pages=None))]
fn extract_pages_markdown(
path: &str,
pages: Option<Vec<u32>>,
) -> PyResult<PyPagesExtractionResult> {
let result = crate::extract_pages_markdown(path, pages.as_deref()).map_err(to_py_err)?;
Ok(to_py_pages_result(result))
}
/// Extract formatted markdown for pages of a PDF from bytes.
///
/// See [`extract_pages_markdown`] for details.
#[pyfunction]
#[pyo3(signature = (data, pages=None))]
fn extract_pages_markdown_bytes(
data: &[u8],
pages: Option<Vec<u32>>,
) -> PyResult<PyPagesExtractionResult> {
let result = crate::extract_pages_markdown_mem(data, pages.as_deref()).map_err(to_py_err)?;
Ok(to_py_pages_result(result))
}
/// Python module definition.
#[pymodule]
fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
m.add_class::<PyPdfResult>()?;
m.add_class::<PyPageOcrReasons>()?;
m.add_class::<PyPdfClassification>()?;
m.add_class::<PyTextItem>()?;
m.add_class::<PyRegionText>()?;
m.add_class::<PyPageRegionTexts>()?;
m.add_class::<PyPageMarkdown>()?;
m.add_class::<PyPagesExtractionResult>()?;
m.add_function(wrap_pyfunction!(process_pdf, m)?)?;
m.add_function(wrap_pyfunction!(process_pdf_bytes, m)?)?;
m.add_function(wrap_pyfunction!(detect_pdf, m)?)?;
@@ -636,7 +462,5 @@ fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
m.add_function(wrap_pyfunction!(extract_text_with_positions_bytes, m)?)?;
m.add_function(wrap_pyfunction!(extract_text_in_regions, m)?)?;
m.add_function(wrap_pyfunction!(extract_text_in_regions_bytes, m)?)?;
m.add_function(wrap_pyfunction!(extract_pages_markdown, m)?)?;
m.add_function(wrap_pyfunction!(extract_pages_markdown_bytes, m)?)?;
Ok(())
}
-94
View File
@@ -76,52 +76,6 @@ pub enum StructRole {
}
impl StructRole {
/// Content roles whose text must never be promoted to a heading by the
/// visual heuristic. These carry an explicit non-heading meaning in the
/// struct tree (lists, quotes, notes, references, captions, formulas,
/// forms, ToC entries), yet their text is often short and visually
/// isolated — exactly what the heuristic keys on. Heading roles (H, H1H6)
/// and generic container/flow roles (P, Div, Sect, Span, …) are excluded
/// so the heuristic can still fire there.
///
/// `Figure` is deliberately NOT in this set: cover/banner pages routinely
/// tag the document title inside a Figure (alongside a seal or logo), and
/// that title is a real heading. `Formula` and `Form` stay — a line
/// explicitly tagged as an equation or form field is never a heading.
///
/// Table roles (Table/TR/TH/TD/THead/TBody/TFoot) are included so that
/// when table reconstruction falls back and cells reach the line loop as
/// plain text, a short isolated cell — a `TH` column header especially —
/// is not promoted to a heading.
pub(crate) fn is_non_heading_content(&self) -> bool {
matches!(
self,
Self::L
| Self::LI
| Self::Lbl
| Self::LBody
| Self::BlockQuote
| Self::Quote
| Self::Caption
| Self::TOC
| Self::TOCI
| Self::Index
| Self::Note
| Self::Reference
| Self::BibEntry
| Self::Code
| Self::Formula
| Self::Form
| Self::Table
| Self::TR
| Self::TH
| Self::TD
| Self::THead
| Self::TBody
| Self::TFoot
)
}
fn from_name(name: &str) -> Self {
match name {
"Document" => Self::Document,
@@ -902,54 +856,6 @@ fn contains_bytes(haystack: &[u8], needle: &[u8]) -> bool {
mod tests {
use super::*;
#[test]
fn non_heading_content_roles() {
for r in [
StructRole::L,
StructRole::LI,
StructRole::BlockQuote,
StructRole::Quote,
StructRole::Caption,
StructRole::TOC,
StructRole::TOCI,
StructRole::Index,
StructRole::Note,
StructRole::Reference,
StructRole::BibEntry,
StructRole::Code,
StructRole::Formula,
StructRole::Form,
StructRole::Table,
StructRole::TR,
StructRole::TH,
StructRole::TD,
StructRole::THead,
StructRole::TBody,
StructRole::TFoot,
] {
assert!(
r.is_non_heading_content(),
"{r:?} should block heading promotion"
);
}
// Heading and generic container/flow roles must NOT block promotion
for r in [
StructRole::H,
StructRole::H1,
StructRole::H3,
StructRole::P,
StructRole::Div,
StructRole::Sect,
StructRole::Span,
StructRole::Figure,
] {
assert!(
!r.is_non_heading_content(),
"{r:?} should allow heading promotion"
);
}
}
#[test]
fn test_struct_role_from_name() {
assert_eq!(StructRole::from_name("H1"), StructRole::H1);
+64 -552
View File
@@ -104,8 +104,6 @@ pub(crate) fn merge_adjacent_items(items: &[TextItem]) -> (Vec<TextItem>, Vec<Ve
page: first_item.page,
is_bold: first_item.is_bold,
is_italic: first_item.is_italic,
is_underline: first_item.is_underline,
is_strikeout: first_item.is_strikeout,
item_type: first_item.item_type.clone(),
mcid: first_item.mcid,
});
@@ -583,14 +581,8 @@ fn detect_table_in_region(items: &[(usize, &TextItem)], mode: TableDetectionMode
// Validation 1: some rows should have content in first column.
// Use a lower threshold (25%) for tables with wrapped cells where
// continuation lines leave the first column empty.
// Skip when cells form a narrow TOC pattern: hierarchical entries indented
// across multiple X levels leave the leftmost column sparse (only top-level
// chapters land there) but the structure is still a valid TOC. Narrow only
// (<=5 cols) — wide multi-column TOCs (e.g. 2-up indices) would render
// poorly through format_toc_as_list, which assumes one entry per row.
let rows_with_first_col = cells.iter().filter(|row| !row[0].is_empty()).count();
let is_narrow_toc = columns.len() <= 5 && is_table_of_contents(&cells);
if rows_with_first_col < rows.len() / 4 && !is_narrow_toc {
if rows_with_first_col < rows.len() / 4 {
log::debug!(
" validation 1 fail: {}/{} rows have first col",
rows_with_first_col,
@@ -661,24 +653,15 @@ fn detect_table_in_region(items: &[(usize, &TextItem)], mode: TableDetectionMode
return None;
}
// Validation 8: Reject paragraph-like content falsely detected as tables.
// TOC pages with deep indentation (top-level chapters in col 0, subsections
// in cols 1-3, page numbers in last col) leave most cells empty and trip
// the paragraph heuristic; TOC shape is a safer signal here. Narrow only
// — see narrow-TOC rationale at validation 1.
if is_paragraph_content(&cells) && !is_narrow_toc {
log::debug!(" validation 9 fail: paragraph content");
// Validation 8: Check for Table of Contents pattern
if is_table_of_contents(&cells) {
log::debug!(" validation 8 fail: table of contents");
return None;
}
// Validation 9: Reject wide "index" layouts where every cell carries a
// full "label ... page" fragment (back-of-book IRS-style indices).
// These render poorly in any structured form; text flow is the best
// fallback. Narrow dot-leader TOCs (2-3 cols) are kept so format.rs
// can emit them as a per-row flat list with titles tab-joined to page
// numbers.
if is_inline_leader_index(&cells) {
log::debug!(" validation 9 fail: inline-leader index");
// Validation 9: Reject paragraph-like content falsely detected as tables
if is_paragraph_content(&cells) {
log::debug!(" validation 9 fail: paragraph content");
return None;
}
@@ -689,7 +672,12 @@ fn detect_table_in_region(items: &[(usize, &TextItem)], mode: TableDetectionMode
item_indices.len()
);
Some(Table::new(columns, rows, cells, item_indices))
Some(Table {
columns,
rows,
cells,
item_indices,
})
}
/// Check if this looks like a key-value pair layout rather than a table
@@ -910,247 +898,77 @@ fn looks_like_number(s: &str) -> bool {
&& s.chars().any(|c| c.is_ascii_digit())
}
/// Check if this looks like a Table of Contents (either style).
///
/// Used by format.rs to render TOCs as flat lists instead of markdown tables.
pub fn is_table_of_contents(cells: &[Vec<String>]) -> bool {
is_dot_leader_toc(cells) || is_tabular_toc(cells)
}
/// Dot-leader TOC: any "Chapter 1 ........ 42" style with explicit leader
/// dots. Covers both narrow 2-3 col TOCs (where the leader is a dedicated
/// cell) and wide indices (where each cell encodes a full "label ... page"
/// fragment). Used by format.rs to render as a flat list.
pub(super) fn is_dot_leader_toc(cells: &[Vec<String>]) -> bool {
has_structural_dot_leader(cells) || is_inline_leader_index(cells)
}
/// Rows with a dedicated dots-only cell flanked by label + number (2-3 col
/// TOC layout). Format.rs handles these well via per-row flat-list
/// rendering; they should NOT be rejected at detect time.
fn has_structural_dot_leader(cells: &[Vec<String>]) -> bool {
/// Check if this looks like a Table of Contents
/// TOCs have characteristic patterns: leader dots, page numbers, section names
fn is_table_of_contents(cells: &[Vec<String>]) -> bool {
if cells.is_empty() {
return false;
}
let structural_rows = cells.iter().filter(|row| row_has_dot_leader(row)).count();
structural_rows as f32 / cells.len() as f32 >= 0.3
}
/// Wide index layout: each cell holds a full "label ... page" fragment
/// because the column detector kept multi-column indices as single cells.
/// These render poorly both as markdown tables (column boundaries are
/// arbitrary) and as flat lists (each row holds 3+ separate index
/// entries). Reject these at detect time so they fall back to the page's
/// normal text flow.
pub(super) fn is_inline_leader_index(cells: &[Vec<String>]) -> bool {
let mut inline_cells = 0;
let mut total_nonempty = 0;
let num_cols = cells[0].len();
let mut dot_cells = 0;
let mut page_number_cells = 0;
let mut total_cells = 0;
// Track which columns contain dots vs numbers to distinguish
// TOC (dots span middle, page number at end) from data tables
// (dots only in label column, many number columns).
let mut dot_cols = vec![0u32; num_cols];
let mut numeric_cols = vec![0u32; num_cols];
for row in cells {
for cell in row {
for (ci, cell) in row.iter().enumerate() {
let trimmed = cell.trim();
if trimmed.is_empty() {
continue;
}
total_nonempty += 1;
if cell_is_inline_leader(trimmed) {
inline_cells += 1;
total_cells += 1;
// Check for leader dots (sequences of periods)
// TOCs often have "........" or ". . . ." patterns
let dot_count = trimmed.chars().filter(|&c| c == '.').count();
let is_mostly_dots = dot_count > trimmed.len() / 2 && dot_count >= 3;
if is_mostly_dots {
dot_cells += 1;
if ci < num_cols {
dot_cols[ci] += 1;
}
}
// Check for standalone page numbers (1-4 digits, possibly with spaces)
let digits_only: String = trimmed.chars().filter(|c| !c.is_whitespace()).collect();
if digits_only.len() <= 4
&& !digits_only.is_empty()
&& digits_only.chars().all(|c| c.is_ascii_digit())
{
page_number_cells += 1;
if ci < num_cols {
numeric_cols[ci] += 1;
}
}
}
}
total_nonempty >= 4 && inline_cells as f32 / total_nonempty as f32 >= 0.25
}
/// A row with a dot-leader. Accepts two layouts:
/// 1. A dedicated dots-only cell ("....") with a text label somewhere
/// to its left and a page number somewhere to its right.
/// 2. A "title ... " cell (trailing leader dots glued to the title)
/// with a page number elsewhere in the same row.
fn row_has_dot_leader(row: &[String]) -> bool {
let has_page_number = row.iter().any(|c| row_cell_is_page_number(c));
for (ci, cell) in row.iter().enumerate() {
let trimmed = cell.trim();
// Pattern 1: dedicated dots-only cell.
let dot_count = trimmed.chars().filter(|&c| c == '.').count();
let is_mostly_dots = dot_count >= 3
&& dot_count > trimmed.len() / 2
&& trimmed.chars().all(|c| c == '.' || c.is_whitespace());
if is_mostly_dots {
let has_label_left = row[..ci].iter().any(|c| {
let t = c.trim();
!t.is_empty() && t.chars().any(|ch| ch.is_alphabetic())
});
if has_label_left && has_page_number {
return true;
}
continue;
}
// Pattern 2: cell ends with a trailing " ... " run after a label.
if has_page_number && cell_has_trailing_leader(trimmed) {
return true;
}
}
false
}
/// Cell ends with a run of ≥3 dots preceded by alphabetic text and a
/// space — the "Title ... " layout where the leader is glued to the name.
/// Alphabetic (not alphanumeric) so that data-table row labels like
/// "1973 ... " do not register as titles.
fn cell_has_trailing_leader(cell: &str) -> bool {
let trimmed = cell.trim_end();
if !trimmed.ends_with('.') {
return false;
}
let without_dots = trimmed.trim_end_matches('.');
let dot_run = trimmed.len() - without_dots.len();
if dot_run < 3 {
return false;
}
// Require a space before the dot run (rules out "etc..." / "Mr...") and
// at least one alphabetic char (rules out "1973 ... " data-row labels).
without_dots.ends_with(' ') && without_dots.trim().chars().any(|c| c.is_alphabetic())
}
/// Page-number shape: single ≤4-digit integer, a ", "-separated list of
/// ≤4-digit integers ("18, 36, 107"), or a dashed section-page ID
/// ("A-1", "5-21"). Rejects decimal cells ("4. 0"), thousands-separated
/// values ("189,164"), and other long numeric data that appears in
/// statistical tables.
fn row_cell_is_page_number(cell: &str) -> bool {
let t = cell.trim();
if t.is_empty() {
return false;
}
if looks_like_section_page_id(t) {
return true;
}
// Page list: ", " separator (with space) distinguishes real page lists
// from thousands-separated numbers like "189,164".
let parts: Vec<&str> = t.split(", ").collect();
parts
.iter()
.all(|p| !p.is_empty() && p.len() <= 4 && p.chars().all(|c| c.is_ascii_digit()))
}
/// A cell shaped like an index leader fragment. Accepts two forms:
/// - "text ... number" — label + dots + page number in one cell
/// - "... number" — bare leader + number (row where the label
/// landed in a separate column)
///
/// Both only count if followed by pure numeric content (optionally
/// comma-separated page lists like "127, 213").
fn cell_is_inline_leader(cell: &str) -> bool {
let cell = cell.trim();
// Find the first "..." run. Surrounding-whitespace checks below
// reject intra-word ellipses ("etc...").
let idx = match cell.match_indices("...").next() {
Some((i, _)) => i,
None => return false,
};
let before = &cell[..idx];
let after_dots = &cell[idx + 3..];
// Allow extra dots (e.g. "....") by skipping any additional '.'
let after = after_dots.trim_start_matches('.');
// Require space (or start-of-cell) before the dots and space/digit
// after — blocks intra-word ellipses.
let before_ok = before.is_empty() || before.ends_with(' ');
let after_ok = after.starts_with(' ') || after.is_empty();
if !before_ok || !after_ok {
if total_cells == 0 {
return false;
}
let after_trim = after.trim();
if after_trim.is_empty() {
return false;
}
// Tail must be purely numeric/page-list content.
let tail_numeric = after_trim
.chars()
.all(|c| c.is_ascii_digit() || matches!(c, ',' | ' ' | '.' | '-' | '$'))
&& after_trim.chars().any(|c| c.is_ascii_digit());
if !tail_numeric {
// Data tables with dot leaders (e.g. "1973....") have dots concentrated
// in one column (the label column) while many other columns contain numbers.
// True TOCs have dots spanning the middle and one page-number column at the end.
// If dots are confined to ≤1 column AND there are ≥3 columns with numbers,
// this is a data table, not a TOC.
let cols_with_dots = dot_cols.iter().filter(|&&c| c >= 2).count();
let cols_with_numbers = numeric_cols.iter().filter(|&&c| c >= 2).count();
if cols_with_dots <= 1 && cols_with_numbers >= 3 {
return false;
}
// Either we have a label before, or the leader is bare (starts the cell)
// — both are legitimate index fragments.
before.chars().any(|c| c.is_alphabetic()) || before.trim().is_empty()
}
// If a significant portion of cells are dots or page numbers, it's likely a TOC
let dot_ratio = dot_cells as f32 / total_cells as f32;
let page_num_ratio = page_number_cells as f32 / total_cells as f32;
/// Dot-less tabular TOC: tagged PDFs emit entries as rows where the first
/// column starts with a dotted section number (e.g. "4.3.1 Something") and
/// the last column is one or more page numbers. These have no leader dots
/// and benefit from flat-list formatting (page numbers aligned to titles).
pub(super) fn is_tabular_toc(cells: &[Vec<String>]) -> bool {
if cells.is_empty() {
return false;
}
let num_cols = cells[0].len();
if num_cols < 2 || cells.len() < 4 {
return false;
}
let section_rows = cells
.iter()
.filter(|row| {
row.iter()
.find(|c| !c.trim().is_empty())
.is_some_and(|c| starts_with_section_number(c.trim()))
})
.count();
let last_col = num_cols - 1;
let (last_filled, last_page_num) = cells.iter().fold((0u32, 0u32), |(f, n), row| {
let cell = row.get(last_col).map(|s| s.trim()).unwrap_or("");
if cell.is_empty() {
return (f, n);
}
let is_page_nums = cell
.split_whitespace()
.all(|tok| !tok.is_empty() && tok.chars().all(|c| c.is_ascii_digit()));
(f + 1, n + if is_page_nums { 1 } else { 0 })
});
let section_ratio = section_rows as f32 / cells.len() as f32;
let page_num_last_ratio = if last_filled > 0 {
last_page_num as f32 / last_filled as f32
} else {
0.0
};
section_ratio >= 0.6 && last_filled >= 3 && page_num_last_ratio >= 0.7
}
/// Matches dashed section-page identifiers used in technical manuals:
/// "5-21", "A-1", "B--3", "TC-2". At least one ASCII digit is required.
fn looks_like_section_page_id(s: &str) -> bool {
let ok = s
.chars()
.all(|c| c.is_ascii_digit() || c.is_ascii_uppercase() || c == '-');
ok && s.chars().any(|c| c.is_ascii_digit())
}
/// Returns true when the leading token looks like a dotted section number:
/// "1", "1.2", "1.2.3", "4.3.1.2" — integer components joined by dots,
/// with at least one dot (single-number prefixes are too ambiguous).
fn starts_with_section_number(s: &str) -> bool {
let Some(first) = s.split_whitespace().next() else {
return false;
};
let first = first.trim_end_matches('.');
let parts: Vec<&str> = first.split('.').collect();
if parts.len() < 2 || parts.len() > 6 {
return false;
}
parts
.iter()
.all(|p| !p.is_empty() && p.len() <= 3 && p.chars().all(|c| c.is_ascii_digit()))
// TOC typically has >15% dot cells and >10% page number cells
dot_ratio > 0.15 || (dot_ratio > 0.05 && page_num_ratio > 0.15)
}
/// Check if detected "table" cells are actually paragraph text fragments.
@@ -1326,33 +1144,6 @@ pub(crate) fn find_first_table_row(
continue;
}
// Skip rows that have duplicate non-empty cells. These are spanning
// super-headers (e.g., "First Degree | First Degree | Higher Degree")
// that sit above the real column header row. Using them as the markdown
// header produces duplicate column names that downstream validation
// rejects. Only skip if a subsequent row looks like a better header
// (denser fill or has data).
if filled_count >= 2 && !has_data {
let mut text_counts: std::collections::HashMap<&str, usize> =
std::collections::HashMap::new();
for cell in &filled_cells {
*text_counts.entry(cell.trim()).or_insert(0) += 1;
}
let has_duplicates = text_counts.values().any(|&count| count >= 2);
if has_duplicates {
// Check if a later row is a better header candidate
let has_better_below = cells.iter().skip(row_idx + 1).take(3).any(|r| {
let next_filled = r.iter().filter(|c| !c.trim().is_empty()).count();
let next_fill = next_filled as f32 / total_cols as f32;
let next_numeric = r.iter().filter(|c| looks_like_number(c.trim())).count();
next_fill >= 0.4 || next_numeric >= 2
});
if has_better_below {
continue;
}
}
}
// Data rows are definitely table content
if has_data {
first_table_row = row_idx;
@@ -1607,283 +1398,4 @@ mod tests {
"data table with dot-leader labels should not be rejected as TOC"
);
}
#[test]
fn is_table_of_contents_accepts_hierarchical_indented_toc() {
// Mythos system card pages 4-5: top-level chapters indent at col 0,
// subsections at cols 1-2, leaving col 0 mostly empty (only ~10% of
// rows). Validation 1 was rejecting these even though the structure
// is unambiguously a TOC.
let cells = vec![
vec!["Abstract".to_string(), String::new(), "3".to_string()],
vec![
"1 Introduction".to_string(),
String::new(),
"10".to_string(),
],
vec![
String::new(),
"1.1 Model training".to_string(),
"11".to_string(),
],
vec![
String::new(),
"1.1.1 Training data".to_string(),
"11".to_string(),
],
vec![
String::new(),
"1.1.2 Crowd workers".to_string(),
"12".to_string(),
],
vec![
String::new(),
"1.2 Release decision".to_string(),
"13".to_string(),
],
vec![
"2 RSP evaluations".to_string(),
String::new(),
"16".to_string(),
],
vec![
String::new(),
"2.1 RSP risk assessment".to_string(),
"16".to_string(),
],
vec![String::new(), "2.1.1 Context".to_string(), "16".to_string()],
vec![
String::new(),
"2.2 CB evaluations".to_string(),
"20".to_string(),
],
];
assert!(
is_table_of_contents(&cells),
"hierarchical TOC with sparse col 0 should still be detected"
);
}
#[test]
fn is_table_of_contents_rejects_dotless_toc() {
// Tabular TOC without leader dots: first column starts with dotted
// section numbers, last column is page numbers. Pattern from
// Mythos system card pages 6-8.
let cells = vec![
vec![
"4.3 Case studies and targeted evaluations".to_string(),
String::new(),
"86".to_string(),
],
vec![
"4.3.1 Destructive or reckless actions".to_string(),
"4.3.1.1 Synthetic-backend evaluation".to_string(),
"86 86".to_string(),
],
vec![
"4.3.2 Adherence to constitution".to_string(),
"4.3.2.1 Overview".to_string(),
"89 89".to_string(),
],
vec![
"4.3.3 Honesty and hallucinations".to_string(),
"4.3.3.1 Factual hallucinations".to_string(),
"93 94".to_string(),
],
vec![
"4.4 Capability evaluations".to_string(),
String::new(),
"101".to_string(),
],
];
assert!(
is_table_of_contents(&cells),
"dot-less TOC with section numbers + page numbers should be rejected"
);
}
#[test]
fn dot_leader_toc_accepts_short_inline_leaders() {
// Index-style cells where the full "label ... number" pattern is
// preserved in a single cell (IRS Publication 17 back-of-book index).
let cells = vec![
vec!["Child tax credit ... 235".to_string(), String::new()],
vec!["Church employee ... 252".to_string(), String::new()],
vec!["Citizens outside the U.S ... 6".to_string(), String::new()],
vec![
"Claim for refund ... 18, 36, 107".to_string(),
String::new(),
],
vec!["Clergy ... 7, 52".to_string(), String::new()],
];
assert!(is_dot_leader_toc(&cells));
}
#[test]
fn dot_leader_toc_allows_ellipsis_data_table() {
// Data tables using "..." as a row-omission marker must not be
// mistaken for dot-leader TOCs. Based on MCF5235RM QSPI RAM layout.
let cells = vec![
vec![
"0x00".to_string(),
"QTR0".to_string(),
"Transmit RAM".to_string(),
],
vec!["0x01".to_string(), "QTR1".to_string(), String::new()],
vec![
"...".to_string(),
"...".to_string(),
"16 bits wide".to_string(),
],
vec!["0x0F".to_string(), "QTR15".to_string(), String::new()],
vec![
"0x10".to_string(),
"QRR0".to_string(),
"Receive RAM".to_string(),
],
vec!["0x11".to_string(), "QRR1".to_string(), String::new()],
vec![
"...".to_string(),
"...".to_string(),
"16 bits wide".to_string(),
],
vec!["0x1F".to_string(), "QRR15".to_string(), String::new()],
];
assert!(
!is_dot_leader_toc(&cells),
"ellipsis markers in a data table should not match TOC detection"
);
}
#[test]
fn dot_leader_toc_rejects_year_row_data_table() {
// ERP-2025 economic data tables: year labels with trailing " ... ",
// a final " ... " column, and decimal-looking numeric cells. The
// detection previously classified these as dot-leader TOCs and
// routed them through flat-list formatting, destroying the grid.
let cells = vec![
vec![
"1973 ... ".to_string(),
"4. 0".to_string(),
"1. 8".to_string(),
"0. 4".to_string(),
"3. 2".to_string(),
" ... ".to_string(),
],
vec![
"1974 ... ".to_string(),
"1. 9".to_string(),
"1. 6".to_string(),
"5. 6".to_string(),
"2. 4".to_string(),
" ... ".to_string(),
],
vec![
"1975 ... ".to_string(),
"2. 6".to_string(),
"5. 1".to_string(),
"6. 1".to_string(),
"4. 1".to_string(),
" ... ".to_string(),
],
vec![
"1976 ... ".to_string(),
"4. 3".to_string(),
"5. 4".to_string(),
"6. 4".to_string(),
"4. 5".to_string(),
" ... ".to_string(),
],
];
assert!(
!is_dot_leader_toc(&cells),
"year-indexed data tables with decimal cells must not match TOC detection"
);
}
#[test]
fn dot_leader_toc_rejects_monthly_data_table() {
// ERP-2025 Table B-22: monthly labor-force rows with "Jan ... ",
// "Feb ... " labels and thousands-separated cells ("189,164").
// Previously matched TOC detection because "Jan ..." has alphabetic
// text and "189,164" passed the page-number shape check.
let cells = vec![
vec![
"2023: Jan ... ".to_string(),
"265,962".to_string(),
"165,871".to_string(),
"160,152".to_string(),
"62. 4".to_string(),
],
vec![
"Feb ... ".to_string(),
"266,112".to_string(),
"166,263".to_string(),
"160,301".to_string(),
"62. 5".to_string(),
],
vec![
"Mar ... ".to_string(),
"266,272".to_string(),
"166,690".to_string(),
"160,824".to_string(),
"62. 6".to_string(),
],
vec![
"Apr ... ".to_string(),
"266,443".to_string(),
"166,678".to_string(),
"160,962".to_string(),
"62. 6".to_string(),
],
];
assert!(
!is_dot_leader_toc(&cells),
"monthly labor-force rows with thousands-separated data must not match TOC detection"
);
}
#[test]
fn tabular_toc_requires_section_numbers_and_pages() {
// Dot-less tabular TOC matches is_tabular_toc but not dot-leader.
let cells = vec![
vec![
"4.3 Case studies".to_string(),
String::new(),
"86".to_string(),
],
vec![
"4.3.1 Destructive actions".to_string(),
String::new(),
"86".to_string(),
],
vec![
"4.3.2 Adherence".to_string(),
String::new(),
"89".to_string(),
],
vec!["4.3.3 Honesty".to_string(), String::new(), "93".to_string()],
];
assert!(is_tabular_toc(&cells));
assert!(!is_dot_leader_toc(&cells));
}
#[test]
fn starts_with_section_number_matches_dotted() {
assert!(starts_with_section_number("1.2"));
assert!(starts_with_section_number("4.3.1"));
assert!(starts_with_section_number("4.3.1.2"));
assert!(starts_with_section_number("4.3 Case studies"));
assert!(starts_with_section_number("2.2.5.1 Expert red teaming"));
}
#[test]
fn starts_with_section_number_rejects_non_sections() {
assert!(!starts_with_section_number("Chapter 1"));
assert!(!starts_with_section_number("1973"));
assert!(!starts_with_section_number("1.5M"));
assert!(!starts_with_section_number("10.0%"));
assert!(!starts_with_section_number(""));
assert!(!starts_with_section_number("Hello world"));
}
}
+33 -267
View File
@@ -4,74 +4,11 @@
//! gridlines. Many IRS forms and government PDFs use these instead of
//! `re` (rectangle) operators.
use std::collections::HashSet;
use crate::tables::Table;
use crate::types::{PdfLine, TextItem};
use super::detect_rects::{assign_items_to_grid, snap_edges};
/// Derive column edges from the x-endpoints of horizontal-rule
/// segments when no vertical lines were drawn.
///
/// Catalog and archival-finding-aid tables are commonly drawn with
/// per-row horizontal rules broken into N segments (one segment per
/// cell), with no vertical dividers at all. The segment break points
/// (e.g. `[50, 127], [127, 485], [485, 562]` per row) implicitly
/// encode the column boundaries.
///
/// Returns column edges if ≥3 distinct x-positions each show up as a
/// segment endpoint on ≥50% of the unique horizontal-line rows.
/// Returns `None` otherwise — decorative rules with varying widths
/// shouldn't be mistaken for a table.
fn derive_columns_from_horizontal_segments(horizontals: &[(f32, f32, f32)]) -> Option<Vec<f32>> {
if horizontals.len() < 3 {
return None;
}
let mut endpoints: Vec<f32> = Vec::with_capacity(horizontals.len() * 2);
for &(_, x_min, x_max) in horizontals {
endpoints.push(x_min);
endpoints.push(x_max);
}
let clusters = snap_edges(&endpoints, 5.0);
if clusters.len() < 3 {
return None;
}
// Bucket y-values to count unique rows. Tolerance ~0.1pt (×10
// rounding) tolerates the snap_edges 3pt clustering used later
// for row edges.
let unique_rows: HashSet<i32> = horizontals
.iter()
.map(|&(y, _, _)| (y * 10.0).round() as i32)
.collect();
if unique_rows.len() < 2 {
return None;
}
let min_rows = (unique_rows.len() as f32 * 0.5).ceil() as usize;
let qualifying: Vec<f32> = clusters
.iter()
.copied()
.filter(|&cluster_x| {
let rows_touched: HashSet<i32> = horizontals
.iter()
.filter(|&&(_, x_min, x_max)| {
(x_min - cluster_x).abs() < 5.0 || (x_max - cluster_x).abs() < 5.0
})
.map(|&(y, _, _)| (y * 10.0).round() as i32)
.collect();
rows_touched.len() >= min_rows
})
.collect();
if qualifying.len() < 3 {
return None;
}
Some(qualifying)
}
/// Detect tables from line segments on a given page.
///
/// Lines are classified as horizontal or vertical, snapped into grid edges,
@@ -115,50 +52,25 @@ pub fn detect_tables_from_lines(items: &[TextItem], lines: &[PdfLine], page: u32
// Diagonal lines are ignored
}
if horizontals.len() < 3 {
if horizontals.len() < 3 || verticals.len() < 2 {
return Vec::new();
}
// If no/very-few vertical lines are drawn, try to derive column edges
// from the x-endpoints of the horizontal-rule segments. Catalog and
// archival-finding-aid layouts commonly draw each row's horizontal
// rule as N segments (one per cell), with no vertical dividers at
// all — the segment break points encode the column boundaries.
let implicit_col_edges: Option<Vec<f32>> = if verticals.len() < 2 {
derive_columns_from_horizontal_segments(&horizontals)
} else {
None
};
if verticals.len() < 2 && implicit_col_edges.is_none() {
return Vec::new();
}
let cols_from_segments = implicit_col_edges.is_some();
log::debug!(
"detect_lines p{}: {} horiz, {} vert lines (of {} total on page){}",
"detect_lines p{}: {} horiz, {} vert lines (of {} total on page)",
page,
horizontals.len(),
verticals.len(),
page_lines.len(),
if cols_from_segments {
" — columns from horizontal segments"
} else {
""
}
page_lines.len()
);
// Snap Y-values of horizontal lines → row edges
let h_ys: Vec<f32> = horizontals.iter().map(|(y, _, _)| *y).collect();
let row_edges = snap_edges(&h_ys, 3.0);
// Column edges from drawn verticals when present, else from the
// horizontal-segment endpoints derived above.
let col_edges = if let Some(c) = implicit_col_edges {
c
} else {
let v_xs: Vec<f32> = verticals.iter().map(|(x, _, _)| *x).collect();
snap_edges(&v_xs, 3.0)
};
// Snap X-values of vertical lines → column edges
let v_xs: Vec<f32> = verticals.iter().map(|(x, _, _)| *x).collect();
let col_edges = snap_edges(&v_xs, 3.0);
log::debug!(
"detect_lines p{}: {} row edges, {} col edges after snap",
@@ -198,21 +110,15 @@ pub fn detect_tables_from_lines(items: &[TextItem], lines: &[PdfLine], page: u32
return Vec::new();
}
// Reject page-spanning frames: a decorative outer border has just 4
// edges (top/bottom/left/right). Real full-page tables — common in
// governmental ledgers, financial reports, etc. — span the same A4 /
// Letter dimensions but have many internal row/column rules. Only
// reject when the line set looks like a bare frame, not a grid.
// Reject page-spanning frames: if the grid covers >90% of a standard page
// dimension in both axes, it's a border frame, not a table.
// Standard pages are ~595×842 (A4) or ~612×792 (Letter).
if table_width > 500.0 && table_height > 700.0 && horizontals.len() <= 4 && verticals.len() <= 4
{
if table_width > 500.0 && table_height > 700.0 {
log::debug!(
"detect_lines p{}: rejected — page-spanning frame ({:.0}×{:.0}, {} h + {} v)",
"detect_lines p{}: rejected — page-spanning frame ({:.0}×{:.0})",
page,
table_width,
table_height,
horizontals.len(),
verticals.len()
table_height
);
return Vec::new();
}
@@ -240,33 +146,24 @@ pub fn detect_tables_from_lines(items: &[TextItem], lines: &[PdfLine], page: u32
// Validate vertical lines: at least 2 should span a meaningful height.
// Full spanning (>30%) is ideal, but accept many shorter lines (>10%)
// for tables with partial column separators. Skipped entirely when
// columns came from horizontal-segment endpoints — there are no
// vertical lines to validate against, and the segment-endpoint
// consistency check in `derive_columns_from_horizontal_segments`
// is the equivalent guard.
let spanning_v = if cols_from_segments {
0
} else {
let s = verticals
.iter()
.filter(|(_, y_min, y_max)| (y_max - y_min) > table_height * 0.3)
.count();
let p = verticals
.iter()
.filter(|(_, y_min, y_max)| (y_max - y_min) > table_height * 0.10)
.count();
if s < 2 && p < 4 {
log::debug!(
"detect_lines p{}: rejected — {} spanning + {} partial V lines",
page,
s,
p
);
return Vec::new();
}
s
};
// for tables with partial column separators.
let spanning_v = verticals
.iter()
.filter(|(_, y_min, y_max)| (y_max - y_min) > table_height * 0.3)
.count();
let partial_v = verticals
.iter()
.filter(|(_, y_min, y_max)| (y_max - y_min) > table_height * 0.10)
.count();
if spanning_v < 2 && partial_v < 4 {
log::debug!(
"detect_lines p{}: rejected — {} spanning + {} partial V lines",
page,
spanning_v,
partial_v
);
return Vec::new();
}
// Row edges need to be in descending order (top of page = higher Y first)
let mut row_edges_desc = row_edges;
@@ -368,12 +265,12 @@ pub fn detect_tables_from_lines(items: &[TextItem], lines: &[PdfLine], page: u32
page, num_rows, num_cols, item_indices.len(), page_item_count, non_empty_rows, cols_with_content
);
vec![Table::new(
col_edges,
row_edges_desc[..num_rows].to_vec(),
vec![Table {
columns: col_edges,
rows: row_edges_desc[..num_rows].to_vec(),
cells,
item_indices,
)]
}]
}
#[cfg(test)]
@@ -393,8 +290,6 @@ mod tests {
page,
is_bold: false,
is_italic: false,
is_underline: false,
is_strikeout: false,
item_type: ItemType::Text,
mcid: None,
}
@@ -515,135 +410,6 @@ mod tests {
assert!(tables.is_empty());
}
#[test]
fn test_horizontal_segments_only_implicit_columns_accepted() {
// Catalog/finding-aid pattern: each row's horizontal rule is
// drawn as 3 segments at consistent x-endpoints (50, 127, 485,
// 562), with no vertical lines anywhere. The segment break
// points must be inferred as column edges.
let mut lines = Vec::new();
// Slightly uneven row spacing so the chart-gridline rejector
// (CV < 0.02) doesn't fire.
let row_ys = [80.0_f32, 145.0, 215.0, 280.0, 350.0, 415.0, 485.0];
for &y in &row_ys {
lines.push(make_hline(y, 50.0, 127.0, 1));
lines.push(make_hline(y, 127.0, 485.0, 1));
lines.push(make_hline(y, 485.0, 562.0, 1));
}
// Populate every cell so capture / density checks pass.
let mut items = Vec::new();
for w in row_ys.windows(2) {
let row_y = (w[0] + w[1]) / 2.0;
items.push(make_item("id", 80.0, row_y, 1));
items.push(make_item("description here", 200.0, row_y, 1));
items.push(make_item("date", 510.0, row_y, 1));
}
let tables = detect_tables_from_lines(&items, &lines, 1);
assert_eq!(
tables.len(),
1,
"horizontal-segment-only grid should be accepted"
);
let t = &tables[0];
assert!(
t.cells.len() >= 4,
"expected ≥4 rows, got {}",
t.cells.len()
);
assert_eq!(t.cells[0].len(), 3, "expected 3 columns");
}
#[test]
fn test_horizontal_segments_with_inconsistent_endpoints_rejected() {
// Decorative rules of varying widths shouldn't be detected as a
// table — each line has its own x-endpoints, no consistent
// column boundary survives the 50%-of-rows threshold.
let lines = vec![
make_hline(100.0, 50.0, 150.0, 1),
make_hline(200.0, 50.0, 220.0, 1),
make_hline(300.0, 50.0, 310.0, 1),
make_hline(400.0, 50.0, 470.0, 1),
];
let items = vec![
make_item("decorative", 100.0, 150.0, 1),
make_item("text", 100.0, 250.0, 1),
];
let tables = detect_tables_from_lines(&items, &lines, 1);
assert!(
tables.is_empty(),
"varying-width decorative rules should not be detected"
);
}
#[test]
fn test_page_spanning_bare_frame_rejected() {
// Just an outer A4-sized rectangle: 2 horizontals + 2 verticals.
// No internal structure → decorative border, not a table.
let lines = vec![
make_hline(20.0, 20.0, 575.0, 1), // top
make_hline(820.0, 20.0, 575.0, 1), // bottom
make_vline(20.0, 20.0, 820.0, 1), // left
make_vline(575.0, 20.0, 820.0, 1), // right
];
let items = vec![
make_item("title", 100.0, 100.0, 1),
make_item("body", 100.0, 200.0, 1),
];
let tables = detect_tables_from_lines(&items, &lines, 1);
assert!(
tables.is_empty(),
"Page-sized 4-edge frame should be rejected as decoration"
);
}
#[test]
fn test_page_spanning_grid_with_internal_lines_accepted() {
// Full-page table (governmental-ledger pattern): A4-sized grid
// that previously hit the "page-spanning frame" early reject
// before downstream validation could even look at it.
// Verticals span the full table height so we isolate the
// frame-vs-grid decision under test.
let mut lines = Vec::new();
// 13 horizontal rules: header + 12 row separators
let h_ys = [
22.5, 37.9, 95.5, 144.5, 184.9, 233.9, 291.7, 340.7, 415.8, 499.6, 574.7, 623.7, 698.8,
];
for &y in &h_ys {
lines.push(make_hline(y, 22.6, 566.6, 1));
}
// 7 column dividers spanning full table height.
let v_xs = [22.6, 66.3, 116.3, 186.6, 263.1, 493.5, 566.5];
for &x in &v_xs {
lines.push(make_vline(x, 22.5, 698.8, 1));
}
// Populate every cell so the capture-ratio + density checks pass.
let mut items = Vec::new();
for r in 0..(h_ys.len() - 1) {
let row_y = (h_ys[r] + h_ys[r + 1]) / 2.0;
for c in 0..(v_xs.len() - 1) {
let col_x = (v_xs[c] + v_xs[c + 1]) / 2.0;
items.push(make_item("x", col_x, row_y, 1));
}
}
let tables = detect_tables_from_lines(&items, &lines, 1);
assert_eq!(
tables.len(),
1,
"Full-page table with internal grid should be accepted"
);
let t = &tables[0];
assert!(
t.cells.len() >= 6,
"expected ≥6 rows, got {}",
t.cells.len()
);
assert!(
t.cells[0].len() >= 3,
"expected ≥3 columns, got {}",
t.cells[0].len()
);
}
#[test]
fn test_single_column_rejected() {
// Only 2 col edges (1 column) — not a table even with verticals
File diff suppressed because it is too large Load Diff
+62 -848
View File
@@ -4,385 +4,15 @@
//! elements linked to MCIDs, this module builds `Table` structs directly from
//! the semantic hierarchy — no geometry heuristics needed.
use std::collections::{HashMap, HashSet};
use std::collections::HashMap;
use log::debug;
use crate::structure_tree::{StructTable, StructTableRow};
use crate::structure_tree::StructTable;
use crate::types::TextItem;
use super::Table;
#[derive(Debug, Clone)]
struct MatchedCell {
text: String,
item_indices: Vec<usize>,
x: Option<f32>,
y: Option<f32>,
}
fn legacy_column_positions(
page_rows: &[&StructTableRow],
mcid_to_items: &HashMap<i64, Vec<usize>>,
items: &[TextItem],
page: u32,
num_cols: usize,
) -> Vec<f32> {
let mut col_positions: Vec<f32> = vec![0.0; num_cols];
for (col, col_pos) in col_positions.iter_mut().enumerate() {
for row in page_rows {
if col < row.cells.len() {
if let Some(x) = row.cells[col]
.mcids
.iter()
.filter(|(_, p)| *p == page)
.filter_map(|(mcid, _)| mcid_to_items.get(mcid))
.flatten()
.map(|&idx| items[idx].x)
.reduce(f32::min)
{
*col_pos = x;
break;
}
}
}
}
col_positions
}
fn infer_column_positions(
raw_rows: &[Vec<MatchedCell>],
fallback_positions: &[f32],
num_cols: usize,
) -> Vec<f32> {
const SAME_COLUMN_TOLERANCE: f32 = 18.0;
let mut anchors = raw_rows
.iter()
.max_by_key(|row| row.iter().filter(|cell| cell.x.is_some()).count())
.map(|row| row.iter().filter_map(|cell| cell.x).collect::<Vec<_>>())
.unwrap_or_default();
if anchors.len() > num_cols {
anchors.truncate(num_cols);
}
let mut additional_positions: Vec<f32> = raw_rows
.iter()
.flat_map(|row| row.iter().filter_map(|cell| cell.x))
.collect();
additional_positions.sort_by(|a, b| a.total_cmp(b));
for x in additional_positions {
if anchors.len() >= num_cols {
break;
}
if anchors
.iter()
.all(|existing| (x - *existing).abs() > SAME_COLUMN_TOLERANCE)
{
anchors.push(x);
anchors.sort_by(|a, b| a.total_cmp(b));
}
}
if anchors.len() < num_cols {
for &x in fallback_positions {
if anchors.len() >= num_cols {
break;
}
if anchors
.iter()
.all(|existing| (x - *existing).abs() > SAME_COLUMN_TOLERANCE)
{
anchors.push(x);
anchors.sort_by(|a, b| a.total_cmp(b));
}
}
}
if anchors.is_empty() {
return fallback_positions.to_vec();
}
while anchors.len() < num_cols {
anchors.push(*anchors.last().unwrap());
}
anchors
}
fn align_positions_to_columns(cell_xs: &[f32], columns: &[f32]) -> Vec<usize> {
if cell_xs.is_empty() || columns.is_empty() {
return Vec::new();
}
if cell_xs.len() >= columns.len() {
return (0..cell_xs.len().min(columns.len())).collect();
}
let mut dp = vec![vec![f32::INFINITY; columns.len() + 1]; cell_xs.len() + 1];
let mut take = vec![vec![false; columns.len() + 1]; cell_xs.len() + 1];
for value in &mut dp[0] {
*value = 0.0;
}
for i in 1..=cell_xs.len() {
for j in 1..=columns.len() {
let skip_cost = dp[i][j - 1];
let take_cost = dp[i - 1][j - 1] + (cell_xs[i - 1] - columns[j - 1]).abs();
if take_cost <= skip_cost {
dp[i][j] = take_cost;
take[i][j] = true;
} else {
dp[i][j] = skip_cost;
}
}
}
let mut assignments_rev = Vec::with_capacity(cell_xs.len());
let mut i = cell_xs.len();
let mut j = columns.len();
while i > 0 && j > 0 {
if take[i][j] {
assignments_rev.push(j - 1);
i -= 1;
j -= 1;
} else {
j -= 1;
}
}
assignments_rev.reverse();
assignments_rev
}
fn align_struct_rows(
raw_rows: &[Vec<MatchedCell>],
col_positions: &[f32],
) -> (Vec<Vec<String>>, Vec<f32>, Vec<usize>) {
let mut cells: Vec<Vec<String>> = Vec::with_capacity(raw_rows.len());
let mut row_positions: Vec<f32> = Vec::with_capacity(raw_rows.len());
let mut all_item_indices: Vec<usize> = Vec::new();
for row in raw_rows {
let present_cells: Vec<&MatchedCell> = row
.iter()
.filter(|cell| {
!cell.item_indices.is_empty() || !cell.text.is_empty() || cell.x.is_some()
})
.collect();
let cell_xs: Vec<f32> = present_cells.iter().filter_map(|cell| cell.x).collect();
let assignments = if cell_xs.len() == present_cells.len() {
align_positions_to_columns(&cell_xs, col_positions)
} else {
(0..present_cells.len().min(col_positions.len())).collect()
};
let mut row_cells = vec![String::new(); col_positions.len()];
for (cell, &col_idx) in present_cells.iter().zip(assignments.iter()) {
if !cell.text.is_empty() {
if !row_cells[col_idx].is_empty() {
row_cells[col_idx].push(' ');
}
row_cells[col_idx].push_str(&cell.text);
}
all_item_indices.extend(cell.item_indices.iter().copied());
}
let row_y = row
.iter()
.filter_map(|cell| cell.y)
.reduce(f32::max)
.unwrap_or(0.0);
cells.push(row_cells);
row_positions.push(row_y);
}
(cells, row_positions, all_item_indices)
}
fn left_align_struct_rows(
raw_rows: &[Vec<MatchedCell>],
num_cols: usize,
) -> (Vec<Vec<String>>, Vec<f32>, Vec<usize>) {
let mut cells: Vec<Vec<String>> = Vec::with_capacity(raw_rows.len());
let mut row_positions: Vec<f32> = Vec::with_capacity(raw_rows.len());
let mut all_item_indices: Vec<usize> = Vec::new();
for row in raw_rows {
let mut row_cells: Vec<String> = row.iter().map(|cell| cell.text.clone()).collect();
row_cells.truncate(num_cols);
while row_cells.len() < num_cols {
row_cells.push(String::new());
}
cells.push(row_cells);
all_item_indices.extend(
row.iter()
.flat_map(|cell| cell.item_indices.iter().copied()),
);
row_positions.push(
row.iter()
.filter_map(|cell| cell.y)
.reduce(f32::max)
.unwrap_or(0.0),
);
}
(cells, row_positions, all_item_indices)
}
fn recover_unclaimed_header_row(table: &mut Table, items: &[TextItem], has_ragged_rows: bool) {
if !has_ragged_rows || table.rows.is_empty() || table.columns.len() < 3 {
return;
}
const MAX_HEADER_DISTANCE: f32 = 90.0;
const MAX_GAP_TO_TABLE: f32 = 35.0;
const MAX_INTER_HEADER_GAP: f32 = 25.0;
const MAX_HEADER_ROWS: usize = 3;
const Y_TOLERANCE: f32 = 5.0;
let top_row_y = table.rows[0];
let x_min = table.columns.first().copied().unwrap_or(0.0) - 25.0;
let x_max = table.columns.last().copied().unwrap_or(0.0) + 120.0;
let claimed: HashSet<usize> = table.item_indices.iter().copied().collect();
let mut candidate_rows: Vec<(f32, Vec<(usize, &TextItem)>)> = Vec::new();
for (idx, item) in items.iter().enumerate() {
if claimed.contains(&idx)
|| item.text.trim().is_empty()
|| item.y <= top_row_y
|| item.y - top_row_y > MAX_HEADER_DISTANCE
|| item.x < x_min
|| item.x > x_max
{
continue;
}
if let Some((_, row_items)) = candidate_rows
.iter_mut()
.find(|(row_y, _)| (item.y - *row_y).abs() < Y_TOLERANCE)
{
row_items.push((idx, item));
} else {
candidate_rows.push((item.y, vec![(idx, item)]));
}
}
if candidate_rows.is_empty() {
return;
}
for (_, row_items) in &mut candidate_rows {
row_items.sort_by(|a, b| a.1.x.total_cmp(&b.1.x));
}
candidate_rows.sort_by(|a, b| a.0.total_cmp(&b.0));
if candidate_rows[0].0 - top_row_y > MAX_GAP_TO_TABLE {
return;
}
let mut candidate_iter = candidate_rows.into_iter();
let Some(first_row) = candidate_iter.next() else {
return;
};
let mut selected_rows: Vec<(f32, Vec<(usize, &TextItem)>)> = vec![first_row];
let mut prev_y = selected_rows[0].0;
for (row_y, row_items) in candidate_iter {
if selected_rows.len() >= MAX_HEADER_ROWS {
break;
}
if row_y - prev_y > MAX_INTER_HEADER_GAP {
break;
}
prev_y = row_y;
selected_rows.push((row_y, row_items));
}
if selected_rows.is_empty() {
return;
}
let mut assigned_rows: Vec<(f32, Vec<String>, Vec<usize>)> = Vec::new();
let mut closest_row_populated = 0usize;
let mut combined_cols: HashSet<usize> = HashSet::new();
for (row_idx, (row_y, row_items)) in selected_rows.iter().enumerate() {
if row_items.len() > table.columns.len() {
return;
}
let row_xs: Vec<f32> = row_items.iter().map(|(_, item)| item.x).collect();
let assignments = align_positions_to_columns(&row_xs, &table.columns);
if assignments.len() != row_items.len() {
return;
}
let mut row_cells = vec![String::new(); table.columns.len()];
let mut row_indices = Vec::with_capacity(row_items.len());
let mut populated_cols: HashSet<usize> = HashSet::new();
for ((idx, item), &col_idx) in row_items.iter().zip(assignments.iter()) {
let text = item.text.trim();
if text.is_empty() {
continue;
}
if !row_cells[col_idx].is_empty() {
row_cells[col_idx].push(' ');
}
row_cells[col_idx].push_str(text);
row_indices.push(*idx);
populated_cols.insert(col_idx);
}
if row_idx == 0 {
closest_row_populated = populated_cols.len();
}
combined_cols.extend(populated_cols.iter().copied());
assigned_rows.push((*row_y, row_cells, row_indices));
}
let required_cols = if table.columns.len() <= 4 {
table.columns.len()
} else {
table.columns.len() - 1
};
if closest_row_populated < 2 || combined_cols.len() < required_cols {
return;
}
let mut header_cells = vec![String::new(); table.columns.len()];
let mut header_indices = Vec::new();
for (_, row_cells, row_indices) in assigned_rows.iter().rev() {
for (col_idx, cell_text) in row_cells.iter().enumerate() {
if cell_text.is_empty() {
continue;
}
if !header_cells[col_idx].is_empty() {
header_cells[col_idx].push(' ');
}
header_cells[col_idx].push_str(cell_text);
}
header_indices.extend(row_indices.iter().copied());
}
table.rows.insert(
0,
assigned_rows
.iter()
.map(|(row_y, _, _)| *row_y)
.reduce(f32::max)
.unwrap_or(top_row_y),
);
table.cells.insert(0, header_cells);
table.item_indices.extend(header_indices);
table.item_indices.sort_unstable();
table.item_indices.dedup();
}
/// Build tables from structure-tree table descriptors by matching MCIDs to TextItems.
///
/// Returns tables for the given page. Tables where fewer than 50% of cells
@@ -437,14 +67,18 @@ pub fn detect_tables_from_struct_tree(
continue;
}
// Build cell text and geometry for alignment and header recovery.
let mut raw_rows: Vec<Vec<MatchedCell>> = Vec::new();
// Build cell text and collect item indices
let mut cells: Vec<Vec<String>> = Vec::new();
let mut all_item_indices: Vec<usize> = Vec::new();
let mut total_cells = 0u32;
let mut matched_cells = 0u32;
for row in &page_rows {
let mut row_cells = Vec::with_capacity(row.cells.len());
for cell in &row.cells {
let mut row_cells = Vec::with_capacity(num_cols);
for (col_idx, cell) in row.cells.iter().enumerate() {
if col_idx >= num_cols {
break;
}
total_cells += 1;
// Collect all items for this cell's MCIDs
@@ -481,18 +115,18 @@ pub fn detect_tables_from_struct_tree(
.collect::<Vec<_>>()
.join(" ");
let item_indices = cell_items.iter().map(|(idx, _)| *idx).collect::<Vec<_>>();
let x = cell_items.iter().map(|(_, item)| item.x).reduce(f32::min);
let y = cell_items.iter().map(|(_, item)| item.y).reduce(f32::max);
for (idx, _) in &cell_items {
all_item_indices.push(*idx);
}
row_cells.push(MatchedCell {
text,
item_indices,
x,
y,
});
row_cells.push(text);
}
raw_rows.push(row_cells);
// Pad to num_cols
while row_cells.len() < num_cols {
row_cells.push(String::new());
}
cells.push(row_cells);
}
// Reject if too few cells matched (stale structure tree)
@@ -514,54 +148,51 @@ pub fn detect_tables_from_struct_tree(
continue;
}
let has_ragged_rows = raw_rows
.iter()
.any(|row| row.iter().filter(|cell| cell.x.is_some()).count() < num_cols);
let first_row_has_tagged_header = page_rows.first().is_some_and(|row| {
let header_cells = row.cells.iter().filter(|cell| cell.is_header).count();
header_cells * 2 >= row.cells.len()
});
let fallback_col_positions =
legacy_column_positions(&page_rows, &mcid_to_items, items, page, num_cols);
let (legacy_cells, legacy_row_positions, mut legacy_item_indices) =
left_align_struct_rows(&raw_rows, num_cols);
legacy_item_indices.sort_unstable();
legacy_item_indices.dedup();
let legacy_table = Table::new(
fallback_col_positions.clone(),
legacy_row_positions,
legacy_cells,
legacy_item_indices,
);
// Derive row/column positions from item geometry
let mut row_positions: Vec<f32> = Vec::new();
for row in &page_rows {
let y = row
.cells
.iter()
.flat_map(|c| c.mcids.iter())
.filter(|(_, p)| *p == page)
.filter_map(|(mcid, _)| mcid_to_items.get(mcid))
.flatten()
.map(|&idx| items[idx].y)
.reduce(f32::max)
.unwrap_or(0.0);
row_positions.push(y);
}
let col_positions = infer_column_positions(&raw_rows, &fallback_col_positions, num_cols);
let (aligned_cells, aligned_row_positions, mut aligned_item_indices) =
align_struct_rows(&raw_rows, &col_positions);
aligned_item_indices.sort_unstable();
aligned_item_indices.dedup();
// Column positions: use X positions of first non-empty cell in each column
let mut col_positions: Vec<f32> = vec![0.0; num_cols];
for (col, col_pos) in col_positions.iter_mut().enumerate() {
for row in &page_rows {
if col < row.cells.len() {
if let Some(x) = row.cells[col]
.mcids
.iter()
.filter(|(_, p)| *p == page)
.filter_map(|(mcid, _)| mcid_to_items.get(mcid))
.flatten()
.map(|&idx| items[idx].x)
.reduce(f32::min)
{
*col_pos = x;
break;
}
}
}
}
let mut aligned_table = Table::new(
col_positions,
aligned_row_positions,
aligned_cells,
aligned_item_indices,
);
let item_count_before_header = aligned_table.item_indices.len();
let row_count_before_header = aligned_table.cells.len();
recover_unclaimed_header_row(
&mut aligned_table,
items,
has_ragged_rows && !first_row_has_tagged_header,
);
all_item_indices.sort_unstable();
all_item_indices.dedup();
let recovered_header = aligned_table.item_indices.len() > item_count_before_header
|| aligned_table.cells.len() > row_count_before_header;
let prefer_aligned = recovered_header;
tables.push(if prefer_aligned {
aligned_table
} else {
legacy_table
tables.push(Table {
columns: col_positions,
rows: row_positions,
cells,
item_indices: all_item_indices,
});
}
@@ -586,8 +217,6 @@ mod tests {
page,
is_bold: false,
is_italic: false,
is_underline: false,
is_strikeout: false,
item_type: ItemType::Text,
mcid,
}
@@ -745,419 +374,4 @@ mod tests {
let tables = detect_tables_from_struct_tree(&items, &struct_tables, 2);
assert_eq!(tables.len(), 1);
}
#[test]
fn realigns_ragged_rows_and_recovers_untagged_header() {
let items = vec![
make_item("Category", 50.0, 120.0, 1, None),
make_item("Potentially", 150.0, 120.0, 1, None),
make_item("Summary", 250.0, 120.0, 1, None),
make_item("Most commonly", 350.0, 120.0, 1, None),
make_item("concerning aspect", 150.0, 110.0, 1, None),
make_item("suggested", 350.0, 110.0, 1, None),
make_item("of circumstances", 150.0, 100.0, 1, None),
make_item("intervention", 350.0, 100.0, 1, None),
make_item("Existence of red-teaming", 150.0, 80.0, 1, Some(10)),
make_item("Important for safety", 250.0, 80.0, 1, Some(11)),
make_item("Ensure welfare interviews", 350.0, 80.0, 1, Some(12)),
make_item("Identity & self-knowledge", 50.0, 60.0, 1, Some(20)),
make_item("Lack of knowledge", 150.0, 60.0, 1, Some(21)),
make_item("Overall negative", 250.0, 60.0, 1, Some(22)),
make_item("Describe training process", 350.0, 60.0, 1, Some(23)),
make_item("Uncertainty around other copies", 150.0, 40.0, 1, Some(30)),
make_item("High uncertainty", 250.0, 40.0, 1, Some(31)),
make_item("No intervention suggested", 350.0, 40.0, 1, Some(32)),
];
let struct_tables = vec![StructTable {
rows: vec![
StructTableRow {
cells: vec![
StructTableCell {
is_header: false,
mcids: vec![(10, 1)],
},
StructTableCell {
is_header: false,
mcids: vec![(11, 1)],
},
StructTableCell {
is_header: false,
mcids: vec![(12, 1)],
},
],
},
StructTableRow {
cells: vec![
StructTableCell {
is_header: false,
mcids: vec![(20, 1)],
},
StructTableCell {
is_header: false,
mcids: vec![(21, 1)],
},
StructTableCell {
is_header: false,
mcids: vec![(22, 1)],
},
StructTableCell {
is_header: false,
mcids: vec![(23, 1)],
},
],
},
StructTableRow {
cells: vec![
StructTableCell {
is_header: false,
mcids: vec![(30, 1)],
},
StructTableCell {
is_header: false,
mcids: vec![(31, 1)],
},
StructTableCell {
is_header: false,
mcids: vec![(32, 1)],
},
],
},
],
}];
let tables = detect_tables_from_struct_tree(&items, &struct_tables, 1);
assert_eq!(tables.len(), 1);
let table = &tables[0];
assert_eq!(table.cells.len(), 4);
assert_eq!(
table.cells[0],
vec![
"Category",
"Potentially concerning aspect of circumstances",
"Summary",
"Most commonly suggested intervention",
]
);
assert_eq!(table.cells[1][0], "");
assert_eq!(table.cells[1][1], "Existence of red-teaming");
assert_eq!(table.cells[2][0], "Identity & self-knowledge");
assert_eq!(table.cells[3][0], "");
assert_eq!(table.columns.len(), 4);
assert!(table.columns.windows(2).all(|w| w[0] < w[1]));
assert_eq!(table.item_indices.len(), items.len());
}
#[test]
fn does_not_absorb_caption_above_ragged_struct_table() {
let items = vec![
make_item("Table 5-7: Summary of responses", 50.0, 120.0, 1, None),
make_item("Aspect one", 150.0, 80.0, 1, Some(10)),
make_item("Summary one", 250.0, 80.0, 1, Some(11)),
make_item("Category", 50.0, 60.0, 1, Some(20)),
make_item("Aspect two", 150.0, 60.0, 1, Some(21)),
make_item("Summary two", 250.0, 60.0, 1, Some(22)),
make_item("Aspect three", 150.0, 40.0, 1, Some(30)),
make_item("Summary three", 250.0, 40.0, 1, Some(31)),
];
let struct_tables = vec![StructTable {
rows: vec![
StructTableRow {
cells: vec![
StructTableCell {
is_header: false,
mcids: vec![(10, 1)],
},
StructTableCell {
is_header: false,
mcids: vec![(11, 1)],
},
],
},
StructTableRow {
cells: vec![
StructTableCell {
is_header: false,
mcids: vec![(20, 1)],
},
StructTableCell {
is_header: false,
mcids: vec![(21, 1)],
},
StructTableCell {
is_header: false,
mcids: vec![(22, 1)],
},
],
},
StructTableRow {
cells: vec![
StructTableCell {
is_header: false,
mcids: vec![(30, 1)],
},
StructTableCell {
is_header: false,
mcids: vec![(31, 1)],
},
],
},
],
}];
let tables = detect_tables_from_struct_tree(&items, &struct_tables, 1);
assert_eq!(tables.len(), 1);
let table = &tables[0];
assert_eq!(table.cells.len(), 3);
assert!(
table
.cells
.iter()
.flatten()
.all(|cell| !cell.contains("Table 5-7")),
"caption must stay outside the table"
);
assert!(!table.item_indices.contains(&0));
}
#[test]
fn keeps_existing_tagged_header_without_absorbing_intro_or_caption() {
let items = vec![
make_item(
"Eighteen people left other comments regarding the Project.",
50.0,
130.0,
1,
None,
),
make_item("Table 5-1:", 220.0, 130.0, 1, None),
make_item("Other Comments", 350.0, 130.0, 1, None),
make_item("Theme", 50.0, 110.0, 1, Some(10)),
make_item("Specific Concern/Inquiry", 200.0, 110.0, 1, Some(11)),
make_item("Response", 420.0, 110.0, 1, Some(12)),
make_item("Traffic", 50.0, 90.0, 1, Some(20)),
make_item("Road conditions", 200.0, 90.0, 1, Some(21)),
make_item("Maintenance response", 420.0, 90.0, 1, Some(22)),
make_item("Noise", 50.0, 70.0, 1, Some(30)),
make_item("Dust concerns", 200.0, 70.0, 1, Some(31)),
make_item("Mitigation response", 420.0, 70.0, 1, Some(32)),
make_item("Resource Use", 50.0, 50.0, 1, Some(40)),
make_item("Snowmobile trails", 200.0, 50.0, 1, Some(41)),
make_item("Access response", 420.0, 50.0, 1, Some(42)),
];
let struct_tables = vec![StructTable {
rows: vec![
StructTableRow {
cells: vec![
StructTableCell {
is_header: true,
mcids: vec![(10, 1)],
},
StructTableCell {
is_header: true,
mcids: vec![(11, 1)],
},
StructTableCell {
is_header: true,
mcids: vec![(12, 1)],
},
],
},
StructTableRow {
cells: vec![
StructTableCell {
is_header: false,
mcids: vec![(20, 1)],
},
StructTableCell {
is_header: false,
mcids: vec![(21, 1)],
},
StructTableCell {
is_header: false,
mcids: vec![(22, 1)],
},
],
},
StructTableRow {
cells: vec![
StructTableCell {
is_header: false,
mcids: vec![(30, 1)],
},
StructTableCell {
is_header: false,
mcids: vec![(31, 1)],
},
StructTableCell {
is_header: false,
mcids: vec![(32, 1)],
},
],
},
StructTableRow {
cells: vec![
StructTableCell {
is_header: false,
mcids: vec![(40, 1)],
},
StructTableCell {
is_header: false,
mcids: vec![(41, 1)],
},
StructTableCell {
is_header: false,
mcids: vec![(42, 1)],
},
],
},
],
}];
let tables = detect_tables_from_struct_tree(&items, &struct_tables, 1);
assert_eq!(tables.len(), 1);
let table = &tables[0];
assert_eq!(table.cells.len(), 4);
assert_eq!(
table.cells[0],
vec!["Theme", "Specific Concern/Inquiry", "Response"]
);
assert!(!table.item_indices.contains(&0));
assert!(!table.item_indices.contains(&1));
assert!(!table.item_indices.contains(&2));
}
#[test]
fn does_not_recover_header_for_narrow_two_column_table() {
let items = vec![
make_item("Alpha", 50.0, 120.0, 1, None),
make_item("Beta", 200.0, 120.0, 1, None),
make_item("First value", 200.0, 80.0, 1, Some(10)),
make_item("Only labeled row", 50.0, 60.0, 1, Some(20)),
make_item("Second value", 200.0, 60.0, 1, Some(21)),
make_item("Third value", 200.0, 40.0, 1, Some(30)),
];
let struct_tables = vec![StructTable {
rows: vec![
StructTableRow {
cells: vec![StructTableCell {
is_header: false,
mcids: vec![(10, 1)],
}],
},
StructTableRow {
cells: vec![
StructTableCell {
is_header: false,
mcids: vec![(20, 1)],
},
StructTableCell {
is_header: false,
mcids: vec![(21, 1)],
},
],
},
StructTableRow {
cells: vec![StructTableCell {
is_header: false,
mcids: vec![(30, 1)],
}],
},
],
}];
let tables = detect_tables_from_struct_tree(&items, &struct_tables, 1);
assert_eq!(tables.len(), 1);
let table = &tables[0];
assert_eq!(table.cells.len(), 3);
assert_eq!(table.cells[0], vec!["First value", ""]);
assert_eq!(table.cells[1], vec!["Only labeled row", "Second value"]);
assert_eq!(table.cells[2], vec!["Third value", ""]);
assert!(!table.item_indices.contains(&0));
assert!(!table.item_indices.contains(&1));
}
#[test]
fn ragged_rows_without_recovered_header_keep_legacy_alignment() {
let items = vec![
make_item("Date", 150.0, 120.0, 1, Some(10)),
make_item("Title", 250.0, 120.0, 1, Some(11)),
make_item("PE", 350.0, 120.0, 1, Some(12)),
make_item("Bidder", 450.0, 120.0, 1, Some(13)),
make_item("Amount", 550.0, 120.0, 1, Some(14)),
make_item("1", 50.0, 100.0, 1, Some(20)),
make_item("8/1", 150.0, 100.0, 1, Some(21)),
make_item("Procurement", 250.0, 100.0, 1, Some(22)),
make_item("PUC", 350.0, 100.0, 1, Some(23)),
make_item("Vendor", 450.0, 100.0, 1, Some(24)),
make_item("SR1", 550.0, 100.0, 1, Some(25)),
];
let struct_tables = vec![StructTable {
rows: vec![
StructTableRow {
cells: vec![
StructTableCell {
is_header: true,
mcids: vec![(10, 1)],
},
StructTableCell {
is_header: true,
mcids: vec![(11, 1)],
},
StructTableCell {
is_header: true,
mcids: vec![(12, 1)],
},
StructTableCell {
is_header: true,
mcids: vec![(13, 1)],
},
StructTableCell {
is_header: true,
mcids: vec![(14, 1)],
},
],
},
StructTableRow {
cells: vec![
StructTableCell {
is_header: false,
mcids: vec![(20, 1)],
},
StructTableCell {
is_header: false,
mcids: vec![(21, 1)],
},
StructTableCell {
is_header: false,
mcids: vec![(22, 1)],
},
StructTableCell {
is_header: false,
mcids: vec![(23, 1)],
},
StructTableCell {
is_header: false,
mcids: vec![(24, 1)],
},
StructTableCell {
is_header: false,
mcids: vec![(25, 1)],
},
],
},
],
}];
let tables = detect_tables_from_struct_tree(&items, &struct_tables, 1);
assert_eq!(tables.len(), 1);
let table = &tables[0];
assert_eq!(table.cells[0][0], "Date");
assert_eq!(table.cells[0][4], "Amount");
assert_eq!(table.cells[0][5], "");
}
}
-2
View File
@@ -108,8 +108,6 @@ pub(crate) fn try_split_financial_item(item: &TextItem) -> Option<Vec<TextItem>>
page: item.page,
is_bold: item.is_bold,
is_italic: item.is_italic,
is_underline: item.is_underline,
is_strikeout: item.is_strikeout,
item_type: item.item_type.clone(),
mcid: item.mcid,
});
+8 -495
View File
@@ -1,22 +1,12 @@
//! Table-to-markdown formatting and cell cleanup.
use super::{Table, TableKind};
use super::Table;
pub fn table_to_markdown(table: &Table) -> String {
if table.cells.is_empty() || table.cells[0].is_empty() {
return String::new();
}
// TOCs render poorly as markdown tables — emit a flat per-row text list
// instead so the page numbers stay aligned with their section titles
// rather than drifting to a separate column. Format from raw cells
// because continuation-row merging in clean_table_cells collapses
// separate TOC entries (e.g. "6.2 Contamination" + "6.2.1 SWE-bench")
// into one line where sub-entries leave column 0 empty.
if table.kind == TableKind::Toc {
return format_toc_as_list(&table.cells, &[]);
}
// Clean up the table: merge continuation rows, extract footnotes, remove empty rows
let (cleaned_cells, footnotes) = clean_table_cells(&table.cells);
@@ -59,182 +49,6 @@ pub fn table_to_markdown(table: &Table) -> String {
output
}
/// Render a table-of-contents as a flat per-row text block.
///
/// Each row becomes one line: non-empty cells joined with spaces, and the
/// last cell (typically a page number) is separated by a tab so the page
/// numbers stay aligned with their titles instead of being pulled into a
/// separate column by the column-aware reader.
fn format_toc_as_list(cells: &[Vec<String>], footnotes: &[String]) -> String {
let mut output = String::new();
for row in cells {
let trimmed: Vec<&str> = row.iter().map(|c| c.trim()).collect();
let last_idx = trimmed.iter().rposition(|c| !c.is_empty());
let Some(last_idx) = last_idx else {
continue;
};
let last_cell = trimmed[last_idx];
let last_is_page = is_page_number_cell(last_cell);
let (title_cells, trailing) = if last_is_page && last_idx > 0 {
(&trimmed[..last_idx], Some(last_cell))
} else {
(&trimmed[..=last_idx], None)
};
// Skip dots-only cells when joining the title — in a detected TOC
// layout, a "...." cell is a leader separator, not part of the
// entry name.
let title = title_cells
.iter()
.filter(|c| !c.is_empty() && !is_dots_only(c))
.copied()
.collect::<Vec<_>>()
.join(" ");
if title.is_empty() && trailing.is_none() {
continue;
}
if !title.is_empty() {
output.push_str(&title);
}
if let Some(page) = trailing {
if !title.is_empty() {
output.push('\t');
}
output.push_str(page);
}
output.push('\n');
}
if !footnotes.is_empty() {
output.push('\n');
for footnote in footnotes {
output.push_str(footnote);
output.push('\n');
}
}
output
}
/// True when the cell looks like a page number. Accepts:
/// - plain digit tokens: "42", "86 86"
/// - dashed section-page IDs: "5-21", "A-1", "B--3", "TC-2" (common in
/// technical manuals)
fn is_page_number_cell(cell: &str) -> bool {
let tokens: Vec<&str> = cell.split_whitespace().collect();
if tokens.is_empty() {
return false;
}
tokens.iter().all(|t| {
if t.is_empty() || t.len() > 8 {
return false;
}
let all_digits = t.chars().all(|c| c.is_ascii_digit());
if all_digits {
return t.len() <= 4;
}
// Section-page form: uppercase letters, digits, dashes; at least
// one digit present.
t.chars()
.all(|c| c.is_ascii_digit() || c.is_ascii_uppercase() || c == '-')
&& t.chars().any(|c| c.is_ascii_digit())
})
}
/// True when the cell is purely leader dots (any length ≥ 3) with optional
/// whitespace.
fn is_dots_only(cell: &str) -> bool {
let t = cell.trim();
let dots = t.chars().filter(|&c| c == '.').count();
dots >= 3 && t.chars().all(|c| c == '.' || c.is_whitespace())
}
fn starts_with_uppercase_word(cell: &str) -> bool {
cell.chars()
.find(|c| c.is_alphanumeric())
.is_some_and(|c| c.is_uppercase())
}
fn starts_with_uppercase_alpha(cell: &str) -> bool {
cell.chars()
.find(|c| c.is_alphabetic())
.is_some_and(|c| c.is_uppercase())
}
fn starts_with_lowercase_alpha(cell: &str) -> bool {
cell.chars()
.find(|c| c.is_alphabetic())
.is_some_and(|c| c.is_lowercase())
}
fn starts_with_numbered_label(cell: &str) -> bool {
let trimmed = cell.trim_start();
let digit_count = trimmed.chars().take_while(|c| c.is_ascii_digit()).count();
digit_count > 0
&& digit_count <= 3
&& trimmed
.chars()
.nth(digit_count)
.is_some_and(|c| matches!(c, '.' | ')' | '-' | ':'))
}
fn alpha_word_count(cell: &str) -> usize {
cell.split_whitespace()
.filter(|word| word.chars().any(|c| c.is_alphabetic()))
.count()
}
fn looks_like_compact_entry_label(cell: &str) -> bool {
let trimmed = cell.trim();
if trimmed.len() < 3 || trimmed.len() > 80 {
return false;
}
if !starts_with_uppercase_alpha(trimmed) && !starts_with_numbered_label(trimmed) {
return false;
}
if trimmed.ends_with(['.', ',', ';', ':']) {
return false;
}
let words = alpha_word_count(trimmed);
(1..=6).contains(&words)
}
fn looks_like_plain_section_label(cell: &str) -> bool {
let trimmed = cell.trim();
if trimmed.len() < 4 || trimmed.len() > 40 {
return false;
}
if trimmed.ends_with(['.', ',', ';', ':']) || trimmed.contains(|ch: char| ch.is_ascii_digit()) {
return false;
}
if trimmed.len() <= 4 && trimmed.chars().all(|ch| !ch.is_lowercase()) {
return false;
}
trimmed
.chars()
.all(|ch| ch.is_alphabetic() || ch.is_whitespace() || matches!(ch, '&' | '/' | '-'))
&& starts_with_uppercase_alpha(trimmed)
&& (1..=4).contains(&alpha_word_count(trimmed))
}
fn ends_like_incomplete_phrase(cell: &str) -> bool {
let lower = cell.trim_end().to_ascii_lowercase();
lower.ends_with(" and")
|| lower.ends_with(" or")
|| lower.ends_with(',')
|| lower.ends_with('-')
|| lower.ends_with('/')
}
/// Clean up table cells: merge continuation rows, extract footnotes, remove empty rows
fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
let mut cleaned: Vec<Vec<String>> = Vec::new();
@@ -260,9 +74,6 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
continue;
}
let num_cols = row.len();
let filled_cells = row.iter().filter(|c| !c.trim().is_empty()).count();
// Check if this is a continuation row (first column is empty but others have content).
// A row with only 1 short non-empty cell (besides the first) is more likely a
// section sub-header (e.g. "JAN", "FEB") than overflow text — don't merge it.
@@ -296,80 +107,26 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
let looks_like_data_row = non_first_cells.len() >= 2
&& avg_cell_len <= 10.0
&& numeric_cells > non_first_cells.len() / 2;
let uppercase_leading_cells = non_first_cells
.iter()
.filter(|cell| starts_with_uppercase_word(cell))
.count();
let first_non_empty_col = row.iter().position(|c| !c.trim().is_empty());
let first_non_empty_cell = first_non_empty_col
.and_then(|idx| row.get(idx))
.map(|c| c.trim())
.unwrap_or("");
let title_like_later_cells = first_non_empty_col
.map(|idx| {
row.iter()
.skip(idx + 1)
.map(|c| c.trim())
.filter(|c| !c.is_empty() && starts_with_uppercase_alpha(c))
.count()
})
.unwrap_or(0);
let prev_first_cell_empty = cleaned
.last()
.and_then(|r| r.first())
.is_some_and(|c| c.trim().is_empty());
let prev_first_cell = cleaned
.last()
.and_then(|r| r.first())
.map(|c| c.trim())
.unwrap_or("");
let header_filled = cleaned
.first()
.map(|r| r.iter().filter(|c| !c.trim().is_empty()).count())
.unwrap_or(num_cols);
let looks_like_spanning_first_column_row = first_cell.is_empty()
&& row.len() >= 4
&& non_first_cells.len() == row.len().saturating_sub(1)
&& uppercase_leading_cells >= non_first_cells.len().saturating_sub(1);
// Hierarchical tables often use a row-spanned first column: sub-rows
// leave column 0 blank, then start a compact title-like label in
// column 1. Wrapped continuations in the existing fixtures start
// mid-sentence/lowercase ("continued text here", "with 3.5%...") or
// carry lowercase fragments in the later cells, so keep those mergeable.
let looks_like_hierarchical_subrow = first_cell.is_empty()
&& row.len() >= 3
&& first_non_empty_col == Some(1)
&& looks_like_compact_entry_label(first_non_empty_cell)
&& ((non_first_cells.len() >= 2 && title_like_later_cells > 0)
|| (non_first_cells.len() == 1
&& prev_first_cell_empty
&& alpha_word_count(first_non_empty_cell) >= 2));
let looks_like_new_first_column_entry = !first_cell.is_empty()
&& (starts_with_numbered_label(first_cell) || starts_with_uppercase_alpha(first_cell))
&& filled_cells >= 2
&& non_first_cells
.iter()
.any(|cell| looks_like_compact_entry_label(cell));
let looks_like_section_label_row = !first_cell.is_empty()
&& filled_cells == 1
&& header_filled >= 3
&& looks_like_plain_section_label(first_cell);
// Classic continuation: first cell empty, content in other cells
let is_classic_continuation = first_cell.is_empty()
&& !non_first_cells.is_empty()
&& !is_short_subheader
&& !looks_like_data_row
&& !looks_like_spanning_first_column_row
&& !looks_like_hierarchical_subrow
&& cleaned.len() > 1;
// Wrapped-cell continuation: row has fewer filled cells than the header
// row, suggesting it's overflow text from the previous row's cells.
// Only trigger when the previous row has significantly more filled cells.
let num_cols = row.len();
let filled_cells = row.iter().filter(|c| !c.trim().is_empty()).count();
let prev_filled = cleaned
.last()
.map(|r| r.iter().filter(|c| !c.trim().is_empty()).count())
.unwrap_or(0);
let header_filled = cleaned
.first()
.map(|r| r.iter().filter(|c| !c.trim().is_empty()).count())
.unwrap_or(num_cols);
// Merge when the row has significantly fewer filled cells than header.
// For wide tables (5+ cols), require ≤50% of header cells.
// For narrow tables (2-4 cols), require fewer than header cells.
@@ -380,18 +137,10 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
} else {
header_filled.saturating_sub(1)
};
let continues_wrapped_first_column_label = !first_cell.is_empty()
&& starts_with_lowercase_alpha(first_cell)
&& ends_like_incomplete_phrase(prev_first_cell);
let is_wrapped_continuation = cleaned.len() > 1
&& filled_cells <= max_filled_for_merge
&& (prev_filled > filled_cells
|| (continues_wrapped_first_column_label && prev_filled >= filled_cells))
&& prev_filled > filled_cells
&& !looks_like_data_row
&& !looks_like_spanning_first_column_row
&& !looks_like_hierarchical_subrow
&& !looks_like_new_first_column_entry
&& !looks_like_section_label_row
&& !is_short_subheader;
let is_continuation = is_classic_continuation || is_wrapped_continuation;
@@ -537,44 +286,6 @@ mod tests {
assert!(cleaned[1][1].contains("continued text here"));
}
#[test]
fn test_clean_table_cells_first_column_section_label_not_merged() {
let cells = vec![
vec![
"Properties".into(),
"Conditions".into(),
"Method".into(),
"Typical values".into(),
"Units".into(),
],
vec![
"Melt Flow Rate".into(),
"230 C/2.16 kg".into(),
"ASTM D1238".into(),
"3.0".into(),
"g/10 min".into(),
],
vec![
"Mechanical".into(),
"".into(),
"".into(),
"".into(),
"".into(),
],
vec![
"Tensile Stress at Yield".into(),
"50 mm/min".into(),
"ASTM D638".into(),
"31".into(),
"MPa".into(),
],
];
let (cleaned, _) = clean_table_cells(&cells);
assert_eq!(cleaned.len(), 4);
assert_eq!(cleaned[2][0], "Mechanical");
}
#[test]
fn test_clean_table_cells_short_subheader_not_merged() {
let cells = vec![
@@ -599,159 +310,6 @@ mod tests {
assert_eq!(cleaned.len(), 3);
}
#[test]
fn test_clean_table_cells_spanning_first_column_row_not_merged() {
let cells = vec![
vec![
"Category".into(),
"Potentially concerning aspect".into(),
"Summary".into(),
"Intervention".into(),
],
vec![
"Identity & self-knowledge".into(),
"Lack of knowledge".into(),
"Overall negative".into(),
"Describe training".into(),
],
vec![
"".into(),
"Uncertainty around other copies".into(),
"High uncertainty".into(),
"No intervention suggested".into(),
],
];
let (cleaned, _) = clean_table_cells(&cells);
assert_eq!(cleaned.len(), 3);
assert_eq!(cleaned[2][0], "");
assert_eq!(cleaned[2][1], "Uncertainty around other copies");
}
#[test]
fn test_clean_table_cells_numbered_hierarchy_rows_not_overmerged() {
let cells = vec![
vec![
"Group".into(),
"Task".into(),
"Detail".into(),
"Benefit".into(),
],
vec![
"1. Group alpha".into(),
"Task setup and".into(),
"Begin setup".into(),
"Faster start".into(),
],
vec![
"".into(),
"management".into(),
"recommended profile".into(),
"with saved defaults".into(),
],
vec![
"2. Group beta and".into(),
"Storage setup".into(),
"Provides upload tools".into(),
"".into(),
],
vec![
"fine-tuning".into(),
"".into(),
"for filtered inputs".into(),
"service".into(),
],
vec![
"".into(),
"Label workspace".into(),
"Creates review sets".into(),
"Lets teams review".into(),
],
vec![
"".into(),
"Model training".into(),
"".into(),
"Supports custom model".into(),
],
];
let (cleaned, _) = clean_table_cells(&cells);
assert_eq!(cleaned.len(), 5);
assert_eq!(cleaned[1][0], "1. Group alpha");
assert_eq!(cleaned[1][1], "Task setup and management");
assert_eq!(cleaned[2][0], "2. Group beta and fine-tuning");
assert_eq!(cleaned[2][1], "Storage setup");
assert_eq!(cleaned[3][1], "Label workspace");
assert_eq!(cleaned[4][1], "Model training");
}
#[test]
fn test_clean_table_cells_partial_hierarchical_subrow_not_merged() {
let cells = vec![
vec![
"Group".into(),
"Task".into(),
"Detail".into(),
"Benefit".into(),
],
vec![
"Group A".into(),
"Alpha task".into(),
"Initial detail".into(),
"Initial benefit".into(),
],
vec![
"".into(),
"Beta task".into(),
"Parallel detail".into(),
"".into(),
],
vec![
"".into(),
"second line".into(),
"additional detail".into(),
"".into(),
],
];
let (cleaned, _) = clean_table_cells(&cells);
assert_eq!(cleaned.len(), 3);
assert_eq!(cleaned[1][1], "Alpha task");
assert_eq!(cleaned[2][0], "");
assert_eq!(cleaned[2][1], "Beta task second line");
assert_eq!(cleaned[2][2], "Parallel detail additional detail");
}
#[test]
fn test_clean_table_cells_full_width_continuation_row_still_merges_when_lowercase() {
let cells = vec![
vec![
"Classification".into(),
"Before tax".into(),
"After tax".into(),
"Standard equipment".into(),
"Options".into(),
],
vec![
"Exclusive Special".into(),
"83,500,000".into(),
"79,275,000".into(),
"Standard equipment".into(),
"Option A".into(),
],
vec![
"".into(),
"with 3.5% individual consumption tax applied".into(),
"with 3.5% individual consumption tax applied".into(),
"lighting(crash pad)".into(),
"sound system".into(),
],
];
let (cleaned, _) = clean_table_cells(&cells);
assert_eq!(cleaned.len(), 2);
assert!(cleaned[1][1].contains("83,500,000"));
assert!(cleaned[1][1].contains("with 3.5%"));
}
#[test]
fn test_clean_table_cells_header_row_not_merged() {
// Continuation requires cleaned.len() > 1 (don't merge into header)
@@ -800,7 +358,6 @@ mod tests {
vec!["Bob".into(), "25".into()],
],
item_indices: vec![],
kind: TableKind::Data,
};
let md = table_to_markdown(&table);
assert!(md.contains("|Name|"));
@@ -816,7 +373,6 @@ mod tests {
rows: vec![500.0],
cells: vec![vec!["Only".into(), "Row".into()]],
item_indices: vec![],
kind: TableKind::Data,
};
let md = table_to_markdown(&table);
assert!(md.contains("|Only|"));
@@ -830,7 +386,6 @@ mod tests {
rows: vec![],
cells: vec![],
item_indices: vec![],
kind: TableKind::Data,
};
assert_eq!(table_to_markdown(&table), "");
}
@@ -846,7 +401,6 @@ mod tests {
vec!["(1)".into(), "Footnote text".into()],
],
item_indices: vec![],
kind: TableKind::Data,
};
let md = table_to_markdown(&table);
assert!(md.contains("(1) Footnote text"));
@@ -862,7 +416,6 @@ mod tests {
vec!["太郎".into(), "25".into()],
],
item_indices: vec![],
kind: TableKind::Data,
};
let md = table_to_markdown(&table);
assert!(md.contains("名前"));
@@ -876,47 +429,7 @@ mod tests {
rows: vec![500.0],
cells: vec![vec![]],
item_indices: vec![],
kind: TableKind::Data,
};
assert_eq!(table_to_markdown(&table), "");
}
#[test]
fn test_table_to_markdown_toc_renders_as_flat_list() {
// A TOC-shaped table with section numbers in col 0 and page numbers
// in the last column should render as a flat list, not a markdown
// table, so the page numbers stay on the same line as their titles.
let table = Table::new(
vec![50.0, 80.0, 300.0],
vec![500.0; 5],
vec![
vec![
"4.3".into(),
"Case studies and targeted evaluations".into(),
"86".into(),
],
vec![
"4.3.1".into(),
"Destructive or reckless actions".into(),
"86".into(),
],
vec![
"4.3.2".into(),
"Adherence to its constitution".into(),
"89".into(),
],
vec!["4.4".into(), "Capability evaluations".into(), "101".into()],
vec!["4.5".into(), "White-box analyses".into(), "113".into()],
],
vec![],
);
assert_eq!(table.kind, TableKind::Toc);
let md = table_to_markdown(&table);
assert!(
!md.contains("|---|"),
"TOC should not render as a markdown table: {md}"
);
assert!(md.contains("4.3 Case studies and targeted evaluations\t86"));
assert!(md.contains("4.5 White-box analyses\t113"));
}
}
+12 -162
View File
@@ -82,42 +82,33 @@ pub(crate) fn find_column_boundaries(
}
}
// Track cluster membership: for each cluster, store the list of x positions
let mut cluster_xs: Vec<Vec<f32>> = vec![vec![x_positions[0]]];
let mut columns = Vec::new();
let mut cluster_items: Vec<f32> = vec![x_positions[0]];
for &x in &x_positions[1..] {
let last_cluster = cluster_xs.last().unwrap();
// For dense columns (gap-histogram triggered), use edge-based clustering:
// compare with the last item to avoid center-drift that merges adjacent
// narrow columns. For normal tables, use center-based (original behavior).
let reference = if use_edge_clustering {
*last_cluster.last().unwrap()
*cluster_items.last().unwrap()
} else {
last_cluster.iter().sum::<f32>() / last_cluster.len() as f32
cluster_items.iter().sum::<f32>() / cluster_items.len() as f32
};
if x - reference > cluster_threshold {
cluster_xs.push(vec![x]);
let cluster_center = cluster_items.iter().sum::<f32>() / cluster_items.len() as f32;
columns.push(cluster_center);
cluster_items = vec![x];
} else {
cluster_xs.last_mut().unwrap().push(x);
cluster_items.push(x);
}
}
// Numeric column merge pass: when a sparse cluster (few items, typically
// header text) is adjacent to a dense numeric cluster and within 1.5×
// threshold, merge them. This fixes tables where multi-line wrapped
// headers have slightly different X positions than the data columns,
// causing the header and data to split into separate clusters.
let columns_before_merge = cluster_xs.len();
if columns_before_merge >= 3 {
cluster_xs = merge_numeric_adjacent_clusters(cluster_xs, items, cluster_threshold);
// Don't forget last cluster
if !cluster_items.is_empty() {
columns.push(cluster_items.iter().sum::<f32>() / cluster_items.len() as f32);
}
let columns: Vec<f32> = cluster_xs
.iter()
.map(|xs| xs.iter().sum::<f32>() / xs.len() as f32)
.collect();
// Filter columns - each should have multiple items
let min_items_per_col = (items.len() / columns.len().max(1) / 4).max(2);
let columns: Vec<f32> = columns
@@ -132,9 +123,8 @@ pub(crate) fn find_column_boundaries(
.collect();
log::debug!(
" find_column_boundaries: {} columns (merged from {}), threshold={:.1}, {} items",
" find_column_boundaries: {} columns before filter, threshold={:.1}, {} items",
columns.len(),
columns_before_merge,
cluster_threshold,
items.len()
);
@@ -158,116 +148,6 @@ pub(crate) fn find_column_boundaries(
columns
}
/// Check if a text string looks like a number (digits, decimals, sign, comma).
fn is_numeric_text(s: &str) -> bool {
let s = s.trim();
if s.is_empty() {
return false;
}
// Match patterns like: 8.23, -1.05, 9.99, 7.12, 100, 3,456.78, +5%, ---
// But NOT: BIO, Department, Core Courses
s.chars()
.all(|c| c.is_ascii_digit() || c == '.' || c == ',' || c == '-' || c == '+' || c == '%')
&& s.chars().any(|c| c.is_ascii_digit())
}
/// Merge adjacent X-position clusters when one is a sparse header cluster
/// and the other is a dense numeric data cluster. This prevents multi-line
/// wrapped headers from splitting a logical column into two clusters.
fn merge_numeric_adjacent_clusters(
mut clusters: Vec<Vec<f32>>,
items: &[(usize, &TextItem)],
threshold: f32,
) -> Vec<Vec<f32>> {
// For each cluster, compute: center, item count, numeric fraction
struct ClusterInfo {
center: f32,
count: usize,
numeric_frac: f32,
}
let compute_info = |xs: &[f32]| -> ClusterInfo {
let center = xs.iter().sum::<f32>() / xs.len() as f32;
// Count items and numeric fraction for items near this cluster center
let mut total = 0;
let mut numeric = 0;
for (_, item) in items {
if (item.x - center).abs() < threshold {
total += 1;
if is_numeric_text(&item.text) {
numeric += 1;
}
}
}
ClusterInfo {
center,
count: total,
numeric_frac: if total > 0 {
numeric as f32 / total as f32
} else {
0.0
},
}
};
// Merge distance: allow merging clusters that are slightly beyond the
// original threshold. Use 1.5× threshold to catch header-vs-data splits.
let merge_dist = threshold * 1.5;
// Iterate and merge adjacent pairs. Use a simple left-to-right scan.
let mut merged = true;
while merged {
merged = false;
let mut i = 0;
while i + 1 < clusters.len() {
let info_a = compute_info(&clusters[i]);
let info_b = compute_info(&clusters[i + 1]);
let dist = (info_b.center - info_a.center).abs();
if dist > merge_dist {
i += 1;
continue;
}
// Determine if one cluster is sparse (header) and the other
// is dense and numeric (data). A cluster is "sparse" if it has
// significantly fewer items than the other.
let (sparse, dense) = if info_a.count < info_b.count {
(&info_a, &info_b)
} else {
(&info_b, &info_a)
};
// Merge if the dense cluster is predominantly numeric (>50%)
// and the sparse cluster has at most 1/3 the items of the dense one.
let should_merge =
dense.numeric_frac > 0.50 && sparse.count <= dense.count / 2 && sparse.count <= 5;
if should_merge {
log::debug!(
" merging column clusters: center {:.1} ({} items, {:.0}% numeric) + {:.1} ({} items, {:.0}% numeric), dist={:.1}",
info_a.center,
info_a.count,
info_a.numeric_frac * 100.0,
info_b.center,
info_b.count,
info_b.numeric_frac * 100.0,
dist,
);
// Merge cluster i+1 into cluster i
let next = clusters.remove(i + 1);
clusters[i].extend(next);
merged = true;
// Don't increment i — check if the merged cluster can merge further
} else {
i += 1;
}
}
}
clusters
}
/// Find row boundaries by clustering Y positions
pub(crate) fn find_row_boundaries(items: &[(usize, &TextItem)]) -> Vec<f32> {
let mut y_positions: Vec<f32> = items.iter().map(|(_, i)| i.y).collect();
@@ -369,10 +249,6 @@ pub(crate) fn join_cell_items(items: &[&TextItem]) -> String {
let prev_ends_with_hyphen = result.ends_with('-');
let curr_is_hyphen = text == "-";
let curr_starts_with_hyphen = text.starts_with('-');
let prev_ends_with_open_delimiter =
result.ends_with('(') || result.ends_with('[') || result.ends_with('{');
let curr_starts_with_close_delimiter =
text.starts_with(')') || text.starts_with(']') || text.starts_with('}');
// Detect subscript/superscript: smaller font size and/or Y offset
let font_ratio = item.font_size / prev_item.font_size;
@@ -389,8 +265,6 @@ pub(crate) fn join_cell_items(items: &[&TextItem]) -> String {
|| curr_starts_with_hyphen
|| is_sub_super
|| was_sub_super
|| prev_ends_with_open_delimiter
|| curr_starts_with_close_delimiter
{
result.push_str(text);
} else {
@@ -505,7 +379,6 @@ pub(crate) fn recover_header_row(
#[cfg(test)]
mod tests {
use super::*;
use crate::tables::TableKind;
use crate::types::ItemType;
fn make_item(text: &str, x: f32, y: f32, font_size: f32) -> TextItem {
@@ -520,8 +393,6 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
is_strikeout: false,
item_type: ItemType::Text,
mcid: None,
}
@@ -735,18 +606,6 @@ mod tests {
assert_eq!(join_cell_items(&[&a, &b, &c]), "pre-fix");
}
#[test]
fn test_join_cell_items_parenthetical_no_inner_spaces() {
let a = make_item("The first sentence", 100.0, 500.0, 10.0);
let b = make_item("(", 190.0, 500.0, 10.0);
let c = make_item("twice", 195.0, 500.0, 10.0);
let d = make_item(")", 220.0, 500.0, 10.0);
assert_eq!(
join_cell_items(&[&a, &b, &c, &d]),
"The first sentence (twice)"
);
}
#[test]
fn test_join_cell_items_subscript_no_space() {
let a = make_item("H", 100.0, 500.0, 12.0);
@@ -777,7 +636,6 @@ mod tests {
rows: vec![500.0, 480.0],
cells: vec![vec!["A".into(), "B".into()], vec!["C".into(), "D".into()]],
item_indices: vec![2, 3],
kind: TableKind::Data,
};
recover_header_row(&mut table, &all_items, 9.0);
@@ -796,7 +654,6 @@ mod tests {
rows: vec![500.0],
cells: vec![vec!["A".into(), "B".into()]],
item_indices: vec![0, 1],
kind: TableKind::Data,
};
let rows_before = table.rows.len();
@@ -817,7 +674,6 @@ mod tests {
rows: vec![500.0, 480.0],
cells: vec![vec!["A".into(), "B".into()], vec!["C".into(), "D".into()]],
item_indices: vec![2, 3],
kind: TableKind::Data,
};
let rows_before = table.rows.len();
@@ -838,7 +694,6 @@ mod tests {
rows: vec![500.0],
cells: vec![vec!["A".into(), "B".into()]],
item_indices: vec![1, 2],
kind: TableKind::Data,
};
let rows_before = table.rows.len();
@@ -854,7 +709,6 @@ mod tests {
rows: vec![],
cells: vec![],
item_indices: vec![],
kind: TableKind::Data,
};
recover_header_row(&mut table, &all_items, 9.0);
@@ -886,8 +740,6 @@ mod tests {
font: String::new(),
is_bold: false,
is_italic: false,
is_underline: false,
is_strikeout: false,
item_type: ItemType::Text,
mcid: None,
page: 1,
@@ -924,8 +776,6 @@ mod tests {
font: String::new(),
is_bold: false,
is_italic: false,
is_underline: false,
is_strikeout: false,
item_type: ItemType::Text,
mcid: None,
page: 1,
+9 -1290
View File
File diff suppressed because it is too large Load Diff
-972
View File
@@ -1,972 +0,0 @@
//! Structure-recovery-aware (TSR) table assembly.
//!
//! Consumes the raw output of an external table-structure recognition model
//! (e.g. SLANet on PaddleOCR): a flat list of HTML structure tokens plus a
//! parallel list of per-cell bboxes. Pairs each cell open-tag with its bbox
//! in document order, tracks row/column position with rowspan/colspan
//! awareness, and emits a markdown pipe-table.
//!
//! No real HTML parser is needed — the token grammar is restricted (see
//! [`parse_structure`]), so a small state machine is enough.
//!
//! Cell text is supplied separately by the caller (typically by overlap-
//! testing PDF text items against each cell's page-PDF-pt bbox).
use std::collections::{HashMap, HashSet};
/// A single resolved cell, with both structural metadata and its bbox in
/// page PDF-points (top-left origin).
#[derive(Debug, Clone)]
pub struct StructuredCell {
/// 0-indexed grid row.
pub row: usize,
/// 0-indexed grid column.
pub col: usize,
/// 1 for a normal cell.
pub rowspan: usize,
/// 1 for a normal cell.
pub colspan: usize,
/// `true` when the cell is a `<th>` or sits inside `<thead>`.
pub is_header: bool,
/// Cell text (filled in by the caller after overlap-testing PDF items).
pub text: String,
/// Axis-aligned bbox `[x1, y1, x2, y2]` in page PDF-points, top-left origin.
pub page_pt_bbox: [f32; 4],
}
/// Intermediate parse result before the caller fills in text + page coords.
#[derive(Debug, Clone)]
pub(crate) struct CellSlot {
pub row: usize,
pub col: usize,
pub rowspan: usize,
pub colspan: usize,
pub is_header: bool,
/// Index into the parallel `cell_bboxes` array.
pub bbox_idx: usize,
}
/// Parse a sequence of SLANet structure tokens into ordered cell slots.
///
/// Token grammar (no real HTML parsing required):
/// - Section markers: `<thead>`, `</thead>`, `<tbody>`, `</tbody>` and
/// wrapper tokens (`<html>`, `<body>`, `<table>`, plus closing variants)
/// are tracked or skipped.
/// - Row markers: `<tr>` opens a new row, `</tr>` is informational.
/// - Empty cell, single token: `<td></td>` or `<th></th>`.
/// - Cell with attributes, multi-token sequence: `<td` (or `<th`), then
/// attribute fragments like ` colspan="4"`, then `>`, then later `</td>`
/// (or `</th>`). Cells get paired with the next bbox in document order.
///
/// Cells inside `<thead>` and any `<th>` cells are flagged as headers.
/// rowspan/colspan attributes are honoured and prior-row rowspans push
/// later-row cells to the right.
pub(crate) fn parse_structure(tokens: &[String]) -> Vec<CellSlot> {
let mut slots: Vec<CellSlot> = Vec::new();
let mut occupied: HashSet<(usize, usize)> = HashSet::new();
let mut row: usize = 0;
let mut col: usize = 0;
let mut bbox_idx: usize = 0;
let mut in_thead = false;
let mut started_first_row = false;
let mut i = 0;
while i < tokens.len() {
let tok = tokens[i].trim();
match tok {
"<thead>" => {
in_thead = true;
}
"</thead>" => {
in_thead = false;
}
"<tr>" => {
if started_first_row {
row += 1;
}
col = 0;
started_first_row = true;
}
"<td></td>" | "<th></th>" => {
let is_th = tok == "<th></th>";
while occupied.contains(&(row, col)) {
col += 1;
}
slots.push(CellSlot {
row,
col,
rowspan: 1,
colspan: 1,
is_header: in_thead || is_th,
bbox_idx,
});
bbox_idx += 1;
col += 1;
}
"<td" | "<th" => {
let is_th = tok == "<th";
let mut rowspan: usize = 1;
let mut colspan: usize = 1;
// Consume attribute fragments until we hit ">".
i += 1;
while i < tokens.len() && tokens[i].trim() != ">" {
let attr = tokens[i].as_str();
if let Some(v) = parse_int_attr(attr, "rowspan") {
rowspan = v.max(1);
} else if let Some(v) = parse_int_attr(attr, "colspan") {
colspan = v.max(1);
}
i += 1;
}
// i now points at the `>` token (or off the end if malformed).
while occupied.contains(&(row, col)) {
col += 1;
}
slots.push(CellSlot {
row,
col,
rowspan,
colspan,
is_header: in_thead || is_th,
bbox_idx,
});
for r in row..row + rowspan {
for c in col..col + colspan {
occupied.insert((r, c));
}
}
bbox_idx += 1;
col += colspan;
}
// Wrapper / informational tokens — no-op.
_ => {}
}
i += 1;
}
slots
}
/// Parse an attribute fragment like ` colspan="4"` or `rowspan='2'`.
///
/// Tolerates leading whitespace and either single or double quotes.
fn parse_int_attr(s: &str, name: &str) -> Option<usize> {
let trimmed = s.trim();
if !trimmed.starts_with(name) {
return None;
}
let rest = trimmed[name.len()..].trim_start();
let rest = rest.strip_prefix('=')?.trim_start();
let value = rest
.trim_start_matches(['"', '\''])
.trim_end_matches(['"', '\'']);
value.parse().ok()
}
/// Convert a SLANet polygon (4 or 8 elements) into an axis-aligned
/// `[x1, y1, x2, y2]` rect.
///
/// 8-element form: `[x1,y1, x2,y1, x2,y2, x1,y2]` (4 corners). We ignore the
/// implicit corner order and just take min/max so rotated polygons collapse
/// to a sane bounding box.
///
/// 4-element form: `[x1, y1, x2, y2]` (axis-aligned, older SLANet variants).
pub(crate) fn polygon_to_aabb(coords: &[f32]) -> Option<[f32; 4]> {
match coords.len() {
4 => {
let x1 = coords[0].min(coords[2]);
let y1 = coords[1].min(coords[3]);
let x2 = coords[0].max(coords[2]);
let y2 = coords[1].max(coords[3]);
Some([x1, y1, x2, y2])
}
8 => {
let xs = [coords[0], coords[2], coords[4], coords[6]];
let ys = [coords[1], coords[3], coords[5], coords[7]];
let x1 = xs.iter().copied().fold(f32::INFINITY, f32::min);
let y1 = ys.iter().copied().fold(f32::INFINITY, f32::min);
let x2 = xs.iter().copied().fold(f32::NEG_INFINITY, f32::max);
let y2 = ys.iter().copied().fold(f32::NEG_INFINITY, f32::max);
if x1.is_finite() && y1.is_finite() && x2.is_finite() && y2.is_finite() {
Some([x1, y1, x2, y2])
} else {
None
}
}
_ => None,
}
}
/// Convert a cell rect from crop image-pixel space to page PDF-points
/// (top-left origin), given the crop's PDF-point offset on the page and the
/// DPI the crop image was rendered at.
pub(crate) fn cell_px_to_page_pt(
cell_px: [f32; 4],
render_dpi: f32,
crop_origin_pt: [f32; 2],
) -> [f32; 4] {
let pt_per_px = if render_dpi > 0.0 {
72.0 / render_dpi
} else {
1.0
};
let [x_off, y_off] = crop_origin_pt;
[
cell_px[0] * pt_per_px + x_off,
cell_px[1] * pt_per_px + y_off,
cell_px[2] * pt_per_px + x_off,
cell_px[3] * pt_per_px + y_off,
]
}
/// Refine TSR cell bboxes into non-overlapping row/column bands.
///
/// SLANet-style bboxes are often plausible but too tall on dense borderless
/// tables. Native PDF text assignment is more reliable when each parsed row
/// owns the band between neighboring row centers instead of the full model box.
pub(crate) fn normalize_cell_bands(cells: &mut [StructuredCell]) {
if cells.len() < 2 {
return;
}
let row_bands = derive_axis_bands(cells, Axis::Y);
let col_bands = derive_axis_bands(cells, Axis::X);
for cell in cells {
let row_end = cell.row + cell.rowspan.max(1).saturating_sub(1);
if let (Some(&(y1, _)), Some(&(_, y2))) =
(row_bands.get(&cell.row), row_bands.get(&row_end))
{
let clamped_y1 = cell.page_pt_bbox[1].max(y1);
let clamped_y2 = cell.page_pt_bbox[3].min(y2);
if clamped_y1 < clamped_y2 {
cell.page_pt_bbox[1] = clamped_y1;
cell.page_pt_bbox[3] = clamped_y2;
}
}
let col_end = cell.col + cell.colspan.max(1).saturating_sub(1);
if let (Some(&(x1, _)), Some(&(_, x2))) =
(col_bands.get(&cell.col), col_bands.get(&col_end))
{
let clamped_x1 = cell.page_pt_bbox[0].max(x1);
let clamped_x2 = cell.page_pt_bbox[2].min(x2);
if clamped_x1 < clamped_x2 {
cell.page_pt_bbox[0] = clamped_x1;
cell.page_pt_bbox[2] = clamped_x2;
}
}
}
}
#[derive(Clone, Copy)]
enum Axis {
X,
Y,
}
fn derive_axis_bands(cells: &[StructuredCell], axis: Axis) -> HashMap<usize, (f32, f32)> {
let mut by_index: HashMap<usize, Vec<(f32, f32)>> = HashMap::new();
// Prefer non-spanning cells so colspan/rowspan boxes do not skew a single
// column/row center. If an axis has no non-spanning examples for an index,
// fall back to anchored cells below.
for cell in cells {
let span = match axis {
Axis::X => cell.colspan.max(1),
Axis::Y => cell.rowspan.max(1),
};
if span == 1 {
let idx = match axis {
Axis::X => cell.col,
Axis::Y => cell.row,
};
by_index
.entry(idx)
.or_default()
.push(axis_bounds(cell.page_pt_bbox, axis));
}
}
for cell in cells {
let idx = match axis {
Axis::X => cell.col,
Axis::Y => cell.row,
};
if !by_index.contains_key(&idx) {
by_index
.entry(idx)
.or_default()
.push(axis_bounds(cell.page_pt_bbox, axis));
}
}
let mut rows: Vec<(usize, f32, f32, f32)> = by_index
.into_iter()
.filter_map(|(idx, bounds)| {
let mut min_edge = f32::INFINITY;
let mut max_edge = f32::NEG_INFINITY;
let mut center_sum = 0.0;
let mut count = 0usize;
for (lo, hi) in bounds {
if lo.is_finite() && hi.is_finite() && lo < hi {
min_edge = min_edge.min(lo);
max_edge = max_edge.max(hi);
center_sum += (lo + hi) * 0.5;
count += 1;
}
}
(count > 0).then_some((idx, center_sum / count as f32, min_edge, max_edge))
})
.collect();
if rows.len() < 2 {
return rows
.into_iter()
.map(|(idx, _center, lo, hi)| (idx, (lo, hi)))
.collect();
}
rows.sort_by_key(|(idx, _, _, _)| *idx);
let mut bands = HashMap::new();
for i in 0..rows.len() {
let (idx, _center, min_edge, max_edge) = rows[i];
let lo = if i == 0 {
min_edge
} else {
(rows[i - 1].1 + rows[i].1) * 0.5
};
let hi = if i + 1 == rows.len() {
max_edge
} else {
(rows[i].1 + rows[i + 1].1) * 0.5
};
if lo.is_finite() && hi.is_finite() && lo < hi {
bands.insert(idx, (lo, hi));
}
}
bands
}
fn axis_bounds(bbox: [f32; 4], axis: Axis) -> (f32, f32) {
match axis {
Axis::X => (bbox[0].min(bbox[2]), bbox[0].max(bbox[2])),
Axis::Y => (bbox[1].min(bbox[3]), bbox[1].max(bbox[3])),
}
}
/// Sanitize cell text for inclusion in a markdown pipe-table cell:
/// collapse whitespace runs, drop newlines/tabs (cells must be one line),
/// and escape pipes that would otherwise break the table.
fn sanitize_cell(text: &str) -> String {
let mut s = String::with_capacity(text.len());
let mut prev_space = false;
for c in text.chars() {
match c {
'|' => {
s.push_str("\\|");
prev_space = false;
}
'\n' | '\r' | '\t' | ' ' => {
if !prev_space {
s.push(' ');
}
prev_space = true;
}
other => {
s.push(other);
prev_space = false;
}
}
}
s.trim().to_string()
}
/// Render a list of explicitly-positioned cells as a markdown pipe-table.
///
/// Grid dimensions are inferred from the cells' (row, col, rowspan, colspan)
/// extents. A cell with colspan/rowspan > 1 is rendered in its top-left
/// position; the absorbed grid positions are emitted as empty cells so the
/// markdown stays a valid rectangular grid that downstream readers can
/// column-count correctly.
///
/// The separator row (`|---|...|`) is emitted after the **last** row that
/// contains a header cell (`is_header == true`). When no cells are flagged
/// as headers — e.g. the upstream TSR model didn't emit `<thead>`/`<th>` —
/// the separator falls back to "after row 0" so the output is still a
/// valid pipe-table.
pub fn cells_to_markdown(cells: &[StructuredCell]) -> String {
if cells.is_empty() {
return String::new();
}
let num_rows = cells
.iter()
.map(|c| c.row + c.rowspan.max(1))
.max()
.unwrap_or(0);
let num_cols = cells
.iter()
.map(|c| c.col + c.colspan.max(1))
.max()
.unwrap_or(0);
if num_rows == 0 || num_cols == 0 {
return String::new();
}
// Separator goes after the last header row, falling back to row 0 when
// no header cells exist. Clamped into range so a malformed cell with
// row >= num_rows can't push it past the table.
let separator_after_row = cells
.iter()
.filter(|c| c.is_header)
.map(|c| c.row)
.max()
.unwrap_or(0)
.min(num_rows.saturating_sub(1));
let mut grid: Vec<Vec<String>> = vec![vec![String::new(); num_cols]; num_rows];
for cell in cells {
if cell.row < num_rows && cell.col < num_cols {
grid[cell.row][cell.col] = sanitize_cell(&cell.text);
}
}
let mut output = String::new();
for (row_idx, row) in grid.iter().enumerate() {
output.push('|');
for cell in row {
output.push_str(cell);
output.push('|');
}
output.push('\n');
if row_idx == separator_after_row {
output.push('|');
for _ in 0..num_cols {
output.push_str("---|");
}
output.push('\n');
}
}
output
}
#[cfg(test)]
mod tests {
use super::*;
fn t(s: &str) -> String {
s.to_string()
}
/// Tokens for the synthetic 3×3 grid example (one colspan-4 row + two
/// data rows of 4 cells each = 9 cells total, 3 rows × 4 cols).
fn synthetic_3x3_tokens() -> Vec<String> {
vec![
"<html>",
"<body>",
"<table>",
"<tbody>",
"<tr>",
"<td",
" colspan=\"4\"",
">",
"</td>",
"</tr>",
"<tr>",
"<td></td>",
"<td></td>",
"<td></td>",
"<td></td>",
"</tr>",
"<tr>",
"<td></td>",
"<td></td>",
"<td></td>",
"<td></td>",
"</tr>",
"</tbody>",
"</table>",
"</body>",
"</html>",
]
.into_iter()
.map(t)
.collect()
}
/// Bboxes for the synthetic 3×3 grid (8-element polygon form), all
/// within a 400×120 px crop.
fn synthetic_3x3_bboxes() -> Vec<Vec<f32>> {
vec![
vec![3.0, 2.0, 395.0, 2.0, 396.0, 59.0, 3.0, 59.0],
vec![26.0, 62.0, 140.0, 62.0, 141.0, 120.0, 26.0, 120.0],
vec![149.0, 64.0, 248.0, 64.0, 248.0, 119.0, 149.0, 119.0],
vec![257.0, 64.0, 350.0, 64.0, 350.0, 119.0, 257.0, 119.0],
vec![359.0, 64.0, 395.0, 64.0, 395.0, 119.0, 359.0, 119.0],
vec![26.0, 122.0, 140.0, 122.0, 140.0, 178.0, 26.0, 178.0],
vec![149.0, 124.0, 248.0, 124.0, 248.0, 179.0, 149.0, 179.0],
vec![257.0, 124.0, 350.0, 124.0, 350.0, 179.0, 257.0, 179.0],
vec![359.0, 124.0, 395.0, 124.0, 395.0, 179.0, 359.0, 179.0],
]
}
#[test]
fn parse_structure_synthetic_3x3() {
let tokens = synthetic_3x3_tokens();
let slots = parse_structure(&tokens);
assert_eq!(slots.len(), 9, "should parse 9 cells");
// Cell 0: row 0 col 0, colspan 4
assert_eq!(slots[0].row, 0);
assert_eq!(slots[0].col, 0);
assert_eq!(slots[0].colspan, 4);
assert_eq!(slots[0].rowspan, 1);
// Cells 1..5: row 1, cols 0..3
for (i, slot) in slots.iter().enumerate().skip(1).take(4) {
assert_eq!(slot.row, 1, "cell {i}: row should be 1");
assert_eq!(slot.col, i - 1, "cell {i}: col should be {}", i - 1);
assert_eq!(slot.colspan, 1);
assert_eq!(slot.rowspan, 1);
}
// Cells 5..9: row 2, cols 0..3
for (i, slot) in slots.iter().enumerate().skip(5).take(4) {
assert_eq!(slot.row, 2, "cell {i}: row should be 2");
assert_eq!(slot.col, i - 5);
assert_eq!(slot.colspan, 1);
}
}
#[test]
fn polygon_to_aabb_8elt() {
// Synthetic cell bbox 0
let coords = vec![3.0, 2.0, 395.0, 2.0, 396.0, 59.0, 3.0, 59.0];
let aabb = polygon_to_aabb(&coords).unwrap();
assert_eq!(aabb, [3.0, 2.0, 396.0, 59.0]);
}
#[test]
fn polygon_to_aabb_4elt() {
let coords = vec![5.0, 10.0, 50.0, 60.0];
let aabb = polygon_to_aabb(&coords).unwrap();
assert_eq!(aabb, [5.0, 10.0, 50.0, 60.0]);
}
#[test]
fn polygon_to_aabb_4elt_unordered() {
// Caller may pass corners in any order; min/max should normalise.
let coords = vec![50.0, 60.0, 5.0, 10.0];
let aabb = polygon_to_aabb(&coords).unwrap();
assert_eq!(aabb, [5.0, 10.0, 50.0, 60.0]);
}
#[test]
fn polygon_to_aabb_invalid_len() {
assert!(polygon_to_aabb(&[1.0, 2.0, 3.0]).is_none());
assert!(polygon_to_aabb(&[1.0; 6]).is_none());
assert!(polygon_to_aabb(&[]).is_none());
}
#[test]
fn synthetic_3x3_aabbs_inside_crop() {
// All 9 bboxes should produce valid (x1<x2, y1<y2) rects within the
// crop bounds (400 wide, ~180 tall by inspection of the fixture).
let bboxes = synthetic_3x3_bboxes();
assert_eq!(bboxes.len(), 9);
for (i, bb) in bboxes.iter().enumerate() {
let aabb = polygon_to_aabb(bb).unwrap_or_else(|| panic!("bbox {i} invalid"));
assert!(aabb[0] < aabb[2], "bbox {i}: x1 < x2");
assert!(aabb[1] < aabb[3], "bbox {i}: y1 < y2");
assert!(aabb[0] >= 0.0 && aabb[2] <= 500.0, "bbox {i}: within crop");
assert!(aabb[1] >= 0.0 && aabb[3] <= 200.0, "bbox {i}: within crop");
}
}
#[test]
fn normalize_cell_bands_splits_overlapping_slanet_rows() {
let mut cells = vec![
StructuredCell {
row: 0,
col: 0,
rowspan: 1,
colspan: 1,
is_header: true,
text: String::new(),
page_pt_bbox: [10.0, 100.0, 90.0, 120.0],
},
StructuredCell {
row: 0,
col: 1,
rowspan: 1,
colspan: 1,
is_header: true,
text: String::new(),
page_pt_bbox: [90.0, 100.0, 170.0, 120.0],
},
StructuredCell {
row: 1,
col: 0,
rowspan: 1,
colspan: 1,
is_header: false,
text: String::new(),
page_pt_bbox: [10.0, 116.0, 90.0, 136.0],
},
StructuredCell {
row: 1,
col: 1,
rowspan: 1,
colspan: 1,
is_header: false,
text: String::new(),
page_pt_bbox: [90.0, 116.0, 170.0, 136.0],
},
];
normalize_cell_bands(&mut cells);
assert_eq!(cells[0].page_pt_bbox[3], cells[2].page_pt_bbox[1]);
assert_eq!(cells[1].page_pt_bbox[3], cells[3].page_pt_bbox[1]);
assert!(
(cells[0].page_pt_bbox[3] - 118.0).abs() < 0.01,
"row separator should be midpoint between row centers: {:?}",
cells
);
}
#[test]
fn normalize_cell_bands_preserves_colspan_extent() {
let mut cells = vec![
StructuredCell {
row: 0,
col: 0,
rowspan: 1,
colspan: 2,
is_header: true,
text: String::new(),
page_pt_bbox: [8.0, 80.0, 172.0, 98.0],
},
StructuredCell {
row: 1,
col: 0,
rowspan: 1,
colspan: 1,
is_header: false,
text: String::new(),
page_pt_bbox: [10.0, 96.0, 90.0, 114.0],
},
StructuredCell {
row: 1,
col: 1,
rowspan: 1,
colspan: 1,
is_header: false,
text: String::new(),
page_pt_bbox: [88.0, 96.0, 170.0, 114.0],
},
];
normalize_cell_bands(&mut cells);
assert!(
cells[0].page_pt_bbox[0] <= cells[1].page_pt_bbox[0],
"spanning cell should retain the first column's left edge"
);
assert!(
cells[0].page_pt_bbox[2] >= cells[2].page_pt_bbox[2],
"spanning cell should retain the last column's right edge"
);
}
#[test]
fn parse_int_attr_basic() {
assert_eq!(parse_int_attr(" colspan=\"4\"", "colspan"), Some(4));
assert_eq!(parse_int_attr(" rowspan=\"2\"", "rowspan"), Some(2));
assert_eq!(parse_int_attr("colspan='3'", "colspan"), Some(3));
assert_eq!(parse_int_attr(" colspan=\"4\"", "rowspan"), None);
assert_eq!(parse_int_attr(" class=\"foo\"", "colspan"), None);
}
#[test]
fn parse_structure_rowspan_pushes_next_row_right() {
// <tr><td rowspan="2">A</td><td>B</td></tr><tr><td>C</td></tr>
// Expected: A at (0,0), B at (0,1), C at (1,1) — col 0 of row 1
// is occupied by A's rowspan.
let tokens: Vec<String> = vec![
"<table>",
"<tbody>",
"<tr>",
"<td",
" rowspan=\"2\"",
">",
"</td>",
"<td></td>",
"</tr>",
"<tr>",
"<td></td>",
"</tr>",
"</tbody>",
"</table>",
]
.into_iter()
.map(t)
.collect();
let slots = parse_structure(&tokens);
assert_eq!(slots.len(), 3);
assert_eq!((slots[0].row, slots[0].col), (0, 0));
assert_eq!(slots[0].rowspan, 2);
assert_eq!((slots[1].row, slots[1].col), (0, 1));
// C should be at (1, 1) because (1, 0) is occupied by A's rowspan.
assert_eq!((slots[2].row, slots[2].col), (1, 1));
}
#[test]
fn parse_structure_thead_marks_headers() {
// <thead><tr><th>H1</th><th>H2</th></tr></thead>
// <tbody><tr><td>D1</td><td>D2</td></tr></tbody>
let tokens: Vec<String> = vec![
"<table>",
"<thead>",
"<tr>",
"<th></th>",
"<th></th>",
"</tr>",
"</thead>",
"<tbody>",
"<tr>",
"<td></td>",
"<td></td>",
"</tr>",
"</tbody>",
"</table>",
]
.into_iter()
.map(t)
.collect();
let slots = parse_structure(&tokens);
assert_eq!(slots.len(), 4);
assert!(slots[0].is_header && slots[1].is_header);
assert!(!slots[2].is_header && !slots[3].is_header);
}
#[test]
fn parse_structure_th_outside_thead_still_header() {
// A row-header style: leading <th> in tbody.
let tokens: Vec<String> = vec![
"<table>",
"<tbody>",
"<tr>",
"<th></th>",
"<td></td>",
"</tr>",
"</tbody>",
"</table>",
]
.into_iter()
.map(t)
.collect();
let slots = parse_structure(&tokens);
assert_eq!(slots.len(), 2);
assert!(slots[0].is_header);
assert!(!slots[1].is_header);
}
#[test]
fn parse_structure_th_with_attrs() {
let tokens: Vec<String> = vec![
"<table>",
"<thead>",
"<tr>",
"<th",
" colspan=\"2\"",
">",
"</th>",
"</tr>",
"</thead>",
"</table>",
]
.into_iter()
.map(t)
.collect();
let slots = parse_structure(&tokens);
assert_eq!(slots.len(), 1);
assert_eq!(slots[0].colspan, 2);
assert!(slots[0].is_header);
}
#[test]
fn cells_to_markdown_synthetic_3x3() {
// Build the cells the parser would produce for the synthetic grid,
// and provide some sample text so we can sanity-check output.
let cells = vec![
StructuredCell {
row: 0,
col: 0,
rowspan: 1,
colspan: 4,
is_header: false,
text: "Title".into(),
page_pt_bbox: [0.0, 0.0, 0.0, 0.0],
},
StructuredCell {
row: 1,
col: 0,
rowspan: 1,
colspan: 1,
is_header: false,
text: "a".into(),
page_pt_bbox: [0.0, 0.0, 0.0, 0.0],
},
StructuredCell {
row: 1,
col: 1,
rowspan: 1,
colspan: 1,
is_header: false,
text: "b".into(),
page_pt_bbox: [0.0, 0.0, 0.0, 0.0],
},
StructuredCell {
row: 1,
col: 2,
rowspan: 1,
colspan: 1,
is_header: false,
text: "c".into(),
page_pt_bbox: [0.0, 0.0, 0.0, 0.0],
},
StructuredCell {
row: 1,
col: 3,
rowspan: 1,
colspan: 1,
is_header: false,
text: "d".into(),
page_pt_bbox: [0.0, 0.0, 0.0, 0.0],
},
];
let md = cells_to_markdown(&cells);
// Header row contains the spanning cell text in col 0 and pads to 4 cols.
// Absorbed-by-colspan positions render as empty cells (no padding).
assert!(md.starts_with("|Title||||\n"), "got: {md}");
assert!(md.contains("|---|---|---|---|\n"));
assert!(md.contains("|a|b|c|d|\n"));
}
#[test]
fn cells_to_markdown_escapes_pipes() {
let cells = vec![
StructuredCell {
row: 0,
col: 0,
rowspan: 1,
colspan: 1,
is_header: false,
text: "a|b".into(),
page_pt_bbox: [0.0, 0.0, 0.0, 0.0],
},
StructuredCell {
row: 0,
col: 1,
rowspan: 1,
colspan: 1,
is_header: false,
text: "x".into(),
page_pt_bbox: [0.0, 0.0, 0.0, 0.0],
},
];
let md = cells_to_markdown(&cells);
assert!(md.contains("|a\\|b|x|"));
}
#[test]
fn cells_to_markdown_collapses_whitespace_and_newlines() {
let cells = vec![StructuredCell {
row: 0,
col: 0,
rowspan: 1,
colspan: 1,
is_header: false,
text: "foo \n bar\tbaz".into(),
page_pt_bbox: [0.0, 0.0, 0.0, 0.0],
}];
let md = cells_to_markdown(&cells);
assert!(md.contains("|foo bar baz|"));
}
fn cell(row: usize, col: usize, is_header: bool, text: &str) -> StructuredCell {
StructuredCell {
row,
col,
rowspan: 1,
colspan: 1,
is_header,
text: text.into(),
page_pt_bbox: [0.0, 0.0, 0.0, 0.0],
}
}
#[test]
fn cells_to_markdown_separator_after_last_header_row() {
// Two-row header (a multi-row thead), then two body rows. Separator
// should land after row 1 (the LAST header row), not after row 0.
let cells = vec![
cell(0, 0, true, "H0a"),
cell(0, 1, true, "H0b"),
cell(1, 0, true, "H1a"),
cell(1, 1, true, "H1b"),
cell(2, 0, false, "d0a"),
cell(2, 1, false, "d0b"),
cell(3, 0, false, "d1a"),
cell(3, 1, false, "d1b"),
];
let md = cells_to_markdown(&cells);
let expected = "|H0a|H0b|\n|H1a|H1b|\n|---|---|\n|d0a|d0b|\n|d1a|d1b|\n";
assert_eq!(md, expected, "got: {md}");
}
#[test]
fn cells_to_markdown_separator_when_row_0_not_header() {
// Row 0 is not flagged as a header but row 1 is. Separator should
// follow row 1 (the header), demonstrating that we don't blindly
// emit after row 0.
let cells = vec![
cell(0, 0, false, "x0a"),
cell(0, 1, false, "x0b"),
cell(1, 0, true, "Hdr1"),
cell(1, 1, true, "Hdr2"),
cell(2, 0, false, "data1"),
cell(2, 1, false, "data2"),
];
let md = cells_to_markdown(&cells);
// Confirm the separator is NOT after row 0.
assert!(!md.starts_with("|x0a|x0b|\n|---|"), "got: {md}");
// Confirm it IS after row 1.
assert!(
md.contains("|Hdr1|Hdr2|\n|---|---|\n|data1|data2|"),
"got: {md}"
);
}
#[test]
fn cells_to_markdown_no_headers_falls_back_to_row_0() {
// No header cells at all — fallback: separator after row 0 so the
// output is still a valid markdown pipe-table.
let cells = vec![
cell(0, 0, false, "a"),
cell(0, 1, false, "b"),
cell(1, 0, false, "c"),
cell(1, 1, false, "d"),
];
let md = cells_to_markdown(&cells);
assert_eq!(md, "|a|b|\n|---|---|\n|c|d|\n");
}
}
-520
View File
@@ -1,520 +0,0 @@
//! Text-quality detection: deciding when an extracted text layer is too broken
//! to serve and a page should fall back to OCR.
//!
//! Extraction can produce plausible-looking bytes that are actually garbage —
//! failed CID→Unicode mappings, broken ToUnicode CMaps, mojibake. These
//! detectors catch that and let callers set `needs_ocr`. They come in two
//! layers, sharing the same primitives:
//!
//! - **Markdown-level** ([`detect_encoding_issues`], [`is_garbage_text`],
//! [`is_cid_garbage`]) run on a page's final markdown string. Used as a
//! backstop on the region-extraction and whole-document paths.
//! - **Item/span-level** ([`analyze_text_quality`],
//! [`region_items_have_decoding_issue`]) run on individual `TextItem`s and
//! accumulate per-page evidence, so localized garbled spans on an otherwise
//! clean page are caught without a single span having to condemn the page.
//!
//! Detection classes, roughly by signal:
//! - **Replacement runs**: U+FFFD clusters ([`has_replacement_text_run`]).
//! - **Private-use / C1-control runs**: CID passthrough landing in PUA or the
//! C1 block ([`has_private_use_text_run`], [`has_cid_control_token`]).
//! - **Dollar-as-space**: `Word$Word$Word` from broken CMaps
//! ([`has_dollar_as_space_pattern`]).
//! - **Non-alphanumeric dominance**: symbol soup ([`is_garbage_text`]).
//! - **Substitution-cipher letter statistics**: pure-ASCII output whose letter
//! distribution is a permutation of natural language ([`CipherGarbleStats`]).
use crate::types::TextItem;
use crate::{add_ocr_reason, OCR_REASON_SUSPECTED_GARBLED_TEXT};
use std::collections::BTreeMap;
/// Detect broken font encodings in extracted markdown text.
///
/// Two heuristics:
/// 1. **U+FFFD**: Any replacement character indicates decode failures.
/// 2. **Dollar-as-space**: Pattern like `Word$Word$Word` where `$` is used as a
/// word separator due to broken ToUnicode CMaps. Triggers when either:
/// - More than 50% of `$` are between letters (clear substitution pattern), OR
/// - More than 20 letter-dollar-letter occurrences (even if some `$` are also
/// used as trailing/leading separators, 20+ is far beyond normal financial text).
pub(crate) fn detect_encoding_issues(markdown: &str) -> bool {
// Heuristic 1: U+FFFD replacement characters
if markdown.contains('\u{FFFD}') {
return true;
}
// Heuristic 2: dollar-as-space pattern
if has_dollar_as_space_pattern(markdown) {
return true;
}
// Heuristic 3: substitution-cipher letter statistics (broken ToUnicode)
let mut stats = CipherGarbleStats::default();
stats.add_text(markdown);
stats.looks_garbled()
}
fn has_dollar_as_space_pattern(markdown: &str) -> bool {
let total_dollars = markdown.matches('$').count();
if total_dollars > 10 {
let bytes = markdown.as_bytes();
let mut letter_dollar_letter = 0usize;
for i in 1..bytes.len().saturating_sub(1) {
if bytes[i] == b'$'
&& bytes[i - 1].is_ascii_alphabetic()
&& bytes[i + 1].is_ascii_alphabetic()
{
letter_dollar_letter += 1;
}
}
if letter_dollar_letter > 20 || letter_dollar_letter * 2 > total_dollars {
return true;
}
}
false
}
/// English letter frequencies (percent, az). Used as a natural-language
/// reference: every Latin-script language in the eval corpus (Swedish,
/// Finnish, Turkish, German, romaji) scores ≥ 0.80 cosine similarity against
/// it, while substitution-cipher text scores ~0.53.
const ENGLISH_LETTER_FREQ: [f64; 26] = [
8.2, 1.5, 2.8, 4.3, 12.7, 2.2, 2.0, 6.1, 7.0, 0.15, 0.8, 4.0, 2.4, 6.7, 7.5, 1.9, 0.1, 6.0,
6.3, 9.1, 2.8, 1.0, 2.4, 0.15, 2.0, 0.07,
];
/// Letter statistics for detecting substitution-cipher garbling: broken
/// ToUnicode CMaps that shift every character by a per-range constant (e.g.
/// `Certificate` extracted as `8VceZWZTReV`). Such text is 100% printable
/// ASCII with word-like token lengths, so it defeats `is_garbage_text` and
/// produces no replacement characters — it needs its own discriminator.
#[derive(Debug, Default)]
struct CipherGarbleStats {
/// Case-folded ASCII letter histogram.
letter_counts: [u32; 26],
ascii_letters: usize,
ascii_vowels: usize,
/// Accented Latin letters (Latin-1 Supplement through Latin Extended-B,
/// plus Latin Extended Additional). Count toward Latin dominance only.
latin_ext_letters: usize,
non_latin_letters: usize,
/// Adjacent ASCII-letter pairs, and how many of them switch from
/// lowercase straight to uppercase mid-word.
letter_bigrams: usize,
case_shift_bigrams: usize,
}
impl CipherGarbleStats {
fn add_text(&mut self, text: &str) {
let mut prev: Option<char> = None;
for ch in text.chars() {
if ch.is_ascii_alphabetic() {
let idx = (ch.to_ascii_lowercase() as u8 - b'a') as usize;
self.letter_counts[idx] += 1;
self.ascii_letters += 1;
if matches!(ch.to_ascii_lowercase(), 'a' | 'e' | 'i' | 'o' | 'u') {
self.ascii_vowels += 1;
}
if let Some(p) = prev {
self.letter_bigrams += 1;
if p.is_ascii_lowercase() && ch.is_ascii_uppercase() {
self.case_shift_bigrams += 1;
}
}
prev = Some(ch);
} else {
if ch.is_alphabetic() {
if matches!(ch as u32, 0xC0..=0x24F | 0x1E00..=0x1EFF) {
self.latin_ext_letters += 1;
} else {
self.non_latin_letters += 1;
}
}
prev = None;
}
}
}
/// Cosine similarity between the observed letter histogram and English
/// letter frequencies. A shifted alphabet permutes the histogram, which
/// destroys the similarity regardless of the shift amount.
fn english_cosine(&self) -> f64 {
if self.ascii_letters == 0 {
return 1.0;
}
let n = self.ascii_letters as f64;
let mut dot = 0.0;
let mut norm_obs = 0.0;
for (count, freq) in self.letter_counts.iter().zip(ENGLISH_LETTER_FREQ) {
let p = *count as f64 / n;
dot += p * freq;
norm_obs += p * p;
}
let norm_en = ENGLISH_LETTER_FREQ
.iter()
.map(|f| f * f)
.sum::<f64>()
.sqrt();
dot / (norm_obs.sqrt() * norm_en)
}
/// Cosine similarity between the observed histogram and English
/// frequencies after sorting BOTH descending — i.e. comparing the *shape*
/// of the frequency profile, ignoring which letter sits where. A
/// substitution cipher is a bijection, so it preserves this shape exactly
/// (att10k 0.97, arbitrary shifts 0.99) regardless of case or offset.
/// Non-linguistic ASCII has a different profile: a small alphabet is far
/// steeper (random DNA 0.74, hex dumps 0.81), so the shape diverges.
fn english_shape_cosine(&self) -> f64 {
if self.ascii_letters == 0 {
return 1.0;
}
let n = self.ascii_letters as f64;
let mut obs: [f64; 26] = std::array::from_fn(|i| self.letter_counts[i] as f64 / n);
obs.sort_unstable_by(|a, b| b.total_cmp(a));
let mut en = ENGLISH_LETTER_FREQ;
en.sort_unstable_by(|a, b| b.total_cmp(a));
let dot: f64 = obs.iter().zip(en).map(|(o, e)| o * e).sum();
let norm_obs = obs.iter().map(|o| o * o).sum::<f64>().sqrt();
let norm_en = en.iter().map(|e| e * e).sum::<f64>().sqrt();
dot / (norm_obs * norm_en)
}
/// Thresholds validated against the 380-document pdf-evals snapshot
/// corpus (0 false positives) and the garbled ParseBench `att10k` page
/// (vowel ratio 0.245, case-shift rate 0.225, cosine 0.532). Closest
/// legitimate document on each axis: vowel ratio 0.264 (circuit
/// schematic), case-shift rate 0.021, cosine 0.801.
fn looks_garbled(&self) -> bool {
// Need a statistically meaningful, Latin-dominant sample.
if self.ascii_letters < 200
|| self.non_latin_letters > self.ascii_letters + self.latin_ext_letters
{
return false;
}
// Real Latin-script text keeps vowels above ~30% of letters even in
// acronym- and part-number-heavy documents; shifted text starves them.
let vowel_ratio = self.ascii_vowels as f64 / self.ascii_letters as f64;
if vowel_ratio > 0.30 {
return false;
}
// Signal 1: lowercase→uppercase transitions inside words. A shifted
// lowercase alphabet straddles the ASCII uppercase block ('i'→'Z',
// 't'→'e'), so garbled words flip case constantly. Real documents
// stay ≤ 0.02 even with camelCase identifiers.
let case_shifts = self.letter_bigrams >= 100
&& self.case_shift_bigrams as f64 >= self.letter_bigrams as f64 * 0.10;
// Signal 2: the histogram is a permutation of natural language — an
// English-like frequency SHAPE (sorted cosine high) but with letters
// in the wrong POSITIONS (unsorted cosine low). This is the signature
// of a substitution cipher and is case-independent, so it catches
// all-lowercase and all-uppercase shifts as well as case-straddling
// ones. Genuinely non-linguistic ASCII that is merely "unlike English"
// fails one of the two halves: DNA/hex dumps have too steep a profile
// (shape cosine < 0.90), while protein sequences, ticker symbols and
// base64 are not sufficiently unlike English in position (unsorted
// cosine ≥ 0.60) — so none of them are routed to OCR.
let permuted_language = self.english_cosine() < 0.60 && self.english_shape_cosine() >= 0.90;
case_shifts || permuted_language
}
}
#[derive(Debug, Default)]
pub(crate) struct TextQualityReport {
pub(crate) pages_needing_ocr: Vec<u32>,
pub(crate) has_encoding_issues: bool,
pub(crate) reasons_by_page: BTreeMap<u32, Vec<String>>,
}
#[derive(Debug, Default)]
struct PageTextQualityEvidence {
chars: usize,
replacement_chars: usize,
replacement_spans: usize,
longest_replacement_run: usize,
cipher_garble: CipherGarbleStats,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
enum TextSpanIssueKind {
Replacement,
Strong,
}
pub(crate) fn analyze_text_quality(items: &[TextItem]) -> TextQualityReport {
let mut reasons_by_page = BTreeMap::new();
let mut evidence_by_page = BTreeMap::<u32, PageTextQualityEvidence>::new();
for item in items {
if !matches!(item.item_type, crate::types::ItemType::Text) {
continue;
}
let evidence = evidence_by_page.entry(item.page).or_default();
evidence.chars += item.text.chars().filter(|ch| !ch.is_whitespace()).count();
evidence.cipher_garble.add_text(&item.text);
match text_span_decoding_issue_kind(&item.text) {
Some(TextSpanIssueKind::Strong) => {
add_ocr_reason(
&mut reasons_by_page,
item.page,
OCR_REASON_SUSPECTED_GARBLED_TEXT,
);
}
Some(TextSpanIssueKind::Replacement) => {
let stats = replacement_text_stats(&item.text);
evidence.replacement_chars += stats.0;
evidence.replacement_spans += 1;
evidence.longest_replacement_run = evidence.longest_replacement_run.max(stats.1);
}
None => {}
}
}
for (page, evidence) in evidence_by_page {
if reasons_by_page.contains_key(&page) {
continue;
}
if page_replacement_evidence_needs_ocr(&evidence) || evidence.cipher_garble.looks_garbled()
{
add_ocr_reason(
&mut reasons_by_page,
page,
OCR_REASON_SUSPECTED_GARBLED_TEXT,
);
}
}
let pages_needing_ocr: Vec<u32> = reasons_by_page.keys().copied().collect();
TextQualityReport {
has_encoding_issues: !pages_needing_ocr.is_empty(),
pages_needing_ocr,
reasons_by_page,
}
}
pub(crate) fn region_items_have_decoding_issue(items: &[TextItem]) -> bool {
items.iter().any(|item| {
matches!(item.item_type, crate::types::ItemType::Text)
&& text_span_has_decoding_issue(&item.text)
})
}
fn text_span_has_decoding_issue(text: &str) -> bool {
text_span_decoding_issue_kind(text).is_some()
}
fn text_span_decoding_issue_kind(text: &str) -> Option<TextSpanIssueKind> {
let text = text.trim();
if text.is_empty() {
return None;
}
if has_dollar_as_space_pattern(text)
|| has_private_use_text_run(text)
|| is_cid_garbage(text)
|| has_cid_control_token(text)
{
return Some(TextSpanIssueKind::Strong);
}
if has_replacement_text_run(text) {
return Some(TextSpanIssueKind::Replacement);
}
None
}
fn replacement_text_stats(text: &str) -> (usize, usize) {
let mut replacement = 0usize;
let mut current_run = 0usize;
let mut longest_run = 0usize;
for ch in text.chars() {
if ch == '\u{FFFD}' {
replacement += 1;
current_run += 1;
longest_run = longest_run.max(current_run);
} else {
current_run = 0;
}
}
(replacement, longest_run)
}
fn page_replacement_evidence_needs_ocr(evidence: &PageTextQualityEvidence) -> bool {
if evidence.replacement_chars == 0 || evidence.chars == 0 {
return false;
}
// If the entire page is only a short broken text layer, even a short
// replacement run is enough evidence. On otherwise text-heavy pages,
// require density so math formulas do not force full-page OCR.
if evidence.chars <= 80 && evidence.longest_replacement_run >= 2 {
return true;
}
let replacement_density_bps = evidence.replacement_chars * 10_000 / evidence.chars;
let enough_bad_text = evidence.replacement_chars >= 12 && replacement_density_bps >= 500;
let repeated_bad_spans = evidence.replacement_spans >= 3 && replacement_density_bps >= 250;
let long_bad_run = evidence.longest_replacement_run >= 8 && replacement_density_bps >= 250;
enough_bad_text || repeated_bad_spans || long_bad_run
}
fn has_replacement_text_run(text: &str) -> bool {
let (replacement, longest_run) = replacement_text_stats(text);
longest_run >= 2 || replacement >= 3
}
fn has_private_use_text_run(text: &str) -> bool {
let mut total = 0usize;
let mut private_use = 0usize;
let mut current_run = 0usize;
let mut longest_run = 0usize;
for ch in text.chars() {
if ch.is_whitespace() {
current_run = 0;
continue;
}
total += 1;
if is_private_use_char(ch) {
private_use += 1;
current_run += 1;
longest_run = longest_run.max(current_run);
} else {
current_run = 0;
}
}
if private_use == 0 {
return false;
}
longest_run >= 3 || (total >= 5 && private_use >= 2 && private_use * 2 >= total)
}
fn has_cid_control_token(text: &str) -> bool {
text.split_whitespace().any(token_has_cid_control)
}
fn token_has_cid_control(token: &str) -> bool {
let mut total = 0usize;
let mut c1_control = 0usize;
for ch in token.chars() {
total += 1;
if ('\u{0080}'..='\u{009F}').contains(&ch) {
c1_control += 1;
}
}
total >= 5 && c1_control >= 2 && c1_control * 20 >= total
}
fn is_private_use_char(ch: char) -> bool {
matches!(
ch as u32,
0xE000..=0xF8FF | 0xF0000..=0xFFFFD | 0x100000..=0x10FFFD
)
}
/// Check if extracted text is predominantly garbage (non-alphanumeric).
///
/// Broken font encodings produce text like "----1-.-.-.___ --.-. .._ I_---."
/// where most characters are punctuation/symbols. Real text in any language
/// has >50% alphanumeric characters.
pub(crate) fn is_garbage_text(markdown: &str) -> bool {
let mut alphanum = 0usize;
let mut non_alphanum = 0usize;
let chars: Vec<char> = markdown.chars().collect();
let mut i = 0usize;
while i < chars.len() {
let ch = chars[i];
let mut run_end = i + 1;
while run_end < chars.len() && chars[run_end] == ch {
run_end += 1;
}
let is_decorative_leader = matches!(ch, '.' | '_' | '·') && run_end - i >= 3;
if !is_decorative_leader {
for &run_ch in &chars[i..run_end] {
if run_ch.is_whitespace() {
continue;
}
// Skip markdown syntax chars that we add (not from the PDF)
if matches!(run_ch, '#' | '*' | '|' | '-' | '\n') {
continue;
}
if run_ch.is_alphanumeric() {
alphanum += 1;
} else {
non_alphanum += 1;
}
}
}
i = run_end;
}
let total = alphanum + non_alphanum;
total >= 50 && alphanum * 2 < total
}
/// Detect garbage from failed CID-to-Unicode mapping on Identity-H fonts.
///
/// When CID values don't correspond to Unicode codepoints, the raw bytes often
/// produce characters in the C1 control range (U+0080U+009F) or Private Use
/// Area, mixed with random Latin Extended characters. Valid text in any
/// language almost never contains C1 controls. We also fall back to the
/// general `is_garbage_text` check for non-alphanumeric-heavy patterns.
pub(crate) fn is_cid_garbage(text: &str) -> bool {
if is_garbage_text(text) {
return true;
}
let mut total = 0usize;
let mut c1_control = 0usize;
let mut high_latin = 0usize;
for ch in text.chars() {
if ch.is_whitespace() {
continue;
}
total += 1;
// C1 control characters (U+0080U+009F) — almost never in real text
if ch == '·' {
continue;
}
if ('\u{0080}'..='\u{009F}').contains(&ch) {
c1_control += 1;
}
// High Latin-1 (U+00A0U+00FF) — legitimate in Western European text
// but when combined with ASCII in CID passthrough, indicates mojibake
// from CID values being misinterpreted as Latin-1 characters.
if ('\u{00A0}'..='\u{00FF}').contains(&ch) {
high_latin += 1;
}
}
if total < 5 {
return false;
}
// If ≥5% of non-whitespace chars are C1 controls, it's garbage
if c1_control >= 2 && c1_control * 20 >= total {
return true;
}
// If ≥40% of non-whitespace chars are high Latin-1 AND the text has few
// ASCII letters, it's likely CID-as-Latin-1 mojibake (Japanese/CJK PDFs
// where CID values 0x80-0xFF become accented Latin characters). Keep a
// minimum length so short math tokens like "2×()×" do not route a clean
// page to OCR.
let ascii_letters = text.chars().filter(|c| c.is_ascii_alphabetic()).count();
total >= 20 && high_latin * 5 >= total * 2 && ascii_letters * 3 < total
}
-20
View File
@@ -93,9 +93,6 @@ pub fn is_bold_font(font_name: &str) -> bool {
|| lower.contains("extrabold")
|| lower.contains("ultrabold")
|| lower.contains("medium") && !lower.contains("mediumitalic") // Some fonts use Medium for semi-bold
// URW Type 1 fonts abbreviate Medium as "Medi" (e.g. NimbusRomNo9L-Medi,
// the Times-Bold substitute in LaTeX documents; -MediItal is bold italic).
|| lower.contains("-medi") && !lower.contains("mediumital")
}
/// Detect if a font name indicates italic/oblique style
@@ -765,17 +762,6 @@ mod tests {
use super::*;
use crate::types::ItemType;
#[test]
fn bold_font_urw_medi_abbreviation() {
// URW Type 1 fonts (LaTeX default Times) abbreviate Medium as "Medi"
assert!(is_bold_font("NROFIU+NimbusRomNo9L-Medi"));
assert!(is_bold_font("NimbusRomNo9L-MediItal"));
assert!(!is_bold_font("DSSZWN+NimbusRomNo9L-Regu"));
assert!(!is_bold_font("NimbusRomNo9L-ReguItal"));
// Medium-Italic exclusion still holds
assert!(!is_bold_font("Foo-MediumItalic"));
}
#[test]
fn strip_soft_hyphen() {
assert_eq!(expand_ligatures("con\u{00AD}tent"), "content");
@@ -897,8 +883,6 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
is_strikeout: false,
item_type: ItemType::Text,
mcid: None,
}
@@ -1018,8 +1002,6 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
is_strikeout: false,
item_type: ItemType::Text,
mcid: None,
});
@@ -1096,8 +1078,6 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
is_strikeout: false,
item_type: ItemType::Text,
mcid: None,
}
+20 -449
View File
@@ -329,7 +329,7 @@ impl ToUnicodeCMap {
if let (Some(start), Some(end), Some(base)) = (
parse_hex_u16(&start_hex),
parse_hex_u16(&end_hex),
hex_to_unicode_scalar(&base_hex),
parse_hex_u32(&base_hex),
) {
self.ranges.push((start, end, base));
}
@@ -520,18 +520,6 @@ impl ToUnicodeCMap {
}
}
/// Get the maximum source CID across all mappings (char_map + ranges).
fn max_source_cid(&self) -> Option<u16> {
let char_max = self.char_map.keys().copied().max();
let range_max = self.ranges.iter().map(|&(_, end, _)| end).max();
match (char_max, range_max) {
(Some(a), Some(b)) => Some(a.max(b)),
(a @ Some(_), None) => a,
(None, b @ Some(_)) => b,
(None, None) => None,
}
}
/// Remap a CMap that references pre-subsetting GIDs to sequential post-subsetting GIDs.
/// Collects all source CIDs, sorts them, and reassigns to 1, 2, 3, ...
pub fn remap_to_sequential(&self) -> ToUnicodeCMap {
@@ -575,86 +563,32 @@ fn parse_hex_u16(hex: &str) -> Option<u16> {
u16::from_str_radix(hex.trim(), 16).ok()
}
/// Convert a ToUnicode destination hex string to Unicode.
///
/// PDF ToUnicode destinations are UTF-16BE strings. Supplementary-plane
/// characters are encoded as surrogate pairs, so treating each 4-hex chunk as
/// a scalar drops emoji like D83CDF1F.
/// Parse a hex string to u32
fn parse_hex_u32(hex: &str) -> Option<u32> {
u32::from_str_radix(hex.trim(), 16).ok()
}
/// Convert a hex string to a Unicode string
/// Handles both 2-byte (BMP) and 4-byte (supplementary) codepoints
fn hex_to_unicode_string(hex: &str) -> Option<String> {
let hex: String = hex.chars().filter(|ch| !ch.is_ascii_whitespace()).collect();
if hex.is_empty() || !hex.len().is_multiple_of(2) {
return None;
}
let hex = hex.trim();
let mut result = String::new();
let bytes: Option<Vec<u8>> = (0..hex.len())
.step_by(2)
.map(|i| u8::from_str_radix(&hex[i..i + 2], 16).ok())
.collect();
let bytes = bytes?;
if bytes.len().is_multiple_of(2) {
let units: Vec<u16> = bytes
.chunks_exact(2)
.map(|chunk| u16::from_be_bytes([chunk[0], chunk[1]]))
.collect();
if let Ok(result) = String::from_utf16(&units) {
if !result.is_empty() {
return Some(normalize_tounicode_destination(result));
// Process 4 hex digits at a time
let mut i = 0;
while i + 4 <= hex.len() {
if let Ok(cp) = u32::from_str_radix(&hex[i..i + 4], 16) {
if let Some(c) = char::from_u32(cp) {
result.push(c);
}
}
i += 4;
}
// Be permissive for non-standard one-byte destinations.
if bytes.len() == 1 {
let ch = bytes[0] as char;
if !ch.is_control() || ch == '\t' || ch == '\n' {
return Some(ch.to_string());
}
}
None
}
fn normalize_tounicode_destination(text: String) -> String {
let is_multi_char = text.chars().nth(1).is_some();
// Some malformed producer CMaps put a list of alternative whitespace or
// hyphen codepoints into one destination. Keep ordinary multi-character
// mappings intact unless that malformed signature is present.
if is_multi_char
&& text.chars().all(char::is_whitespace)
&& text.chars().any(|ch| matches!(ch, '\t' | '\n' | '\r'))
{
return if text.contains('\t') {
"\t".to_string()
} else {
" ".to_string()
};
}
if is_multi_char
&& text.contains('\u{00ad}')
&& text.chars().all(|ch| {
matches!(
ch,
'-' | '\u{00ad}' | '\u{2010}' | '\u{2011}' | '\u{2012}' | '\u{2013}' | '\u{2212}'
)
})
{
return "-".to_string();
}
text
}
fn hex_to_unicode_scalar(hex: &str) -> Option<u32> {
let text = hex_to_unicode_string(hex)?;
let mut chars = text.chars();
let ch = chars.next()?;
if chars.next().is_none() {
Some(ch as u32)
} else {
if result.is_empty() {
None
} else {
Some(result)
}
}
@@ -723,81 +657,6 @@ fn get_w_array_start_cid(cid_font_dict: &lopdf::Dictionary, doc: &Document) -> O
}
}
/// Return true if the CIDFont's W (widths) array explicitly covers the given CID.
///
/// The W array uses two formats (PDF 32000-1:2008, §9.7.4.3):
/// 1. `c [w1 w2 ... wn]` — widths for CIDs c, c+1, ..., c+n-1
/// 2. `c_first c_last w` — CIDs c_first..c_last all have width w
fn w_array_covers_cid(cid_font_dict: &lopdf::Dictionary, doc: &Document, target: u16) -> bool {
let Ok(w_obj) = cid_font_dict.get(b"W") else {
return false;
};
let arr = match w_obj {
Object::Array(arr) => arr,
Object::Reference(r) => match doc.get_object(*r) {
Ok(Object::Array(arr)) => arr,
_ => return false,
},
_ => return false,
};
let resolve_int = |o: &Object| -> Option<i64> {
match o {
Object::Integer(n) => Some(*n),
Object::Reference(r) => match doc.get_object(*r) {
Ok(Object::Integer(n)) => Some(*n),
_ => None,
},
_ => None,
}
};
let resolve_arr = |o: &Object| -> Option<Vec<Object>> {
match o {
Object::Array(a) => Some(a.clone()),
Object::Reference(r) => match doc.get_object(*r) {
Ok(Object::Array(a)) => Some(a.clone()),
_ => None,
},
_ => None,
}
};
let target = target as i64;
let mut i = 0usize;
while i < arr.len() {
let Some(first) = resolve_int(&arr[i]) else {
break;
};
i += 1;
if i >= arr.len() {
break;
}
// Peek at arr[i] to decide format.
if let Some(widths) = resolve_arr(&arr[i]) {
// Format 1: c [w1 ... wn]
let last = first + widths.len() as i64 - 1;
if target >= first && target <= last {
return true;
}
i += 1;
} else if let Some(last) = resolve_int(&arr[i]) {
// Format 2: c_first c_last w
i += 1;
if i < arr.len() {
i += 1; // skip the width value
}
if target >= first && target <= last {
return true;
}
} else {
// Unknown token — abort parsing safely
break;
}
}
false
}
/// Extract CIDToGIDMap as a vector of GIDs (u16) indexed by CID.
fn get_cid_to_gid_map(cid_font_dict: &lopdf::Dictionary, doc: &Document) -> Option<Vec<u16>> {
let obj = cid_font_dict.get(b"CIDToGIDMap").ok()?;
@@ -893,20 +752,6 @@ fn try_remap_subset_cmap(
_ => return (cmap, None),
};
// If the W array actually covers the CMap's max source CID, the CMap is
// aligned with the font — no sequential renumbering happened. A sparse W
// array starting at CID 0 (for .notdef) with additional high-CID entries
// matching the CMap is the normal subset layout, not a mismatch.
if let Some(max_cid) = cmap.max_source_cid() {
if w_array_covers_cid(cid_font_dict, doc, max_cid) {
debug!(
"Subset remap skipped for obj={}: W array covers CMap max CID {}",
obj_num, max_cid
);
return (cmap, None);
}
}
debug!(
"Subset GID mismatch detected for obj={}: W starts at CID {}, CMap min CID {}. Remapping to sequential.",
obj_num, w_start, min_cid
@@ -2661,97 +2506,6 @@ endbfrange
assert_eq!(cmap.lookup(0x0005), Some("C".to_string()));
}
#[test]
fn test_parse_bfchar_surrogate_pair_emoji() {
let cmap_content = r#"
1 begincodespacerange
<00> <FF>
endcodespacerange
2 beginbfchar
<16> <D83CDF1F>
<9D> <D83CDFAD>
endbfchar
"#;
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
assert_eq!(cmap.code_byte_length, 1);
assert_eq!(cmap.lookup(0x16), Some("🌟".to_string()));
assert_eq!(cmap.lookup(0x9D), Some("🎭".to_string()));
}
#[test]
fn test_parse_bfrange_surrogate_pair_base() {
let cmap_content = r#"
1 begincodespacerange
<00> <FF>
endcodespacerange
1 beginbfrange
<C8> <C9> <D83CDFD8>
endbfrange
"#;
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
assert_eq!(cmap.code_byte_length, 1);
assert_eq!(cmap.lookup(0xC8), Some("🏘".to_string()));
assert_eq!(cmap.lookup(0xC9), Some("🏙".to_string()));
}
#[test]
fn test_parse_bfrange_preserves_single_hyphen_like_base() {
let cmap_content = r#"
1 begincodespacerange
<00> <FF>
endcodespacerange
1 beginbfrange
<21> <22> <2013>
endbfrange
"#;
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
assert_eq!(cmap.lookup(0x21), Some("".to_string()));
assert_eq!(cmap.lookup(0x22), Some("".to_string()));
}
#[test]
fn test_parse_spaced_destination_hex_without_control_noise() {
let cmap_content = r#"
1 begincodespacerange
<00> <FF>
endcodespacerange
3 beginbfchar
<21> < 0009 000d 0020 00a0 >
<22> < 002d 00ad 2010 >
<23> <00a0>
endbfchar
"#;
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
assert_eq!(cmap.lookup(0x21), Some("\t".to_string()));
assert_eq!(cmap.lookup(0x22), Some("-".to_string()));
assert_eq!(cmap.lookup(0x23), Some("\u{00a0}".to_string()));
}
#[test]
fn test_parse_preserves_valid_multi_character_destinations() {
let cmap_content = r#"
1 begincodespacerange
<00> <FF>
endcodespacerange
4 beginbfchar
<21> <002d002d>
<22> <20132013>
<23> <002000a0>
<24> <00660069>
endbfchar
"#;
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
assert_eq!(cmap.lookup(0x21), Some("--".to_string()));
assert_eq!(cmap.lookup(0x22), Some("––".to_string()));
assert_eq!(cmap.lookup(0x23), Some(" \u{00a0}".to_string()));
assert_eq!(cmap.lookup(0x24), Some("fi".to_string()));
}
#[test]
fn test_remap_to_sequential() {
// Simulate a broken CMap where GIDs are from pre-subsetting:
@@ -2963,187 +2717,4 @@ endbfchar
assert_eq!(remapped.unwrap().char_map.len(), 50);
assert_eq!(fallback.unwrap().char_map.len(), 10);
}
#[test]
fn test_max_source_cid() {
let cmap_content = r#"
1 begincodespacerange
<0000><FFFF>
endcodespacerange
2 beginbfchar
<0003> <0020>
<0031> <004E>
endbfchar
1 beginbfrange
<0208> <0227> <0430>
endbfrange
"#;
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
assert_eq!(cmap.min_source_cid(), Some(0x0003));
assert_eq!(cmap.max_source_cid(), Some(0x0227));
}
/// Helper: build a minimal CIDFont dict with a W array and check coverage.
fn cid_font_dict_with_w(w_items: Vec<lopdf::Object>) -> lopdf::Dictionary {
let mut d = lopdf::Dictionary::new();
d.set("W", lopdf::Object::Array(w_items));
d
}
#[test]
fn test_w_array_covers_cid_format1() {
// Format 1: `c [w1 w2 ... wn]` — widths for CIDs c..c+n-1.
// Mimics the 16.pdf Tahoma W array: 0[1000] 3[313] 5[401] 11[383 383] 16[363 303 382]
let doc = Document::new();
let d = cid_font_dict_with_w(vec![
lopdf::Object::Integer(0),
lopdf::Object::Array(vec![lopdf::Object::Integer(1000)]),
lopdf::Object::Integer(3),
lopdf::Object::Array(vec![lopdf::Object::Integer(313)]),
lopdf::Object::Integer(5),
lopdf::Object::Array(vec![lopdf::Object::Integer(401)]),
lopdf::Object::Integer(11),
lopdf::Object::Array(vec![
lopdf::Object::Integer(383),
lopdf::Object::Integer(383),
]),
lopdf::Object::Integer(16),
lopdf::Object::Array(vec![
lopdf::Object::Integer(363),
lopdf::Object::Integer(303),
lopdf::Object::Integer(382),
]),
lopdf::Object::Integer(570),
lopdf::Object::Array(vec![lopdf::Object::Integer(667); 26]),
]);
assert!(w_array_covers_cid(&d, &doc, 0));
assert!(w_array_covers_cid(&d, &doc, 3));
assert!(w_array_covers_cid(&d, &doc, 5));
assert!(w_array_covers_cid(&d, &doc, 11));
assert!(w_array_covers_cid(&d, &doc, 12));
assert!(w_array_covers_cid(&d, &doc, 16));
assert!(w_array_covers_cid(&d, &doc, 18));
assert!(w_array_covers_cid(&d, &doc, 570));
assert!(w_array_covers_cid(&d, &doc, 595));
// Gaps are NOT covered
assert!(!w_array_covers_cid(&d, &doc, 1));
assert!(!w_array_covers_cid(&d, &doc, 4));
assert!(!w_array_covers_cid(&d, &doc, 19));
assert!(!w_array_covers_cid(&d, &doc, 596));
}
#[test]
fn test_w_array_covers_cid_format2() {
// Format 2: `c_first c_last w` — CIDs c_first..c_last all have width w.
let doc = Document::new();
let d = cid_font_dict_with_w(vec![
lopdf::Object::Integer(100),
lopdf::Object::Integer(120),
lopdf::Object::Integer(500),
]);
assert!(w_array_covers_cid(&d, &doc, 100));
assert!(w_array_covers_cid(&d, &doc, 110));
assert!(w_array_covers_cid(&d, &doc, 120));
assert!(!w_array_covers_cid(&d, &doc, 99));
assert!(!w_array_covers_cid(&d, &doc, 121));
}
#[test]
fn test_w_array_covers_cid_missing_w() {
let doc = Document::new();
let d = lopdf::Dictionary::new();
assert!(!w_array_covers_cid(&d, &doc, 3));
}
#[test]
fn test_try_remap_skipped_when_w_covers_cmap() {
// Simulates 16.pdf: CMap's max source CID (0x0279 = 633) is explicitly
// in the W array, so no subset-renumbering happened — remap must NOT fire.
let cmap_content = r#"
1 begincodespacerange
<0000><FFFF>
endcodespacerange
2 beginbfchar
<0003> <0020>
<0031> <004E>
endbfchar
2 beginbfrange
<023A> <0253> <0410>
<0255> <0279> <042B>
endbfrange
"#;
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
let mut doc = Document::new();
// Build a CIDFont dict with Identity CIDToGIDMap and a W array that
// covers CID 633 via `597 [widths...]`.
let mut cid_font = lopdf::Dictionary::new();
cid_font.set("CIDToGIDMap", lopdf::Object::Name(b"Identity".to_vec()));
cid_font.set(
"W",
lopdf::Object::Array(vec![
lopdf::Object::Integer(0),
lopdf::Object::Array(vec![lopdf::Object::Integer(750)]),
lopdf::Object::Integer(597),
lopdf::Object::Array(vec![lopdf::Object::Integer(500); 37]), // 597..633
]),
);
let cid_font_id = doc.add_object(cid_font);
// Build the Type0 font dict with Identity-H + DescendantFonts ref.
let mut font_dict = lopdf::Dictionary::new();
font_dict.set("Encoding", lopdf::Object::Name(b"Identity-H".to_vec()));
font_dict.set(
"DescendantFonts",
lopdf::Object::Array(vec![lopdf::Object::Reference(cid_font_id)]),
);
let (primary, remapped) = try_remap_subset_cmap(cmap, &font_dict, &doc, 123);
assert!(
remapped.is_none(),
"Remap must be skipped when W covers CMap max CID (this is 16.pdf)"
);
assert_eq!(primary.lookup(0x0003), Some(" ".to_string()));
}
#[test]
fn test_try_remap_fires_for_true_subset_mismatch() {
// True mismatch: CMap has high CIDs (512-544) but W only lists low sequential CIDs.
let cmap_content = r#"
1 begincodespacerange
<0000><FFFF>
endcodespacerange
1 beginbfrange
<0200> <0220> <0410>
endbfrange
"#;
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
let mut doc = Document::new();
let mut cid_font = lopdf::Dictionary::new();
cid_font.set("CIDToGIDMap", lopdf::Object::Name(b"Identity".to_vec()));
cid_font.set(
"W",
lopdf::Object::Array(vec![
lopdf::Object::Integer(0),
lopdf::Object::Array(vec![lopdf::Object::Integer(500); 34]), // 0..33
]),
);
let cid_font_id = doc.add_object(cid_font);
let mut font_dict = lopdf::Dictionary::new();
font_dict.set("Encoding", lopdf::Object::Name(b"Identity-H".to_vec()));
font_dict.set(
"DescendantFonts",
lopdf::Object::Array(vec![lopdf::Object::Reference(cid_font_id)]),
);
let (_primary, remapped) = try_remap_subset_cmap(cmap, &font_dict, &doc, 456);
assert!(
remapped.is_some(),
"Remap must fire when CMap's CIDs are outside W array coverage"
);
}
}
+7 -36
View File
@@ -116,14 +116,6 @@ pub struct TextItem {
pub is_bold: bool,
/// Whether the font is italic
pub is_italic: bool,
/// Whether the text is underlined (drawn rule/thin rect under the
/// baseline — PDFs have no underline font flag, so this is detected
/// geometrically after extraction; see `extractor::underline`).
pub is_underline: bool,
/// Whether the text is struck out (drawn rule/thin rect crossing the
/// glyphs at mid x-height). Same geometric detection as underline,
/// different vertical window; see `extractor::underline`.
pub is_strikeout: bool,
/// Type of item (text, image, link)
pub item_type: ItemType,
/// Marked Content ID from the content stream's BDC/BMC operator.
@@ -145,17 +137,12 @@ pub struct TextLine {
impl TextLine {
pub fn text(&self) -> String {
self.text_with_formatting(false, false, false)
self.text_with_formatting(false, false)
}
/// Get text with optional bold/italic/underline markdown formatting
pub fn text_with_formatting(
&self,
format_bold: bool,
format_italic: bool,
format_underline: bool,
) -> String {
if !format_bold && !format_italic && !format_underline {
/// Get text with optional bold/italic markdown formatting
pub fn text_with_formatting(&self, format_bold: bool, format_italic: bool) -> String {
if !format_bold && !format_italic {
return self.text_plain();
}
@@ -164,7 +151,6 @@ impl TextLine {
let mut result = String::new();
let mut current_bold = false;
let mut current_italic = false;
let mut current_underline = false;
for (i, item) in self.items.iter().enumerate() {
let text = item.text.as_str();
@@ -190,13 +176,9 @@ impl TextLine {
// we push text_trimmed below (which strips it).
let has_leading_space = text.starts_with(' ');
// Check for style changes. Underline is exclusive: `<u>` content
// stays free of `**`/`*` markers — consumers (and the eval
// harnesses this feeds) match the tag content literally, and
// mixed `<u>**x**</u>` nesting breaks that.
let item_underline = format_underline && item.is_underline;
let item_bold = format_bold && item.is_bold && !item_underline;
let item_italic = format_italic && item.is_italic && !item_underline;
// Check for style changes
let item_bold = format_bold && item.is_bold;
let item_italic = format_italic && item.is_italic;
// Close previous styles if they change
if current_italic && !item_italic {
@@ -207,10 +189,6 @@ impl TextLine {
result.push_str("**");
current_bold = false;
}
if current_underline && !item_underline {
result.push_str("</u>");
current_underline = false;
}
// Add space: either from spacing logic or preserved from item text
if needs_space || (has_leading_space && !result.is_empty() && !result.ends_with(' ')) {
@@ -218,10 +196,6 @@ impl TextLine {
}
// Open new styles
if item_underline && !current_underline {
result.push_str("<u>");
current_underline = true;
}
if item_bold && !current_bold {
result.push_str("**");
current_bold = true;
@@ -241,9 +215,6 @@ impl TextLine {
if current_bold {
result.push_str("**");
}
if current_underline {
result.push_str("</u>");
}
result
}
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
+18 -2072
View File
File diff suppressed because it is too large Load Diff
+2 -2
View File
@@ -188,7 +188,7 @@
|156|23/7|General renovation works to Block B at Belonie Secondary School|MOE|Belvedere Builders|SR869,505.75|
|157|30/7|Procurement of Engine Block and Crankshaft for Engine A11|PUC|Ras Tek Pvt Ltd|Euro798,650.00|
|158|30/7|procurement of Wartsila Engine spares|PUC|Wartsila Eastern Africa ltd|Euro158,424.00|
|159|30/7|Proposed walkway, Drain, rock armoring, road and Bridge widening at Anse Talbot( Ex-Golden Egg)|SLTA|G&S Enterpise|SR1,113,010.00|
|159|30/7|Proposed walkway, Drain, rock armoring , road and Bridge widening at Anse Talbot( Ex-Golden Egg)|SLTA|G&S Enterpise|SR1,113,010.00|
|160|30/7|Procurement of transfer pump control panel|PUC|CA Engineering Consultancy Pte Ltd|SGD14,600.00|
|161|30/7|Consultancy service for North to South Victoria Bye- Pass road and utilities organisation|MLUH|Sonnel Seychelles LTD|SR1,332,000.00|
|162 AUG|30/7|Procurement of the supply of sodium cardonate|PUC|HPL Chemical LTD|USD42,600.00|
@@ -237,7 +237,7 @@
|201|24/9|Procurement of vehicle x 2|SLTA|Abhaye Valabhji Pty Ltd|SR1000.000.00|
||OCT|||||
|202|1/10|Proposed new traffic lane to 5th June Avenue|SLTA|Divy Constrution|SR2,864,589.00|
|203|1/10||Proposed Walkway, Drain, rock armoring, road and Bridge widening at Anse Talbot(Ex-Golden Egg) - Variations SLTA|G & S Enterprise|SR200,448.00|
|203|1/10||Proposed Walkway, Drain, rock armoring , road and Bridge widening at Anse Talbot( Ex-Golden Egg) - Variations SLTA|G & S Enterprise|SR200,448.00|
|204|1/10|Proposed Reconstrcution of Burnt House-Au Cap|MLUH|Furui Construction|SR946,130.00|
|205|1/10|Variation on the project associated with the procurement of seven 100m3/day containerised plant|PUC|Tornado Group (UAE)|USD172,500.00|
|206|1/10|Works on the breaker system at Bel Omber desalination plant|PUC|United Concrete Products (Sey)Ltd|SR1,998,993.11|
+4 -4
View File
@@ -1,9 +1,9 @@
본 가격표는 국내 거주 중인 외국인을 위한 한국어 가격표의 비공식 번역본입니다. ※ The post-tax benefit sales price is provided for your reference only, reflecting the current tax benefits and eco-friendly vehicle individual consumption tax reductions. 본 가격표와 한국어 가격표의 내용이 상이한 경우 한국어 가격표의 내용이 우선하므로, 반드시 한국어 가격표의 내용을 확인하십시오. The final sales price may vary depending on the addition of optional items and whether the eco-friendly vehicle criteria are met, so please be sure to check the quotation. This price list is an unofficial translation of the Korean price list for the convenience of foreign residents in South Korea. ※ Please check the Korean price list for information on colors, details, and fuel consumption for each model. If the price list differs from the Korean price list, please check the contents of the Korean price list first. ※ All optional item prices are listed based on pre-tax reduction amounts. The actual sales price, which reflects the total individual consumption tax reduction including optional items, may differ depending on applicable tax benefits. ※ The items (specifications, colors, etc.) and prices listed in this pricing table are subject to change without prior notice depending on the holding of new car launch events, improvements made in automobile performance, introduction of related laws and regulations, and changes in company circumstances. The all-new NEXO Release Date: June 10, 2025 / (Unit: KRW)
|Classification|Selling price before tax benefit Supply value(surtax)|Selling price after tax benefit|Standard equipment|Options (before tax benefit)|
|Classification Exclusive Exclusive|Selling price before tax benefit Supply value(surtax) 80,509,000 73,190,000(7,319,000)|Selling price after tax benefit 76,435,000|Standard equipment • Powertrain/Performance: Fuel cell system(150kW drive motor, lithium-ion battery, and reducer), Regenerative braking system, Column-Type Shift By Wire(vibration warning), Drive mode select • Safety: 9 airbag system(1st-row advanced/center side airbags, 1st/2nd-row side airbags, and rollover-resistant curtain airbags), Multi-Collision Brake System, Active hood system(for pedestrian protection), Safety unlock function, Artificial engine sound(for pedestrian protection), Child seat fastening system (2 in 2nd-row), Fire extinguisher for vehicles, Pedal Misapplication Safety Assist • Smart Safety Technology: Forward Collision-avoidance Assist(vehicles/ pedestrians/two-wheeled vehicles/junction turning/front oncoming), Smart Cruise Control with Stop & Go, Lane Keeping Assist, Lane Following Assist 2, Blind-spot Collision Warning(driving), Blind-spot Collision-avoidance Assist(forward exit), Rear Cross-traffic Collision-avoidance Assist, Safety Exit Assist, Driver Attention Warning, High Beam Assist, Advanced Rear Occupant Alert, Intelligent Speed Limit Assist, Hands-On Detection, Highway Driving Assist, Navigation-based Smart Cruise Control(safety speed zone/curve control), Vibration warning steering wheel • Exterior: Full LED headlamps(projection type), LED turn signal lamps (front and rear), LED Daytime Running Lights, LED positioning lights, LED rear combination lamps, LED third brake lights, 18-inch alloy wheels & tires, Solar glass(windshield), Double-glazed soundproof glass(windshield, and 1st/ 2nd-row doors), Outside mirror(heating, power-folding, power adjustment,|Options (before tax benefit) ▶ Hi-pass(e hi-pass) [200,000]|
|---|---|---|---|---|
|Exclusive|80,509,000 73,190,000(7,319,000) with 3.5% individual consumption tax applied 79,287,000|76,435,000 with 3.5% individual consumption tax applied 76,435,000|• Powertrain/Performance: Fuel cell system(150kW drive motor, lithium-ion battery, and reducer), Regenerative braking system, Column-Type Shift By Wire(vibration warning), Drive mode select • Safety: 9 airbag system(1st-row advanced/center side airbags, 1st/2nd-row side airbags, and rollover-resistant curtain airbags), Multi-Collision Brake System, Active hood system(for pedestrian protection), Safety unlock function, Artificial engine sound(for pedestrian protection), Child seat fastening system (2 in 2nd-row), Fire extinguisher for vehicles, Pedal Misapplication Safety Assist • Smart Safety Technology: Forward Collision-avoidance Assist(vehicles/ pedestrians/two-wheeled vehicles/junction turning/front oncoming), Smart Cruise Control with Stop & Go, Lane Keeping Assist, Lane Following Assist 2, Blind-spot Collision Warning(driving), Blind-spot Collision-avoidance Assist(forward exit), Rear Cross-traffic Collision-avoidance Assist, Safety Exit Assist, Driver Attention Warning, High Beam Assist, Advanced Rear Occupant Alert, Intelligent Speed Limit Assist, Hands-On Detection, Highway Driving Assist, Navigation-based Smart Cruise Control(safety speed zone/curve control), Vibration warning steering wheel • Exterior: Full LED headlamps(projection type), LED turn signal lamps (front and rear), LED Daytime Running Lights, LED positioning lights, LED rear combination lamps, LED third brake lights, 18-inch alloy wheels & tires, Solar glass(windshield), Double-glazed soundproof glass(windshield, and 1st/ 2nd-row doors), Outside mirror(heating, power-folding, power adjustment, and LED turn signal lamps), Auto flush door handles, Black door garnish • Interior: Panoramic curved display, 12.3-inch color LCD cluster, Leather- upholstered steering wheel(with heating, two-tone color, and Interactive Pixel Lights), LED interior lamp (map lamp, personal lamp, sun visor lamp, and luggage lamp), Metallic door scuff plate • Seat: Synthetic leather seats, 1st-row manual seats, Heated 1st-row seats, 2nd-row 60/40-split folding seats(reclining) • Convenience: Proximity key with push-button start, Smart key remote start, Electronic Parking Brake(with automatic vehicle hold), Paddle shift(regenerative control), Dual-zone full automatic air conditioning(with high-performance antibacterial combination filter, auto defog system, fine dust sensor, air cleaning mode, and after-blow function), 2nd-row seat air vent, Auto light control system, USB Type-C Ports(1×27W switchable charging/data port in 1st-row, and 2×100W charging ports in both 1st and 2nd-row), ECM room mirror(frameless), Rain sensor, Power windows with pinch protection(1st/2nd-row), Power outlet (1 in 1st-row), Parking Distance Warning-Forward/Reverse, Rear View Monitor, Wireless phone charger(single), Walk-away lock, Route planner, Hyundai AI Assistant • Infotainment: 12.3-inch navigation(Bluelink, phone projection, Bluetooth hands-free, and In-car Payment), Audio system(6 speakers), Over-The-Air navigation updates|Hi-pass(e hi-pass) [200,000]|
|Exclusive Special|83,500,000 75,909,091(7,590,909) with 3.5% individual consumption tax applied 82,232,000|79,275,000 with 3.5% individual consumption tax applied 79,275,000|▶ Standard equipment of Exclusive plus • Smart Safety Technology: Forward Collision-avoidance Assist(intersection crossing/changing lanes in oncoming traffic/approaching from either side/ evasive steering assist), Highway Driving Assist 2, Navigation-based Smart Cruise Control(access road)Exterior: Roof rack • Interior: Metallic pedal, Driving mode-dependent ambient mood lighting(crash pad, 1st/2nd-row door trim) • Seat: Synthetic leather seats(patch applied), Power-adjustable driver's seat(8-way, lumbar support, and Integrated Memory System(driver's seat and outside mirror connected)), Power-adjustable front passengers seat(8-way), Ventilated 1st-row seats, Heated 2nd-row seats • Convenience: Hi-pass(e hi-pass), In-car fingerprint authentication system(personalization, startup, payment, and etc.), Smart power tailgate|▶ [600,000] Built-in Cam 2 Plus, Augmented reality navigation ▶ [850,000] Indoor/outdoor V2L ▶ [950,000] Parking Assist ▶ [1,150,000] Audio by BANG & OLUFSEN sound system ▶ [250,000] 19-inch alloy wheels & tires|
|Prestige|87,893,000 79,902,727(7,990,273) with 3.5% individual consumption tax applied 86,559,000|83,445,000 with 3.5% individual consumption tax applied 83,445,000|▶ Standard equipment of Exclusive Special plus • Smart Safety Technology: Remote Smart Parking Assist 2, Parking Collison- avoidance Assist(front/side/rear) • Exterior: Intelligent Front-Lighting System(IFS), Dynamic welcome/escort lighting(1 type), Sequential turn signals(front and rear), Ambient lighting auto flush door handles, Two-tone door garnish, Glossy black rear diffuser • Interior: Recycled PET suede interior materials(headlining/sunvisor), Fabric upholstered crash pad • Seat: BIO-processed natural leather seats(metal patch applied, embossed design punching), Passenger's seat walk-in device, 1st-row relaxation comfort seats(leg rest included), Ventilated 2nd-row seats • Convenience: Parking Distance Warning-Side, Head-Up Display, Digital key 2, Wireless phone charger(dual), Surround View Monitor, Blind-spot View Monitor, LED reverse light guide • Infotainment: Audio by BANG & OLUFSEN sound system(14 speakers, including external amp), Active road noise control, Active Sound Design|▶ [600,000] Built-in Cam 2 Plus, Augmented reality navigation ▶ [850,000] Indoor/outdoor V2L ▶ [900,000] Vision roof ▶ [1,380,000] Digital side mirror ▶ [750,000] Camera package ▶ [250,000] 19-inch alloy wheels & tires|
|Special|with 3.5% individual consumption tax applied 79,287,000 83,500,000 75,909,091(7,590,909) with 3.5% individual consumption tax applied 82,232,000|with 3.5% individual consumption tax applied 76,435,000 79,275,000 with 3.5% individual consumption tax applied 79,275,000|and LED turn signal lamps), Auto flush door handles, Black door garnish • Interior: Panoramic curved display, 12.3-inch color LCD cluster, Leather- upholstered steering wheel(with heating, two-tone color, and Interactive Pixel Lights), LED interior lamp (map lamp, personal lamp, sun visor lamp, and luggage lamp), Metallic door scuff plate • Seat: Synthetic leather seats, 1st-row manual seats, Heated 1st-row seats, 2nd-row 60/40-split folding seats(reclining) • Convenience: Proximity key with push-button start, Smart key remote start, Electronic Parking Brake(with automatic vehicle hold), Paddle shift(regenerative control), Dual-zone full automatic air conditioning(with high-performance antibacterial combination filter, auto defog system, fine dust sensor, air cleaning mode, and after-blow function), 2nd-row seat air vent, Auto light control system, USB Type-C Ports(1×27W switchable charging/data port in 1st-row, and 2×100W charging ports in both 1st and 2nd-row), ECM room mirror(frameless), Rain sensor, Power windows with pinch protection(1st/2nd-row), Power outlet (1 in 1st-row), Parking Distance Warning-Forward/Reverse, Rear View Monitor, Wireless phone charger(single), Walk-away lock, Route planner, Hyundai AI Assistant • Infotainment: 12.3-inch navigation(Bluelink, phone projection, Bluetooth hands-free, and In-car Payment), Audio system(6 speakers), Over-The-Air navigation updates Standard equipment of Exclusive plus • Smart Safety Technology: Forward Collision-avoidance Assist(intersection crossing/changing lanes in oncoming traffic/approaching from either side/ evasive steering assist), Highway Driving Assist 2, Navigation-based Smart Cruise Control(access road) • Exterior: Roof rack • Interior: Metallic pedal, Driving mode-dependent ambient mood lighting(crash pad, 1st/2nd-row door trim) • Seat: Synthetic leather seats(patch applied), Power-adjustable driver's|▶ [600,000] Built-in Cam 2 Plus, Augmented reality navigation ▶ [850,000] Indoor/outdoor V2L ▶ [950,000] Parking Assist ▶ [1,150,000] Audio by BANG & OLUFSEN|
|Prestige|87,893,000 79,902,727(7,990,273) with 3.5% individual consumption tax applied 86,559,000|83,445,000 with 3.5% individual consumption tax applied 83,445,000|seat(8-way, lumbar support, and Integrated Memory System(driver's seat and outside mirror connected)), Power-adjustable front passengers seat(8-way), Ventilated 1st-row seats, Heated 2nd-row seats • Convenience: Hi-pass(e hi-pass), In-car fingerprint authentication system(personalization, startup, payment, and etc.), Smart power tailgate ▶ Standard equipment of Exclusive Special plus • Smart Safety Technology: Remote Smart Parking Assist 2, Parking Collison- avoidance Assist(front/side/rear) • Exterior: Intelligent Front-Lighting System(IFS), Dynamic welcome/escort lighting(1 type), Sequential turn signals(front and rear), Ambient lighting auto flush door handles, Two-tone door garnish, Glossy black rear diffuserInterior: Recycled PET suede interior materials(headlining/sunvisor), Fabric upholstered crash pad • Seat: BIO-processed natural leather seats(metal patch applied, embossed design punching), Passenger's seat walk-in device, 1st-row relaxation comfort seats(leg rest included), Ventilated 2nd-row seats • Convenience: Parking Distance Warning-Side, Head-Up Display, Digital key 2, Wireless phone charger(dual), Surround View Monitor, Blind-spot View Monitor, LED reverse light guide • Infotainment: Audio by BANG & OLUFSEN sound system(14 speakers, including external amp), Active road noise control, Active Sound Design|sound system ▶ [250,000] 19-inch alloy wheels & tires ▶ [600,000] Built-in Cam 2 Plus, Augmented reality navigation ▶ [850,000] Indoor/outdoor V2L ▶ [900,000] Vision roof ▶ [1,380,000] Digital side mirror ▶ [750,000] Camera package ▶ [250,000] 19-inch alloy wheels & tires|
**Classification Details** **Indoor/outdoor V2L** Indoor V2L, Outdoor V2L(connectorless type) **Parking Assist** Surround View Monitor, Blind-spot View Monitor, Parking Distance Warning-Side, Parking Collison-avoidance Assist-Rear **Audio by BANG & OLUFSEN** Audio by BANG & OLUFSEN sound system(14 speakers, including external amp.), Active road noise control, Active Sound Design **sound system** **Camera package** Digital center mirror(with camera sensor cleaning system), Driver monitoring system THE ALL-NEW NEXO /// ECO-FRIENDLY CAR
+16 -15
View File
@@ -10,7 +10,7 @@ Department of the Treasury **Internal Revenue Service**
### This publication contains:
**Form 4070A,** Employees Daily Record of Tips **Form 4070,** Employees Report of Tips to Employer
**Form 4070A, Employees Daily Record of** Tips **Form 4070, Employees Report of Tips to** Employer
For the period
@@ -22,7 +22,7 @@ Name and address of employee
**Publication 1244 (Rev. 7-96)** Cat. No. 44472W
**Instructions** You must keep sufficient proof to show the amount of your tip income for the year. A daily record of your tip income is considered sufficient proof. Keep a daily record for each workday showing the amount of cash and credit card tips received directly from customers or other employees. Also keep a record of the amount of tips, if any, you paid to other employees through tip sharing, tip pooling or other arrangements, and the names of employees to whom you paid tips. Show the date that each entry is made. This date should be on or near the date you received the tip income. You may use **Form 4070A**, Employees Daily Record of Tips, or any other daily record to record your tips. **Reporting Tips to Your Employer.—**If you receive tips that total $20 or more for any month while working for one employer, you must report the tips to your employer. Tips include cash left by customers, tips customers add to credit card charges, and tips you receive from other employees. You must report your tips for any one month by the 10th day of the next month. If the 10th day falls on a Saturday, Sunday, or legal holiday, you may give the report to your employer on the next business day that is not a Saturday, Sunday, or legal holiday. You must report tips that total $20 or more every month regardless of your total wages and tips for the year. You may use **Form 4070**, Employees Report of Tips to Employer, to report your tips to your employer. See the instructions on the back of Form 4070. You must include all tips, including tips not reported to your employer, as wages on your income tax return. You may use the last page of this publication to total your tips for the year. Your employer must withhold income, social security, and Medicare (or railroad retirement) taxes on tips you report. Your employer usually deducts the withholding due on tips from your regular wages.
**Instructions** You must keep sufficient proof to show the amount of your tip income for the year. A daily record of your tip income is considered sufficient proof. Keep a daily record for each workday showing the amount of cash and credit card tips received directly from customers or other employees. Also keep a record of the amount of tips, if any, you paid to other employees through tip sharing, tip pooling or other arrangements, and the names of employees to whom you paid tips. Show the date that each entry is made. This date should be on or near the date you received the tip income. You may use Form 4070A, Employees Daily Record of Tips, or any other daily record to record your tips. **Reporting Tips to Your Employer.—If you** receive tips that total $20 or more for any month while working for one employer, you must report the tips to your employer. Tips include cash left by customers, tips customers add to credit card charges, and tips you receive from other employees. You must report your tips for any one month by the 10th day of the next month. If the 10th day falls on a Saturday, Sunday, or legal holiday, you may give the report to your employer on the next business day that is not a Saturday, Sunday, or legal holiday. You must report tips that total $20 or more every month regardless of your total wages and tips for the year. You may use Form 4070, Employees Report of Tips to Employer, to report your tips to your employer. See the instructions on the back of Form 4070. You must include all tips, including tips not reported to your employer, as wages on your income tax return. You may use the last page of this publication to total your tips for the year. Your employer must withhold income, social security, and Medicare (or railroad retirement) taxes on tips you report. Your employer usually deducts the withholding due on tips from your regular wages.
*(continued on inside of back cover)*
@@ -30,14 +30,14 @@ Form **4070A** Employees Daily Record of Tips (Rev. July 1996) **This is a vo
Establishment name (if different)
Date Date **a.** Tips received
Date Date **a. Tips received**
**b.** Credit card tips **c.** Tips paid out to **d.** Names of employees to whom you
**b. Credit card tips c. Tips paid out to d. Names of employees to whom you**
tips of directly from customers received other employees paid tips recd. entry and other employees 1 2 3 4 5 **Subtotals** **For Paperwork Reduction Act Notice, see Instructions on the back of Form 4070. Page 1**
Date Date **a.** Tips received
Date Date **a. Tips received**
**b.** Credit card tips **c.** Tips paid out to **d.** Names of employees to whom you
**b. Credit card tips c. Tips paid out to d. Names of employees to whom you**
tips of directly from customers received other employees paid tips recd. entry and other employees
7 8 9 10 11 12 13 14 15 **Subtotals**
@@ -48,11 +48,11 @@ tips of directly from customers received other employees paid tips recd. entr
**Page 3**
27 28 29 30 31 **Subtotals from pages** **1, 2, and 3** **Totals**
27 28 29 30 31 **Subtotals** **from pages** **1, 2, and 3** **Totals**
**1.** Report total cash tips (col. **a**) on Form 4070, line **1.**
**2.** Report total credit card tips (col. **b**) on Form 4070, line **2.**
**3.** Report total tips paid out (col. **c**) on Form 4070, line **3.** **Page 4**
**1.** Report total cash tips (col. a) on Form 4070, line 1.
**2.** Report total credit card tips (col. b) on Form 4070, line 2.
**3.** Report total tips paid out (col. c) on Form 4070, line 3. **Page 4**
Form Employees Report (Rev. July 1996)
@@ -66,16 +66,17 @@ Employers name and address (include establishment name, if different) **1** C
**3** Tips paid out
Month or shorter period in which tips were received **4** Net tips (lines **1 + 2 - 3**) from, 19, to, 19 Signature Date
Month or shorter period in which tips were received **4** Net tips (lines 1 + 2 - 3) from, 19, to, 19 Signature Date
**Paperwork Reduction Act Notice.—**We ask for the information on these forms to carry out the Internal Revenue laws of the United States. You are required to give us the information. We need it to ensure that you are complying with these laws and to allow us to figure and collect the right amount of tax. You are not required to provide the information requested on a form that is subject to the Paperwork Reduction Act unless the form displays a valid OMB control number. Books or records relating to a form or its instructions must be retained as long as their contents may become material in the administration of any Internal Revenue law. Generally, tax returns and return information are confidential, as required by Code section 6103. The time needed to complete Forms 4070 and 4070A will vary depending on individual circumstances. The estimated average times are: **Recordkeeping**—Form 4070, 7 min.; Form 4070A, 3 hr. and 23 min.; **Learning** **about the law**—each form, 2 min.; **Preparing** Form 4070, 13 min.; Form 4070A, 55 min.; and **Copying and** **providing** Form 4070, 10 min.; Form 4070A, 14 min. If you have comments concerning the accuracy of these time estimates or suggestions for making these
**Paperwork Reduction Act Notice.—We ask for the** information on these forms to carry out the Internal Revenue laws of the United States. You are required to give us the information. We need it to ensure that you are complying with these laws and to allow us to figure and collect the right amount of tax. You are not required to provide the information requested on a form that is subject to the Paperwork Reduction Act unless the form displays a valid OMB control number. Books or records relating to a form or its instructions must be retained as long as their contents may become material in the administration of any Internal Revenue law. Generally, tax returns and return information are confidential, as required by Code section 6103. The time needed to complete Forms 4070 and 4070A will vary depending on individual circumstances. The estimated average times are: Recordkeeping—Form 4070, 7 min.; Form 4070A, 3 hr. and 23 min.; Learning **about the law—each form, 2 min.; Preparing Form 4070,** 13 min.; Form 4070A, 55 min.; and Copying and **providing Form 4070, 10 min.; Form 4070A, 14 min.** If you have comments concerning the accuracy of these time estimates or suggestions for making these
forms simpler, we would be happy to hear from you. You can write to the Tax Forms Committee, Western Area Distribution Center, Rancho Cordova, CA 95743-0001. **Purpose.—**Use this form to report tips you receive to your employer. This includes cash tips, tips you receive from other employees, and credit card tips. You must report tips every month regardless of your total wages and tips for the year. However, you do not have to report tips to your employer for any month you received less than $20 in tips while working for that employer. Report tips by the 10th day of the month following the month that you receive them. If the 10th day is a Saturday, Sunday, or legal holiday, report tips by the next day that is not a Saturday, Sunday, or legal holiday. See **Pub. 531**, Reporting Tip Income, for more information. You can get additional copies of **Pub. 1244**, Employees Daily Record of Tips and Report to Employer, which contains both Forms 4070A and 4070, by calling 1-800-TAX-FORM (1-800-829-3676).
forms simpler, we would be happy to hear from you. You can write to the Tax Forms Committee, Western Area Distribution Center, Rancho Cordova, CA 95743-0001. **Purpose.—Use this form to report tips you receive to** your employer. This includes cash tips, tips you receive from other employees, and credit card tips. You must report tips every month regardless of your total wages and tips for the year. However, you do not have to report tips to your employer for any month you received less than $20 in tips while working for that employer. Report tips by the 10th day of the month following the month that you receive them. If the 10th day is a Saturday, Sunday, or legal holiday, report tips by the next day that is not a Saturday, Sunday, or legal holiday. See Pub. 531, Reporting Tip Income, for more information. You can get additional copies of Pub. 1244, Employees Daily Record of Tips and Report to Employer, which contains both Forms 4070A and 4070, by calling 1-800-TAX-FORM (1-800-829-3676).
**Instructions** *(continued)*
**Instructions (continued)**
**Unreported Tips.—**If you received tips of $20 or more for any month while working for one employer but did not report them to your employer, you must figure and pay social security and Medicare taxes on the unreported tips when you file your tax return. If you have unreported tips, you **must** use Form 1040 and **Form 4137,** Social Security and Medicare Tax on Unreported Tip Income, to report them. You may **not** use Form 1040A or 1040EZ. Employees subject to the Railroad Retirement Tax Act **cannot** use Form 4137 to pay railroad retirement tax on unreported tips. To get railroad retirement credit, you must report tips to your employer. If you do not report tips to your employer as required, you may be charged a penalty of 50% of the social security and Medicare taxes (or railroad retirement tax) due on the unreported tips unless there was reasonable cause for not reporting them. **Additional Information.—**Get **Pub. 531,** Reporting Tip Income, and Form 4137 for more information on tips. If you are an employee of certain large food or beverage establishments, see Pub. 531 for tip allocation rules. **Recordkeeping.—**If you do not keep a daily record of tips, you must keep other reliable proof of the tip income you received. This proof includes copies of restaurant bills and credit card charges that show amounts customers added as tips. Keep your tip income records for as long as the information on them may be needed in the administration of any Internal Revenue law.
**Unreported Tips.—If you received tips of $20 or** more for any month while working for one employer but did not report them to your employer, you must figure and pay social security and Medicare taxes on the unreported tips when you file your tax return. If you have unreported tips, you must use Form 1040 and Form 4137, Social Security and Medicare Tax on Unreported Tip Income, to report them. You may not use Form 1040A or 1040EZ. Employees subject to the Railroad Retirement Tax Act cannot use Form 4137 to pay railroad retirement tax on unreported tips. To get railroad retirement credit, you must report tips to your employer. If you do not report tips to your employer as required, you may be charged a penalty of 50% of the social security and Medicare taxes (or railroad retirement tax) due on the unreported tips unless there was reasonable cause for not reporting them. **Additional Information.—Get Pub. 531, Reporting** Tip Income, and Form 4137 for more information on tips. If you are an employee of certain large food or beverage establishments, see Pub. 531 for tip allocation rules. **Recordkeeping.—If you do not keep a daily** record of tips, you must keep other reliable proof of the tip income you received. This proof includes copies of restaurant bills and credit card charges that show amounts customers added as tips. Keep your tip income records for as long as the information on them may be needed in the administration of any Internal Revenue law.
### Instructions (continued)
Use this space to total your tips for the year
+3 -5
View File
@@ -6,7 +6,7 @@
8 4 Z E L L / L U R I E R E A L E S T A T E C E N T E R
**Table I:** Cap rate correlations **Cap Rate Correlation With:*** **BBB Corp** **10-Year Bond Yield S&P Dividend** **Treasury (10-15 yr) Yield** Multifamily 0.187 0.771 0.068 Industrial-0.221 0.748-0.307 CBD Office-0.449 0.694-0.458 Retail-0.181 0.649-02.58
**Table I: Cap rate correlations** **Cap Rate Correlation With:*** **BBB Corp** **10-Year Bond Yield S&P Dividend** **Treasury (10-15 yr) Yield** Multifamily 0.187 0.771 0.068 Industrial-0.221 0.748-0.307 CBD Office-0.449 0.694-0.458 Retail-0.181 0.649-02.58
* Based on 25 years of data for the 10-yrT & S&P DivYld; and 14 years for BBB.
**Figure 1:** NCREIF cap rates vs. 10-yearTreasury
@@ -20,9 +20,7 @@ R E V I E W 8 5
**Figure 2:** Capratespreadsover10-yearTreasury
**Basis Points** -200
-400
**Basis Points -200** -400
-600
@@ -34,7 +32,7 @@ R E V I E W 8 5
1982 1986 1990 1994 1998 2002 2006
**Table II:** Correlationsofspreadsbypropertytype **Correlation of Cap Rate Spreads Over Treasury** **Multifamily Industrial CBD Office**
**Table II: Correlationsofspreadsbypropertytype** **Correlation of Cap Rate Spreads Over Treasury** **Multifamily Industrial CBD Office**
||Multifamily|Industrial|CBD Office|
|---|---|---|---|
+52 -65
View File
@@ -1,8 +1,8 @@
(e) [Reserved]. For further guidance, see §1.1563-3T(e)(1). Par. 50. Section 1.1563-3T is added to read as follows:
<u>§1.1563-3T Rules for determining stock ownership (temporary)</u>.
§1.1563-3T Rules for determining stock ownership (temporary).
(a) through (d)(2)(iii) [Reserved]. For further guidance, see §1.1563-3(a)
through (d)(2)(iii). (iv) <u>Statement</u>. If the application of paragraph (d)(2)(ii) or (iii) of §1.1563-3 does not result in a corporation being treated as a component member of only one controlled group of corporations on a December 31, then such corporation will be treated as a component member of only one such group on such date. Such corporation may elect the group in which it is to be included by including on or with its income tax return a statement entitled, “STATEMENT TO ELECT CONTROLLED GROUP PURSUANT TO §1.1563-3T(d)(2)(iv).” The statement must include--
through (d)(2)(iii). (iv) Statement. If the application of paragraph (d)(2)(ii) or (iii) of §1.1563-3 does not result in a corporation being treated as a component member of only one controlled group of corporations on a December 31, then such corporation will be treated as a component member of only one such group on such date. Such corporation may elect the group in which it is to be included by including on or with its income tax return a statement entitled, “STATEMENT TO ELECT CONTROLLED GROUP PURSUANT TO §1.1563-3T(d)(2)(iv).” The statement must include--
(A) A description of each of the controlled groups in which the corporation
could be included. The description must include the name and employer identification number of each component member of each such group and the stock ownership of the component members of each such group; and
@@ -10,13 +10,11 @@ could be included. The description must include the name and employer identifica
(B) The following representation: [INSERT NAME AND EMPLOYER
IDENTIFICATION NUMBER OF CORPORATION] ELECTS TO BE TREATED AS A COMPONENT MEMBER OF THE [INSERT DESIGNATION OF GROUP].
(v) <u>Election</u>-- (A) <u>Election filed</u>. An election filed under paragraph (d)(2)(iv) of
(v) Election-- (A) Election filed. An election filed under paragraph (d)(2)(iv) of
this section is irrevocable and effective until paragraph (d)(2)(ii) or (iii) of §1.1563-3 applies or until a change in the stock ownership of the corporation results in
|termination of membership in the controlled group in which such corporation has||
|termination of membership in the controlled group in which such corporation has been included. (B) Election not filed.|In the event no election is filed in accordance with the|
|---|---|
|been included.||
|(B) Election not filed.|In the event no election is filed in accordance with the|
|provisions of paragraph (d)(2)(iv) of this section, then the Internal Revenue Service||
|will determine the group in which such corporation is to be included. Such||
|determination will be binding for all subsequent years unless the corporation files a||
@@ -30,47 +28,42 @@ Federal income tax return (including any amended return filed on or before the d
2006.
(2) Expiration date. The applicability of this section will expire on May 26,
2009. Par. 51. Section 1.6012-2 is amended by revising paragraph (c) and adding paragraph (k) to read as follows: <u>§1.6012-2 Corporations required to make returns of income</u>.
2009. Par. 51. Section 1.6012-2 is amended by revising paragraph (c) and adding paragraph (k) to read as follows: §1.6012-2 Corporations required to make returns of income.
* * * * *
(c) [Reserved]. For further guidance, see §1.6012-2T(c).
* * * * *
(k) [Reserved]. For further guidance, see §1.6012-2T(k)(1).
Par. 52. Section 1.6012-2T is added to read as follows: <u>§1.6012-2T Corporations required to make returns of income (temporary)</u>.
Par. 52. Section 1.6012-2T is added to read as follows: §1.6012-2T Corporations required to make returns of income (temporary).
(a) through (b) [Reserved]. For further guidance, see §1.6012-2(a) through
(b).
(c) Insurance companies-- (1) Domestic life insurance companies-- (i) In
<u>general</u>. A life insurance company subject to tax under section 801 shall make a return on Form 1120L. Except as provided in paragraph (c)(4) of this section, such company shall file with its return--
general. A life insurance company subject to tax under section 801 shall make a return on Form 1120L. Except as provided in paragraph (c)(4) of this section, such company shall file with its return--
(A) A copy of its annual statement which shows the reserves used by the
company in computing the taxable income reported on its return; and
(B) A copy of Schedule A (real estate) and of Schedule D (bonds and stocks),
or any successor thereto, of such annual statement. (ii) <u>Mutual savings banks</u>. Mutual savings banks conducting life insurance business and meeting the requirements of section 594 are subject to partial tax computed on Form 1120 and partial tax computed on Form 1120L. The Form 1120L is attached as a schedule to Form 1120, together with the annual statement and schedules required to be filed with Form 1120L.
or any successor thereto, of such annual statement. (ii) Mutual savings banks. Mutual savings banks conducting life insurance business and meeting the requirements of section 594 are subject to partial tax computed on Form 1120 and partial tax computed on Form 1120L. The Form 1120L is attached as a schedule to Form 1120, together with the annual statement and schedules required to be filed with Form 1120L.
(2) <u>Domestic nonlife insurance companies</u>. Every domestic insurance
(2) Domestic nonlife insurance companies. Every domestic insurance
company other than a life insurance company shall make a return on Form 1120PC. This includes organizations described in section 501(m)(1) that provide commercial- type insurance and organizations described in section 833. Except as provided in paragraph (c)(4) of this section, such company shall file with its return a copy of its
annual statement (or a pro forma annual statement), including the underwriting and investment exhibit for the year covered by such return.
(3) <u>Foreign insurance companies</u>. The provisions of paragraphs (c)(1) and
(c)(2) of this section concerning the returns and statements of insurance companies subject to tax under section 801 or section 831 also apply to foreign insurance companies subject to tax under those sections, except that the copy of the annual statement required to be submitted with the return shall, in the case of a foreign insurance company that is not required to file an annual statement, be a copy of the pro forma annual statement relating to the United States business of such company.
(4) <u>Exception for insurance companies filing their Federal income tax returns</u>
<u>electronically</u>. If an insurance company described in paragraph (c)(1), (c)(2), or
(c)(3) of this section files its Federal income tax return electronically, it should not include on or with such return its annual statement (or pro forma annual statement), or any portion thereof. Such statement must be available at all times for inspection by authorized Internal Revenue Service officers or employees and retained for so long as such statements may be material in the administration of any internal revenue law. See §1.6001-1(e).
(5) <u>Definition</u>. For purposes of this section, the term <u>annual statement</u> means
the annual statement, the form of which is approved by the National Association of Insurance Commissioners (NAIC), which is filed by an insurance company for the year with the insurance departments of States, Territories, and the District of
||(3) Foreign insurance companies. The provisions of paragraphs (c)(1) and|
|---|---|
||(c)(2) of this section concerning the returns and statements of insurance companies subject to tax under section 801 or section 831 also apply to foreign insurance companies subject to tax under those sections, except that the copy of the annual statement required to be submitted with the return shall, in the case of a foreign insurance company that is not required to file an annual statement, be a copy of the pro forma annual statement relating to the United States business of such company. (4) Exception for insurance companies filing their Federal income tax returns electronically. If an insurance company described in paragraph (c)(1), (c)(2), or (c)(3) of this section files its Federal income tax return electronically, it should not include on or with such return its annual statement (or pro forma annual statement), or any portion thereof. Such statement must be available at all times for inspection by authorized Internal Revenue Service officers or employees and retained for so long as such statements may be material in the administration of any internal revenue law. See §1.6001-1(e). (5) Definition. For purposes of this section, the term annual statement means the annual statement, the form of which is approved by the National Association of Insurance Commissioners (NAIC), which is filed by an insurance company for the year with the insurance departments of States, Territories, and the District of|
Columbia. The term annual statement also includes a pro forma annual statement if the insurance company is not required to file the NAIC annual statement.
(d) through (j) [Reserved]. For further guidance, see §1.6012-2(d) through (j).
(k) <u>Effective date</u>-- (1) <u>Applicability date</u>. This section applies to any original
(k) Effective date-- (1) Applicability date. This section applies to any original
Federal income tax return (including any amended return filed on or before the due date (including extensions) of such original return) timely filed on or after May 30,
2006.
(2) <u>Expiration date</u>. The applicability of this section will expire on May 26,
(2) Expiration date. The applicability of this section will expire on May 26,
2009.
|||Par. 53. For each entry in the “Location” column of the following table,|
@@ -108,32 +101,45 @@ section and paragraph
(c)(4), and (c)(5) of this section, and paragraph
(c)(2) of §1.382-8T
||§1.382-8(g), Example|
|---|---|
||The first sentence of §1.382-8(g), Example §1.382-8(g), Example §1.382-8(g), Example|
(2)(c)
(2)(e)
(3)(b)
(3)(c)(1)(B)
||The second sentence of|
|---|---|
||§1.382-8(g), Example The second sentence of §1.382-8(g), Example The first sentence of §1.1502-32(b)(4)(v)(A) The first sentence of §1.1502-32(b)(4)(v)(B)|
(4)(c)
(5)(c)
|The fifth sentence of|paragraph (c) of this|paragraphs (c)(1), (c)(3),|
|---|---|---|
|§1.382-8(f)|section|(c)(4), and (c)(5) of this section, and paragraph (c)(2) of §1.382-8T|
|§1.382-8(g), Example|paragraph (c) of this|paragraphs (c)(1), (c)(3), section, and paragraph (c)(2) of §1.382-8T|
|The second sentence of|paragraph (c) of this|paragraphs (c)(1), (c)(3),|
|§1.382-8(g), Example|section paragraph (c)(2) of this section paragraph (c)(2) of this section paragraph (c)(2) of this section paragraphs (c)(1) and (2) of this section paragraph (c)(2) of this section paragraph (c)(2) of this section paragraph (b)(4)(iv) of this section paragraph (b)(4)(iv) of this section|(c)(4), and (c)(5) of this (c)(2) of §1.382-8T paragraph (c)(2) of §1.382-8T paragraph (c)(2) of §1.382-8T paragraph (c)(2) of §1.382-8T paragraph (c)(1) of this section and paragraph (c)(2) of §1.382-8T paragraph (c)(2) of §1.382-8T paragraph (c)(2) of §1.382-8T paragraph (b)(4)(iv) of §1.1502-32T paragraph (b)(4)(iv) of §1.1502-32T|
|§1.382-8(g), Example|section|(c)(4), and (c)(5) of this|
(1)(b)(2) section (c)(4), and (c)(5) of this
(1)(c) section, and paragraph
(c)(2) of §1.382-8T
|§1.382-8(g), Example|paragraph (c)(2) of this|paragraph (c)(2) of|
|---|---|---|
|The first sentence of|paragraph (c)(2) of this|paragraph (c)(2) of|
|§1.382-8(g), Example|section|§1.382-8T|
|§1.382-8(g), Example|paragraph (c)(2) of this|paragraph (c)(2) of|
|§1.382-8(g), Example|paragraphs (c)(1) and (2)|paragraph (c)(1) of this|
(2)(c) section §1.382-8T
(2)(e)
(3)(b) section §1.382-8T
(3)(c)(1)(B) of this section section and paragraph
(c)(2) of §1.382-8T
|The second sentence of|paragraph (c)(2) of this|paragraph (c)(2) of|
|---|---|---|
|§1.382-8(g), Example|section|§1.382-8T|
|The second sentence of|paragraph (c)(2) of this|paragraph (c)(2) of|
|§1.382-8(g), Example|section|§1.382-8T|
|The first sentence of|paragraph (b)(4)(iv) of|paragraph (b)(4)(iv) of|
|§1.1502-32(b)(4)(v)(A)|this section|§1.1502-32T|
|The first sentence of|paragraph (b)(4)(iv) of|paragraph (b)(4)(iv) of|
|§1.1502-32(b)(4)(v)(B)|this section|§1.1502-32T|
(4)(c)
(5)(c)
|§1.1502-35(c)(4)(ii)(B)|§1.1502-76(b)(2)(ii)(D)|§1.1502-76T(b)(2)(ii)(D)|
|---|---|---|
|§1.1502-76(b)(2)(ii)(A)(2)|paragraph (b)(2)(ii)(D) of this section|paragraph (b)(2)(ii)(D) of §1.1502-76T|
@@ -162,33 +168,14 @@ section and paragraph
|§1.6043-2(a)|or 1.1081-11|3T(a), or §1.1081-11T|
|The first sentence of §301.6011-5T(a) (twice)|§1.6012-2|paragraphs (a), (b) and (d) through (j) of §1.6012- 2, and paragraph (c) of §1.6012-2T|
PART 602--OMB CONTROL NUMBERS UNDER THE PAPERWORK REDUCTION ACT Par. 54. The authority citation for part 602 continues to read as follows: Authority: 26 U.S.C. 7805. Par. 55. In §602.101, paragraph (b) is amended to read as follows:
1. The following entries to the table are removed:
<u>§602.101 OMB Control numbers</u>.
* * * * *
(b) * * *
CFR part or section where Current OMB identified or described control No.
* * * * *
1.332-6…………………………………………………………………. 1545-2019
1.382-11……………………………………………………………….. 1545-2019
1.351-3…………………………………………………………………. 1545-2019
1.355-5…………………………………………………………………. 1545-2019
1.368-3…………………………………………………………………. 1545-2019
1.1081-11………………………………………………………………. 1545-2019
* * * * * **______________________________________________________________**
2. The following entries are added in numerical order to the table:
<u>§602.101 OMB Control numbers</u>.
* * * * *
(b) * * *
CFR part or section where Current OMB identified or described control No.
* * * * *
1.302-2T………………………………………………………………… 1545-2019
1.302-4T………………………………………………………………… 1545-2019
|||PART 602--OMB CONTROL NUMBERS UNDER THE PAPERWORK||
|---|---|---|---|
||REDUCTION ACT Authority: 26 U.S.C. 7805. 1. The following entries to the table are removed: §602.101 OMB Control numbers.|Par. 54. The authority citation for part 602 continues to read as follows: Par. 55. In §602.101, paragraph (b) is amended to read as follows:||
|* * * * *|(b) * * * CFR part or section where identified or described||Current OMB control No.|
|* * * * *|1.332-6………………………………………………………………….|1.382-11……………………………………………………………….. 1545-2019 1.351-3…………………………………………………………………. 1545-2019 1.355-5…………………………………………………………………. 1545-2019 1.368-3…………………………………………………………………. 1545-2019 1.1081-11………………………………………………………………. 1545-2019|1545-2019|
|* * * * *|§602.101 OMB Control numbers.|______________________________________________________________ 2. The following entries are added in numerical order to the table:||
|* * * * *|(b) * * * CFR part or section where identified or described||Current OMB control No.|
|* * * * *|1.302-2T………………………………………………………………… 1545 1.302-4T………………………………………………………………… 1545||-2019 -2019|
|1.331-1T………………………………………………………………… 1545|-2019|
|---|---|
+2 -2
View File
@@ -26,7 +26,7 @@ A.P., NIST Standard Reference in cubic meters per kilogram Database 23, NIST the
##### Physical Properties
|Chemical Formula|CCl₂F₂|
|Chemical Formula|CCl2F2|
|---|---|
|Molecular mass|120.91|
|Boiling Point At one atmosphere|-29.75°C|
@@ -45,7 +45,7 @@ l
|Temp|Pressure||Volume|||Density||Enthalpy|||Entropy|Temp|
|---|---|---|---|---|---|---|---|---|---|---|---|---|
|°C|[kPa]|[m³ Liquid v f|/kg]|Vapour v g|Liquid d f|[kg/m³] Vapour d g|Liquid H f|[kJ/kg] Latent H fg|Vapour H g|Liquid S f|[kJ/K-kg] Vapour S g|°C|
|°C|[kPa]|[m3 Liquid v f|/kg]|Vapour v g|Liquid d f|[kg/m3] Vapour d g|Liquid H f|[kJ/kg] Latent H fg|Vapour H g|Liquid S f|[kJ/K-kg] Vapour S g|°C|
|-100|1.2|0.0006|10.0000|1679.0|0.100|113.3|192.8|306.1|0.6077|1.7210|-100|
|---|---|---|---|---|---|---|---|---|---|---|---|
-72
View File
@@ -270,78 +270,6 @@ class TestExtractTextInRegions:
)
# ---------------------------------------------------------------------------
# extract_pages_markdown / extract_pages_markdown_bytes
# ---------------------------------------------------------------------------
class TestExtractPagesMarkdown:
def test_default_returns_all_pages(self):
result = pdf_inspector.extract_pages_markdown(
fixture_path("thermo-freon12.pdf")
)
assert len(result.pages) == 3
assert [p.page for p in result.pages] == [0, 1, 2]
assert all(isinstance(p.markdown, str) for p in result.pages)
def test_bytes_default_returns_all_pages(self):
data = fixture_bytes("thermo-freon12.pdf")
result = pdf_inspector.extract_pages_markdown_bytes(data)
assert len(result.pages) == 3
def test_selected_pages_preserve_order(self):
result = pdf_inspector.extract_pages_markdown(
fixture_path("thermo-freon12.pdf"), pages=[2, 0]
)
assert [p.page for p in result.pages] == [2, 0]
def test_bytes_selected_pages_preserve_order(self):
data = fixture_bytes("thermo-freon12.pdf")
result = pdf_inspector.extract_pages_markdown_bytes(data, pages=[1])
assert len(result.pages) == 1
assert result.pages[0].page == 1
def test_page_fields(self):
result = pdf_inspector.extract_pages_markdown(
fixture_path("thermo-freon12.pdf"), pages=[0]
)
page = result.pages[0]
assert isinstance(page.page, int)
assert isinstance(page.markdown, str)
assert isinstance(page.needs_ocr, bool)
assert not page.needs_ocr # text-based fixture
assert len(page.markdown) > 0
def test_result_fields(self):
result = pdf_inspector.extract_pages_markdown(
fixture_path("thermo-freon12.pdf")
)
assert isinstance(result.pages, list)
assert isinstance(result.pages_with_tables, list)
assert isinstance(result.pages_with_columns, list)
assert isinstance(result.pages_needing_ocr, list)
assert isinstance(result.is_complex, bool)
def test_out_of_range_page_marks_needs_ocr(self):
result = pdf_inspector.extract_pages_markdown(
fixture_path("thermo-freon12.pdf"), pages=[9999]
)
assert len(result.pages) == 1
assert result.pages[0].needs_ocr
assert result.pages[0].markdown == ""
def test_repr(self):
result = pdf_inspector.extract_pages_markdown(
fixture_path("thermo-freon12.pdf"), pages=[0]
)
assert "PagesExtractionResult" in repr(result)
assert "PageMarkdown" in repr(result.pages[0])
def test_not_a_pdf(self):
with pytest.raises(ValueError):
pdf_inspector.extract_pages_markdown_bytes(b"not a pdf")
# ---------------------------------------------------------------------------
# Error handling
# ---------------------------------------------------------------------------