Compare commits
61
Commits
@@ -49,13 +49,23 @@ jobs:
|
||||
working-directory: napi
|
||||
run: bunx napi build --platform --release
|
||||
|
||||
- name: Upload artifact
|
||||
- name: Upload native binary
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: bindings-${{ matrix.target }}
|
||||
path: napi/*.node
|
||||
if-no-files-found: error
|
||||
|
||||
- name: Upload generated JS bindings
|
||||
if: matrix.target == 'x86_64-unknown-linux-gnu'
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: js-bindings
|
||||
path: |
|
||||
napi/index.js
|
||||
napi/index.d.ts
|
||||
if-no-files-found: error
|
||||
|
||||
publish:
|
||||
name: Publish to npm
|
||||
needs: build
|
||||
@@ -79,8 +89,9 @@ jobs:
|
||||
- name: Collect binaries and publish
|
||||
working-directory: napi
|
||||
run: |
|
||||
# Copy all .node binaries into the package directory
|
||||
cp artifacts/bindings-*/*.node .
|
||||
cp artifacts/js-bindings/index.js .
|
||||
cp artifacts/js-bindings/index.d.ts .
|
||||
|
||||
echo "=== Package contents ==="
|
||||
ls -la *.node index.js index.d.ts
|
||||
|
||||
@@ -24,6 +24,10 @@ Thumbs.db
|
||||
# Build cache
|
||||
**/*.rs.bk
|
||||
|
||||
# NAPI generated (regenerated by `napi prepublish` during CI)
|
||||
napi/index.js
|
||||
napi/index.d.ts
|
||||
|
||||
# Local samples and scripts
|
||||
samples/
|
||||
scripts/
|
||||
|
||||
+9
-3
@@ -8,9 +8,16 @@ description = "Fast PDF inspection, classification, and text extraction with sma
|
||||
license = "MIT"
|
||||
repository = "https://github.com/firecrawl/pdf-inspector"
|
||||
|
||||
[lib]
|
||||
name = "pdf_inspector"
|
||||
crate-type = ["lib", "cdylib"]
|
||||
|
||||
[dependencies]
|
||||
# Python bindings
|
||||
pyo3 = { version = "0.25", features = ["extension-module"], optional = true }
|
||||
|
||||
# PDF parsing
|
||||
lopdf = { git = "https://github.com/J-F-Liu/lopdf", rev = "052674053814a9f4897af94f0b8e46a545c9b329", features = ["rayon"] }
|
||||
lopdf = { git = "https://github.com/J-F-Liu/lopdf", rev = "7a05512d831415b1f2b1ce522391d6beab8a1284", features = ["rayon"] }
|
||||
|
||||
# Error handling
|
||||
thiserror = "2.0"
|
||||
@@ -35,6 +42,7 @@ tempfile = "3.3"
|
||||
|
||||
[features]
|
||||
default = []
|
||||
python = ["pyo3"]
|
||||
|
||||
[[bin]]
|
||||
name = "pdf2md"
|
||||
@@ -47,5 +55,3 @@ path = "src/bin/detect_pdf.rs"
|
||||
[[bin]]
|
||||
name = "dump_ops"
|
||||
path = "src/bin/dump_ops.rs"
|
||||
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# pdf-inspector
|
||||
|
||||
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR.
|
||||
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Includes bindings for [Python](docs/python.md) and [Node.js](napi/README.md).
|
||||
|
||||
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
|
||||
|
||||
@@ -12,90 +12,81 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
|
||||
- **Table detection** — Dual-mode: rectangle-based detection from PDF drawing ops, plus heuristic detection from text alignment. Handles financial tables, footnotes, and continuation tables across pages.
|
||||
- **CID font support** — ToUnicode CMap decoding for Type0/Identity-H fonts, UTF-16BE, UTF-8, and Latin-1 encodings.
|
||||
- **Multi-column layout** — Automatic detection of newspaper-style columns, sequential reading order, and RTL text support.
|
||||
- **Encoding issue detection** — Automatically flags broken font encodings (garbled text, replacement characters) so callers can fall back to OCR.
|
||||
- **Encoding issue detection** — Automatically flags broken font encodings so callers can fall back to OCR.
|
||||
- **Single document load** — The document is parsed once and shared between detection and extraction, avoiding redundant I/O.
|
||||
- **Lightweight** — Pure Rust, no ML models, no external services. Single dependency on `lopdf` for PDF parsing.
|
||||
|
||||
## Benchmark
|
||||
|
||||
Evaluated on the [opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs). Only direct text extraction engines are shown — no OCR, no ML models. Scores are 0-1, higher is better.
|
||||
|
||||
| Engine | Overall | Reading Order (NID) | Tables (TEDS) | Headings (MHS) | Speed (200 docs) |
|
||||
|---|---|---|---|---|---|
|
||||
| pdf-inspector | 0.78 | 0.87 | 0.59 | 0.57 | 4s |
|
||||
| opendataloader | 0.84 | 0.91 | 0.49 | 0.74 | 11s |
|
||||
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
|
||||
| markitdown | 0.58 | 0.88 | 0.00 | 0.00 | 8s |
|
||||
|
||||
For context, engines that use OCR/ML (docling, marker, mineru) score 0.83-0.88 overall but take 2-180 minutes on the same corpus.
|
||||
|
||||
**Where we do well:** Speed (fastest of all engines), reading order, table detection vs other direct-text tools.
|
||||
|
||||
**Where we lag:** Heading detection trails opendataloader — many PDFs use bold text at body font size for headings, or headings that are only slightly larger than body text. Table detection trails OCR-based engines that can see visual table structure.
|
||||
|
||||
## Quick start
|
||||
|
||||
### As a library
|
||||
### Python
|
||||
|
||||
Add to your `Cargo.toml`:
|
||||
```bash
|
||||
pip install maturin
|
||||
maturin develop --release
|
||||
```
|
||||
|
||||
```python
|
||||
import pdf_inspector
|
||||
|
||||
result = pdf_inspector.process_pdf("document.pdf")
|
||||
print(result.pdf_type) # "text_based", "scanned", "image_based", "mixed"
|
||||
print(result.markdown) # Markdown string or None
|
||||
```
|
||||
|
||||
> Full API reference: [docs/python.md](docs/python.md)
|
||||
|
||||
### Node.js
|
||||
|
||||
```bash
|
||||
npm install @firecrawl/pdf-inspector-js
|
||||
```
|
||||
|
||||
```javascript
|
||||
import { readFileSync } from 'fs';
|
||||
import { processPdf, classifyPdf } from '@firecrawl/pdf-inspector-js';
|
||||
|
||||
const result = processPdf(readFileSync('document.pdf'));
|
||||
console.log(result.pdfType); // "TextBased", "Scanned", "ImageBased", "Mixed"
|
||||
console.log(result.markdown); // Markdown string or null
|
||||
```
|
||||
|
||||
> Full API reference: [napi/README.md](napi/README.md)
|
||||
|
||||
### Rust
|
||||
|
||||
```toml
|
||||
[dependencies]
|
||||
pdf-inspector = { git = "https://github.com/firecrawl/pdf-inspector" }
|
||||
```
|
||||
|
||||
Detect and extract in one call:
|
||||
|
||||
```rust
|
||||
use pdf_inspector::process_pdf;
|
||||
|
||||
let result = process_pdf("document.pdf")?;
|
||||
|
||||
println!("Type: {:?}", result.pdf_type); // TextBased, Scanned, ImageBased, Mixed
|
||||
println!("Confidence: {:.0}%", result.confidence * 100.0);
|
||||
println!("Pages: {}", result.page_count);
|
||||
|
||||
println!("Type: {:?}", result.pdf_type);
|
||||
if let Some(markdown) = &result.markdown {
|
||||
println!("{}", markdown);
|
||||
}
|
||||
```
|
||||
|
||||
Fast metadata-only detection (no text extraction or markdown generation):
|
||||
|
||||
```rust
|
||||
use pdf_inspector::detect_pdf;
|
||||
|
||||
let info = detect_pdf("document.pdf")?;
|
||||
|
||||
match info.pdf_type {
|
||||
pdf_inspector::PdfType::TextBased => {
|
||||
// Extract locally — fast and free
|
||||
}
|
||||
_ => {
|
||||
// Route to OCR service
|
||||
// info.pages_needing_ocr tells you exactly which pages
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
Customize processing with `PdfOptions`:
|
||||
|
||||
```rust
|
||||
use pdf_inspector::{process_pdf_with_options, PdfOptions, ProcessMode, DetectionConfig, ScanStrategy};
|
||||
|
||||
// Analyze layout without generating markdown
|
||||
let result = process_pdf_with_options(
|
||||
"document.pdf",
|
||||
PdfOptions::new().mode(ProcessMode::Analyze),
|
||||
)?;
|
||||
|
||||
// Full extraction with custom detection strategy
|
||||
let result = process_pdf_with_options(
|
||||
"large.pdf",
|
||||
PdfOptions::new().detection(DetectionConfig {
|
||||
strategy: ScanStrategy::Sample(5),
|
||||
..Default::default()
|
||||
}),
|
||||
)?;
|
||||
|
||||
// Process only specific pages
|
||||
let result = process_pdf_with_options(
|
||||
"document.pdf",
|
||||
PdfOptions::new().pages([1, 3, 5]),
|
||||
)?;
|
||||
```
|
||||
|
||||
Process from a byte buffer (no filesystem needed):
|
||||
|
||||
```rust
|
||||
use pdf_inspector::process_pdf_mem;
|
||||
|
||||
let bytes = std::fs::read("document.pdf")?;
|
||||
let result = process_pdf_mem(&bytes)?;
|
||||
```
|
||||
> Full API reference: [docs/rust-api.md](docs/rust-api.md)
|
||||
|
||||
### CLI
|
||||
|
||||
@@ -120,7 +111,6 @@ cargo run --bin detect-pdf -- document.pdf
|
||||
cargo run --bin detect-pdf -- document.pdf --json
|
||||
|
||||
# Detection + layout analysis (tables, columns)
|
||||
cargo run --bin detect-pdf -- document.pdf --analyze
|
||||
cargo run --bin detect-pdf -- document.pdf --analyze --json
|
||||
```
|
||||
|
||||
@@ -159,6 +149,7 @@ The document is loaded **once** via `load_document_from_path` / `load_document_f
|
||||
```
|
||||
src/
|
||||
lib.rs — Public API, PdfOptions builder, convenience functions
|
||||
python.rs — PyO3 Python bindings
|
||||
types.rs — Shared types: TextItem, TextLine, PdfRect, ItemType
|
||||
text_utils.rs — Character/text helpers (CJK, RTL, ligatures, bold/italic)
|
||||
process_mode.rs — ProcessMode enum (DetectOnly, Analyze, Full)
|
||||
@@ -169,6 +160,7 @@ src/
|
||||
tables/ — Table detection and formatting
|
||||
markdown/ — Markdown conversion and structure detection
|
||||
bin/ — CLI tools (pdf2md, detect_pdf)
|
||||
napi/ — Node.js/Bun bindings (napi-rs)
|
||||
```
|
||||
|
||||
## How classification works
|
||||
@@ -189,50 +181,6 @@ This detects 300+ page PDFs in milliseconds. The result includes `pages_needing_
|
||||
| `Sample(n)` | Sample `n` evenly distributed pages (first, last, middle) | Very large PDFs where speed matters more than precision |
|
||||
| `Pages(vec)` | Only scan specific 1-indexed page numbers | When the caller knows which pages to check |
|
||||
|
||||
## API
|
||||
|
||||
### Processing modes
|
||||
|
||||
| Mode | What it does | Returns |
|
||||
|---|---|---|
|
||||
| `ProcessMode::Full` (default) | Detect + extract + convert to Markdown | Everything populated |
|
||||
| `ProcessMode::Analyze` | Detect + extract + layout analysis (no Markdown) | `markdown` is `None`, `layout` is populated |
|
||||
| `ProcessMode::DetectOnly` | Classification only (fastest) | `markdown` is `None`, `layout` is default |
|
||||
|
||||
### Functions
|
||||
|
||||
| Function | Description |
|
||||
|---|---|
|
||||
| `process_pdf(path)` | Full processing with defaults |
|
||||
| `detect_pdf(path)` | Fast metadata-only detection (no extraction) |
|
||||
| `process_pdf_with_options(path, options)` | Process with custom `PdfOptions` |
|
||||
| `process_pdf_mem(bytes)` | Full processing from a byte buffer |
|
||||
| `detect_pdf_mem(bytes)` | Fast detection from a byte buffer |
|
||||
| `process_pdf_mem_with_options(bytes, options)` | Process from bytes with custom options |
|
||||
| `extract_text(path)` | Plain text extraction |
|
||||
| `extract_text_with_positions(path)` | Text with X/Y coordinates and font info |
|
||||
| `to_markdown(text, options)` | Convert plain text to Markdown |
|
||||
| `to_markdown_from_items(items, options)` | Markdown from pre-extracted `TextItem`s |
|
||||
| `to_markdown_from_items_with_rects(items, options, rects)` | Markdown with rectangle-based table detection |
|
||||
|
||||
Low-level detection functions are also available via the `detector` module (`detect_pdf_type`, `detect_pdf_type_with_config`, etc.) for callers who need `PdfTypeResult` instead of `PdfProcessResult`.
|
||||
|
||||
### Types
|
||||
|
||||
| Type | Description |
|
||||
|---|---|
|
||||
| `PdfOptions` | Builder for processing configuration (mode, detection, markdown, page filter) |
|
||||
| `ProcessMode` | `DetectOnly`, `Analyze`, `Full` |
|
||||
| `PdfType` | `TextBased`, `Scanned`, `ImageBased`, `Mixed` |
|
||||
| `PdfProcessResult` | Full result: pdf_type, markdown, page_count, confidence, layout, has_encoding_issues, timing |
|
||||
| `PdfTypeResult` | Low-level detection result: type, confidence, page count, pages needing OCR |
|
||||
| `DetectionConfig` | Configuration for detection: scan strategy, thresholds |
|
||||
| `ScanStrategy` | `EarlyExit`, `Full`, `Sample(n)`, `Pages(vec)` |
|
||||
| `LayoutComplexity` | Layout analysis: is_complex, pages_with_tables, pages_with_columns |
|
||||
| `TextItem` | Text with position, font info, and page number |
|
||||
| `MarkdownOptions` | Configuration for Markdown formatting (page numbers, etc.) |
|
||||
| `PdfError` | `Io`, `Parse`, `Encrypted`, `InvalidStructure`, `NotAPdf` |
|
||||
|
||||
## Markdown output
|
||||
|
||||
The converter handles:
|
||||
@@ -255,36 +203,6 @@ The converter handles:
|
||||
| Drop caps | Large initial letters merged with following text |
|
||||
| Dot leaders | TOC-style dots collapsed to " ... " |
|
||||
|
||||
## Debugging with RUST_LOG
|
||||
|
||||
Structured logging via `RUST_LOG` replaces the former debug binaries. Set the environment variable to control which sections emit debug output on stderr:
|
||||
|
||||
```bash
|
||||
# Raw PDF content stream operators (replaces dump_ops)
|
||||
RUST_LOG=pdf_inspector::extractor::content_stream=trace cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Font metadata, encodings, ligatures (replaces debug_fonts / debug_ligatures)
|
||||
RUST_LOG=pdf_inspector::extractor::fonts=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# ToUnicode CMap parsing
|
||||
RUST_LOG=pdf_inspector::tounicode=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Text items per page with x/y/width (replaces debug_spaces / debug_pages)
|
||||
RUST_LOG=pdf_inspector::extractor=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Column detection and reading order (replaces debug_order)
|
||||
RUST_LOG=pdf_inspector::extractor::layout=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Y-gap analysis and paragraph thresholds (replaces debug_ygaps)
|
||||
RUST_LOG=pdf_inspector::markdown::analysis=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Table detection
|
||||
RUST_LOG=pdf_inspector::tables=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Everything
|
||||
RUST_LOG=pdf_inspector=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
```
|
||||
|
||||
## Use case: smart PDF routing
|
||||
|
||||
pdf-inspector was built for pipelines that process PDFs at scale. Instead of sending every PDF through OCR:
|
||||
@@ -299,6 +217,10 @@ PDF arrives
|
||||
|
||||
This saves cost and latency for the majority of PDFs that are already text-based (reports, papers, invoices, legal docs).
|
||||
|
||||
## Debugging
|
||||
|
||||
See [docs/debugging.md](docs/debugging.md) for `RUST_LOG` environment variable usage.
|
||||
|
||||
## License
|
||||
|
||||
MIT
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
# Debugging with RUST_LOG
|
||||
|
||||
Structured logging via `RUST_LOG` replaces the former debug binaries. Set the environment variable to control which sections emit debug output on stderr:
|
||||
|
||||
```bash
|
||||
# Raw PDF content stream operators (replaces dump_ops)
|
||||
RUST_LOG=pdf_inspector::extractor::content_stream=trace cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Font metadata, encodings, ligatures (replaces debug_fonts / debug_ligatures)
|
||||
RUST_LOG=pdf_inspector::extractor::fonts=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# ToUnicode CMap parsing
|
||||
RUST_LOG=pdf_inspector::tounicode=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Text items per page with x/y/width (replaces debug_spaces / debug_pages)
|
||||
RUST_LOG=pdf_inspector::extractor=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Column detection and reading order (replaces debug_order)
|
||||
RUST_LOG=pdf_inspector::extractor::layout=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Y-gap analysis and paragraph thresholds (replaces debug_ygaps)
|
||||
RUST_LOG=pdf_inspector::markdown::analysis=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Table detection
|
||||
RUST_LOG=pdf_inspector::tables=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Everything
|
||||
RUST_LOG=pdf_inspector=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
```
|
||||
@@ -0,0 +1,74 @@
|
||||
# Python API
|
||||
|
||||
Python bindings via [PyO3](https://pyo3.rs). Requires Rust toolchain for building from source.
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
pip install maturin
|
||||
maturin develop --release
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
```python
|
||||
import pdf_inspector
|
||||
|
||||
# Full processing: detect + extract + convert to Markdown
|
||||
result = pdf_inspector.process_pdf("document.pdf")
|
||||
print(result.pdf_type) # "text_based", "scanned", "image_based", "mixed"
|
||||
print(result.confidence) # 0.0 - 1.0
|
||||
print(result.page_count) # number of pages
|
||||
print(result.markdown) # Markdown string or None
|
||||
|
||||
# Process specific pages only
|
||||
result = pdf_inspector.process_pdf("document.pdf", pages=[1, 3, 5])
|
||||
|
||||
# Process from bytes (no filesystem needed)
|
||||
with open("document.pdf", "rb") as f:
|
||||
result = pdf_inspector.process_pdf_bytes(f.read())
|
||||
|
||||
# Fast detection only (no text extraction)
|
||||
result = pdf_inspector.detect_pdf("document.pdf")
|
||||
if result.pdf_type == "text_based":
|
||||
print("Can extract locally!")
|
||||
else:
|
||||
print(f"Pages needing OCR: {result.pages_needing_ocr}")
|
||||
|
||||
# Plain text extraction
|
||||
text = pdf_inspector.extract_text("document.pdf")
|
||||
|
||||
# Positioned text items with font info
|
||||
items = pdf_inspector.extract_text_with_positions("document.pdf")
|
||||
for item in items[:5]:
|
||||
print(f"'{item.text}' at ({item.x:.0f}, {item.y:.0f}) size={item.font_size}")
|
||||
```
|
||||
|
||||
## API reference
|
||||
|
||||
| Function | Description |
|
||||
|---|---|
|
||||
| `process_pdf(path, pages=None)` | Full processing (detect + extract + markdown) |
|
||||
| `process_pdf_bytes(data, pages=None)` | Full processing from bytes |
|
||||
| `detect_pdf(path)` | Fast detection only (returns PdfResult) |
|
||||
| `detect_pdf_bytes(data)` | Fast detection from bytes |
|
||||
| `classify_pdf(path)` | Lightweight classification (returns PdfClassification) |
|
||||
| `classify_pdf_bytes(data)` | Lightweight classification from bytes |
|
||||
| `extract_text(path)` | Plain text extraction |
|
||||
| `extract_text_bytes(data)` | Plain text extraction from bytes |
|
||||
| `extract_text_with_positions(path, pages=None)` | Text with X/Y coords and font info |
|
||||
| `extract_text_with_positions_bytes(data, pages=None)` | Text with positions from bytes |
|
||||
| `extract_text_in_regions(path, page_regions)` | Extract text in bounding-box regions |
|
||||
| `extract_text_in_regions_bytes(data, page_regions)` | Region extraction from bytes |
|
||||
|
||||
## Types
|
||||
|
||||
**`PdfResult` fields:** `pdf_type`, `markdown`, `page_count`, `processing_time_ms`, `pages_needing_ocr`, `title`, `confidence`, `is_complex_layout`, `pages_with_tables`, `pages_with_columns`, `has_encoding_issues`
|
||||
|
||||
**`PdfClassification` fields:** `pdf_type`, `page_count`, `pages_needing_ocr` (0-indexed), `confidence`
|
||||
|
||||
**`TextItem` fields:** `text`, `x`, `y`, `width`, `height`, `font`, `font_size`, `page`, `is_bold`, `is_italic`, `item_type`
|
||||
|
||||
**`RegionText` fields:** `text`, `needs_ocr`
|
||||
|
||||
**`PageRegionTexts` fields:** `page` (0-indexed), `regions` (list of RegionText)
|
||||
@@ -0,0 +1,122 @@
|
||||
# Rust API
|
||||
|
||||
Add to your `Cargo.toml`:
|
||||
|
||||
```toml
|
||||
[dependencies]
|
||||
pdf-inspector = { git = "https://github.com/firecrawl/pdf-inspector" }
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
Detect and extract in one call:
|
||||
|
||||
```rust
|
||||
use pdf_inspector::process_pdf;
|
||||
|
||||
let result = process_pdf("document.pdf")?;
|
||||
|
||||
println!("Type: {:?}", result.pdf_type); // TextBased, Scanned, ImageBased, Mixed
|
||||
println!("Confidence: {:.0}%", result.confidence * 100.0);
|
||||
println!("Pages: {}", result.page_count);
|
||||
|
||||
if let Some(markdown) = &result.markdown {
|
||||
println!("{}", markdown);
|
||||
}
|
||||
```
|
||||
|
||||
Fast metadata-only detection (no text extraction or markdown generation):
|
||||
|
||||
```rust
|
||||
use pdf_inspector::detect_pdf;
|
||||
|
||||
let info = detect_pdf("document.pdf")?;
|
||||
|
||||
match info.pdf_type {
|
||||
pdf_inspector::PdfType::TextBased => {
|
||||
// Extract locally — fast and free
|
||||
}
|
||||
_ => {
|
||||
// Route to OCR service
|
||||
// info.pages_needing_ocr tells you exactly which pages
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
Customize processing with `PdfOptions`:
|
||||
|
||||
```rust
|
||||
use pdf_inspector::{process_pdf_with_options, PdfOptions, ProcessMode, DetectionConfig, ScanStrategy};
|
||||
|
||||
// Analyze layout without generating markdown
|
||||
let result = process_pdf_with_options(
|
||||
"document.pdf",
|
||||
PdfOptions::new().mode(ProcessMode::Analyze),
|
||||
)?;
|
||||
|
||||
// Full extraction with custom detection strategy
|
||||
let result = process_pdf_with_options(
|
||||
"large.pdf",
|
||||
PdfOptions::new().detection(DetectionConfig {
|
||||
strategy: ScanStrategy::Sample(5),
|
||||
..Default::default()
|
||||
}),
|
||||
)?;
|
||||
|
||||
// Process only specific pages
|
||||
let result = process_pdf_with_options(
|
||||
"document.pdf",
|
||||
PdfOptions::new().pages([1, 3, 5]),
|
||||
)?;
|
||||
```
|
||||
|
||||
Process from a byte buffer (no filesystem needed):
|
||||
|
||||
```rust
|
||||
use pdf_inspector::process_pdf_mem;
|
||||
|
||||
let bytes = std::fs::read("document.pdf")?;
|
||||
let result = process_pdf_mem(&bytes)?;
|
||||
```
|
||||
|
||||
## Processing modes
|
||||
|
||||
| Mode | What it does | Returns |
|
||||
|---|---|---|
|
||||
| `ProcessMode::Full` (default) | Detect + extract + convert to Markdown | Everything populated |
|
||||
| `ProcessMode::Analyze` | Detect + extract + layout analysis (no Markdown) | `markdown` is `None`, `layout` is populated |
|
||||
| `ProcessMode::DetectOnly` | Classification only (fastest) | `markdown` is `None`, `layout` is default |
|
||||
|
||||
## Functions
|
||||
|
||||
| Function | Description |
|
||||
|---|---|
|
||||
| `process_pdf(path)` | Full processing with defaults |
|
||||
| `detect_pdf(path)` | Fast metadata-only detection (no extraction) |
|
||||
| `process_pdf_with_options(path, options)` | Process with custom `PdfOptions` |
|
||||
| `process_pdf_mem(bytes)` | Full processing from a byte buffer |
|
||||
| `detect_pdf_mem(bytes)` | Fast detection from a byte buffer |
|
||||
| `process_pdf_mem_with_options(bytes, options)` | Process from bytes with custom options |
|
||||
| `extract_text(path)` | Plain text extraction |
|
||||
| `extract_text_with_positions(path)` | Text with X/Y coordinates and font info |
|
||||
| `to_markdown(text, options)` | Convert plain text to Markdown |
|
||||
| `to_markdown_from_items(items, options)` | Markdown from pre-extracted `TextItem`s |
|
||||
| `to_markdown_from_items_with_rects(items, options, rects)` | Markdown with rectangle-based table detection |
|
||||
|
||||
Low-level detection functions are also available via the `detector` module (`detect_pdf_type`, `detect_pdf_type_with_config`, etc.) for callers who need `PdfTypeResult` instead of `PdfProcessResult`.
|
||||
|
||||
## Types
|
||||
|
||||
| Type | Description |
|
||||
|---|---|
|
||||
| `PdfOptions` | Builder for processing configuration (mode, detection, markdown, page filter) |
|
||||
| `ProcessMode` | `DetectOnly`, `Analyze`, `Full` |
|
||||
| `PdfType` | `TextBased`, `Scanned`, `ImageBased`, `Mixed` |
|
||||
| `PdfProcessResult` | Full result: pdf_type, markdown, page_count, confidence, layout, has_encoding_issues, timing |
|
||||
| `PdfTypeResult` | Low-level detection result: type, confidence, page count, pages needing OCR |
|
||||
| `DetectionConfig` | Configuration for detection: scan strategy, thresholds |
|
||||
| `ScanStrategy` | `EarlyExit`, `Full`, `Sample(n)`, `Pages(vec)` |
|
||||
| `LayoutComplexity` | Layout analysis: is_complex, pages_with_tables, pages_with_columns |
|
||||
| `TextItem` | Text with position, font info, and page number |
|
||||
| `MarkdownOptions` | Configuration for Markdown formatting (page numbers, etc.) |
|
||||
| `PdfError` | `Io`, `Parse`, `Encrypted`, `InvalidStructure`, `NotAPdf` |
|
||||
@@ -0,0 +1,98 @@
|
||||
"""Basic usage examples for pdf-inspector Python library."""
|
||||
|
||||
import sys
|
||||
import pdf_inspector
|
||||
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 2:
|
||||
print("Usage: python basic_usage.py <path-to-pdf>")
|
||||
sys.exit(1)
|
||||
|
||||
path = sys.argv[1]
|
||||
|
||||
# 1. Full processing: detect + extract + markdown
|
||||
print("=" * 60)
|
||||
print("Full processing")
|
||||
print("=" * 60)
|
||||
result = pdf_inspector.process_pdf(path)
|
||||
print(f"Type: {result.pdf_type}")
|
||||
print(f"Pages: {result.page_count}")
|
||||
print(f"Confidence: {result.confidence:.0%}")
|
||||
print(f"Time: {result.processing_time_ms}ms")
|
||||
print(f"Title: {result.title}")
|
||||
print(f"Complex: {result.is_complex_layout}")
|
||||
print(f"Tables on: {result.pages_with_tables}")
|
||||
print(f"Columns on: {result.pages_with_columns}")
|
||||
print(f"Encoding: {'issues detected' if result.has_encoding_issues else 'ok'}")
|
||||
print(f"OCR needed: {result.pages_needing_ocr or 'none'}")
|
||||
if result.markdown:
|
||||
print(f"\n--- Markdown ({len(result.markdown)} chars) ---")
|
||||
print(result.markdown[:500])
|
||||
if len(result.markdown) > 500:
|
||||
print(f"\n... ({len(result.markdown) - 500} more chars)")
|
||||
|
||||
# 2. Fast detection only
|
||||
print("\n" + "=" * 60)
|
||||
print("Detection only")
|
||||
print("=" * 60)
|
||||
info = pdf_inspector.detect_pdf(path)
|
||||
print(f"Type: {info.pdf_type}")
|
||||
print(f"Confidence: {info.confidence:.0%}")
|
||||
print(f"Time: {info.processing_time_ms}ms")
|
||||
|
||||
# 3. From bytes
|
||||
print("\n" + "=" * 60)
|
||||
print("From bytes")
|
||||
print("=" * 60)
|
||||
with open(path, "rb") as f:
|
||||
data = f.read()
|
||||
result = pdf_inspector.process_pdf_bytes(data)
|
||||
print(f"Type: {result.pdf_type}, Pages: {result.page_count}")
|
||||
|
||||
# 4. Plain text
|
||||
print("\n" + "=" * 60)
|
||||
print("Plain text extraction")
|
||||
print("=" * 60)
|
||||
text = pdf_inspector.extract_text(path)
|
||||
print(text[:300])
|
||||
|
||||
# 5. Positioned items
|
||||
print("\n" + "=" * 60)
|
||||
print("Positioned text items (first 10)")
|
||||
print("=" * 60)
|
||||
items = pdf_inspector.extract_text_with_positions(path, pages=[1])
|
||||
for item in items[:10]:
|
||||
bold = " [B]" if item.is_bold else ""
|
||||
italic = " [I]" if item.is_italic else ""
|
||||
print(
|
||||
f" p{item.page} ({item.x:6.1f}, {item.y:6.1f}) "
|
||||
f"size={item.font_size:5.1f}{bold}{italic} "
|
||||
f"'{item.text}'"
|
||||
)
|
||||
|
||||
# 6. Lightweight classification
|
||||
print("\n" + "=" * 60)
|
||||
print("Lightweight classification")
|
||||
print("=" * 60)
|
||||
cls = pdf_inspector.classify_pdf(path)
|
||||
print(f"Type: {cls.pdf_type}")
|
||||
print(f"Pages: {cls.page_count}")
|
||||
print(f"Confidence: {cls.confidence:.0%}")
|
||||
print(f"OCR pages: {cls.pages_needing_ocr or 'none'} (0-indexed)")
|
||||
|
||||
# 7. Region-based text extraction
|
||||
print("\n" + "=" * 60)
|
||||
print("Region-based text extraction (page 0, top region)")
|
||||
print("=" * 60)
|
||||
regions = pdf_inspector.extract_text_in_regions(
|
||||
path, [(0, [[0.0, 0.0, 600.0, 200.0]])]
|
||||
)
|
||||
for page_result in regions:
|
||||
for i, region in enumerate(page_result.regions):
|
||||
print(f" Region {i}: needs_ocr={region.needs_ocr}")
|
||||
print(f" Text: {region.text[:200]}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Generated
+1
-19
@@ -129,12 +129,6 @@ version = "3.20.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5d20789868f4b01b2f2caec9f5c4e0213b41e3e5702a50157d699ae31ced2fcb"
|
||||
|
||||
[[package]]
|
||||
name = "bytecount"
|
||||
version = "0.6.9"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "175812e0be2bccb6abe50bb8d566126198344f707e304f45c648fd8f2cc0365e"
|
||||
|
||||
[[package]]
|
||||
name = "cbc"
|
||||
version = "0.1.2"
|
||||
@@ -679,7 +673,7 @@ checksum = "5e5032e24019045c762d3c0f28f5b6b8bbf38563a65908389bf7978758920897"
|
||||
[[package]]
|
||||
name = "lopdf"
|
||||
version = "0.40.0"
|
||||
source = "git+https://github.com/J-F-Liu/lopdf?rev=052674053814a9f4897af94f0b8e46a545c9b329#052674053814a9f4897af94f0b8e46a545c9b329"
|
||||
source = "git+https://github.com/J-F-Liu/lopdf?rev=7a05512d831415b1f2b1ce522391d6beab8a1284#7a05512d831415b1f2b1ce522391d6beab8a1284"
|
||||
dependencies = [
|
||||
"aes",
|
||||
"bitflags",
|
||||
@@ -695,7 +689,6 @@ dependencies = [
|
||||
"log",
|
||||
"md-5",
|
||||
"nom",
|
||||
"nom_locate",
|
||||
"rand",
|
||||
"rangemap",
|
||||
"rayon",
|
||||
@@ -807,17 +800,6 @@ dependencies = [
|
||||
"memchr",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "nom_locate"
|
||||
version = "5.0.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0b577e2d69827c4740cba2b52efaad1c4cc7c73042860b199710b3575c68438d"
|
||||
dependencies = [
|
||||
"bytecount",
|
||||
"memchr",
|
||||
"nom",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "num-conv"
|
||||
version = "0.2.1"
|
||||
|
||||
+1
-1
@@ -1,3 +1,3 @@
|
||||
fn main() {
|
||||
napi_build::setup();
|
||||
napi_build::setup();
|
||||
}
|
||||
|
||||
Vendored
-49
@@ -1,49 +0,0 @@
|
||||
/* auto-generated by NAPI-RS */
|
||||
/* eslint-disable */
|
||||
/**
|
||||
* Classify a PDF: detect type (TextBased/Scanned/Mixed/ImageBased),
|
||||
* page count, and which pages need OCR. Takes PDF bytes as Buffer.
|
||||
*/
|
||||
export declare function classifyPdf(buffer: Buffer): PdfClassification
|
||||
|
||||
/**
|
||||
* Extract text within bounding-box regions from a PDF.
|
||||
*
|
||||
* For hybrid OCR: layout model detects regions in rendered images,
|
||||
* this extracts PDF text within those regions — skipping GPU OCR
|
||||
* for text-based pages.
|
||||
*
|
||||
* Each region result includes `needs_ocr` — set when the extracted text
|
||||
* is unreliable (empty, GID-encoded fonts, garbage, encoding issues).
|
||||
*
|
||||
* Coordinates are PDF points with top-left origin.
|
||||
*/
|
||||
export declare function extractTextInRegions(buffer: Buffer, pageRegions: Array<PageRegions>): Array<PageRegionTexts>
|
||||
|
||||
/** A page's regions for text extraction: (page_index_0based, bboxes). */
|
||||
export interface PageRegions {
|
||||
page: number
|
||||
/** Each bbox is [x1, y1, x2, y2] in PDF points, top-left origin. */
|
||||
regions: Array<Array<number>>
|
||||
}
|
||||
|
||||
/** Extracted text for one page's regions. */
|
||||
export interface PageRegionTexts {
|
||||
page: number
|
||||
regions: Array<RegionText>
|
||||
}
|
||||
|
||||
/** Lightweight PDF classification result. */
|
||||
export interface PdfClassification {
|
||||
pdfType: string
|
||||
pageCount: number
|
||||
pagesNeedingOcr: Array<number>
|
||||
confidence: number
|
||||
}
|
||||
|
||||
/** Extracted text for a single region. */
|
||||
export interface RegionText {
|
||||
text: string
|
||||
/** `true` when the text should not be trusted (empty, GID fonts, garbage, encoding issues). */
|
||||
needsOcr: boolean
|
||||
}
|
||||
-580
@@ -1,580 +0,0 @@
|
||||
// prettier-ignore
|
||||
/* eslint-disable */
|
||||
// @ts-nocheck
|
||||
/* auto-generated by NAPI-RS */
|
||||
|
||||
const { readFileSync } = require('node:fs')
|
||||
let nativeBinding = null
|
||||
const loadErrors = []
|
||||
|
||||
const isMusl = () => {
|
||||
let musl = false
|
||||
if (process.platform === 'linux') {
|
||||
musl = isMuslFromFilesystem()
|
||||
if (musl === null) {
|
||||
musl = isMuslFromReport()
|
||||
}
|
||||
if (musl === null) {
|
||||
musl = isMuslFromChildProcess()
|
||||
}
|
||||
}
|
||||
return musl
|
||||
}
|
||||
|
||||
const isFileMusl = (f) => f.includes('libc.musl-') || f.includes('ld-musl-')
|
||||
|
||||
const isMuslFromFilesystem = () => {
|
||||
try {
|
||||
return readFileSync('/usr/bin/ldd', 'utf-8').includes('musl')
|
||||
} catch {
|
||||
return null
|
||||
}
|
||||
}
|
||||
|
||||
const isMuslFromReport = () => {
|
||||
let report = null
|
||||
if (typeof process.report?.getReport === 'function') {
|
||||
process.report.excludeNetwork = true
|
||||
report = process.report.getReport()
|
||||
}
|
||||
if (!report) {
|
||||
return null
|
||||
}
|
||||
if (report.header && report.header.glibcVersionRuntime) {
|
||||
return false
|
||||
}
|
||||
if (Array.isArray(report.sharedObjects)) {
|
||||
if (report.sharedObjects.some(isFileMusl)) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
const isMuslFromChildProcess = () => {
|
||||
try {
|
||||
return require('child_process').execSync('ldd --version', { encoding: 'utf8' }).includes('musl')
|
||||
} catch (e) {
|
||||
// If we reach this case, we don't know if the system is musl or not, so is better to just fallback to false
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
function requireNative() {
|
||||
if (process.env.NAPI_RS_NATIVE_LIBRARY_PATH) {
|
||||
try {
|
||||
return require(process.env.NAPI_RS_NATIVE_LIBRARY_PATH);
|
||||
} catch (err) {
|
||||
loadErrors.push(err)
|
||||
}
|
||||
} else if (process.platform === 'android') {
|
||||
if (process.arch === 'arm64') {
|
||||
try {
|
||||
return require('./pdf-inspector.android-arm64.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-android-arm64')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-android-arm64/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else if (process.arch === 'arm') {
|
||||
try {
|
||||
return require('./pdf-inspector.android-arm-eabi.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-android-arm-eabi')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-android-arm-eabi/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
loadErrors.push(new Error(`Unsupported architecture on Android ${process.arch}`))
|
||||
}
|
||||
} else if (process.platform === 'win32') {
|
||||
if (process.arch === 'x64') {
|
||||
if (process.config?.variables?.shlib_suffix === 'dll.a' || process.config?.variables?.node_target_type === 'shared_library') {
|
||||
try {
|
||||
return require('./pdf-inspector.win32-x64-gnu.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-win32-x64-gnu')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-win32-x64-gnu/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
try {
|
||||
return require('./pdf-inspector.win32-x64-msvc.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-win32-x64-msvc')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-win32-x64-msvc/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
}
|
||||
} else if (process.arch === 'ia32') {
|
||||
try {
|
||||
return require('./pdf-inspector.win32-ia32-msvc.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-win32-ia32-msvc')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-win32-ia32-msvc/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else if (process.arch === 'arm64') {
|
||||
try {
|
||||
return require('./pdf-inspector.win32-arm64-msvc.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-win32-arm64-msvc')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-win32-arm64-msvc/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
loadErrors.push(new Error(`Unsupported architecture on Windows: ${process.arch}`))
|
||||
}
|
||||
} else if (process.platform === 'darwin') {
|
||||
try {
|
||||
return require('./pdf-inspector.darwin-universal.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-darwin-universal')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-darwin-universal/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
if (process.arch === 'x64') {
|
||||
try {
|
||||
return require('./pdf-inspector.darwin-x64.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-darwin-x64')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-darwin-x64/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else if (process.arch === 'arm64') {
|
||||
try {
|
||||
return require('./pdf-inspector.darwin-arm64.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-darwin-arm64')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-darwin-arm64/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
loadErrors.push(new Error(`Unsupported architecture on macOS: ${process.arch}`))
|
||||
}
|
||||
} else if (process.platform === 'freebsd') {
|
||||
if (process.arch === 'x64') {
|
||||
try {
|
||||
return require('./pdf-inspector.freebsd-x64.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-freebsd-x64')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-freebsd-x64/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else if (process.arch === 'arm64') {
|
||||
try {
|
||||
return require('./pdf-inspector.freebsd-arm64.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-freebsd-arm64')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-freebsd-arm64/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
loadErrors.push(new Error(`Unsupported architecture on FreeBSD: ${process.arch}`))
|
||||
}
|
||||
} else if (process.platform === 'linux') {
|
||||
if (process.arch === 'x64') {
|
||||
if (isMusl()) {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-x64-musl.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-x64-musl')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-x64-musl/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-x64-gnu.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-x64-gnu')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-x64-gnu/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
}
|
||||
} else if (process.arch === 'arm64') {
|
||||
if (isMusl()) {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-arm64-musl.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-arm64-musl')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-arm64-musl/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-arm64-gnu.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-arm64-gnu')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-arm64-gnu/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
}
|
||||
} else if (process.arch === 'arm') {
|
||||
if (isMusl()) {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-arm-musleabihf.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-arm-musleabihf')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-arm-musleabihf/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-arm-gnueabihf.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-arm-gnueabihf')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-arm-gnueabihf/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
}
|
||||
} else if (process.arch === 'loong64') {
|
||||
if (isMusl()) {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-loong64-musl.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-loong64-musl')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-loong64-musl/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-loong64-gnu.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-loong64-gnu')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-loong64-gnu/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
}
|
||||
} else if (process.arch === 'riscv64') {
|
||||
if (isMusl()) {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-riscv64-musl.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-riscv64-musl')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-riscv64-musl/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-riscv64-gnu.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-riscv64-gnu')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-riscv64-gnu/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
}
|
||||
} else if (process.arch === 'ppc64') {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-ppc64-gnu.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-ppc64-gnu')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-ppc64-gnu/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else if (process.arch === 's390x') {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-s390x-gnu.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-s390x-gnu')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-s390x-gnu/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
loadErrors.push(new Error(`Unsupported architecture on Linux: ${process.arch}`))
|
||||
}
|
||||
} else if (process.platform === 'openharmony') {
|
||||
if (process.arch === 'arm64') {
|
||||
try {
|
||||
return require('./pdf-inspector.openharmony-arm64.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-openharmony-arm64')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-openharmony-arm64/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else if (process.arch === 'x64') {
|
||||
try {
|
||||
return require('./pdf-inspector.openharmony-x64.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-openharmony-x64')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-openharmony-x64/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else if (process.arch === 'arm') {
|
||||
try {
|
||||
return require('./pdf-inspector.openharmony-arm.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-openharmony-arm')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-openharmony-arm/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
loadErrors.push(new Error(`Unsupported architecture on OpenHarmony: ${process.arch}`))
|
||||
}
|
||||
} else {
|
||||
loadErrors.push(new Error(`Unsupported OS: ${process.platform}, architecture: ${process.arch}`))
|
||||
}
|
||||
}
|
||||
|
||||
nativeBinding = requireNative()
|
||||
|
||||
if (!nativeBinding || process.env.NAPI_RS_FORCE_WASI) {
|
||||
let wasiBinding = null
|
||||
let wasiBindingError = null
|
||||
try {
|
||||
wasiBinding = require('./pdf-inspector.wasi.cjs')
|
||||
nativeBinding = wasiBinding
|
||||
} catch (err) {
|
||||
if (process.env.NAPI_RS_FORCE_WASI) {
|
||||
wasiBindingError = err
|
||||
}
|
||||
}
|
||||
if (!nativeBinding || process.env.NAPI_RS_FORCE_WASI) {
|
||||
try {
|
||||
wasiBinding = require('firecrawl-pdf-inspector-wasm32-wasi')
|
||||
nativeBinding = wasiBinding
|
||||
} catch (err) {
|
||||
if (process.env.NAPI_RS_FORCE_WASI) {
|
||||
if (!wasiBindingError) {
|
||||
wasiBindingError = err
|
||||
} else {
|
||||
wasiBindingError.cause = err
|
||||
}
|
||||
loadErrors.push(err)
|
||||
}
|
||||
}
|
||||
}
|
||||
if (process.env.NAPI_RS_FORCE_WASI === 'error' && !wasiBinding) {
|
||||
const error = new Error('WASI binding not found and NAPI_RS_FORCE_WASI is set to error')
|
||||
error.cause = wasiBindingError
|
||||
throw error
|
||||
}
|
||||
}
|
||||
|
||||
if (!nativeBinding) {
|
||||
if (loadErrors.length > 0) {
|
||||
throw new Error(
|
||||
`Cannot find native binding. ` +
|
||||
`npm has a bug related to optional dependencies (https://github.com/npm/cli/issues/4828). ` +
|
||||
'Please try `npm i` again after removing both package-lock.json and node_modules directory.',
|
||||
{
|
||||
cause: loadErrors.reduce((err, cur) => {
|
||||
cur.cause = err
|
||||
return cur
|
||||
}),
|
||||
},
|
||||
)
|
||||
}
|
||||
throw new Error(`Failed to load native binding`)
|
||||
}
|
||||
|
||||
module.exports = nativeBinding
|
||||
module.exports.classifyPdf = nativeBinding.classifyPdf
|
||||
module.exports.extractTextInRegions = nativeBinding.extractTextInRegions
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "firecrawl-pdf-inspector",
|
||||
"version": "0.2.3",
|
||||
"version": "0.7.1",
|
||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||
"main": "index.js",
|
||||
"types": "index.d.ts",
|
||||
|
||||
+367
-69
@@ -2,58 +2,269 @@
|
||||
|
||||
use napi::bindgen_prelude::*;
|
||||
use napi_derive::napi;
|
||||
use std::collections::HashSet;
|
||||
use std::panic;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Enums
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// PDF document type classification.
|
||||
#[napi(string_enum)]
|
||||
pub enum PdfType {
|
||||
TextBased,
|
||||
Scanned,
|
||||
ImageBased,
|
||||
Mixed,
|
||||
}
|
||||
|
||||
/// Type of a positioned text item.
|
||||
#[napi(string_enum)]
|
||||
pub enum ItemType {
|
||||
Text,
|
||||
Image,
|
||||
Link,
|
||||
FormField,
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Result types
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Full PDF processing result with markdown and metadata.
|
||||
#[napi(object)]
|
||||
pub struct PdfResult {
|
||||
pub pdf_type: PdfType,
|
||||
pub markdown: Option<String>,
|
||||
pub page_count: u32,
|
||||
pub processing_time_ms: u32,
|
||||
/// 1-indexed page numbers that need OCR.
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
pub title: Option<String>,
|
||||
pub confidence: f64,
|
||||
pub is_complex_layout: bool,
|
||||
pub pages_with_tables: Vec<u32>,
|
||||
pub pages_with_columns: Vec<u32>,
|
||||
pub has_encoding_issues: bool,
|
||||
}
|
||||
|
||||
/// Lightweight PDF classification result.
|
||||
#[napi(object)]
|
||||
pub struct PdfClassification {
|
||||
pub pdf_type: String,
|
||||
pub page_count: u32,
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
pub confidence: f64,
|
||||
pub pdf_type: PdfType,
|
||||
pub page_count: u32,
|
||||
/// 0-indexed page numbers that need OCR.
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
pub confidence: f64,
|
||||
}
|
||||
|
||||
/// A positioned text item extracted from a PDF.
|
||||
#[napi(object)]
|
||||
pub struct TextItem {
|
||||
pub text: String,
|
||||
pub x: f64,
|
||||
pub y: f64,
|
||||
pub width: f64,
|
||||
pub height: f64,
|
||||
pub font: String,
|
||||
pub font_size: f64,
|
||||
pub page: u32,
|
||||
pub is_bold: bool,
|
||||
pub is_italic: bool,
|
||||
pub item_type: ItemType,
|
||||
/// URL for link items, `None` for other types.
|
||||
pub link_url: Option<String>,
|
||||
}
|
||||
|
||||
/// A page's regions for text extraction: (page_index_0based, bboxes).
|
||||
#[napi(object)]
|
||||
pub struct PageRegions {
|
||||
pub page: u32,
|
||||
/// Each bbox is [x1, y1, x2, y2] in PDF points, top-left origin.
|
||||
pub regions: Vec<Vec<f64>>,
|
||||
pub page: u32,
|
||||
/// Each bbox is [x1, y1, x2, y2] in PDF points, top-left origin.
|
||||
pub regions: Vec<Vec<f64>>,
|
||||
}
|
||||
|
||||
/// Extracted text for a single region.
|
||||
#[napi(object)]
|
||||
pub struct RegionText {
|
||||
pub text: String,
|
||||
/// `true` when the text should not be trusted (empty, GID fonts, garbage, encoding issues).
|
||||
pub needs_ocr: bool,
|
||||
pub text: String,
|
||||
/// `true` when the text should not be trusted (empty, GID fonts, garbage, encoding issues).
|
||||
pub needs_ocr: bool,
|
||||
}
|
||||
|
||||
/// Extracted text for one page's regions.
|
||||
#[napi(object)]
|
||||
pub struct PageRegionTexts {
|
||||
pub page: u32,
|
||||
pub regions: Vec<RegionText>,
|
||||
pub page: u32,
|
||||
pub regions: Vec<RegionText>,
|
||||
}
|
||||
|
||||
/// Classify a PDF: detect type (TextBased/Scanned/Mixed/ImageBased),
|
||||
/// page count, and which pages need OCR. Takes PDF bytes as Buffer.
|
||||
// ---------------------------------------------------------------------------
|
||||
// Helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn convert_pdf_type(t: pdf_inspector::PdfType) -> PdfType {
|
||||
match t {
|
||||
pdf_inspector::PdfType::TextBased => PdfType::TextBased,
|
||||
pdf_inspector::PdfType::Scanned => PdfType::Scanned,
|
||||
pdf_inspector::PdfType::ImageBased => PdfType::ImageBased,
|
||||
pdf_inspector::PdfType::Mixed => PdfType::Mixed,
|
||||
}
|
||||
}
|
||||
|
||||
fn to_napi_result(r: pdf_inspector::PdfProcessResult) -> PdfResult {
|
||||
PdfResult {
|
||||
pdf_type: convert_pdf_type(r.pdf_type),
|
||||
markdown: r.markdown,
|
||||
page_count: r.page_count,
|
||||
processing_time_ms: r.processing_time_ms as u32,
|
||||
pages_needing_ocr: r.pages_needing_ocr,
|
||||
title: r.title,
|
||||
confidence: r.confidence as f64,
|
||||
is_complex_layout: r.layout.is_complex,
|
||||
pages_with_tables: r.layout.pages_with_tables,
|
||||
pages_with_columns: r.layout.pages_with_columns,
|
||||
has_encoding_issues: r.has_encoding_issues,
|
||||
}
|
||||
}
|
||||
|
||||
fn convert_item_type(t: &pdf_inspector::types::ItemType) -> (ItemType, Option<String>) {
|
||||
match t {
|
||||
pdf_inspector::types::ItemType::Text => (ItemType::Text, None),
|
||||
pdf_inspector::types::ItemType::Image => (ItemType::Image, None),
|
||||
pdf_inspector::types::ItemType::Link(url) => (ItemType::Link, Some(url.clone())),
|
||||
pdf_inspector::types::ItemType::FormField => (ItemType::FormField, None),
|
||||
}
|
||||
}
|
||||
|
||||
fn to_napi_err(e: impl std::fmt::Display, ctx: &str) -> Error {
|
||||
Error::new(Status::GenericFailure, format!("{ctx}: {e}"))
|
||||
}
|
||||
|
||||
/// Run a closure, catching any Rust panic and converting it to a NAPI error.
|
||||
/// Prevents process abort from unwind panics in the native module.
|
||||
fn catch_panic<F, T>(ctx: &str, f: F) -> Result<T>
|
||||
where
|
||||
F: FnOnce() -> Result<T> + panic::UnwindSafe,
|
||||
{
|
||||
match panic::catch_unwind(f) {
|
||||
Ok(result) => result,
|
||||
Err(payload) => {
|
||||
let msg = if let Some(s) = payload.downcast_ref::<&str>() {
|
||||
s.to_string()
|
||||
} else if let Some(s) = payload.downcast_ref::<String>() {
|
||||
s.clone()
|
||||
} else {
|
||||
"unknown panic".to_string()
|
||||
};
|
||||
Err(Error::new(
|
||||
Status::GenericFailure,
|
||||
format!("{ctx}: Rust panic: {msg}"),
|
||||
))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Public NAPI API
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Process a PDF from a Buffer: detect type, extract text, and convert to Markdown.
|
||||
#[napi]
|
||||
pub fn process_pdf(buffer: Buffer, pages: Option<Vec<u32>>) -> Result<PdfResult> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
catch_panic("process_pdf", move || {
|
||||
let mut opts = pdf_inspector::PdfOptions::new();
|
||||
if let Some(p) = pages {
|
||||
opts = opts.pages(p);
|
||||
}
|
||||
let result = pdf_inspector::process_pdf_mem_with_options(&bytes, opts)
|
||||
.map_err(|e| to_napi_err(e, "process_pdf"))?;
|
||||
Ok(to_napi_result(result))
|
||||
})
|
||||
}
|
||||
|
||||
/// Fast detection only — no text extraction or markdown.
|
||||
#[napi]
|
||||
pub fn detect_pdf(buffer: Buffer) -> Result<PdfResult> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
catch_panic("detect_pdf", move || {
|
||||
let result =
|
||||
pdf_inspector::detect_pdf_mem(&bytes).map_err(|e| to_napi_err(e, "detect_pdf"))?;
|
||||
Ok(to_napi_result(result))
|
||||
})
|
||||
}
|
||||
|
||||
/// Lightweight PDF classification — returns type, page count, and OCR pages.
|
||||
/// Faster than detectPdf as it skips building the full PdfResult.
|
||||
/// Pages in pagesNeedingOcr are 0-indexed.
|
||||
#[napi]
|
||||
pub fn classify_pdf(buffer: Buffer) -> Result<PdfClassification> {
|
||||
let result = pdf_inspector::classify_pdf_mem(&buffer).map_err(|e| {
|
||||
Error::new(Status::GenericFailure, format!("classify_pdf failed: {e}"))
|
||||
})?;
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
catch_panic("classify_pdf", move || {
|
||||
let result =
|
||||
pdf_inspector::classify_pdf_mem(&bytes).map_err(|e| to_napi_err(e, "classify_pdf"))?;
|
||||
Ok(PdfClassification {
|
||||
pdf_type: convert_pdf_type(result.pdf_type),
|
||||
page_count: result.page_count,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
confidence: result.confidence as f64,
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
Ok(PdfClassification {
|
||||
pdf_type: match result.pdf_type {
|
||||
pdf_inspector::PdfType::TextBased => "TextBased".to_string(),
|
||||
pdf_inspector::PdfType::Scanned => "Scanned".to_string(),
|
||||
pdf_inspector::PdfType::ImageBased => "ImageBased".to_string(),
|
||||
pdf_inspector::PdfType::Mixed => "Mixed".to_string(),
|
||||
},
|
||||
page_count: result.page_count,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
confidence: result.confidence as f64,
|
||||
})
|
||||
/// Extract plain text from a PDF Buffer.
|
||||
#[napi]
|
||||
pub fn extract_text(buffer: Buffer) -> Result<String> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
catch_panic("extract_text", move || {
|
||||
pdf_inspector::extractor::extract_text_mem(&bytes)
|
||||
.map_err(|e| to_napi_err(e, "extract_text"))
|
||||
})
|
||||
}
|
||||
|
||||
/// Extract text with position information from a PDF Buffer.
|
||||
#[napi]
|
||||
pub fn extract_text_with_positions(
|
||||
buffer: Buffer,
|
||||
pages: Option<Vec<u32>>,
|
||||
) -> Result<Vec<TextItem>> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
catch_panic("extract_text_with_positions", move || {
|
||||
let items = match pages {
|
||||
Some(p) => {
|
||||
let page_set: HashSet<u32> = p.into_iter().collect();
|
||||
pdf_inspector::extractor::extract_text_with_positions_mem_pages(
|
||||
&bytes,
|
||||
Some(&page_set),
|
||||
)
|
||||
.map_err(|e| to_napi_err(e, "extract_text_with_positions"))?
|
||||
}
|
||||
None => pdf_inspector::extractor::extract_text_with_positions_mem(&bytes)
|
||||
.map_err(|e| to_napi_err(e, "extract_text_with_positions"))?,
|
||||
};
|
||||
|
||||
Ok(items
|
||||
.into_iter()
|
||||
.map(|item| {
|
||||
let (item_type, link_url) = convert_item_type(&item.item_type);
|
||||
TextItem {
|
||||
text: item.text,
|
||||
x: item.x as f64,
|
||||
y: item.y as f64,
|
||||
width: item.width as f64,
|
||||
height: item.height as f64,
|
||||
font: item.font,
|
||||
font_size: item.font_size as f64,
|
||||
page: item.page,
|
||||
is_bold: item.is_bold,
|
||||
is_italic: item.is_italic,
|
||||
item_type,
|
||||
link_url,
|
||||
}
|
||||
})
|
||||
.collect())
|
||||
})
|
||||
}
|
||||
|
||||
/// Extract text within bounding-box regions from a PDF.
|
||||
@@ -62,55 +273,142 @@ pub fn classify_pdf(buffer: Buffer) -> Result<PdfClassification> {
|
||||
/// this extracts PDF text within those regions — skipping GPU OCR
|
||||
/// for text-based pages.
|
||||
///
|
||||
/// Each region result includes `needs_ocr` — set when the extracted text
|
||||
/// Each region result includes `needsOcr` — set when the extracted text
|
||||
/// is unreliable (empty, GID-encoded fonts, garbage, encoding issues).
|
||||
///
|
||||
/// Coordinates are PDF points with top-left origin.
|
||||
#[napi]
|
||||
pub fn extract_text_in_regions(
|
||||
buffer: Buffer,
|
||||
page_regions: Vec<PageRegions>,
|
||||
buffer: Buffer,
|
||||
page_regions: Vec<PageRegions>,
|
||||
) -> Result<Vec<PageRegionTexts>> {
|
||||
// Convert from napi types to the Rust API's expected format
|
||||
let regions: Vec<(u32, Vec<[f32; 4]>)> = page_regions
|
||||
.iter()
|
||||
.map(|pr| {
|
||||
let bboxes: Vec<[f32; 4]> = pr
|
||||
.regions
|
||||
.iter()
|
||||
.map(|r| {
|
||||
if r.len() != 4 {
|
||||
[0.0, 0.0, 0.0, 0.0]
|
||||
} else {
|
||||
[r[0] as f32, r[1] as f32, r[2] as f32, r[3] as f32]
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
(pr.page, bboxes)
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
let regions = parse_page_regions(&page_regions);
|
||||
|
||||
catch_panic("extract_text_in_regions", move || {
|
||||
let results = pdf_inspector::extract_text_in_regions_mem(&bytes, ®ions)
|
||||
.map_err(|e| to_napi_err(e, "extract_text_in_regions"))?;
|
||||
Ok(to_page_region_texts(results))
|
||||
})
|
||||
.collect();
|
||||
}
|
||||
|
||||
let results = pdf_inspector::extract_text_in_regions_mem(&buffer, ®ions).map_err(|e| {
|
||||
Error::new(
|
||||
Status::GenericFailure,
|
||||
format!("extract_text_in_regions failed: {e}"),
|
||||
)
|
||||
})?;
|
||||
/// Extract markdown tables within bounding-box regions from a PDF.
|
||||
///
|
||||
/// Like `extractTextInRegions` but runs table detection on items within each
|
||||
/// region and returns markdown pipe-tables instead of flat text.
|
||||
///
|
||||
/// When table structure is detected, `text` contains a markdown pipe-table and
|
||||
/// `needsOcr` is `false`. When no table is found, `text` is empty and
|
||||
/// `needsOcr` is `true` so the caller can fall back to GPU OCR.
|
||||
///
|
||||
/// Coordinates are PDF points with top-left origin.
|
||||
#[napi]
|
||||
pub fn extract_tables_in_regions(
|
||||
buffer: Buffer,
|
||||
page_regions: Vec<PageRegions>,
|
||||
) -> Result<Vec<PageRegionTexts>> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
let regions = parse_page_regions(&page_regions);
|
||||
|
||||
Ok(
|
||||
catch_panic("extract_tables_in_regions", move || {
|
||||
let results = pdf_inspector::extract_tables_in_regions_mem(&bytes, ®ions)
|
||||
.map_err(|e| to_napi_err(e, "extract_tables_in_regions"))?;
|
||||
Ok(to_page_region_texts(results))
|
||||
})
|
||||
}
|
||||
|
||||
/// Per-page markdown extraction result.
|
||||
#[napi(object)]
|
||||
pub struct PageMarkdownResult {
|
||||
/// 0-indexed page number.
|
||||
pub page: u32,
|
||||
/// Formatted markdown for this page.
|
||||
pub markdown: String,
|
||||
/// `true` when text on this page is unreliable.
|
||||
pub needs_ocr: bool,
|
||||
}
|
||||
|
||||
/// Combined per-page markdown extraction and layout classification result.
|
||||
#[napi(object)]
|
||||
pub struct PagesExtractionResult {
|
||||
/// Per-page markdown results.
|
||||
pub pages: Vec<PageMarkdownResult>,
|
||||
/// 1-indexed pages where tables were detected.
|
||||
pub pages_with_tables: Vec<u32>,
|
||||
/// 1-indexed pages where multi-column layout was detected.
|
||||
pub pages_with_columns: Vec<u32>,
|
||||
/// 1-indexed pages that need OCR (scanned/image-based).
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
/// True if any page has tables or columns.
|
||||
pub is_complex: bool,
|
||||
}
|
||||
|
||||
/// Extract formatted markdown for specific pages of a PDF, with layout
|
||||
/// classification metadata.
|
||||
///
|
||||
/// Returns per-page markdown and classification data (tables, columns,
|
||||
/// OCR needs) from a single parse. Font statistics are computed from the
|
||||
/// full document so header detection is consistent across pages.
|
||||
#[napi]
|
||||
pub fn extract_pages_markdown(
|
||||
buffer: Buffer,
|
||||
pages: Vec<u32>,
|
||||
) -> Result<PagesExtractionResult> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
catch_panic("extract_pages_markdown", move || {
|
||||
let result = pdf_inspector::extract_pages_markdown_mem(&bytes, &pages)
|
||||
.map_err(|e| to_napi_err(e, "extract_pages_markdown"))?;
|
||||
Ok(PagesExtractionResult {
|
||||
pages: result
|
||||
.pages
|
||||
.into_iter()
|
||||
.map(|r| PageMarkdownResult {
|
||||
page: r.page,
|
||||
markdown: r.markdown,
|
||||
needs_ocr: r.needs_ocr,
|
||||
})
|
||||
.collect(),
|
||||
pages_with_tables: result.pages_with_tables,
|
||||
pages_with_columns: result.pages_with_columns,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
is_complex: result.is_complex,
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
fn parse_page_regions(page_regions: &[PageRegions]) -> Vec<(u32, Vec<[f32; 4]>)> {
|
||||
page_regions
|
||||
.iter()
|
||||
.map(|pr| {
|
||||
let bboxes: Vec<[f32; 4]> = pr
|
||||
.regions
|
||||
.iter()
|
||||
.map(|r| {
|
||||
if r.len() != 4 {
|
||||
[0.0, 0.0, 0.0, 0.0]
|
||||
} else {
|
||||
[r[0] as f32, r[1] as f32, r[2] as f32, r[3] as f32]
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
(pr.page, bboxes)
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn to_page_region_texts(results: Vec<pdf_inspector::PageRegionResult>) -> Vec<PageRegionTexts> {
|
||||
results
|
||||
.into_iter()
|
||||
.map(|page_result| PageRegionTexts {
|
||||
page: page_result.page,
|
||||
regions: page_result
|
||||
.regions
|
||||
.into_iter()
|
||||
.map(|r| RegionText {
|
||||
text: r.text,
|
||||
needs_ocr: r.needs_ocr,
|
||||
})
|
||||
.collect(),
|
||||
})
|
||||
.collect(),
|
||||
)
|
||||
.into_iter()
|
||||
.map(|page_result| PageRegionTexts {
|
||||
page: page_result.page,
|
||||
regions: page_result
|
||||
.regions
|
||||
.into_iter()
|
||||
.map(|r| RegionText {
|
||||
text: r.text,
|
||||
needs_ocr: r.needs_ocr,
|
||||
})
|
||||
.collect(),
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
@@ -0,0 +1,98 @@
|
||||
import { readFileSync } from 'fs';
|
||||
import { strict as assert } from 'assert';
|
||||
import {
|
||||
processPdf,
|
||||
detectPdf,
|
||||
classifyPdf,
|
||||
extractText,
|
||||
extractTextWithPositions,
|
||||
extractTextInRegions,
|
||||
} from './index.js';
|
||||
|
||||
const fixture = readFileSync('../tests/fixtures/thermo-freon12.pdf');
|
||||
|
||||
// --- processPdf ---
|
||||
console.log('Testing processPdf...');
|
||||
const result = processPdf(fixture);
|
||||
assert.equal(result.pdfType, 'TextBased');
|
||||
assert.equal(result.pageCount, 3);
|
||||
assert.ok(result.confidence > 0);
|
||||
assert.ok(result.markdown && result.markdown.length > 0);
|
||||
assert.equal(typeof result.isComplexLayout, 'boolean');
|
||||
assert.ok(Array.isArray(result.pagesWithTables));
|
||||
assert.ok(Array.isArray(result.pagesWithColumns));
|
||||
assert.equal(typeof result.hasEncodingIssues, 'boolean');
|
||||
console.log(' processPdf: OK');
|
||||
|
||||
// processPdf with pages
|
||||
const result2 = processPdf(fixture, [1]);
|
||||
assert.ok(result2.markdown && result2.markdown.length > 0);
|
||||
console.log(' processPdf with pages: OK');
|
||||
|
||||
// --- detectPdf ---
|
||||
console.log('Testing detectPdf...');
|
||||
const detected = detectPdf(fixture);
|
||||
assert.equal(detected.pdfType, 'TextBased');
|
||||
assert.equal(detected.pageCount, 3);
|
||||
assert.equal(detected.markdown, undefined);
|
||||
console.log(' detectPdf: OK');
|
||||
|
||||
// --- classifyPdf ---
|
||||
console.log('Testing classifyPdf...');
|
||||
const classified = classifyPdf(fixture);
|
||||
assert.equal(classified.pdfType, 'TextBased');
|
||||
assert.equal(classified.pageCount, 3);
|
||||
assert.ok(classified.confidence > 0);
|
||||
assert.ok(Array.isArray(classified.pagesNeedingOcr));
|
||||
console.log(' classifyPdf: OK');
|
||||
|
||||
// --- extractText ---
|
||||
console.log('Testing extractText...');
|
||||
const text = extractText(fixture);
|
||||
assert.equal(typeof text, 'string');
|
||||
assert.ok(text.length > 0);
|
||||
console.log(' extractText: OK');
|
||||
|
||||
// --- extractTextWithPositions ---
|
||||
console.log('Testing extractTextWithPositions...');
|
||||
const items = extractTextWithPositions(fixture);
|
||||
assert.ok(items.length > 0);
|
||||
const item = items[0];
|
||||
assert.equal(typeof item.text, 'string');
|
||||
assert.equal(typeof item.x, 'number');
|
||||
assert.equal(typeof item.y, 'number');
|
||||
assert.equal(typeof item.width, 'number');
|
||||
assert.equal(typeof item.height, 'number');
|
||||
assert.equal(typeof item.font, 'string');
|
||||
assert.equal(typeof item.fontSize, 'number');
|
||||
assert.equal(typeof item.page, 'number');
|
||||
assert.equal(typeof item.isBold, 'boolean');
|
||||
assert.equal(typeof item.isItalic, 'boolean');
|
||||
assert.equal(typeof item.itemType, 'string');
|
||||
console.log(' extractTextWithPositions: OK');
|
||||
|
||||
// with pages filter
|
||||
const page1Items = extractTextWithPositions(fixture, [1]);
|
||||
assert.ok(page1Items.length > 0);
|
||||
assert.ok(page1Items.every(i => i.page === 1));
|
||||
console.log(' extractTextWithPositions with pages: OK');
|
||||
|
||||
// --- extractTextInRegions ---
|
||||
console.log('Testing extractTextInRegions...');
|
||||
const regionResults = extractTextInRegions(fixture, [
|
||||
{ page: 0, regions: [[0, 0, 600, 100]] },
|
||||
]);
|
||||
assert.equal(regionResults.length, 1);
|
||||
assert.equal(regionResults[0].page, 0);
|
||||
assert.equal(regionResults[0].regions.length, 1);
|
||||
assert.equal(typeof regionResults[0].regions[0].text, 'string');
|
||||
assert.equal(typeof regionResults[0].regions[0].needsOcr, 'boolean');
|
||||
console.log(' extractTextInRegions: OK');
|
||||
|
||||
// --- Error handling ---
|
||||
console.log('Testing error handling...');
|
||||
assert.throws(() => processPdf(Buffer.from('not a pdf')), /process_pdf/);
|
||||
assert.throws(() => classifyPdf(Buffer.from('')), /classify_pdf/);
|
||||
console.log(' error handling: OK');
|
||||
|
||||
console.log('\nAll NAPI tests passed!');
|
||||
@@ -0,0 +1,117 @@
|
||||
"""Type stubs for pdf_inspector."""
|
||||
|
||||
from typing import Optional
|
||||
|
||||
class PdfResult:
|
||||
"""Result of processing a PDF file."""
|
||||
pdf_type: str
|
||||
"""'text_based', 'scanned', 'image_based', or 'mixed'."""
|
||||
markdown: Optional[str]
|
||||
page_count: int
|
||||
processing_time_ms: int
|
||||
pages_needing_ocr: list[int]
|
||||
title: Optional[str]
|
||||
confidence: float
|
||||
is_complex_layout: bool
|
||||
pages_with_tables: list[int]
|
||||
pages_with_columns: list[int]
|
||||
has_encoding_issues: bool
|
||||
|
||||
class PdfClassification:
|
||||
"""Lightweight PDF classification result."""
|
||||
pdf_type: str
|
||||
"""'text_based', 'scanned', 'image_based', or 'mixed'."""
|
||||
page_count: int
|
||||
pages_needing_ocr: list[int]
|
||||
"""0-indexed page numbers that need OCR."""
|
||||
confidence: float
|
||||
|
||||
class TextItem:
|
||||
"""A positioned text item extracted from a PDF."""
|
||||
text: str
|
||||
x: float
|
||||
y: float
|
||||
width: float
|
||||
height: float
|
||||
font: str
|
||||
font_size: float
|
||||
page: int
|
||||
is_bold: bool
|
||||
is_italic: bool
|
||||
item_type: str
|
||||
|
||||
class RegionText:
|
||||
"""Extracted text for a single region."""
|
||||
text: str
|
||||
needs_ocr: bool
|
||||
"""True when the text should not be trusted."""
|
||||
|
||||
class PageRegionTexts:
|
||||
"""Extracted text for one page's regions."""
|
||||
page: int
|
||||
"""0-indexed page number."""
|
||||
regions: list[RegionText]
|
||||
|
||||
def process_pdf(path: str, pages: Optional[list[int]] = None) -> PdfResult:
|
||||
"""Process a PDF: detect type, extract text, convert to Markdown."""
|
||||
...
|
||||
|
||||
def process_pdf_bytes(data: bytes, pages: Optional[list[int]] = None) -> PdfResult:
|
||||
"""Process a PDF from bytes in memory."""
|
||||
...
|
||||
|
||||
def detect_pdf(path: str) -> PdfResult:
|
||||
"""Fast detection only — no text extraction."""
|
||||
...
|
||||
|
||||
def detect_pdf_bytes(data: bytes) -> PdfResult:
|
||||
"""Fast detection from bytes."""
|
||||
...
|
||||
|
||||
def classify_pdf(path: str) -> PdfClassification:
|
||||
"""Lightweight classification — type, page count, and OCR pages (0-indexed)."""
|
||||
...
|
||||
|
||||
def classify_pdf_bytes(data: bytes) -> PdfClassification:
|
||||
"""Lightweight classification from bytes."""
|
||||
...
|
||||
|
||||
def extract_text(path: str) -> str:
|
||||
"""Extract plain text from a PDF."""
|
||||
...
|
||||
|
||||
def extract_text_bytes(data: bytes) -> str:
|
||||
"""Extract plain text from PDF bytes."""
|
||||
...
|
||||
|
||||
def extract_text_with_positions(path: str, pages: Optional[list[int]] = None) -> list[TextItem]:
|
||||
"""Extract text with position information."""
|
||||
...
|
||||
|
||||
def extract_text_with_positions_bytes(data: bytes, pages: Optional[list[int]] = None) -> list[TextItem]:
|
||||
"""Extract text with position information from bytes."""
|
||||
...
|
||||
|
||||
def extract_text_in_regions(
|
||||
path: str,
|
||||
page_regions: list[tuple[int, list[list[float]]]],
|
||||
) -> list[PageRegionTexts]:
|
||||
"""Extract text within bounding-box regions from a PDF file.
|
||||
|
||||
Args:
|
||||
path: Path to the PDF file.
|
||||
page_regions: List of (page_0indexed, [[x1, y1, x2, y2], ...]) tuples.
|
||||
"""
|
||||
...
|
||||
|
||||
def extract_text_in_regions_bytes(
|
||||
data: bytes,
|
||||
page_regions: list[tuple[int, list[list[float]]]],
|
||||
) -> list[PageRegionTexts]:
|
||||
"""Extract text within bounding-box regions from PDF bytes.
|
||||
|
||||
Args:
|
||||
data: PDF file contents as bytes.
|
||||
page_regions: List of (page_0indexed, [[x1, y1, x2, y2], ...]) tuples.
|
||||
"""
|
||||
...
|
||||
@@ -0,0 +1,21 @@
|
||||
[build-system]
|
||||
requires = ["maturin>=1.0,<2.0"]
|
||||
build-backend = "maturin"
|
||||
|
||||
[project]
|
||||
name = "pdf-inspector"
|
||||
version = "0.1.0"
|
||||
description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection"
|
||||
license = { text = "MIT" }
|
||||
requires-python = ">=3.8"
|
||||
classifiers = [
|
||||
"Programming Language :: Rust",
|
||||
"Programming Language :: Python :: Implementation :: CPython",
|
||||
"Programming Language :: Python :: 3",
|
||||
"License :: OSI Approved :: MIT License",
|
||||
"Operating System :: OS Independent",
|
||||
"Topic :: Text Processing",
|
||||
]
|
||||
|
||||
[tool.maturin]
|
||||
features = ["python"]
|
||||
+236
-28
@@ -598,7 +598,15 @@ fn page_has_identity_h_no_tounicode(doc: &Document, page_id: ObjectId) -> bool {
|
||||
if font_dict.get(b"ToUnicode").is_ok() {
|
||||
continue;
|
||||
}
|
||||
// Identity-H/V without ToUnicode — flag it
|
||||
|
||||
// Check if fallback decoding paths can handle this font.
|
||||
// The extraction pipeline tries: TrueType cmap → CIDSystemInfo → passthrough.
|
||||
// If any of these would succeed, the font is decodable — don't flag it.
|
||||
if identity_h_font_has_fallback(font_dict, doc) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Identity-H/V without ToUnicode and no fallback — flag it
|
||||
log::debug!(
|
||||
"page has Identity-H/V font without ToUnicode: {:?}",
|
||||
font_dict
|
||||
@@ -612,6 +620,102 @@ fn page_has_identity_h_no_tounicode(doc: &Document, page_id: ObjectId) -> bool {
|
||||
false
|
||||
}
|
||||
|
||||
/// Check whether an Identity-H font without ToUnicode can still be decoded
|
||||
/// via one of the extraction pipeline's fallback paths.
|
||||
fn identity_h_font_has_fallback(font_dict: &lopdf::Dictionary, doc: &Document) -> bool {
|
||||
let desc_fonts_obj = match font_dict.get(b"DescendantFonts").ok() {
|
||||
Some(obj) => obj,
|
||||
None => return false,
|
||||
};
|
||||
let desc_fonts = match desc_fonts_obj {
|
||||
Object::Array(arr) => arr,
|
||||
Object::Reference(r) => match doc.get_object(*r) {
|
||||
Ok(Object::Array(arr)) => arr,
|
||||
_ => return false,
|
||||
},
|
||||
_ => return false,
|
||||
};
|
||||
if desc_fonts.is_empty() {
|
||||
return false;
|
||||
}
|
||||
let cid_font_dict = match &desc_fonts[0] {
|
||||
Object::Reference(r) => match doc.get_dictionary(*r) {
|
||||
Ok(d) => d,
|
||||
_ => return false,
|
||||
},
|
||||
Object::Dictionary(d) => d,
|
||||
_ => return false,
|
||||
};
|
||||
|
||||
// Fallback 1: W array CIDs look like Unicode codepoints → passthrough works.
|
||||
// Many PDF generators (Chromium, wkhtmltopdf) use Identity-H where CID = Unicode.
|
||||
if crate::tounicode::cid_values_look_like_unicode(cid_font_dict) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Fallback 2: Embedded TrueType/OpenType font has a usable cmap table.
|
||||
if let Some(font_descriptor) = cid_font_dict
|
||||
.get(b"FontDescriptor")
|
||||
.ok()
|
||||
.and_then(|o| match o {
|
||||
Object::Reference(r) => doc.get_dictionary(*r).ok(),
|
||||
Object::Dictionary(d) => Some(d),
|
||||
_ => None,
|
||||
})
|
||||
{
|
||||
let font_file_ref = font_descriptor
|
||||
.get(b"FontFile2")
|
||||
.ok()
|
||||
.and_then(|o| o.as_reference().ok())
|
||||
.or_else(|| {
|
||||
font_descriptor
|
||||
.get(b"FontFile3")
|
||||
.ok()
|
||||
.and_then(|o| o.as_reference().ok())
|
||||
});
|
||||
if let Some(ff_ref) = font_file_ref {
|
||||
if embedded_font_has_cmap(doc, ff_ref) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
false
|
||||
}
|
||||
|
||||
/// Quick check whether an embedded TrueType/OpenType font has a cmap table
|
||||
/// that can map GIDs to Unicode codepoints.
|
||||
fn embedded_font_has_cmap(doc: &Document, font_ref: lopdf::ObjectId) -> bool {
|
||||
let stream = match doc.get_object(font_ref).and_then(Object::as_stream) {
|
||||
Ok(s) => s,
|
||||
Err(_) => return false,
|
||||
};
|
||||
let data = match stream.decompressed_content() {
|
||||
Ok(d) => d,
|
||||
Err(_) => return false,
|
||||
};
|
||||
let face = match ttf_parser::Face::parse(&data, 0) {
|
||||
Ok(f) => f,
|
||||
Err(_) => return false,
|
||||
};
|
||||
// Check that the font has a cmap table with at least some Unicode mappings
|
||||
if let Some(cmap) = face.tables().cmap {
|
||||
for subtable in cmap.subtables {
|
||||
if subtable.is_unicode()
|
||||
|| (subtable.platform_id == ttf_parser::PlatformId::Windows
|
||||
&& subtable.encoding_id == 0)
|
||||
{
|
||||
let mut count = 0u32;
|
||||
subtable.codepoints(|_| count += 1);
|
||||
if count > 0 {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
/// Returns true if every font on the page is Type3 (no normal text fonts).
|
||||
/// Type3 fonts render glyphs as custom drawings/bitmaps. Without a ToUnicode
|
||||
/// CMap, character codes can't be mapped to Unicode — the page needs OCR.
|
||||
@@ -730,7 +834,7 @@ fn scan_content_for_text_operators(
|
||||
unique_chars: &mut HashSet<u8>,
|
||||
) -> (u32, u32, u32, u32) {
|
||||
let mut text_ops = 0u32;
|
||||
let mut image_count = 0u32;
|
||||
let image_count = 0u32;
|
||||
let mut path_ops = 0u32;
|
||||
let mut font_changes = 0u32;
|
||||
|
||||
@@ -771,14 +875,10 @@ fn scan_content_for_text_operators(
|
||||
}
|
||||
}
|
||||
|
||||
// Look for 'Do' operator (XObject/image placement)
|
||||
if b == b'D'
|
||||
&& i + 1 < content.len()
|
||||
&& content[i + 1] == b'o'
|
||||
&& (i + 2 >= content.len() || content[i + 2].is_ascii_whitespace())
|
||||
{
|
||||
image_count += 1;
|
||||
}
|
||||
// Note: We do NOT count 'Do' operators here because Do invokes any
|
||||
// XObject — including Form XObjects that contain text. Actual image
|
||||
// detection is handled by scan_xobjects_in_resources (checks Subtype)
|
||||
// and analyze_page_images (measures pixel area).
|
||||
|
||||
// Count path construction/painting operators.
|
||||
// Single-byte: m (moveto), l (lineto), c (curveto), h (closepath),
|
||||
@@ -1185,23 +1285,24 @@ mod tests {
|
||||
// H, e, l, o = 4 unique
|
||||
assert!(uchars.len() >= 4);
|
||||
|
||||
// Content with Do (image)
|
||||
// Content with Do (XObject invocation — not counted as image here;
|
||||
// actual image detection is handled by scan_xobjects_in_resources)
|
||||
uchars.clear();
|
||||
let content3 = b"q 100 0 0 100 50 700 cm /Img1 Do Q";
|
||||
let (ops3, imgs3, _, _) = scan_content_for_text_operators(content3, &mut uchars);
|
||||
assert_eq!(ops3, 0);
|
||||
assert_eq!(imgs3, 1);
|
||||
assert_eq!(imgs3, 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_image_dominated_detection() {
|
||||
// Simulate a page with many Do operators and minimal text
|
||||
// Do operators are no longer counted as images by scan_content_for_text_operators.
|
||||
// Image-dominated detection now relies on scan_xobjects_in_resources which
|
||||
// checks XObject Subtype. Here we verify that Do operators don't inflate image_count.
|
||||
let mut content = Vec::new();
|
||||
// Add 50 Do operators (image-heavy)
|
||||
for i in 0..50 {
|
||||
content.extend_from_slice(format!("/Im{i} Do\n").as_bytes());
|
||||
}
|
||||
// Add a few text operators with only a bullet char
|
||||
content.extend_from_slice(b"BT (x) Tj ET\n");
|
||||
content.extend_from_slice(b"BT (x) Tj ET\n");
|
||||
content.extend_from_slice(b"BT (x) Tj ET\n");
|
||||
@@ -1209,15 +1310,8 @@ mod tests {
|
||||
let mut uchars = HashSet::new();
|
||||
let (ops, imgs, _, _) = scan_content_for_text_operators(&content, &mut uchars);
|
||||
assert_eq!(ops, 3);
|
||||
assert_eq!(imgs, 50);
|
||||
// Only 'x' unique char
|
||||
assert_eq!(imgs, 0); // Do operators are not counted here
|
||||
assert_eq!(uchars.len(), 1);
|
||||
|
||||
// This should be image-dominated: 50 > 10 && 50 > 3*3=9
|
||||
let is_image_dominated = imgs > 10 && imgs > ops * 3;
|
||||
assert!(is_image_dominated);
|
||||
// And fails unique char threshold
|
||||
assert!(uchars.len() < 5);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -1227,12 +1321,9 @@ mod tests {
|
||||
let mut uchars = HashSet::new();
|
||||
let (ops, imgs, _, _) = scan_content_for_text_operators(content, &mut uchars);
|
||||
assert_eq!(ops, 1);
|
||||
assert_eq!(imgs, 2);
|
||||
// Many unique chars from the sentence
|
||||
assert_eq!(imgs, 0); // Do operators not counted here
|
||||
// Many unique chars from the sentence
|
||||
assert!(uchars.len() >= 5);
|
||||
// Not image-dominated: 2 > 10 fails
|
||||
let is_image_dominated = imgs > 10 && imgs > ops * 3;
|
||||
assert!(!is_image_dominated);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -1396,6 +1487,123 @@ mod tests {
|
||||
assert!(!page_has_identity_h_no_tounicode(&doc, page_id));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_identity_h_with_unicode_cids_not_flagged() {
|
||||
// Type0 Identity-H font without ToUnicode but with W array CIDs
|
||||
// that look like Unicode codepoints (e.g. from Chromium/wkhtmltopdf).
|
||||
// The CID-as-Unicode passthrough can decode these — don't flag.
|
||||
use lopdf::dictionary;
|
||||
let mut doc = Document::with_version("1.4");
|
||||
let pages_id = doc.new_object_id();
|
||||
let page_id = doc.new_object_id();
|
||||
// CIDFont with W array containing Unicode-range CIDs (>= 0x41)
|
||||
let cid_font_id = doc.add_object(dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => Object::Name(b"CIDFontType2".to_vec()),
|
||||
"W" => Object::Array(vec![
|
||||
Object::Integer(0x41), // CID 65 = 'A'
|
||||
Object::Array(vec![
|
||||
Object::Integer(600), Object::Integer(600), Object::Integer(600),
|
||||
]),
|
||||
Object::Integer(0x61), // CID 97 = 'a'
|
||||
Object::Array(vec![
|
||||
Object::Integer(500), Object::Integer(500), Object::Integer(500),
|
||||
]),
|
||||
]),
|
||||
});
|
||||
let font_id = doc.add_object(dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => Object::Name(b"Type0".to_vec()),
|
||||
"BaseFont" => Object::Name(b"ABCDEF+ArialMT".to_vec()),
|
||||
"Encoding" => Object::Name(b"Identity-H".to_vec()),
|
||||
"DescendantFonts" => Object::Array(vec![Object::Reference(cid_font_id)]),
|
||||
});
|
||||
let resources = dictionary! {
|
||||
"Font" => dictionary! {
|
||||
"F1" => Object::Reference(font_id),
|
||||
},
|
||||
};
|
||||
doc.objects.insert(
|
||||
page_id,
|
||||
Object::Dictionary(dictionary! {
|
||||
"Type" => "Page",
|
||||
"Parent" => Object::Reference(pages_id),
|
||||
"Resources" => resources,
|
||||
}),
|
||||
);
|
||||
doc.objects.insert(
|
||||
pages_id,
|
||||
Object::Dictionary(dictionary! {
|
||||
"Type" => "Pages",
|
||||
"Kids" => vec![Object::Reference(page_id)],
|
||||
"Count" => Object::Integer(1),
|
||||
}),
|
||||
);
|
||||
assert!(
|
||||
!page_has_identity_h_no_tounicode(&doc, page_id),
|
||||
"Should NOT flag: W array CIDs look like Unicode, passthrough works"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_identity_h_with_low_gid_cids_still_flagged() {
|
||||
// Type0 Identity-H font without ToUnicode and W array CIDs
|
||||
// that are low GID values (subset font, no cmap). These can't
|
||||
// be decoded — should still be flagged.
|
||||
use lopdf::dictionary;
|
||||
let mut doc = Document::with_version("1.4");
|
||||
let pages_id = doc.new_object_id();
|
||||
let page_id = doc.new_object_id();
|
||||
// CIDFont with W array containing low GID values (< 0x41)
|
||||
let cid_font_id = doc.add_object(dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => Object::Name(b"CIDFontType2".to_vec()),
|
||||
"W" => Object::Array(vec![
|
||||
Object::Integer(3), // Low GID
|
||||
Object::Array(vec![
|
||||
Object::Integer(600), Object::Integer(600), Object::Integer(600),
|
||||
Object::Integer(600), Object::Integer(600),
|
||||
]),
|
||||
Object::Integer(10), // Still low
|
||||
Object::Array(vec![
|
||||
Object::Integer(500), Object::Integer(500), Object::Integer(500),
|
||||
]),
|
||||
]),
|
||||
});
|
||||
let font_id = doc.add_object(dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => Object::Name(b"Type0".to_vec()),
|
||||
"BaseFont" => Object::Name(b"GPBCHP+TimesNewRoman".to_vec()),
|
||||
"Encoding" => Object::Name(b"Identity-H".to_vec()),
|
||||
"DescendantFonts" => Object::Array(vec![Object::Reference(cid_font_id)]),
|
||||
});
|
||||
let resources = dictionary! {
|
||||
"Font" => dictionary! {
|
||||
"F1" => Object::Reference(font_id),
|
||||
},
|
||||
};
|
||||
doc.objects.insert(
|
||||
page_id,
|
||||
Object::Dictionary(dictionary! {
|
||||
"Type" => "Page",
|
||||
"Parent" => Object::Reference(pages_id),
|
||||
"Resources" => resources,
|
||||
}),
|
||||
);
|
||||
doc.objects.insert(
|
||||
pages_id,
|
||||
Object::Dictionary(dictionary! {
|
||||
"Type" => "Pages",
|
||||
"Kids" => vec![Object::Reference(page_id)],
|
||||
"Count" => Object::Integer(1),
|
||||
}),
|
||||
);
|
||||
assert!(
|
||||
page_has_identity_h_no_tounicode(&doc, page_id),
|
||||
"Should flag: low GID CIDs, no cmap, no passthrough"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_scan_content_counts_tf_operators() {
|
||||
let mut uchars = HashSet::new();
|
||||
|
||||
@@ -82,7 +82,7 @@ pub(crate) fn extract_page_text_items(
|
||||
page_num: u32,
|
||||
font_cmaps: &FontCMaps,
|
||||
include_invisible: bool,
|
||||
) -> Result<(PageExtraction, bool), PdfError> {
|
||||
) -> Result<(PageExtraction, bool, bool), PdfError> {
|
||||
use lopdf::content::Content;
|
||||
|
||||
let mut items = Vec::new();
|
||||
@@ -182,7 +182,7 @@ pub(crate) fn extract_page_text_items(
|
||||
content.operations.len(),
|
||||
MAX_OPERATIONS
|
||||
);
|
||||
return Ok(((Vec::new(), Vec::new(), Vec::new()), false));
|
||||
return Ok(((Vec::new(), Vec::new(), Vec::new()), false, false));
|
||||
}
|
||||
|
||||
// Graphics state tracking
|
||||
@@ -1009,11 +1009,12 @@ pub(crate) fn extract_page_text_items(
|
||||
// Some PDFs embed landscape content in portrait pages using a rotated text
|
||||
// matrix (e.g. [0, b, -b, 0, tx, ty] for 90° CCW). The layout engine
|
||||
// assumes x=horizontal, y=vertical — so we swap coordinates to match.
|
||||
let (items, rects, lines) = correct_rotated_page(items, rects, lines, &rotation_votes);
|
||||
let (items, rects, lines, coords_rotated) =
|
||||
correct_rotated_page(items, rects, lines, &rotation_votes);
|
||||
|
||||
let items = super::merge_text_items(items);
|
||||
let items = super::merge_subscript_items(items);
|
||||
Ok(((items, rects, lines), has_gid_fonts))
|
||||
Ok(((items, rects, lines), has_gid_fonts, coords_rotated))
|
||||
}
|
||||
|
||||
/// Counts of text operators with horizontal vs rotated combined matrices.
|
||||
@@ -1030,9 +1031,9 @@ fn correct_rotated_page(
|
||||
mut rects: Vec<PdfRect>,
|
||||
mut lines: Vec<PdfLine>,
|
||||
votes: &RotationVotes,
|
||||
) -> (Vec<TextItem>, Vec<PdfRect>, Vec<PdfLine>) {
|
||||
) -> (Vec<TextItem>, Vec<PdfRect>, Vec<PdfLine>, bool) {
|
||||
if items.len() < 2 {
|
||||
return (items, rects, lines);
|
||||
return (items, rects, lines, false);
|
||||
}
|
||||
|
||||
// Use the combined-matrix direction votes collected during extraction.
|
||||
@@ -1041,7 +1042,7 @@ fn correct_rotated_page(
|
||||
let total_votes = votes.horizontal + votes.rotated;
|
||||
if total_votes == 0 || votes.rotated * 3 < total_votes * 2 {
|
||||
// Less than ~67% of text operators are rotated → not a rotated page
|
||||
return (items, rects, lines);
|
||||
return (items, rects, lines, false);
|
||||
}
|
||||
|
||||
log::debug!(
|
||||
@@ -1092,7 +1093,7 @@ fn correct_rotated_page(
|
||||
line.y2 = new_y2;
|
||||
}
|
||||
|
||||
(items, rects, lines)
|
||||
(items, rects, lines, true)
|
||||
}
|
||||
|
||||
/// Remove near-duplicate rects (same coordinates within 0.5 pt tolerance).
|
||||
@@ -1228,7 +1229,7 @@ mod tests {
|
||||
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let result = extract_page_text_items(&doc, page_id, 1, &font_cmaps, false).unwrap();
|
||||
let ((items, rects, lines), _has_gid) = result;
|
||||
let ((items, rects, lines), _has_gid, _coords_rotated) = result;
|
||||
assert!(items.is_empty());
|
||||
assert!(rects.is_empty());
|
||||
assert!(lines.is_empty());
|
||||
|
||||
+202
-25
@@ -35,6 +35,7 @@ pub(crate) fn detect_columns(
|
||||
if page_items.is_empty() {
|
||||
return vec![];
|
||||
}
|
||||
debug!("page {}: detect_columns: {} items", page, page_items.len());
|
||||
|
||||
// Find page bounds
|
||||
let x_min = page_items.iter().map(|i| i.x).fold(f32::INFINITY, f32::min);
|
||||
@@ -163,10 +164,17 @@ pub(crate) fn detect_columns(
|
||||
}
|
||||
}
|
||||
}
|
||||
// Try XY-cut fallback before giving up
|
||||
if let Some(columns) = try_xy_cut_split(&page_items, x_min, x_max, page) {
|
||||
return columns;
|
||||
}
|
||||
return vec![ColumnRegion { x_min, x_max }];
|
||||
}
|
||||
|
||||
return validate_and_build_columns(
|
||||
// Try center-based assignment first (handles asymmetric layouts / sidebars
|
||||
// better than edge-based). Fall back to edge-based if center produces
|
||||
// a degenerate split (one side empty).
|
||||
let result = validate_and_build_columns(
|
||||
&valleys,
|
||||
&page_items,
|
||||
x_min,
|
||||
@@ -175,8 +183,177 @@ pub(crate) fn detect_columns(
|
||||
MIN_ITEMS_PER_COLUMN,
|
||||
MIN_VERTICAL_SPAN_RATIO,
|
||||
page,
|
||||
false, // edge-based assignment for absolute valleys
|
||||
true, // center-based assignment
|
||||
);
|
||||
if result.len() > 1 {
|
||||
return result;
|
||||
}
|
||||
let result = validate_and_build_columns(
|
||||
&valleys,
|
||||
&page_items,
|
||||
x_min,
|
||||
BIN_WIDTH,
|
||||
x_max,
|
||||
MIN_ITEMS_PER_COLUMN,
|
||||
MIN_VERTICAL_SPAN_RATIO,
|
||||
page,
|
||||
false, // edge-based fallback
|
||||
);
|
||||
if result.len() > 1 {
|
||||
return result;
|
||||
}
|
||||
|
||||
// Fallback: XY-cut style gap detection. When the histogram finds no
|
||||
// clear valleys (common with asymmetric/sidebar layouts), look for the
|
||||
// largest horizontal gap between item edges. This is a simplified
|
||||
// single-level XY-cut inspired by opendataloader's XY-Cut++ algorithm.
|
||||
if page_items.len() >= 20 && !page_has_table {
|
||||
if let Some(columns) = try_xy_cut_split(&page_items, x_min, x_max, page) {
|
||||
return columns;
|
||||
}
|
||||
}
|
||||
|
||||
vec![ColumnRegion { x_min, x_max }]
|
||||
}
|
||||
|
||||
/// Simplified single-level XY-cut: find the largest horizontal gap between
|
||||
/// item right-edges and left-edges. If the gap is wide enough and both sides
|
||||
/// have sufficient items with vertical overlap, split into two columns.
|
||||
///
|
||||
/// Inspired by opendataloader's XY-Cut++ algorithm but without full recursion.
|
||||
/// Handles asymmetric layouts (sidebars) that the histogram misses because
|
||||
/// the narrow column has too few items to register in the occupancy profile.
|
||||
fn try_xy_cut_split(
|
||||
page_items: &[&TextItem],
|
||||
page_x_min: f32,
|
||||
page_x_max: f32,
|
||||
page: u32,
|
||||
) -> Option<Vec<ColumnRegion>> {
|
||||
const MIN_GAP: f32 = 15.0; // minimum gap to consider a split
|
||||
const MIN_ITEMS_MAJOR: usize = 10; // major column must have ≥10 items
|
||||
const MIN_ITEMS_MINOR: usize = 3; // minor column (sidebar) must have ≥3
|
||||
|
||||
let page_width = page_x_max - page_x_min;
|
||||
if page_width < 200.0 {
|
||||
return None;
|
||||
}
|
||||
|
||||
// Collect all item edges: (right_edge, left_edge) pairs sorted by right_edge
|
||||
// The gap between one item's right edge and the next item's left edge
|
||||
// reveals column gutters.
|
||||
let mut edges: Vec<(f32, f32)> = page_items
|
||||
.iter()
|
||||
.map(|i| (i.x, i.x + effective_width(i)))
|
||||
.collect();
|
||||
edges.sort_by(|a, b| a.0.total_cmp(&b.0));
|
||||
|
||||
// Find the largest gap between consecutive items (by left edge).
|
||||
// Use a sweep: sort left edges, find max gap between sorted right edges
|
||||
// of items to the left and left edges of items to the right.
|
||||
let mut left_edges: Vec<f32> = page_items.iter().map(|i| i.x).collect();
|
||||
left_edges.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
// Build prefix max of right edges (for items sorted by left edge)
|
||||
let mut sorted_by_left: Vec<(f32, f32)> = page_items
|
||||
.iter()
|
||||
.map(|i| (i.x, i.x + effective_width(i)))
|
||||
.collect();
|
||||
sorted_by_left.sort_by(|a, b| a.0.total_cmp(&b.0));
|
||||
|
||||
let mut best_gap = 0.0f32;
|
||||
let mut best_split = 0.0f32;
|
||||
let mut max_right_so_far = f32::NEG_INFINITY;
|
||||
|
||||
for i in 0..sorted_by_left.len() - 1 {
|
||||
let (_, right) = sorted_by_left[i];
|
||||
max_right_so_far = max_right_so_far.max(right);
|
||||
|
||||
let (next_left, _) = sorted_by_left[i + 1];
|
||||
let gap = next_left - max_right_so_far;
|
||||
if gap > best_gap {
|
||||
best_gap = gap;
|
||||
best_split = (max_right_so_far + next_left) / 2.0;
|
||||
}
|
||||
}
|
||||
|
||||
if best_gap < MIN_GAP {
|
||||
return None;
|
||||
}
|
||||
|
||||
// Don't split at page margins (within 10% of edges)
|
||||
let margin = page_width * 0.10;
|
||||
if best_split - page_x_min < margin || page_x_max - best_split < margin {
|
||||
return None;
|
||||
}
|
||||
|
||||
// Count items on each side
|
||||
let left_count = page_items
|
||||
.iter()
|
||||
.filter(|i| i.x + effective_width(i) / 2.0 <= best_split)
|
||||
.count();
|
||||
let right_count = page_items
|
||||
.iter()
|
||||
.filter(|i| i.x + effective_width(i) / 2.0 > best_split)
|
||||
.count();
|
||||
|
||||
let (minor, major) = if left_count <= right_count {
|
||||
(left_count, right_count)
|
||||
} else {
|
||||
(right_count, left_count)
|
||||
};
|
||||
|
||||
if major < MIN_ITEMS_MAJOR || minor < MIN_ITEMS_MINOR {
|
||||
return None;
|
||||
}
|
||||
|
||||
// Check vertical overlap — both sides should span a meaningful Y range
|
||||
let left_items: Vec<&&TextItem> = page_items
|
||||
.iter()
|
||||
.filter(|i| i.x + effective_width(i) / 2.0 <= best_split)
|
||||
.collect();
|
||||
let right_items: Vec<&&TextItem> = page_items
|
||||
.iter()
|
||||
.filter(|i| i.x + effective_width(i) / 2.0 > best_split)
|
||||
.collect();
|
||||
|
||||
let l_y_min = left_items.iter().map(|i| i.y).fold(f32::INFINITY, f32::min);
|
||||
let l_y_max = left_items
|
||||
.iter()
|
||||
.map(|i| i.y)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let r_y_min = right_items
|
||||
.iter()
|
||||
.map(|i| i.y)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let r_y_max = right_items
|
||||
.iter()
|
||||
.map(|i| i.y)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
|
||||
let overlap_min = l_y_min.max(r_y_min);
|
||||
let overlap_max = l_y_max.min(r_y_max);
|
||||
let overlap = (overlap_max - overlap_min).max(0.0);
|
||||
let y_range = (l_y_max.max(r_y_max) - l_y_min.min(r_y_min)).max(1.0);
|
||||
|
||||
if overlap / y_range < 0.20 {
|
||||
return None;
|
||||
}
|
||||
|
||||
debug!(
|
||||
"page {}: XY-cut split at x={:.1} (gap={:.1}pt, left={}, right={})",
|
||||
page, best_split, best_gap, left_count, right_count
|
||||
);
|
||||
|
||||
Some(vec![
|
||||
ColumnRegion {
|
||||
x_min: page_x_min,
|
||||
x_max: best_split,
|
||||
},
|
||||
ColumnRegion {
|
||||
x_min: best_split,
|
||||
x_max: page_x_max,
|
||||
},
|
||||
])
|
||||
}
|
||||
|
||||
/// Check whether each proposed column contains paragraph-like content.
|
||||
@@ -218,7 +395,7 @@ fn columns_have_prose(columns: &[ColumnRegion], items: &[&TextItem]) -> bool {
|
||||
|
||||
// Sort by Y descending (top of page = higher Y in PDF coords)
|
||||
let mut sorted: Vec<&TextItem> = col_items;
|
||||
sorted.sort_by(|a, b| b.y.partial_cmp(&a.y).unwrap_or(std::cmp::Ordering::Equal));
|
||||
sorted.sort_by(|a, b| b.y.total_cmp(&a.y));
|
||||
|
||||
// Group into lines by Y-proximity and measure fill + item count
|
||||
let mut full_lines = 0usize;
|
||||
@@ -505,7 +682,15 @@ fn validate_and_build_columns(
|
||||
})
|
||||
.collect();
|
||||
|
||||
if left_items.len() < min_items || right_items.len() < min_items {
|
||||
// Require both sides to have items. Symmetric layout needs min_items
|
||||
// on each side. Asymmetric layouts (sidebars) are accepted when the
|
||||
// dominant side has ≥ min_items and the smaller side has ≥ 3 items.
|
||||
let (smaller, larger) = if left_items.len() <= right_items.len() {
|
||||
(left_items.len(), right_items.len())
|
||||
} else {
|
||||
(right_items.len(), left_items.len())
|
||||
};
|
||||
if larger < min_items || smaller < 3 {
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -616,7 +801,7 @@ fn identify_spanning_lines(items: &[TextItem], columns: &[ColumnRegion]) -> Vec<
|
||||
// Build (original_index, y) pairs sorted by Y descending for grouping
|
||||
let mut indexed: Vec<(usize, f32)> =
|
||||
items.iter().enumerate().map(|(i, it)| (i, it.y)).collect();
|
||||
indexed.sort_by(|a, b| b.1.partial_cmp(&a.1).unwrap_or(std::cmp::Ordering::Equal));
|
||||
indexed.sort_by(|a, b| b.1.total_cmp(&a.1));
|
||||
|
||||
// Group by Y-proximity into rough lines (as index sets)
|
||||
let mut groups: Vec<Vec<usize>> = Vec::new();
|
||||
@@ -778,7 +963,7 @@ pub(crate) fn is_newspaper_layout(
|
||||
return 0.0;
|
||||
}
|
||||
let mut ys: Vec<f32> = lines.iter().map(|l| l.y).collect();
|
||||
ys.sort_by(|a, b| a.partial_cmp(b).unwrap());
|
||||
ys.sort_by(|a, b| a.total_cmp(b));
|
||||
let span = ys.last().unwrap() - ys.first().unwrap();
|
||||
span / (lines.len() as f32 - 1.0)
|
||||
};
|
||||
@@ -845,7 +1030,7 @@ fn split_column_stragglers(lines: Vec<TextLine>) -> (Vec<TextLine>, Vec<TextLine
|
||||
|
||||
// Median gap = typical line spacing
|
||||
let mut sorted_gaps = gaps.clone();
|
||||
sorted_gaps.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
sorted_gaps.sort_by(|a, b| a.total_cmp(b));
|
||||
let median_gap = sorted_gaps[sorted_gaps.len() / 2];
|
||||
|
||||
// A gap > 3× median (min 30pt) indicates a break between content clusters
|
||||
@@ -1078,9 +1263,8 @@ pub(crate) fn group_into_lines_with_thresholds(
|
||||
}
|
||||
}
|
||||
|
||||
above.sort_by(|a, b| b.y.partial_cmp(&a.y).unwrap_or(std::cmp::Ordering::Equal));
|
||||
below_spanning
|
||||
.sort_by(|a, b| b.y.partial_cmp(&a.y).unwrap_or(std::cmp::Ordering::Equal));
|
||||
above.sort_by(|a, b| b.y.total_cmp(&a.y));
|
||||
below_spanning.sort_by(|a, b| b.y.total_cmp(&a.y));
|
||||
|
||||
all_lines.extend(above);
|
||||
for col in core_columns {
|
||||
@@ -1101,16 +1285,13 @@ pub(crate) fn group_into_lines_with_thresholds(
|
||||
|
||||
// Sort by Y descending (top-first), then by X for same-Y lines
|
||||
all_page_lines.sort_by(|a, b| {
|
||||
b.y.partial_cmp(&a.y)
|
||||
.unwrap_or(std::cmp::Ordering::Equal)
|
||||
.then(
|
||||
a.items
|
||||
.first()
|
||||
.map(|i| i.x)
|
||||
.unwrap_or(0.0)
|
||||
.partial_cmp(&b.items.first().map(|i| i.x).unwrap_or(0.0))
|
||||
.unwrap_or(std::cmp::Ordering::Equal),
|
||||
)
|
||||
b.y.total_cmp(&a.y).then(
|
||||
a.items
|
||||
.first()
|
||||
.map(|i| i.x)
|
||||
.unwrap_or(0.0)
|
||||
.total_cmp(&b.items.first().map(|i| i.x).unwrap_or(0.0)),
|
||||
)
|
||||
});
|
||||
|
||||
// Merge lines at the same Y (within tolerance) into single lines
|
||||
@@ -1185,11 +1366,7 @@ fn group_single_column(items: Vec<TextItem>, adaptive_threshold: f32) -> Vec<Tex
|
||||
let items = if use_y_sorting {
|
||||
// Sort by Y descending (top to bottom in PDF coords)
|
||||
let mut sorted = items;
|
||||
sorted.sort_by(|a, b| {
|
||||
b.y.partial_cmp(&a.y)
|
||||
.unwrap_or(std::cmp::Ordering::Equal)
|
||||
.then(a.x.partial_cmp(&b.x).unwrap_or(std::cmp::Ordering::Equal))
|
||||
});
|
||||
sorted.sort_by(|a, b| b.y.total_cmp(&a.y).then(a.x.total_cmp(&b.x)));
|
||||
sorted
|
||||
} else {
|
||||
items
|
||||
|
||||
@@ -188,7 +188,7 @@ fn extract_positioned_text_impl(
|
||||
continue;
|
||||
}
|
||||
}
|
||||
let ((mut items, rects, lines), has_gid_fonts) =
|
||||
let ((mut items, rects, lines), has_gid_fonts, _coords_rotated) =
|
||||
extract_page_text_items(doc, page_id, *page_num, font_cmaps, include_invisible)?;
|
||||
if has_gid_fonts {
|
||||
gid_encoded_pages.insert(*page_num);
|
||||
@@ -334,17 +334,14 @@ pub(crate) fn merge_text_items(items: Vec<TextItem>) -> Vec<TextItem> {
|
||||
for (_, _, group) in &mut line_groups {
|
||||
let rtl = is_rtl_text(group.iter().map(|i| &i.text));
|
||||
if rtl {
|
||||
group.sort_by(|a, b| b.x.partial_cmp(&a.x).unwrap_or(std::cmp::Ordering::Equal));
|
||||
group.sort_by(|a, b| b.x.total_cmp(&a.x));
|
||||
} else {
|
||||
group.sort_by(|a, b| a.x.partial_cmp(&b.x).unwrap_or(std::cmp::Ordering::Equal));
|
||||
group.sort_by(|a, b| a.x.total_cmp(&b.x));
|
||||
}
|
||||
}
|
||||
|
||||
// Sort groups by page then Y descending (top of page first)
|
||||
line_groups.sort_by(|a, b| {
|
||||
a.0.cmp(&b.0)
|
||||
.then_with(|| b.1.partial_cmp(&a.1).unwrap_or(std::cmp::Ordering::Equal))
|
||||
});
|
||||
line_groups.sort_by(|a, b| a.0.cmp(&b.0).then_with(|| b.1.total_cmp(&a.1)));
|
||||
|
||||
let mut merged = Vec::new();
|
||||
|
||||
@@ -452,7 +449,7 @@ pub(crate) fn merge_subscript_items(items: Vec<TextItem>) -> Vec<TextItem> {
|
||||
|
||||
for (_, _, mut group) in line_groups {
|
||||
// Sort by X position
|
||||
group.sort_by(|a, b| a.x.partial_cmp(&b.x).unwrap_or(std::cmp::Ordering::Equal));
|
||||
group.sort_by(|a, b| a.x.total_cmp(&b.x));
|
||||
|
||||
// Find the dominant (most common) font size in this group
|
||||
let max_fs = group.iter().map(|i| i.font_size).fold(0.0_f32, f32::max);
|
||||
|
||||
+848
-60
File diff suppressed because it is too large
Load Diff
@@ -8,6 +8,24 @@ use log::debug;
|
||||
/// Font statistics for a document
|
||||
pub(crate) struct FontStats {
|
||||
pub(crate) most_common_size: f32,
|
||||
/// Font size frequency distribution (size_key → line count).
|
||||
/// Used for rarity-based heading detection.
|
||||
pub(crate) size_counts: HashMap<i32, usize>,
|
||||
/// Total number of lines counted.
|
||||
pub(crate) total_lines: usize,
|
||||
}
|
||||
|
||||
/// Compute how rare a font size is in the document (0.0 = most common, 1.0 = unique).
|
||||
/// Mirrors opendataloader's font rarity boosting approach: heading fonts appear on
|
||||
/// far fewer lines than body text, so their percentile rank is high.
|
||||
pub(crate) fn font_size_rarity(font_size: f32, stats: &FontStats) -> f32 {
|
||||
if stats.total_lines == 0 {
|
||||
return 0.0;
|
||||
}
|
||||
let key = (font_size * 10.0) as i32;
|
||||
let count = stats.size_counts.get(&key).copied().unwrap_or(0);
|
||||
// Rarity = 1 - (frequency ratio). A size used on 1/100 lines has rarity ~0.99.
|
||||
1.0 - (count as f32 / stats.total_lines as f32)
|
||||
}
|
||||
|
||||
/// Calculate font stats directly from items (before grouping into lines)
|
||||
@@ -21,6 +39,8 @@ pub(crate) fn calculate_font_stats_from_items(items: &[TextItem]) -> FontStats {
|
||||
}
|
||||
}
|
||||
|
||||
let total_lines = size_counts.values().sum();
|
||||
|
||||
// Break ties by preferring the smaller font size for deterministic output
|
||||
let most_common_size = size_counts
|
||||
.iter()
|
||||
@@ -30,7 +50,11 @@ pub(crate) fn calculate_font_stats_from_items(items: &[TextItem]) -> FontStats {
|
||||
.map(|(size, _)| *size as f32 / 10.0)
|
||||
.unwrap_or(12.0);
|
||||
|
||||
FontStats { most_common_size }
|
||||
FontStats {
|
||||
most_common_size,
|
||||
size_counts,
|
||||
total_lines,
|
||||
}
|
||||
}
|
||||
|
||||
/// Calculate font stats from grouped lines
|
||||
@@ -48,6 +72,8 @@ pub(crate) fn calculate_font_stats(lines: &[TextLine]) -> FontStats {
|
||||
}
|
||||
}
|
||||
|
||||
let total_lines = size_counts.values().sum();
|
||||
|
||||
// Break ties by preferring the smaller font size for deterministic output
|
||||
let most_common_size = size_counts
|
||||
.iter()
|
||||
@@ -57,7 +83,23 @@ pub(crate) fn calculate_font_stats(lines: &[TextLine]) -> FontStats {
|
||||
.map(|(size, _)| *size as f32 / 10.0)
|
||||
.unwrap_or(12.0);
|
||||
|
||||
FontStats { most_common_size }
|
||||
FontStats {
|
||||
most_common_size,
|
||||
size_counts,
|
||||
total_lines,
|
||||
}
|
||||
}
|
||||
|
||||
/// Determine the heading level for a bold-only line that didn't meet the font-size
|
||||
/// threshold. These are common in academic papers where section headings are bold
|
||||
/// at the same size as body text.
|
||||
///
|
||||
/// Returns a level below the lowest font-size tier (or H2 when no tiers exist).
|
||||
pub(crate) fn bold_heading_level(heading_tiers: &[f32]) -> usize {
|
||||
let level = heading_tiers.len() + 1;
|
||||
// Clamp to 1..=6 — if no font-size tiers, bold headings become H2
|
||||
// (H1 is reserved for titles which are typically larger)
|
||||
level.clamp(2, 6)
|
||||
}
|
||||
|
||||
/// Detect TOC-style lines that contain dot leaders (e.g., "Section Name .... 42").
|
||||
@@ -121,7 +163,7 @@ pub(crate) fn compute_paragraph_threshold(lines: &[TextLine], base_size: f32) ->
|
||||
return fallback;
|
||||
}
|
||||
|
||||
gaps.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
gaps.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
let median = gaps[gaps.len() / 2];
|
||||
|
||||
@@ -221,7 +263,7 @@ pub(crate) fn compute_heading_tiers(lines: &[TextLine], base_size: f32) -> Vec<f
|
||||
}
|
||||
|
||||
// Sort descending
|
||||
heading_sizes.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal));
|
||||
heading_sizes.sort_by(|a, b| b.total_cmp(a));
|
||||
|
||||
// Cluster sizes within 0.5pt into same tier (use first value as representative)
|
||||
let mut tiers: Vec<f32> = Vec::new();
|
||||
|
||||
@@ -4,13 +4,11 @@
|
||||
pub(crate) fn is_caption_line(text: &str) -> bool {
|
||||
let trimmed = text.trim();
|
||||
|
||||
// Common caption prefixes in multiple languages
|
||||
let caption_prefixes = [
|
||||
"Figure ",
|
||||
// Caption prefixes that always match (always followed by identifiers)
|
||||
let always_prefixes = [
|
||||
"Figura ",
|
||||
"Fig. ",
|
||||
"Fig ",
|
||||
"Table ",
|
||||
"Tabela ",
|
||||
"Source:",
|
||||
"Fonte:",
|
||||
@@ -27,17 +25,39 @@ pub(crate) fn is_caption_line(text: &str) -> bool {
|
||||
"Photo ",
|
||||
"Foto ",
|
||||
];
|
||||
|
||||
// Check if line starts with a caption prefix
|
||||
for prefix in &caption_prefixes {
|
||||
for prefix in &always_prefixes {
|
||||
if trimmed.starts_with(prefix) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// Check case-insensitive patterns
|
||||
// "Figure" and "Table" need a digit/reference after them to distinguish
|
||||
// captions ("Table 1", "Figure 3.2") from headings ("Table of Contents")
|
||||
for prefix in ["Figure ", "Table "] {
|
||||
if let Some(rest) = trimmed.strip_prefix(prefix) {
|
||||
if rest
|
||||
.trim_start()
|
||||
.starts_with(|c: char| c.is_ascii_digit() || c == '(' || c == '#')
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Check case-insensitive patterns — require digit or punctuation after
|
||||
// prefix to avoid matching "Table of Contents" or "Figure drawing" etc.
|
||||
let lower = trimmed.to_lowercase();
|
||||
if lower.starts_with("figure ") || lower.starts_with("table ") || lower.starts_with("source:") {
|
||||
for pfx in ["figure ", "table "] {
|
||||
if let Some(rest) = lower.strip_prefix(pfx) {
|
||||
if rest
|
||||
.trim_start()
|
||||
.starts_with(|c: char| c.is_ascii_digit() || c == '(' || c == '#')
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
if lower.starts_with("source:") {
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
+343
-9
@@ -1,19 +1,152 @@
|
||||
//! Core line-to-markdown conversion loop with table/image interleaving.
|
||||
|
||||
use std::collections::HashSet;
|
||||
use std::collections::{HashMap, HashSet};
|
||||
|
||||
use crate::structure_tree::StructRole;
|
||||
use crate::types::TextLine;
|
||||
|
||||
use super::analysis::{
|
||||
calculate_font_stats, compute_heading_tiers, compute_paragraph_threshold, detect_header_level,
|
||||
has_dot_leaders,
|
||||
bold_heading_level, calculate_font_stats, compute_heading_tiers, compute_paragraph_threshold,
|
||||
detect_header_level, font_size_rarity, has_dot_leaders,
|
||||
};
|
||||
use super::classify::{format_list_item, is_caption_line, is_list_item, is_monospace_font};
|
||||
use super::postprocess::clean_markdown;
|
||||
use super::preprocess::{merge_drop_caps, merge_heading_lines};
|
||||
use super::MarkdownOptions;
|
||||
|
||||
/// Pre-scan struct heading tags to find levels that are overused — i.e., tagged on
|
||||
/// so many lines that they clearly represent body text, not real headings.
|
||||
/// Returns the set of heading levels (1–6) that should be suppressed.
|
||||
///
|
||||
/// Some PDFs (e.g. British Academy grant guidance) tag every numbered paragraph
|
||||
/// line as H2, producing hundreds of false headings. We detect this by checking
|
||||
/// if any heading level accounts for >25% of tagged lines.
|
||||
fn detect_overused_struct_heading_levels(
|
||||
lines: &[TextLine],
|
||||
struct_roles: Option<
|
||||
&std::collections::HashMap<u32, std::collections::HashMap<i64, StructRole>>,
|
||||
>,
|
||||
) -> HashSet<usize> {
|
||||
let mut overused = HashSet::new();
|
||||
let Some(roles) = struct_roles else {
|
||||
return overused;
|
||||
};
|
||||
|
||||
let mut level_counts: HashMap<usize, usize> = HashMap::new();
|
||||
let mut total = 0usize;
|
||||
|
||||
for line in lines {
|
||||
if let Some(role) = resolve_line_struct_role(line, roles) {
|
||||
total += 1;
|
||||
if let Some(level) = struct_role_heading_level(&role) {
|
||||
*level_counts.entry(level).or_insert(0) += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if total < 20 {
|
||||
return overused;
|
||||
}
|
||||
|
||||
for (&level, &count) in &level_counts {
|
||||
let ratio = count as f32 / total as f32;
|
||||
if ratio > 0.15 {
|
||||
log::debug!(
|
||||
"struct heading H{} overused: {}/{} lines ({:.0}%), suppressing",
|
||||
level,
|
||||
count,
|
||||
total,
|
||||
ratio * 100.0
|
||||
);
|
||||
overused.insert(level);
|
||||
}
|
||||
}
|
||||
|
||||
overused
|
||||
}
|
||||
|
||||
/// Pre-scan lines to find "isolated" ones: short lines with paragraph breaks both
|
||||
/// before and after. These are heading candidates even at body font size — common
|
||||
/// in academic papers ("Acknowledgements", "B.3 Prompt Engineering").
|
||||
fn find_isolated_lines(lines: &[TextLine], base_size: f32, para_threshold: f32) -> HashSet<usize> {
|
||||
let mut set = HashSet::new();
|
||||
for i in 0..lines.len() {
|
||||
let line = &lines[i];
|
||||
let plain = line.text();
|
||||
let trimmed = plain.trim();
|
||||
let word_count = trimmed.split_whitespace().count();
|
||||
if !(1..=6).contains(&word_count) || trimmed.len() <= 3 {
|
||||
continue;
|
||||
}
|
||||
let font_size = line.items.first().map(|it| it.font_size).unwrap_or(0.0);
|
||||
if font_size < base_size * 0.95 {
|
||||
continue;
|
||||
}
|
||||
if is_list_item(trimmed) || is_caption_line(trimmed) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Reject lines that look like wrapped paragraph text:
|
||||
// ends with hyphen, comma, preposition, or lowercase continuation
|
||||
let last_char = trimmed.chars().last().unwrap_or(' ');
|
||||
if last_char == '-' || last_char == ',' || last_char == ';' {
|
||||
continue;
|
||||
}
|
||||
// Last word is a common continuation word → wrapped paragraph
|
||||
let last_word = trimmed.split_whitespace().last().unwrap_or("");
|
||||
let continuation_words = [
|
||||
"the", "a", "an", "and", "or", "of", "in", "to", "for", "with", "by", "on", "at",
|
||||
"from", "as", "is", "are", "was", "were", "be", "that", "this", "their", "its", "our",
|
||||
"your", "has", "have", "had", "not",
|
||||
];
|
||||
if continuation_words.contains(&last_word.to_lowercase().as_str()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Paragraph break BEFORE
|
||||
let break_before = if i == 0 {
|
||||
true
|
||||
} else {
|
||||
let prev = &lines[i - 1];
|
||||
prev.page != line.page || (prev.y - line.y).abs() > para_threshold
|
||||
};
|
||||
|
||||
// Paragraph break AFTER
|
||||
let break_after = if i + 1 >= lines.len() {
|
||||
true
|
||||
} else {
|
||||
let next = &lines[i + 1];
|
||||
next.page != line.page || (line.y - next.y).abs() > para_threshold
|
||||
};
|
||||
|
||||
if !break_before || !break_after {
|
||||
continue;
|
||||
}
|
||||
|
||||
set.insert(i);
|
||||
}
|
||||
|
||||
// Density guard: if too many lines on a page are "isolated", they're
|
||||
// all paragraph lines in a multi-column layout, not headings. Real
|
||||
// headings are rare — at most ~20% of lines on a page.
|
||||
let mut page_line_counts: HashMap<u32, (usize, usize)> = HashMap::new(); // (total, isolated)
|
||||
for (i, line) in lines.iter().enumerate() {
|
||||
let entry = page_line_counts.entry(line.page).or_insert((0, 0));
|
||||
entry.0 += 1;
|
||||
if set.contains(&i) {
|
||||
entry.1 += 1;
|
||||
}
|
||||
}
|
||||
for (&page, &(total, isolated)) in &page_line_counts {
|
||||
if total > 0 && isolated as f32 / total as f32 > 0.25 {
|
||||
// Too many isolated lines on this page — remove them all
|
||||
set.retain(|&i| lines[i].page != page);
|
||||
}
|
||||
}
|
||||
|
||||
set
|
||||
}
|
||||
|
||||
/// Resolve the dominant structure role for a text line by looking up its items' MCIDs.
|
||||
///
|
||||
/// Returns the first non-container role found (skipping Document/Part/Sect/Div/NonStruct/Span).
|
||||
@@ -256,6 +389,16 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
// threshold and cause every line to be treated as a paragraph break.
|
||||
let para_threshold = compute_paragraph_threshold(&lines, base_size);
|
||||
|
||||
// Pre-scan: identify isolated lines (paragraph break before AND after).
|
||||
// These are heading candidates even without bold/large font — common in
|
||||
// academic papers where section titles like "Acknowledgements" sit alone
|
||||
// between paragraphs at body font size. Inspired by opendataloader's
|
||||
// lookahead in HeadingProcessor (prevNode/nextNode context).
|
||||
let isolated_lines = find_isolated_lines(&lines, base_size, para_threshold);
|
||||
|
||||
// Detect struct heading levels that are overused (body text mistagged as headings)
|
||||
let overused_heading_levels = detect_overused_struct_heading_levels(&lines, struct_roles);
|
||||
|
||||
let mut output = String::new();
|
||||
let mut current_page = 0u32;
|
||||
let mut prev_y = f32::MAX;
|
||||
@@ -277,7 +420,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
all_content_pages.sort();
|
||||
all_content_pages.dedup();
|
||||
|
||||
for line in lines {
|
||||
for (line_idx, line) in lines.iter().enumerate() {
|
||||
// Page break
|
||||
if line.page != current_page {
|
||||
// Flush current page's remaining tables and images
|
||||
@@ -405,7 +548,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
|
||||
// Detect figure/table captions and source citations
|
||||
// These should be on their own line followed by a paragraph break
|
||||
let struct_role = struct_roles.and_then(|roles| resolve_line_struct_role(&line, roles));
|
||||
let struct_role = struct_roles.and_then(|roles| resolve_line_struct_role(line, roles));
|
||||
|
||||
// Determine if this line is code (struct-tree or font-based) for block accumulation
|
||||
let is_code_line = struct_role
|
||||
@@ -437,13 +580,51 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
// Structure roles ADD headings (e.g. same-size text tagged H2) but do NOT
|
||||
// suppress headings that the font heuristic would detect (some tagged PDFs
|
||||
// mark obvious headings as P or Span).
|
||||
let struct_heading = struct_role.as_ref().and_then(struct_role_heading_level);
|
||||
let struct_heading = struct_role
|
||||
.as_ref()
|
||||
.and_then(struct_role_heading_level)
|
||||
.filter(|level| !overused_heading_levels.contains(level));
|
||||
let heuristic_heading = if options.detect_headers
|
||||
&& plain_trimmed.len() > 3
|
||||
&& plain_trimmed.split_whitespace().count() <= 15
|
||||
{
|
||||
let line_font_size = line.items.first().map(|i| i.font_size).unwrap_or(base_size);
|
||||
detect_header_level(line_font_size, base_size, &heading_tiers)
|
||||
detect_header_level(line_font_size, base_size, &heading_tiers).or_else(|| {
|
||||
// Rarity-based heading detection (inspired by opendataloader).
|
||||
// Heading probability scoring with lookahead context.
|
||||
// Score = rarity * 0.5 + bold * 0.3 + standalone * 0.2
|
||||
// + isolated * 0.3 (paragraph break before AND after)
|
||||
// Only consider lines at or above body font size.
|
||||
if line_font_size < base_size * 0.95 {
|
||||
return None;
|
||||
}
|
||||
let word_count = plain_trimmed.split_whitespace().count();
|
||||
if !(1..=15).contains(&word_count) {
|
||||
return None;
|
||||
}
|
||||
let rarity = font_size_rarity(line_font_size, &font_stats);
|
||||
let all_bold = !line.items.is_empty() && line.items.iter().all(|i| i.is_bold);
|
||||
let standalone = !in_paragraph;
|
||||
let isolated = isolated_lines.contains(&line_idx);
|
||||
|
||||
let score = rarity * 0.5
|
||||
+ if all_bold { 0.3 } else { 0.0 }
|
||||
+ if standalone { 0.2 } else { 0.0 }
|
||||
+ if isolated { 0.3 } else { 0.0 };
|
||||
|
||||
// Require standalone + at least one strong signal.
|
||||
// Non-bold, non-isolated lines need very high rarity (≥0.97)
|
||||
// to avoid classifying ordinary body text as headings in
|
||||
// multi-column layouts where column switches break
|
||||
// paragraph continuity and minor font-size variation
|
||||
// inflates rarity scores.
|
||||
let has_strong_signal = all_bold || isolated || (rarity >= 0.97 && word_count <= 8);
|
||||
if score >= 0.5 && standalone && word_count >= 2 && has_strong_signal {
|
||||
Some(bold_heading_level(&heading_tiers))
|
||||
} else {
|
||||
None
|
||||
}
|
||||
})
|
||||
} else {
|
||||
None
|
||||
};
|
||||
@@ -626,6 +807,8 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
// Compute the typical line spacing for paragraph break detection
|
||||
let para_threshold = compute_paragraph_threshold(&lines, base_size);
|
||||
|
||||
let isolated_lines = find_isolated_lines(&lines, base_size, para_threshold);
|
||||
|
||||
let mut output = String::new();
|
||||
let mut current_page = 0u32;
|
||||
let mut prev_y = f32::MAX;
|
||||
@@ -634,7 +817,7 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
let mut last_list_x: Option<f32> = None;
|
||||
let mut prev_had_dot_leaders = false;
|
||||
|
||||
for line in lines {
|
||||
for (line_idx, line) in lines.iter().enumerate() {
|
||||
// Page break
|
||||
if line.page != current_page {
|
||||
if current_page > 0 {
|
||||
@@ -699,7 +882,27 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
{
|
||||
let line_font_size = line.items.first().map(|i| i.font_size).unwrap_or(base_size);
|
||||
if let Some(header_level) =
|
||||
detect_header_level(line_font_size, base_size, &heading_tiers)
|
||||
detect_header_level(line_font_size, base_size, &heading_tiers).or_else(|| {
|
||||
if line_font_size < base_size * 0.95 {
|
||||
return None;
|
||||
}
|
||||
let word_count = plain_trimmed.split_whitespace().count();
|
||||
if !(1..=15).contains(&word_count) {
|
||||
return None;
|
||||
}
|
||||
let rarity = font_size_rarity(line_font_size, &font_stats);
|
||||
let all_bold = !line.items.is_empty() && line.items.iter().all(|i| i.is_bold);
|
||||
let standalone = !in_paragraph;
|
||||
let isolated = isolated_lines.contains(&line_idx);
|
||||
let score = rarity * 0.5
|
||||
+ if all_bold { 0.3 } else { 0.0 }
|
||||
+ if standalone { 0.2 } else { 0.0 }
|
||||
+ if isolated { 0.3 } else { 0.0 };
|
||||
if score >= 0.5 && standalone && word_count >= 2 {
|
||||
return Some(bold_heading_level(&heading_tiers));
|
||||
}
|
||||
None
|
||||
})
|
||||
{
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
@@ -1016,6 +1219,62 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_rarity_heading_requires_strong_signal() {
|
||||
// Simulate a two-column academic paper where body text lines become
|
||||
// "standalone" due to column switches. Body text at the same font
|
||||
// size as most of the document should NOT be classified as headings
|
||||
// just because of moderate rarity + standalone.
|
||||
//
|
||||
// Regression: previously, lines with rarity ~0.62 and standalone=true
|
||||
// scored 0.51 (>=0.5 threshold), producing hundreds of false ## headings.
|
||||
|
||||
// Create many body-text lines at font_size=10.9 (most common)
|
||||
let mut lines = Vec::new();
|
||||
for i in 0..20 {
|
||||
let mut item = make_item("This is ordinary body text in a paragraph.", 1, None);
|
||||
item.font_size = 10.9;
|
||||
item.y = 700.0 - i as f32 * 14.0;
|
||||
lines.push(make_line(vec![item]));
|
||||
}
|
||||
// A few lines at a slightly different size (simulating column B text)
|
||||
for i in 0..10 {
|
||||
let mut item = make_item("Another body text line from the second column.", 1, None);
|
||||
item.font_size = 11.0; // slightly different → non-zero rarity
|
||||
item.y = 700.0 - i as f32 * 14.0;
|
||||
item.x = 320.0; // right column
|
||||
lines.push(make_line(vec![item]));
|
||||
}
|
||||
// One genuine bold heading
|
||||
let mut heading_item = make_item("3 Philosophical Perspectives", 1, None);
|
||||
heading_item.font_size = 10.9;
|
||||
heading_item.is_bold = true;
|
||||
heading_item.y = 200.0;
|
||||
lines.push(make_line(vec![heading_item]));
|
||||
|
||||
let md = to_markdown_from_lines_with_tables_and_images(
|
||||
lines,
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
None,
|
||||
);
|
||||
|
||||
// The bold heading should be detected
|
||||
assert!(
|
||||
md.contains("## 3 Philosophical Perspectives"),
|
||||
"Bold heading should be detected: {md}"
|
||||
);
|
||||
|
||||
// Body text lines should NOT be headings
|
||||
let heading_count = md.lines().filter(|l| l.starts_with("##")).count();
|
||||
assert!(
|
||||
heading_count <= 2,
|
||||
"Expected at most 2 headings but found {heading_count} in:\n{md}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_struct_role_code_multiline_accumulation() {
|
||||
let mut line1 = make_item("fn main() {", 1, Some(0));
|
||||
@@ -1058,4 +1317,79 @@ mod tests {
|
||||
"Should not have adjacent close/open fences: {md}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_overused_struct_heading_suppressed() {
|
||||
// Simulate a PDF where H2 is mistagged on body text lines.
|
||||
// 30 lines total: 5 tagged H1 (real headings), 20 tagged H2 (mistagged body),
|
||||
// 5 tagged P.
|
||||
let mut lines = Vec::new();
|
||||
let mut page_roles = HashMap::new();
|
||||
let mut mcid = 0i64;
|
||||
|
||||
for i in 0..30 {
|
||||
let mut item = make_item(&format!("Line {i}"), 1, Some(mcid));
|
||||
item.y = 700.0 - (i as f32 * 15.0);
|
||||
lines.push(make_line(vec![item]));
|
||||
|
||||
let role = if i < 5 {
|
||||
StructRole::H1
|
||||
} else if i < 25 {
|
||||
StructRole::H2
|
||||
} else {
|
||||
StructRole::P
|
||||
};
|
||||
page_roles.insert(mcid, role);
|
||||
mcid += 1;
|
||||
}
|
||||
|
||||
let mut roles = HashMap::new();
|
||||
roles.insert(1u32, page_roles);
|
||||
|
||||
let overused = detect_overused_struct_heading_levels(&lines, Some(&roles));
|
||||
// H2 is on 20/30 = 67% of lines — should be suppressed
|
||||
assert!(
|
||||
overused.contains(&2),
|
||||
"H2 should be detected as overused: {:?}",
|
||||
overused
|
||||
);
|
||||
// H1 is on 5/30 = 17% — should also be suppressed at >15% threshold
|
||||
assert!(
|
||||
overused.contains(&1),
|
||||
"H1 at 17% should also be suppressed: {:?}",
|
||||
overused
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_normal_struct_headings_not_suppressed() {
|
||||
// Normal document: a few headings, mostly body text
|
||||
let mut lines = Vec::new();
|
||||
let mut page_roles = HashMap::new();
|
||||
let mut mcid = 0i64;
|
||||
|
||||
for i in 0..50 {
|
||||
let mut item = make_item(&format!("Line {i}"), 1, Some(mcid));
|
||||
item.y = 700.0 - (i as f32 * 14.0);
|
||||
lines.push(make_line(vec![item]));
|
||||
|
||||
let role = if i % 10 == 0 {
|
||||
StructRole::H1 // 5 headings out of 50 = 10%
|
||||
} else {
|
||||
StructRole::P
|
||||
};
|
||||
page_roles.insert(mcid, role);
|
||||
mcid += 1;
|
||||
}
|
||||
|
||||
let mut roles = HashMap::new();
|
||||
roles.insert(1u32, page_roles);
|
||||
|
||||
let overused = detect_overused_struct_heading_levels(&lines, Some(&roles));
|
||||
assert!(
|
||||
overused.is_empty(),
|
||||
"No heading level should be overused: {:?}",
|
||||
overused
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
+74
-9
@@ -42,7 +42,7 @@ pub(crate) fn split_side_by_side(items: &[TextItem]) -> Vec<(f32, f32)> {
|
||||
|
||||
// Sort items by left edge
|
||||
let mut xs: Vec<f32> = items.iter().map(|i| i.x).collect();
|
||||
xs.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
xs.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
// Find all candidate gaps: ≥30pt, in the middle 60% of the X range,
|
||||
// with ≥20 items on each side.
|
||||
@@ -118,7 +118,7 @@ pub(crate) fn split_side_by_side(items: &[TextItem]) -> Vec<(f32, f32)> {
|
||||
})
|
||||
.copied()
|
||||
.collect();
|
||||
balanced_positions.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
balanced_positions.sort_by(|a, b| a.total_cmp(b));
|
||||
balanced_positions.dedup_by(|a, b| (*a - *b).abs() < 50.0);
|
||||
if balanced_positions.len() > 1 {
|
||||
return vec![];
|
||||
@@ -209,7 +209,7 @@ fn split_from_hint_regions(items: &[TextItem], rects: &[PdfRect], page: u32) ->
|
||||
|
||||
// Width outlier filter (same as detect_tables_from_rects)
|
||||
let mut widths: Vec<f32> = page_rects.iter().map(|&(_, _, w, _)| w).collect();
|
||||
widths.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
widths.sort_by(|a, b| a.total_cmp(b));
|
||||
let median_width = widths[widths.len() / 2];
|
||||
page_rects.retain(|&(_, _, w, _)| w <= median_width * 10.0);
|
||||
|
||||
@@ -601,6 +601,15 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
let group = page_groups.get(&page).unwrap();
|
||||
let page_items: Vec<TextItem> = group.iter().map(|(_, item)| (*item).clone()).collect();
|
||||
|
||||
// Detect columns early — on multi-column pages, the merged-band retry
|
||||
// should skip body-font heuristic table detection (which mistakes column
|
||||
// text for tables). Individual band heuristic detection is left enabled
|
||||
// because bands are scoped to single columns.
|
||||
let page_has_columns = {
|
||||
let cols = crate::extractor::detect_columns(&page_items, page, false);
|
||||
cols.len() >= 2
|
||||
};
|
||||
|
||||
// Check for side-by-side layout (e.g. two tables placed left and right)
|
||||
let mut bands = split_side_by_side(&page_items);
|
||||
// Fallback: use rect hint regions to detect side-by-side layout
|
||||
@@ -873,10 +882,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
run_heuristic(&unclaimed_items, &unclaimed_map, 6);
|
||||
}
|
||||
|
||||
// 4. Column-based table detection: last resort for borderless tabular
|
||||
// layouts (e.g. exam/reference grids) when ALL structural methods
|
||||
// found nothing. Only runs when no rects/lines exist (truly borderless)
|
||||
// and no other detection method found tables in this band.
|
||||
// 4. Column-based table detection for borderless tabular layouts.
|
||||
let band_has_tables = band_items.iter().enumerate().any(|(idx, _)| {
|
||||
band_index_map
|
||||
.get(idx)
|
||||
@@ -903,6 +909,65 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
}
|
||||
}
|
||||
|
||||
// 5. Thin-rect border synthesis: last resort for PDFs that draw table
|
||||
// borders as thin filled rectangles (common in spreadsheet exports).
|
||||
// Only runs when ALL other methods found nothing on this page.
|
||||
if !page_tables.contains_key(&page) {
|
||||
let page_rects: Vec<&crate::types::PdfRect> =
|
||||
rects.iter().filter(|r| r.page == page).collect();
|
||||
let mut synth_lines: Vec<crate::types::PdfLine> = Vec::new();
|
||||
for r in &page_rects {
|
||||
let (mut w, mut h) = (r.width, r.height);
|
||||
let (mut x, mut y) = (r.x, r.y);
|
||||
if w < 0.0 {
|
||||
x += w;
|
||||
w = -w;
|
||||
}
|
||||
if h < 0.0 {
|
||||
y += h;
|
||||
h = -h;
|
||||
}
|
||||
if h < 2.0 && w >= 10.0 {
|
||||
let mid_y = y + h / 2.0;
|
||||
synth_lines.push(crate::types::PdfLine {
|
||||
x1: x,
|
||||
y1: mid_y,
|
||||
x2: x + w,
|
||||
y2: mid_y,
|
||||
page,
|
||||
});
|
||||
} else if w < 2.0 && h >= 10.0 {
|
||||
let mid_x = x + w / 2.0;
|
||||
synth_lines.push(crate::types::PdfLine {
|
||||
x1: mid_x,
|
||||
y1: y,
|
||||
x2: mid_x,
|
||||
y2: y + h,
|
||||
page,
|
||||
});
|
||||
}
|
||||
}
|
||||
if synth_lines.len() >= 10 {
|
||||
let page_text: Vec<TextItem> = text_items
|
||||
.iter()
|
||||
.filter(|i| i.page == page)
|
||||
.cloned()
|
||||
.collect();
|
||||
let line_tables = detect_tables_from_lines(&page_text, &synth_lines, page);
|
||||
for table in &line_tables {
|
||||
for &idx in &table.item_indices {
|
||||
table_items.insert(idx);
|
||||
}
|
||||
let table_y = table.rows.first().copied().unwrap_or(0.0);
|
||||
let table_md = table_to_markdown(table);
|
||||
page_tables
|
||||
.entry(page)
|
||||
.or_default()
|
||||
.push((table_y, table_md));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Merged-band retry: if we split into bands but found no tables in
|
||||
// any band, retry heuristic detection with all items as a single band.
|
||||
// This catches borderless tables whose text-column alignment was
|
||||
@@ -915,7 +980,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
band_items.len(),
|
||||
was_split
|
||||
);
|
||||
let heuristic_tables = detect_tables(band_items, base_size, false);
|
||||
let heuristic_tables = detect_tables(band_items, base_size, page_has_columns);
|
||||
for table in &heuristic_tables {
|
||||
for &idx in &table.item_indices {
|
||||
if let Some(&page_idx) = band_index_map.get(idx) {
|
||||
@@ -1037,7 +1102,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
}
|
||||
// Sort by Y descending (top to bottom) so left and right
|
||||
// band lines interleave in visual reading order.
|
||||
page_lines.sort_by(|a, b| b.y.partial_cmp(&a.y).unwrap_or(std::cmp::Ordering::Equal));
|
||||
page_lines.sort_by(|a, b| b.y.total_cmp(&a.y));
|
||||
all_lines.extend(page_lines);
|
||||
}
|
||||
all_lines
|
||||
|
||||
@@ -275,7 +275,7 @@ pub(crate) fn strip_repeated_lines(lines: Vec<TextLine>, page_count: u32) -> Vec
|
||||
page_sorted_ys.entry(line.page).or_default().push(line.y);
|
||||
}
|
||||
for ys in page_sorted_ys.values_mut() {
|
||||
ys.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
ys.sort_by(|a, b| a.total_cmp(b));
|
||||
ys.dedup();
|
||||
}
|
||||
|
||||
|
||||
+466
@@ -0,0 +1,466 @@
|
||||
//! PyO3 Python bindings for pdf-inspector.
|
||||
|
||||
use pyo3::exceptions::PyValueError;
|
||||
use pyo3::prelude::*;
|
||||
use std::collections::HashSet;
|
||||
|
||||
use crate::detector::PdfType;
|
||||
use crate::types::ItemType;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Result wrapper
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Result of processing a PDF file.
|
||||
#[pyclass(name = "PdfResult")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyPdfResult {
|
||||
/// The detected PDF type: "text_based", "scanned", "image_based", or "mixed".
|
||||
#[pyo3(get)]
|
||||
pub pdf_type: String,
|
||||
/// Markdown output (None if detect-only or scanned PDF).
|
||||
#[pyo3(get)]
|
||||
pub markdown: Option<String>,
|
||||
/// Total number of pages.
|
||||
#[pyo3(get)]
|
||||
pub page_count: u32,
|
||||
/// Processing time in milliseconds.
|
||||
#[pyo3(get)]
|
||||
pub processing_time_ms: u64,
|
||||
/// 1-indexed page numbers that need OCR.
|
||||
#[pyo3(get)]
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
/// Title from PDF metadata.
|
||||
#[pyo3(get)]
|
||||
pub title: Option<String>,
|
||||
/// Detection confidence (0.0-1.0).
|
||||
#[pyo3(get)]
|
||||
pub confidence: f32,
|
||||
/// Whether the layout is complex (tables/columns detected).
|
||||
#[pyo3(get)]
|
||||
pub is_complex_layout: bool,
|
||||
/// Pages with tables detected.
|
||||
#[pyo3(get)]
|
||||
pub pages_with_tables: Vec<u32>,
|
||||
/// Pages with multi-column layout.
|
||||
#[pyo3(get)]
|
||||
pub pages_with_columns: Vec<u32>,
|
||||
/// Whether encoding issues were detected.
|
||||
#[pyo3(get)]
|
||||
pub has_encoding_issues: bool,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl PyPdfResult {
|
||||
fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"PdfResult(pdf_type='{}', pages={}, confidence={:.2})",
|
||||
self.pdf_type, self.page_count, self.confidence
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Classification wrapper (lightweight)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Lightweight PDF classification result.
|
||||
#[pyclass(name = "PdfClassification")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyPdfClassification {
|
||||
/// The detected PDF type: "text_based", "scanned", "image_based", or "mixed".
|
||||
#[pyo3(get)]
|
||||
pub pdf_type: String,
|
||||
/// Total number of pages.
|
||||
#[pyo3(get)]
|
||||
pub page_count: u32,
|
||||
/// 0-indexed page numbers that need OCR.
|
||||
#[pyo3(get)]
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
/// Detection confidence (0.0-1.0).
|
||||
#[pyo3(get)]
|
||||
pub confidence: f32,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl PyPdfClassification {
|
||||
fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"PdfClassification(pdf_type='{}', pages={}, confidence={:.2})",
|
||||
self.pdf_type, self.page_count, self.confidence
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Region extraction wrappers
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Extracted text for a single region.
|
||||
#[pyclass(name = "RegionText")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyRegionText {
|
||||
/// Extracted text content.
|
||||
#[pyo3(get)]
|
||||
pub text: String,
|
||||
/// True when the text should not be trusted (empty, GID fonts, garbage, encoding issues).
|
||||
#[pyo3(get)]
|
||||
pub needs_ocr: bool,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl PyRegionText {
|
||||
fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"RegionText(text='{}', needs_ocr={})",
|
||||
self.text.chars().take(40).collect::<String>(),
|
||||
self.needs_ocr
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
/// Extracted text for one page's regions.
|
||||
#[pyclass(name = "PageRegionTexts")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyPageRegionTexts {
|
||||
/// 0-indexed page number.
|
||||
#[pyo3(get)]
|
||||
pub page: u32,
|
||||
/// Per-region results, parallel to the input regions.
|
||||
#[pyo3(get)]
|
||||
pub regions: Vec<PyRegionText>,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl PyPageRegionTexts {
|
||||
fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"PageRegionTexts(page={}, regions={})",
|
||||
self.page,
|
||||
self.regions.len()
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Text item wrapper
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// A positioned text item extracted from a PDF.
|
||||
#[pyclass(name = "TextItem")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyTextItem {
|
||||
#[pyo3(get)]
|
||||
pub text: String,
|
||||
#[pyo3(get)]
|
||||
pub x: f32,
|
||||
#[pyo3(get)]
|
||||
pub y: f32,
|
||||
#[pyo3(get)]
|
||||
pub width: f32,
|
||||
#[pyo3(get)]
|
||||
pub height: f32,
|
||||
#[pyo3(get)]
|
||||
pub font: String,
|
||||
#[pyo3(get)]
|
||||
pub font_size: f32,
|
||||
#[pyo3(get)]
|
||||
pub page: u32,
|
||||
#[pyo3(get)]
|
||||
pub is_bold: bool,
|
||||
#[pyo3(get)]
|
||||
pub is_italic: bool,
|
||||
#[pyo3(get)]
|
||||
pub item_type: String,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl PyTextItem {
|
||||
fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"TextItem(text='{}', page={}, x={:.1}, y={:.1})",
|
||||
self.text.chars().take(40).collect::<String>(),
|
||||
self.page,
|
||||
self.x,
|
||||
self.y,
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn pdf_type_str(t: PdfType) -> String {
|
||||
match t {
|
||||
PdfType::TextBased => "text_based".into(),
|
||||
PdfType::Scanned => "scanned".into(),
|
||||
PdfType::ImageBased => "image_based".into(),
|
||||
PdfType::Mixed => "mixed".into(),
|
||||
}
|
||||
}
|
||||
|
||||
fn to_py_result(r: crate::PdfProcessResult) -> PyPdfResult {
|
||||
PyPdfResult {
|
||||
pdf_type: pdf_type_str(r.pdf_type),
|
||||
markdown: r.markdown,
|
||||
page_count: r.page_count,
|
||||
processing_time_ms: r.processing_time_ms,
|
||||
pages_needing_ocr: r.pages_needing_ocr,
|
||||
title: r.title,
|
||||
confidence: r.confidence,
|
||||
is_complex_layout: r.layout.is_complex,
|
||||
pages_with_tables: r.layout.pages_with_tables,
|
||||
pages_with_columns: r.layout.pages_with_columns,
|
||||
has_encoding_issues: r.has_encoding_issues,
|
||||
}
|
||||
}
|
||||
|
||||
fn to_py_err(e: crate::PdfError) -> PyErr {
|
||||
PyValueError::new_err(e.to_string())
|
||||
}
|
||||
|
||||
fn item_type_str(t: &ItemType) -> String {
|
||||
match t {
|
||||
ItemType::Text => "text".into(),
|
||||
ItemType::Image => "image".into(),
|
||||
ItemType::Link(url) => format!("link:{url}"),
|
||||
ItemType::FormField => "form_field".into(),
|
||||
}
|
||||
}
|
||||
|
||||
fn convert_text_items(items: Vec<crate::TextItem>) -> Vec<PyTextItem> {
|
||||
items
|
||||
.into_iter()
|
||||
.map(|item| PyTextItem {
|
||||
text: item.text,
|
||||
x: item.x,
|
||||
y: item.y,
|
||||
width: item.width,
|
||||
height: item.height,
|
||||
font: item.font,
|
||||
font_size: item.font_size,
|
||||
page: item.page,
|
||||
is_bold: item.is_bold,
|
||||
is_italic: item.is_italic,
|
||||
item_type: item_type_str(&item.item_type),
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn parse_page_regions(
|
||||
page_regions: Vec<(u32, Vec<Vec<f64>>)>,
|
||||
) -> PyResult<Vec<(u32, Vec<[f32; 4]>)>> {
|
||||
page_regions
|
||||
.into_iter()
|
||||
.map(|(page, regions)| {
|
||||
let mut bboxes: Vec<[f32; 4]> = Vec::with_capacity(regions.len());
|
||||
for (idx, region) in regions.into_iter().enumerate() {
|
||||
if region.len() != 4 {
|
||||
return Err(PyValueError::new_err(format!(
|
||||
"Invalid region at page {page}, index {idx}: expected [x1, y1, x2, y2], got {} values",
|
||||
region.len()
|
||||
)));
|
||||
}
|
||||
let [x1, y1, x2, y2] = [region[0], region[1], region[2], region[3]];
|
||||
if !(x1.is_finite() && y1.is_finite() && x2.is_finite() && y2.is_finite()) {
|
||||
return Err(PyValueError::new_err(format!(
|
||||
"Invalid region at page {page}, index {idx}: coordinates must be finite numbers"
|
||||
)));
|
||||
}
|
||||
if x2 < x1 || y2 < y1 {
|
||||
return Err(PyValueError::new_err(format!(
|
||||
"Invalid region at page {page}, index {idx}: expected x2>=x1 and y2>=y1, got [{x1}, {y1}, {x2}, {y2}]"
|
||||
)));
|
||||
}
|
||||
bboxes.push([x1 as f32, y1 as f32, x2 as f32, y2 as f32]);
|
||||
}
|
||||
Ok((page, bboxes))
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn convert_region_results(results: Vec<crate::PageRegionResult>) -> Vec<PyPageRegionTexts> {
|
||||
results
|
||||
.into_iter()
|
||||
.map(|page_result| PyPageRegionTexts {
|
||||
page: page_result.page,
|
||||
regions: page_result
|
||||
.regions
|
||||
.into_iter()
|
||||
.map(|r| PyRegionText {
|
||||
text: r.text,
|
||||
needs_ocr: r.needs_ocr,
|
||||
})
|
||||
.collect(),
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Public Python API
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Process a PDF file: detect type, extract text, and convert to Markdown.
|
||||
#[pyfunction]
|
||||
#[pyo3(signature = (path, pages=None))]
|
||||
fn process_pdf(path: &str, pages: Option<Vec<u32>>) -> PyResult<PyPdfResult> {
|
||||
let mut opts = crate::PdfOptions::new();
|
||||
if let Some(p) = pages {
|
||||
opts = opts.pages(p);
|
||||
}
|
||||
let result = crate::process_pdf_with_options(path, opts).map_err(to_py_err)?;
|
||||
Ok(to_py_result(result))
|
||||
}
|
||||
|
||||
/// Process a PDF from bytes in memory.
|
||||
#[pyfunction]
|
||||
#[pyo3(signature = (data, pages=None))]
|
||||
fn process_pdf_bytes(data: &[u8], pages: Option<Vec<u32>>) -> PyResult<PyPdfResult> {
|
||||
let mut opts = crate::PdfOptions::new();
|
||||
if let Some(p) = pages {
|
||||
opts = opts.pages(p);
|
||||
}
|
||||
let result = crate::process_pdf_mem_with_options(data, opts).map_err(to_py_err)?;
|
||||
Ok(to_py_result(result))
|
||||
}
|
||||
|
||||
/// Fast detection only — no text extraction or markdown.
|
||||
#[pyfunction]
|
||||
fn detect_pdf(path: &str) -> PyResult<PyPdfResult> {
|
||||
let result = crate::detect_pdf(path).map_err(to_py_err)?;
|
||||
Ok(to_py_result(result))
|
||||
}
|
||||
|
||||
/// Fast detection from bytes — no text extraction or markdown.
|
||||
#[pyfunction]
|
||||
fn detect_pdf_bytes(data: &[u8]) -> PyResult<PyPdfResult> {
|
||||
let result = crate::detect_pdf_mem(data).map_err(to_py_err)?;
|
||||
Ok(to_py_result(result))
|
||||
}
|
||||
|
||||
/// Lightweight PDF classification — returns type, page count, and OCR pages.
|
||||
/// Faster than detect_pdf as it skips building the full PdfProcessResult.
|
||||
/// Pages in pages_needing_ocr are 0-indexed.
|
||||
#[pyfunction]
|
||||
fn classify_pdf(path: &str) -> PyResult<PyPdfClassification> {
|
||||
let data = std::fs::read(path).map_err(|e| PyValueError::new_err(e.to_string()))?;
|
||||
classify_pdf_bytes(&data)
|
||||
}
|
||||
|
||||
/// Lightweight PDF classification from bytes.
|
||||
/// Pages in pages_needing_ocr are 0-indexed.
|
||||
#[pyfunction]
|
||||
fn classify_pdf_bytes(data: &[u8]) -> PyResult<PyPdfClassification> {
|
||||
let result = crate::classify_pdf_mem(data).map_err(to_py_err)?;
|
||||
Ok(PyPdfClassification {
|
||||
pdf_type: pdf_type_str(result.pdf_type),
|
||||
page_count: result.page_count,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
confidence: result.confidence,
|
||||
})
|
||||
}
|
||||
|
||||
/// Extract plain text from a PDF file.
|
||||
#[pyfunction]
|
||||
fn extract_text(path: &str) -> PyResult<String> {
|
||||
crate::extract_text(path).map_err(to_py_err)
|
||||
}
|
||||
|
||||
/// Extract plain text from PDF bytes.
|
||||
#[pyfunction]
|
||||
fn extract_text_bytes(data: &[u8]) -> PyResult<String> {
|
||||
crate::extractor::extract_text_mem(data).map_err(to_py_err)
|
||||
}
|
||||
|
||||
/// Extract text with position information from a file.
|
||||
#[pyfunction]
|
||||
#[pyo3(signature = (path, pages=None))]
|
||||
fn extract_text_with_positions(path: &str, pages: Option<Vec<u32>>) -> PyResult<Vec<PyTextItem>> {
|
||||
let items = match pages {
|
||||
Some(p) => {
|
||||
let page_set: HashSet<u32> = p.into_iter().collect();
|
||||
crate::extract_text_with_positions_pages(path, Some(&page_set)).map_err(to_py_err)?
|
||||
}
|
||||
None => crate::extract_text_with_positions(path).map_err(to_py_err)?,
|
||||
};
|
||||
Ok(convert_text_items(items))
|
||||
}
|
||||
|
||||
/// Extract text with position information from bytes.
|
||||
#[pyfunction]
|
||||
#[pyo3(signature = (data, pages=None))]
|
||||
fn extract_text_with_positions_bytes(
|
||||
data: &[u8],
|
||||
pages: Option<Vec<u32>>,
|
||||
) -> PyResult<Vec<PyTextItem>> {
|
||||
let items = match pages {
|
||||
Some(p) => {
|
||||
let page_set: HashSet<u32> = p.into_iter().collect();
|
||||
crate::extractor::extract_text_with_positions_mem_pages(data, Some(&page_set))
|
||||
.map_err(to_py_err)?
|
||||
}
|
||||
None => crate::extractor::extract_text_with_positions_mem(data).map_err(to_py_err)?,
|
||||
};
|
||||
Ok(convert_text_items(items))
|
||||
}
|
||||
|
||||
/// Extract text within bounding-box regions from a PDF file.
|
||||
///
|
||||
/// Args:
|
||||
/// path: Path to the PDF file.
|
||||
/// page_regions: List of (page_0indexed, [[x1, y1, x2, y2], ...]) tuples.
|
||||
/// Coordinates are PDF points with top-left origin.
|
||||
///
|
||||
/// Returns:
|
||||
/// List of PageRegionTexts with per-region text and needs_ocr flag.
|
||||
#[pyfunction]
|
||||
fn extract_text_in_regions(
|
||||
path: &str,
|
||||
page_regions: Vec<(u32, Vec<Vec<f64>>)>,
|
||||
) -> PyResult<Vec<PyPageRegionTexts>> {
|
||||
let data = std::fs::read(path).map_err(|e| PyValueError::new_err(e.to_string()))?;
|
||||
extract_text_in_regions_bytes(&data, page_regions)
|
||||
}
|
||||
|
||||
/// Extract text within bounding-box regions from PDF bytes.
|
||||
///
|
||||
/// Args:
|
||||
/// data: PDF file contents as bytes.
|
||||
/// page_regions: List of (page_0indexed, [[x1, y1, x2, y2], ...]) tuples.
|
||||
/// Coordinates are PDF points with top-left origin.
|
||||
///
|
||||
/// Returns:
|
||||
/// List of PageRegionTexts with per-region text and needs_ocr flag.
|
||||
#[pyfunction]
|
||||
fn extract_text_in_regions_bytes(
|
||||
data: &[u8],
|
||||
page_regions: Vec<(u32, Vec<Vec<f64>>)>,
|
||||
) -> PyResult<Vec<PyPageRegionTexts>> {
|
||||
let regions = parse_page_regions(page_regions)?;
|
||||
let results = crate::extract_text_in_regions_mem(data, ®ions).map_err(to_py_err)?;
|
||||
Ok(convert_region_results(results))
|
||||
}
|
||||
|
||||
/// Python module definition.
|
||||
#[pymodule]
|
||||
fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
|
||||
m.add_class::<PyPdfResult>()?;
|
||||
m.add_class::<PyPdfClassification>()?;
|
||||
m.add_class::<PyTextItem>()?;
|
||||
m.add_class::<PyRegionText>()?;
|
||||
m.add_class::<PyPageRegionTexts>()?;
|
||||
m.add_function(wrap_pyfunction!(process_pdf, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(process_pdf_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(detect_pdf, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(detect_pdf_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(classify_pdf, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(classify_pdf_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text_with_positions, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text_with_positions_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text_in_regions, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text_in_regions_bytes, m)?)?;
|
||||
Ok(())
|
||||
}
|
||||
@@ -48,7 +48,7 @@ pub(crate) fn merge_adjacent_items(items: &[TextItem]) -> (Vec<TextItem>, Vec<Ve
|
||||
}
|
||||
|
||||
// Sort groups by Y descending (top of page first)
|
||||
line_groups.sort_by(|a, b| b.0.partial_cmp(&a.0).unwrap_or(std::cmp::Ordering::Equal));
|
||||
line_groups.sort_by(|a, b| b.0.total_cmp(&a.0));
|
||||
|
||||
let mut merged_items = Vec::new();
|
||||
let mut index_map: Vec<Vec<usize>> = Vec::new();
|
||||
@@ -219,7 +219,7 @@ pub fn detect_tables(items: &[TextItem], base_font_size: f32, skip_body_font: bo
|
||||
body_font_low,
|
||||
body_font_high,
|
||||
);
|
||||
if body_candidates.len() >= 9 {
|
||||
if body_candidates.len() >= 6 {
|
||||
let regions = find_table_regions_strict(&body_candidates);
|
||||
log::debug!("body-font: {} strict regions found", regions.len());
|
||||
|
||||
@@ -241,7 +241,7 @@ pub fn detect_tables(items: &[TextItem], base_font_size: f32, skip_body_font: bo
|
||||
body_candidates.len()
|
||||
);
|
||||
|
||||
if region_items.len() < 9 {
|
||||
if region_items.len() < 6 {
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -284,7 +284,7 @@ fn find_table_regions(items: &[(usize, &TextItem)]) -> Vec<(f32, f32)> {
|
||||
}
|
||||
|
||||
let mut y_positions: Vec<f32> = items.iter().map(|(_, i)| i.y).collect();
|
||||
y_positions.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
y_positions.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
// Find clusters of Y positions (table regions)
|
||||
let mut regions = Vec::new();
|
||||
@@ -347,7 +347,7 @@ fn find_table_regions_strict(items: &[(usize, &TextItem)]) -> Vec<(f32, f32, f32
|
||||
let mut qualifying_rows: Vec<(f32, Vec<f32>)> = Vec::new(); // (y, cluster_starts)
|
||||
for (y, x_positions) in &row_groups {
|
||||
let mut sorted_xs = x_positions.clone();
|
||||
sorted_xs.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
sorted_xs.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
if sorted_xs.is_empty() {
|
||||
continue;
|
||||
@@ -379,14 +379,14 @@ fn find_table_regions_strict(items: &[(usize, &TextItem)]) -> Vec<(f32, f32, f32
|
||||
// Step 3: Find contiguous runs of qualifying rows.
|
||||
// Use adaptive gap: median spacing × 3 (handles wrapped cells where
|
||||
// qualifying rows are spaced further apart), with a floor of 25pt.
|
||||
qualifying_rows.sort_by(|a, b| a.0.partial_cmp(&b.0).unwrap_or(std::cmp::Ordering::Equal));
|
||||
qualifying_rows.sort_by(|a, b| a.0.total_cmp(&b.0));
|
||||
|
||||
let max_gap = if qualifying_rows.len() >= 3 {
|
||||
let mut gaps: Vec<f32> = qualifying_rows
|
||||
.windows(2)
|
||||
.map(|w| (w[1].0 - w[0].0).abs())
|
||||
.collect();
|
||||
gaps.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
gaps.sort_by(|a, b| a.total_cmp(b));
|
||||
let median_gap = gaps[gaps.len() / 2];
|
||||
(median_gap * 3.0).max(25.0)
|
||||
} else {
|
||||
@@ -566,11 +566,9 @@ fn detect_table_in_region(items: &[(usize, &TextItem)], mode: TableDetectionMode
|
||||
// Sort by X position (direction-aware)
|
||||
let rtl = is_rtl_text(col_items.iter().map(|i| &i.text));
|
||||
if rtl {
|
||||
col_items
|
||||
.sort_by(|a, b| b.x.partial_cmp(&a.x).unwrap_or(std::cmp::Ordering::Equal));
|
||||
col_items.sort_by(|a, b| b.x.total_cmp(&a.x));
|
||||
} else {
|
||||
col_items
|
||||
.sort_by(|a, b| a.x.partial_cmp(&b.x).unwrap_or(std::cmp::Ordering::Equal));
|
||||
col_items.sort_by(|a, b| a.x.total_cmp(&b.x));
|
||||
}
|
||||
|
||||
// Join items with subscript-aware spacing
|
||||
@@ -803,7 +801,25 @@ fn has_table_like_content(cells: &[Vec<String>], mode: TableDetectionMode) -> bo
|
||||
// Bypass content check for wide tables (3+ columns) — text-only tables
|
||||
// (category lists, program descriptions) are legitimate if they passed
|
||||
// all structural validations (alignment, consistency, not key-value).
|
||||
pct_data > min_pct || num_cols >= 3
|
||||
// Also bypass for 2-column body-font tables with short cells (avg ≤40 chars),
|
||||
// which are likely definition/category lists, not paragraph text.
|
||||
if pct_data > min_pct || num_cols >= 3 {
|
||||
return true;
|
||||
}
|
||||
if num_cols == 2 && matches!(mode, TableDetectionMode::BodyFont) {
|
||||
let non_empty: Vec<usize> = cells
|
||||
.iter()
|
||||
.skip(1)
|
||||
.flat_map(|row| row.iter())
|
||||
.filter(|c| !c.trim().is_empty())
|
||||
.map(|c| c.trim().len())
|
||||
.collect();
|
||||
if !non_empty.is_empty() {
|
||||
let avg_len = non_empty.iter().sum::<usize>() / non_empty.len();
|
||||
return avg_len <= 25;
|
||||
}
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
/// Check if a cell value looks like table data
|
||||
@@ -1128,6 +1144,33 @@ pub(crate) fn find_first_table_row(
|
||||
continue;
|
||||
}
|
||||
|
||||
// Skip rows that have duplicate non-empty cells. These are spanning
|
||||
// super-headers (e.g., "First Degree | First Degree | Higher Degree")
|
||||
// that sit above the real column header row. Using them as the markdown
|
||||
// header produces duplicate column names that downstream validation
|
||||
// rejects. Only skip if a subsequent row looks like a better header
|
||||
// (denser fill or has data).
|
||||
if filled_count >= 2 && !has_data {
|
||||
let mut text_counts: std::collections::HashMap<&str, usize> =
|
||||
std::collections::HashMap::new();
|
||||
for cell in &filled_cells {
|
||||
*text_counts.entry(cell.trim()).or_insert(0) += 1;
|
||||
}
|
||||
let has_duplicates = text_counts.values().any(|&count| count >= 2);
|
||||
if has_duplicates {
|
||||
// Check if a later row is a better header candidate
|
||||
let has_better_below = cells.iter().skip(row_idx + 1).take(3).any(|r| {
|
||||
let next_filled = r.iter().filter(|c| !c.trim().is_empty()).count();
|
||||
let next_fill = next_filled as f32 / total_cols as f32;
|
||||
let next_numeric = r.iter().filter(|c| looks_like_number(c.trim())).count();
|
||||
next_fill >= 0.4 || next_numeric >= 2
|
||||
});
|
||||
if has_better_below {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Data rows are definitely table content
|
||||
if has_data {
|
||||
first_table_row = row_idx;
|
||||
|
||||
@@ -167,7 +167,7 @@ pub fn detect_tables_from_lines(items: &[TextItem], lines: &[PdfLine], page: u32
|
||||
|
||||
// Row edges need to be in descending order (top of page = higher Y first)
|
||||
let mut row_edges_desc = row_edges;
|
||||
row_edges_desc.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal));
|
||||
row_edges_desc.sort_by(|a, b| b.total_cmp(a));
|
||||
|
||||
log::debug!(
|
||||
"detect_lines p{}: {} row_edges, {} col_edges, table=({:.0},{:.0})-({:.0},{:.0}), spanning_h={}, spanning_v={}",
|
||||
@@ -243,9 +243,11 @@ pub fn detect_tables_from_lines(items: &[TextItem], lines: &[PdfLine], page: u32
|
||||
.map(|s| (s - mean_spacing).powi(2))
|
||||
.sum::<f32>()
|
||||
/ spacings.len() as f32;
|
||||
let cv = variance.sqrt() / mean_spacing; // coefficient of variation
|
||||
// CV < 0.05 means nearly identical spacing — chart grid
|
||||
if cv < 0.05 {
|
||||
let cv = variance.sqrt() / mean_spacing;
|
||||
// CV < 0.02 means nearly identical spacing — likely chart grid.
|
||||
// Spreadsheet-exported tables often have uniform rows (CV 0.03-0.05),
|
||||
// so we use a tighter threshold to avoid false negatives.
|
||||
if cv < 0.02 {
|
||||
return Vec::new();
|
||||
}
|
||||
}
|
||||
|
||||
+17
-17
@@ -144,7 +144,7 @@ fn split_wide_cluster(
|
||||
|
||||
// Build sorted list of X-intervals (x_left, x_right) from each rect
|
||||
let mut intervals: Vec<(f32, f32)> = rects.iter().map(|&(x, _, w, _)| (x, x + w)).collect();
|
||||
intervals.sort_by(|a, b| a.0.partial_cmp(&b.0).unwrap_or(std::cmp::Ordering::Equal));
|
||||
intervals.sort_by(|a, b| a.0.total_cmp(&b.0));
|
||||
|
||||
// Merge overlapping intervals to find contiguous X-bands
|
||||
let mut merged: Vec<(f32, f32)> = Vec::new();
|
||||
@@ -262,7 +262,7 @@ pub fn detect_tables_from_rects(
|
||||
// background fills stand out clearly.
|
||||
if page_rects.len() >= 6 {
|
||||
let mut widths: Vec<f32> = page_rects.iter().map(|&(_, _, w, _)| w).collect();
|
||||
widths.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
widths.sort_by(|a, b| a.total_cmp(b));
|
||||
let median_width = widths[widths.len() / 2];
|
||||
let width_threshold = median_width * 10.0;
|
||||
let before = page_rects.len();
|
||||
@@ -333,7 +333,7 @@ pub fn detect_tables_from_rects(
|
||||
// cluster they overlap, so grid detection still has their edges.
|
||||
let is_page_bg = {
|
||||
let mut heights: Vec<f32> = page_rects.iter().map(|&(_, _, _, h)| h).collect();
|
||||
heights.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
heights.sort_by(|a, b| a.total_cmp(b));
|
||||
let median_height = heights[heights.len() / 2];
|
||||
let height_threshold = median_height * 20.0;
|
||||
let flags: Vec<bool> = page_rects
|
||||
@@ -621,7 +621,7 @@ fn merge_overlapping_hints(mut hints: Vec<RectHintRegion>) -> Vec<RectHintRegion
|
||||
return hints;
|
||||
}
|
||||
loop {
|
||||
hints.sort_by(|a, b| a.x_left.partial_cmp(&b.x_left).unwrap());
|
||||
hints.sort_by(|a, b| a.x_left.total_cmp(&b.x_left));
|
||||
let mut merged: Vec<RectHintRegion> = Vec::new();
|
||||
let mut any_merged = false;
|
||||
for hint in &hints {
|
||||
@@ -685,7 +685,7 @@ fn extract_hint_region(group_rects: &[(f32, f32, f32, f32)]) -> Option<RectHintR
|
||||
|
||||
// Compute median height to identify cell-sized rects
|
||||
let mut heights: Vec<f32> = group_rects.iter().map(|&(_, _, _, h)| h).collect();
|
||||
heights.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
heights.sort_by(|a, b| a.total_cmp(b));
|
||||
let median_h = heights[heights.len() / 2];
|
||||
|
||||
// Keep only cell-sized rects (height ≤ 4× median)
|
||||
@@ -857,9 +857,9 @@ fn try_build_grid(
|
||||
|
||||
// Sort column edges left-to-right, row edges top-to-bottom (highest Y first for PDF)
|
||||
let mut col_edges = x_edges;
|
||||
col_edges.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
col_edges.sort_by(|a, b| a.total_cmp(b));
|
||||
let mut row_edges = y_edges;
|
||||
row_edges.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal));
|
||||
row_edges.sort_by(|a, b| b.total_cmp(a));
|
||||
|
||||
let num_cols = col_edges.len() - 1;
|
||||
let num_rows = row_edges.len() - 1;
|
||||
@@ -1048,7 +1048,7 @@ fn try_build_grid(
|
||||
/// Deduplicate nearby edge values within a tolerance, returning sorted unique edges.
|
||||
pub(crate) fn snap_edges(values: &[f32], tolerance: f32) -> Vec<f32> {
|
||||
let mut sorted: Vec<f32> = values.to_vec();
|
||||
sorted.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
sorted.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
let mut snapped: Vec<f32> = Vec::new();
|
||||
for &v in &sorted {
|
||||
@@ -1204,7 +1204,7 @@ fn is_row_stripe_pattern(rects: &[(f32, f32, f32, f32)]) -> bool {
|
||||
}
|
||||
|
||||
let mut widths: Vec<f32> = rects.iter().map(|&(_, _, w, _)| w).collect();
|
||||
widths.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
widths.sort_by(|a, b| a.total_cmp(b));
|
||||
let median_width = widths[widths.len() / 2];
|
||||
|
||||
// Must be page-spanning (>200pt)
|
||||
@@ -1252,7 +1252,7 @@ fn detect_row_stripe_table(
|
||||
|
||||
// Sort row edges top-to-bottom (highest Y first for PDF)
|
||||
let mut row_edges = y_edges;
|
||||
row_edges.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal));
|
||||
row_edges.sort_by(|a, b| b.total_cmp(a));
|
||||
|
||||
// Compute the bounding box of the stripe region for filtering items
|
||||
let y_top = row_edges[0];
|
||||
@@ -1476,7 +1476,7 @@ fn detect_row_stripe_table_from_cell_rects(
|
||||
// bounding box to scope items and derive rows from text Y-positions.
|
||||
let row_edges = if y_edges.len() >= 4 {
|
||||
let mut edges = y_edges;
|
||||
edges.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal));
|
||||
edges.sort_by(|a, b| b.total_cmp(a));
|
||||
edges
|
||||
} else {
|
||||
// Fall back: gather items in the rect region and cluster by Y
|
||||
@@ -1508,11 +1508,11 @@ fn detect_row_stripe_table_from_cell_rects(
|
||||
// Cluster Y positions using median font height as threshold
|
||||
let median_h = {
|
||||
let mut hs: Vec<f32> = region_items.iter().map(|i| i.height).collect();
|
||||
hs.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
hs.sort_by(|a, b| a.total_cmp(b));
|
||||
hs[hs.len() / 2]
|
||||
};
|
||||
let mut ys: Vec<f32> = region_items.iter().map(|i| i.y).collect();
|
||||
ys.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal));
|
||||
ys.sort_by(|a, b| b.total_cmp(a));
|
||||
let mut edges = Vec::new();
|
||||
let threshold = median_h * 0.8;
|
||||
let mut cluster_start = ys[0];
|
||||
@@ -1536,7 +1536,7 @@ fn detect_row_stripe_table_from_cell_rects(
|
||||
edges.push(center - median_h * 0.5);
|
||||
let _ = cluster_start; // suppress unused warning
|
||||
edges = snap_edges(&edges, 3.0);
|
||||
edges.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal));
|
||||
edges.sort_by(|a, b| b.total_cmp(a));
|
||||
if edges.len() < 4 {
|
||||
return None;
|
||||
}
|
||||
@@ -1546,7 +1546,7 @@ fn detect_row_stripe_table_from_cell_rects(
|
||||
// Compute bounding box from non-full-page rects
|
||||
let median_h = {
|
||||
let mut heights: Vec<f32> = group_rects.iter().map(|&(_, _, _, h)| h).collect();
|
||||
heights.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
heights.sort_by(|a, b| a.total_cmp(b));
|
||||
heights[heights.len() / 2]
|
||||
};
|
||||
let content_rects: Vec<_> = group_rects
|
||||
@@ -1724,7 +1724,7 @@ fn detect_merged_cluster_table(
|
||||
}
|
||||
|
||||
let mut row_edges = y_edges;
|
||||
row_edges.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal));
|
||||
row_edges.sort_by(|a, b| b.total_cmp(a));
|
||||
|
||||
// Bounding box of all rects
|
||||
let y_top = row_edges[0];
|
||||
@@ -1890,7 +1890,7 @@ fn detect_merged_cluster_table(
|
||||
/// (no need for anti-paragraph safeguards).
|
||||
fn cluster_x_positions(items: &[(usize, &TextItem)], min_threshold: f32) -> Vec<f32> {
|
||||
let mut x_positions: Vec<f32> = items.iter().map(|(_, i)| i.x).collect();
|
||||
x_positions.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
x_positions.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
if x_positions.is_empty() {
|
||||
return vec![];
|
||||
|
||||
+136
-16
@@ -9,7 +9,7 @@ pub(crate) fn find_column_boundaries(
|
||||
mode: TableDetectionMode,
|
||||
) -> Vec<f32> {
|
||||
let mut x_positions: Vec<f32> = items.iter().map(|(_, i)| i.x).collect();
|
||||
x_positions.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
x_positions.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
if x_positions.is_empty() {
|
||||
return vec![];
|
||||
@@ -43,7 +43,7 @@ pub(crate) fn find_column_boundaries(
|
||||
.collect();
|
||||
|
||||
if consec_gaps.len() > 2 {
|
||||
consec_gaps.sort_by(|a, b| a.partial_cmp(b).unwrap());
|
||||
consec_gaps.sort_by(|a, b| a.total_cmp(b));
|
||||
// Find the biggest jump in the sorted gap sequence — natural break
|
||||
// between within-column jitter and between-column spacing.
|
||||
// Require at least 3 values on each side to avoid outlier-dominated
|
||||
@@ -82,33 +82,42 @@ pub(crate) fn find_column_boundaries(
|
||||
}
|
||||
}
|
||||
|
||||
let mut columns = Vec::new();
|
||||
let mut cluster_items: Vec<f32> = vec![x_positions[0]];
|
||||
// Track cluster membership: for each cluster, store the list of x positions
|
||||
let mut cluster_xs: Vec<Vec<f32>> = vec![vec![x_positions[0]]];
|
||||
|
||||
for &x in &x_positions[1..] {
|
||||
let last_cluster = cluster_xs.last().unwrap();
|
||||
// For dense columns (gap-histogram triggered), use edge-based clustering:
|
||||
// compare with the last item to avoid center-drift that merges adjacent
|
||||
// narrow columns. For normal tables, use center-based (original behavior).
|
||||
let reference = if use_edge_clustering {
|
||||
*cluster_items.last().unwrap()
|
||||
*last_cluster.last().unwrap()
|
||||
} else {
|
||||
cluster_items.iter().sum::<f32>() / cluster_items.len() as f32
|
||||
last_cluster.iter().sum::<f32>() / last_cluster.len() as f32
|
||||
};
|
||||
|
||||
if x - reference > cluster_threshold {
|
||||
let cluster_center = cluster_items.iter().sum::<f32>() / cluster_items.len() as f32;
|
||||
columns.push(cluster_center);
|
||||
cluster_items = vec![x];
|
||||
cluster_xs.push(vec![x]);
|
||||
} else {
|
||||
cluster_items.push(x);
|
||||
cluster_xs.last_mut().unwrap().push(x);
|
||||
}
|
||||
}
|
||||
|
||||
// Don't forget last cluster
|
||||
if !cluster_items.is_empty() {
|
||||
columns.push(cluster_items.iter().sum::<f32>() / cluster_items.len() as f32);
|
||||
// Numeric column merge pass: when a sparse cluster (few items, typically
|
||||
// header text) is adjacent to a dense numeric cluster and within 1.5×
|
||||
// threshold, merge them. This fixes tables where multi-line wrapped
|
||||
// headers have slightly different X positions than the data columns,
|
||||
// causing the header and data to split into separate clusters.
|
||||
let columns_before_merge = cluster_xs.len();
|
||||
if columns_before_merge >= 3 {
|
||||
cluster_xs = merge_numeric_adjacent_clusters(cluster_xs, items, cluster_threshold);
|
||||
}
|
||||
|
||||
let columns: Vec<f32> = cluster_xs
|
||||
.iter()
|
||||
.map(|xs| xs.iter().sum::<f32>() / xs.len() as f32)
|
||||
.collect();
|
||||
|
||||
// Filter columns - each should have multiple items
|
||||
let min_items_per_col = (items.len() / columns.len().max(1) / 4).max(2);
|
||||
let columns: Vec<f32> = columns
|
||||
@@ -123,8 +132,9 @@ pub(crate) fn find_column_boundaries(
|
||||
.collect();
|
||||
|
||||
log::debug!(
|
||||
" find_column_boundaries: {} columns before filter, threshold={:.1}, {} items",
|
||||
" find_column_boundaries: {} columns (merged from {}), threshold={:.1}, {} items",
|
||||
columns.len(),
|
||||
columns_before_merge,
|
||||
cluster_threshold,
|
||||
items.len()
|
||||
);
|
||||
@@ -148,10 +158,120 @@ pub(crate) fn find_column_boundaries(
|
||||
columns
|
||||
}
|
||||
|
||||
/// Check if a text string looks like a number (digits, decimals, sign, comma).
|
||||
fn is_numeric_text(s: &str) -> bool {
|
||||
let s = s.trim();
|
||||
if s.is_empty() {
|
||||
return false;
|
||||
}
|
||||
// Match patterns like: 8.23, -1.05, 9.99, 7.12, 100, 3,456.78, +5%, ---
|
||||
// But NOT: BIO, Department, Core Courses
|
||||
s.chars()
|
||||
.all(|c| c.is_ascii_digit() || c == '.' || c == ',' || c == '-' || c == '+' || c == '%')
|
||||
&& s.chars().any(|c| c.is_ascii_digit())
|
||||
}
|
||||
|
||||
/// Merge adjacent X-position clusters when one is a sparse header cluster
|
||||
/// and the other is a dense numeric data cluster. This prevents multi-line
|
||||
/// wrapped headers from splitting a logical column into two clusters.
|
||||
fn merge_numeric_adjacent_clusters(
|
||||
mut clusters: Vec<Vec<f32>>,
|
||||
items: &[(usize, &TextItem)],
|
||||
threshold: f32,
|
||||
) -> Vec<Vec<f32>> {
|
||||
// For each cluster, compute: center, item count, numeric fraction
|
||||
struct ClusterInfo {
|
||||
center: f32,
|
||||
count: usize,
|
||||
numeric_frac: f32,
|
||||
}
|
||||
|
||||
let compute_info = |xs: &[f32]| -> ClusterInfo {
|
||||
let center = xs.iter().sum::<f32>() / xs.len() as f32;
|
||||
// Count items and numeric fraction for items near this cluster center
|
||||
let mut total = 0;
|
||||
let mut numeric = 0;
|
||||
for (_, item) in items {
|
||||
if (item.x - center).abs() < threshold {
|
||||
total += 1;
|
||||
if is_numeric_text(&item.text) {
|
||||
numeric += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
ClusterInfo {
|
||||
center,
|
||||
count: total,
|
||||
numeric_frac: if total > 0 {
|
||||
numeric as f32 / total as f32
|
||||
} else {
|
||||
0.0
|
||||
},
|
||||
}
|
||||
};
|
||||
|
||||
// Merge distance: allow merging clusters that are slightly beyond the
|
||||
// original threshold. Use 1.5× threshold to catch header-vs-data splits.
|
||||
let merge_dist = threshold * 1.5;
|
||||
|
||||
// Iterate and merge adjacent pairs. Use a simple left-to-right scan.
|
||||
let mut merged = true;
|
||||
while merged {
|
||||
merged = false;
|
||||
let mut i = 0;
|
||||
while i + 1 < clusters.len() {
|
||||
let info_a = compute_info(&clusters[i]);
|
||||
let info_b = compute_info(&clusters[i + 1]);
|
||||
let dist = (info_b.center - info_a.center).abs();
|
||||
|
||||
if dist > merge_dist {
|
||||
i += 1;
|
||||
continue;
|
||||
}
|
||||
|
||||
// Determine if one cluster is sparse (header) and the other
|
||||
// is dense and numeric (data). A cluster is "sparse" if it has
|
||||
// significantly fewer items than the other.
|
||||
let (sparse, dense) = if info_a.count < info_b.count {
|
||||
(&info_a, &info_b)
|
||||
} else {
|
||||
(&info_b, &info_a)
|
||||
};
|
||||
|
||||
// Merge if the dense cluster is predominantly numeric (>50%)
|
||||
// and the sparse cluster has at most 1/3 the items of the dense one.
|
||||
let should_merge =
|
||||
dense.numeric_frac > 0.50 && sparse.count <= dense.count / 2 && sparse.count <= 5;
|
||||
|
||||
if should_merge {
|
||||
log::debug!(
|
||||
" merging column clusters: center {:.1} ({} items, {:.0}% numeric) + {:.1} ({} items, {:.0}% numeric), dist={:.1}",
|
||||
info_a.center,
|
||||
info_a.count,
|
||||
info_a.numeric_frac * 100.0,
|
||||
info_b.center,
|
||||
info_b.count,
|
||||
info_b.numeric_frac * 100.0,
|
||||
dist,
|
||||
);
|
||||
// Merge cluster i+1 into cluster i
|
||||
let next = clusters.remove(i + 1);
|
||||
clusters[i].extend(next);
|
||||
merged = true;
|
||||
// Don't increment i — check if the merged cluster can merge further
|
||||
} else {
|
||||
i += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
clusters
|
||||
}
|
||||
|
||||
/// Find row boundaries by clustering Y positions
|
||||
pub(crate) fn find_row_boundaries(items: &[(usize, &TextItem)]) -> Vec<f32> {
|
||||
let mut y_positions: Vec<f32> = items.iter().map(|(_, i)| i.y).collect();
|
||||
y_positions.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal)); // Descending
|
||||
y_positions.sort_by(|a, b| b.total_cmp(a)); // Descending
|
||||
|
||||
if y_positions.is_empty() {
|
||||
return vec![];
|
||||
@@ -162,7 +282,7 @@ pub(crate) fn find_row_boundaries(items: &[(usize, &TextItem)]) -> Vec<f32> {
|
||||
// inter-row gaps (≥1× font size), preventing row merging in uniform-spaced PDFs.
|
||||
let cluster_threshold = {
|
||||
let mut font_sizes: Vec<f32> = items.iter().map(|(_, i)| i.font_size).collect();
|
||||
font_sizes.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
font_sizes.sort_by(|a, b| a.total_cmp(b));
|
||||
let median_font = font_sizes[font_sizes.len() / 2];
|
||||
(median_font * 0.8).max(4.0)
|
||||
};
|
||||
|
||||
+6
-6
@@ -33,7 +33,7 @@ pub(crate) fn try_build_rect_guided_table(
|
||||
|
||||
// 1. Derive column boundaries from rect X positions (snapped to 2pt tolerance)
|
||||
let mut x_lefts: Vec<f32> = cluster_rects.iter().map(|&(x, _, _, _)| x).collect();
|
||||
x_lefts.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
x_lefts.sort_by(|a, b| a.total_cmp(b));
|
||||
// Snap: deduplicate within 2pt tolerance
|
||||
let mut col_boundaries: Vec<f32> = Vec::new();
|
||||
for x in &x_lefts {
|
||||
@@ -54,7 +54,7 @@ pub(crate) fn try_build_rect_guided_table(
|
||||
// boundaries so every day gets a column.
|
||||
if col_boundaries.len() >= 2 {
|
||||
let mut spacings: Vec<f32> = col_boundaries.windows(2).map(|w| w[1] - w[0]).collect();
|
||||
spacings.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
spacings.sort_by(|a, b| a.total_cmp(b));
|
||||
let median_spacing = spacings[spacings.len() / 2];
|
||||
let threshold = median_spacing * 1.5;
|
||||
|
||||
@@ -87,7 +87,7 @@ pub(crate) fn try_build_rect_guided_table(
|
||||
|
||||
// 3. Derive row boundaries from item Y positions (5pt tolerance)
|
||||
let mut y_values: Vec<f32> = expanded_items.iter().map(|(item, _)| item.y).collect();
|
||||
y_values.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal)); // descending
|
||||
y_values.sort_by(|a, b| b.total_cmp(a)); // descending
|
||||
let mut row_boundaries: Vec<f32> = Vec::new();
|
||||
for y in &y_values {
|
||||
if row_boundaries
|
||||
@@ -296,7 +296,7 @@ pub(crate) fn try_build_table_from_columns(items: &[TextItem], page: u32) -> Opt
|
||||
|
||||
// Find the top-most row with items in multiple columns (likely the header)
|
||||
let mut ys: Vec<f32> = page_items.iter().map(|i| i.y).collect();
|
||||
ys.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal));
|
||||
ys.sort_by(|a, b| b.total_cmp(a));
|
||||
ys.dedup_by(|a, b| (*a - *b).abs() < y_tol);
|
||||
|
||||
for &header_y in ys.iter().take(5) {
|
||||
@@ -318,7 +318,7 @@ pub(crate) fn try_build_table_from_columns(items: &[TextItem], page: u32) -> Opt
|
||||
if col_items.len() >= 2 {
|
||||
// Sort by X and find the split point
|
||||
let mut sorted: Vec<f32> = col_items.iter().map(|i| i.x).collect();
|
||||
sorted.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
sorted.sort_by(|a, b| a.total_cmp(b));
|
||||
// Split at the midpoint between the two items
|
||||
let split_x = (sorted[0]
|
||||
+ col_items.iter().find(|i| i.x == sorted[0]).unwrap().width
|
||||
@@ -411,7 +411,7 @@ pub(crate) fn try_build_table_from_columns(items: &[TextItem], page: u32) -> Opt
|
||||
}
|
||||
}
|
||||
}
|
||||
row_ys.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal));
|
||||
row_ys.sort_by(|a, b| b.total_cmp(a));
|
||||
|
||||
if row_ys.len() < 3 || row_ys.len() > 40 {
|
||||
return None;
|
||||
|
||||
+4
-4
@@ -68,9 +68,9 @@ where
|
||||
pub(crate) fn sort_line_items(items: &mut [TextItem]) {
|
||||
let rtl = is_rtl_text(items.iter().map(|i| &i.text));
|
||||
if rtl {
|
||||
items.sort_by(|a, b| b.x.partial_cmp(&a.x).unwrap_or(std::cmp::Ordering::Equal));
|
||||
items.sort_by(|a, b| b.x.total_cmp(&a.x));
|
||||
} else {
|
||||
items.sort_by(|a, b| a.x.partial_cmp(&b.x).unwrap_or(std::cmp::Ordering::Equal));
|
||||
items.sort_by(|a, b| a.x.total_cmp(&b.x));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -376,7 +376,7 @@ fn compute_canva_join_threshold(items: &[TextItem]) -> f32 {
|
||||
}
|
||||
|
||||
let mut sorted: Vec<f32> = ratios;
|
||||
sorted.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
sorted.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
if sorted[sorted.len() - 1] < 0.40 || sorted[0] < 0.40 {
|
||||
return DEFAULT;
|
||||
@@ -478,7 +478,7 @@ fn compute_single_char_join_threshold(items: &[TextItem]) -> f32 {
|
||||
return DEFAULT;
|
||||
}
|
||||
|
||||
ratios.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
ratios.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
// If all gaps are tight (max < 0.40), use default — normal PDF
|
||||
let max_ratio = ratios[ratios.len() - 1];
|
||||
|
||||
+82
-12
@@ -1650,7 +1650,7 @@ fn merge_cmaps(mut base: ToUnicodeCMap, overlay: ToUnicodeCMap) -> ToUnicodeCMap
|
||||
///
|
||||
/// Returns true if the median CID is >= 0x41 (letter 'A'), indicating
|
||||
/// the PDF generator likely used Unicode codepoints as CIDs.
|
||||
fn cid_values_look_like_unicode(cid_font_dict: &lopdf::Dictionary) -> bool {
|
||||
pub(crate) fn cid_values_look_like_unicode(cid_font_dict: &lopdf::Dictionary) -> bool {
|
||||
let w_arr = match cid_font_dict.get(b"W").ok() {
|
||||
Some(Object::Array(arr)) => arr,
|
||||
_ => return false,
|
||||
@@ -1771,15 +1771,49 @@ impl FontCMaps {
|
||||
/// Iterates every page, collects fonts (including Form XObject fonts),
|
||||
/// and parses any `/ToUnicode` streams via lopdf's decompression.
|
||||
pub fn from_doc(doc: &Document) -> Self {
|
||||
Self::from_doc_pages(doc, None)
|
||||
}
|
||||
|
||||
/// Build FontCMaps for specific pages only. Pass `None` for all pages.
|
||||
pub fn from_doc_pages(doc: &Document, page_filter: Option<&HashSet<u32>>) -> Self {
|
||||
Self::from_doc_pages_inner(doc, page_filter, false)
|
||||
}
|
||||
|
||||
/// Build FontCMaps in fast mode: skip expensive TrueType font fallback
|
||||
/// parsing. Fonts that can't be decoded from their ToUnicode CMap alone
|
||||
/// will be missing, causing text extraction to produce empty/garbage text
|
||||
/// which triggers `needs_ocr` fallback. This is ideal for hybrid OCR
|
||||
/// pipelines where GPU OCR is always available as a fallback.
|
||||
pub fn from_doc_pages_fast(doc: &Document, page_filter: Option<&HashSet<u32>>) -> Self {
|
||||
Self::from_doc_pages_inner(doc, page_filter, true)
|
||||
}
|
||||
|
||||
fn from_doc_pages_inner(
|
||||
doc: &Document,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
skip_truetype_fallback: bool,
|
||||
) -> Self {
|
||||
let mut by_obj_num: HashMap<u32, CMapEntry> = HashMap::new();
|
||||
|
||||
for (_page_num, &page_id) in doc.get_pages().iter() {
|
||||
for (page_num, &page_id) in doc.get_pages().iter() {
|
||||
if let Some(filter) = page_filter {
|
||||
if !filter.contains(page_num) {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
// Page-level fonts (includes inherited parent resources)
|
||||
let fonts = doc.get_page_fonts(page_id).unwrap_or_default();
|
||||
Self::collect_cmaps_from_fonts(&fonts, doc, &mut by_obj_num);
|
||||
Self::collect_cmaps_from_fonts_inner(
|
||||
&fonts,
|
||||
doc,
|
||||
&mut by_obj_num,
|
||||
skip_truetype_fallback,
|
||||
);
|
||||
|
||||
// Fonts inside Form XObjects referenced by this page
|
||||
Self::collect_cmaps_from_xobjects(doc, page_id, &mut by_obj_num);
|
||||
if !skip_truetype_fallback {
|
||||
// Fonts inside Form XObjects referenced by this page
|
||||
Self::collect_cmaps_from_xobjects(doc, page_id, &mut by_obj_num);
|
||||
}
|
||||
}
|
||||
|
||||
FontCMaps { by_obj_num }
|
||||
@@ -1792,6 +1826,15 @@ impl FontCMaps {
|
||||
fonts: &std::collections::BTreeMap<Vec<u8>, &lopdf::Dictionary>,
|
||||
doc: &Document,
|
||||
by_obj_num: &mut HashMap<u32, CMapEntry>,
|
||||
) {
|
||||
Self::collect_cmaps_from_fonts_inner(fonts, doc, by_obj_num, false);
|
||||
}
|
||||
|
||||
fn collect_cmaps_from_fonts_inner(
|
||||
fonts: &std::collections::BTreeMap<Vec<u8>, &lopdf::Dictionary>,
|
||||
doc: &Document,
|
||||
by_obj_num: &mut HashMap<u32, CMapEntry>,
|
||||
skip_truetype_fallback: bool,
|
||||
) {
|
||||
// First pass: collect ToUnicode CMaps
|
||||
for font_dict in fonts.values() {
|
||||
@@ -1825,13 +1868,32 @@ impl FontCMaps {
|
||||
);
|
||||
let (mut primary, mut remapped) =
|
||||
try_remap_subset_cmap(cmap, font_dict, doc, obj_num);
|
||||
let mut fallback = build_fallback_tounicode_from_encoding(font_dict, doc)
|
||||
.or_else(|| build_fallback_cmap_for_type0(font_dict, doc))
|
||||
.or_else(|| build_fallback_cmap_for_simple(font_dict, doc));
|
||||
|
||||
// If the ToUnicode map is extremely sparse, prefer the fallback
|
||||
// (often a better mapping for Symbol/Wingdings/Arabic CID fonts).
|
||||
// Only build expensive fallbacks when the primary CMap is sparse.
|
||||
// build_fallback_cmap_for_type0 can take seconds on large embedded
|
||||
// TrueType fonts (decompressing + parsing 100K+ byte font files).
|
||||
// Skip entirely when the primary CMap is sufficient.
|
||||
let primary_entries = primary.char_map.len() + primary.ranges.len();
|
||||
let mut fallback = if primary_entries < 10 && !skip_truetype_fallback {
|
||||
// Try cheap fallback first; only attempt expensive TrueType
|
||||
// parsing if cheap fallbacks don't yield results.
|
||||
let cheap = build_fallback_tounicode_from_encoding(font_dict, doc)
|
||||
.or_else(|| build_fallback_cmap_for_simple(font_dict, doc));
|
||||
if cheap.is_some() {
|
||||
cheap
|
||||
} else {
|
||||
build_fallback_cmap_for_type0(font_dict, doc)
|
||||
}
|
||||
} else if primary_entries < 10 {
|
||||
// Fast mode: only try cheap fallbacks, skip TrueType parsing.
|
||||
// Regions using this font will get needs_ocr=true.
|
||||
build_fallback_tounicode_from_encoding(font_dict, doc)
|
||||
.or_else(|| build_fallback_cmap_for_simple(font_dict, doc))
|
||||
} else {
|
||||
// Primary is rich enough; only try the cheap encoding fallback
|
||||
build_fallback_tounicode_from_encoding(font_dict, doc)
|
||||
};
|
||||
|
||||
if primary_entries < 10 {
|
||||
if let Some(fb) = fallback.take() {
|
||||
debug!(
|
||||
@@ -1852,8 +1914,12 @@ impl FontCMaps {
|
||||
);
|
||||
} else {
|
||||
// ToUnicode present but parse failed; try fallbacks to avoid empty decoding.
|
||||
let fallback = build_fallback_cmap_for_type0(font_dict, doc)
|
||||
.or_else(|| build_fallback_cmap_for_simple(font_dict, doc));
|
||||
let fallback = if skip_truetype_fallback {
|
||||
build_fallback_cmap_for_simple(font_dict, doc)
|
||||
} else {
|
||||
build_fallback_cmap_for_type0(font_dict, doc)
|
||||
.or_else(|| build_fallback_cmap_for_simple(font_dict, doc))
|
||||
};
|
||||
if let Some(fb) = fallback {
|
||||
debug!(
|
||||
"ToUnicode CMap obj={} parse failed; using fallback (entries={})",
|
||||
@@ -1874,6 +1940,10 @@ impl FontCMaps {
|
||||
|
||||
// Second pass: Identity-H/V fonts without ToUnicode
|
||||
// Try: (1) embedded TrueType/OpenType cmap, (2) predefined CID→Unicode mapping
|
||||
// Skip entirely in fast mode — these fonts require expensive TrueType parsing.
|
||||
if skip_truetype_fallback {
|
||||
return;
|
||||
}
|
||||
for font_dict in fonts.values() {
|
||||
if font_dict.get(b"ToUnicode").is_ok() {
|
||||
continue;
|
||||
|
||||
BIN
Binary file not shown.
+525
-2
@@ -4,9 +4,12 @@ use pdf_inspector::detector::{DetectionConfig, ScanStrategy};
|
||||
use pdf_inspector::extractor::group_into_lines;
|
||||
use pdf_inspector::types::TextLine;
|
||||
use pdf_inspector::{
|
||||
detect_pdf_type, extract_text, extract_text_with_positions, process_pdf_with_options,
|
||||
to_markdown, MarkdownOptions, PdfError, PdfOptions, PdfType, TextItem,
|
||||
detect_pdf_type, extract_pages_markdown_mem, extract_tables_in_regions_mem, extract_text,
|
||||
extract_text_in_regions_mem, extract_text_with_positions, process_pdf_mem,
|
||||
process_pdf_with_options, to_markdown, MarkdownOptions, PdfError, PdfOptions, PdfType,
|
||||
TextItem,
|
||||
};
|
||||
use std::collections::HashSet;
|
||||
|
||||
// Helper to create test TextItems
|
||||
fn make_text_item(text: &str, x: f32, y: f32, font_size: f32, page: u32) -> TextItem {
|
||||
@@ -1107,3 +1110,523 @@ fn test_rotated_table_layout_correction() {
|
||||
"District data should be in a markdown table row"
|
||||
);
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
// extract_text_in_regions_mem tests
|
||||
// =========================================================================
|
||||
|
||||
/// Build full-page region args for `page_count` pages.
|
||||
/// Uses a generously large bbox (1200x1200) to capture any page size.
|
||||
fn full_page_regions(page_count: u32) -> Vec<(u32, Vec<[f32; 4]>)> {
|
||||
(0..page_count)
|
||||
.map(|p| (p, vec![[0.0, 0.0, 1200.0, 1200.0]]))
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Normalize text for comparison: lowercase, strip non-alphanumeric, split into words.
|
||||
fn normalize_words(text: &str) -> HashSet<String> {
|
||||
text.split(|c: char| !c.is_alphanumeric())
|
||||
.map(|w| w.to_lowercase())
|
||||
.filter(|w| w.len() > 3)
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Fraction of normalized words in `a` that also appear in `b`.
|
||||
fn word_overlap_ratio(a: &str, b: &str) -> f64 {
|
||||
let words_a = normalize_words(a);
|
||||
if words_a.is_empty() {
|
||||
return if normalize_words(b).is_empty() {
|
||||
1.0
|
||||
} else {
|
||||
0.0
|
||||
};
|
||||
}
|
||||
let words_b = normalize_words(b);
|
||||
let overlap = words_a.intersection(&words_b).count();
|
||||
overlap as f64 / words_a.len() as f64
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_regions_mem_basic_text_pdf() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
let result = process_pdf_mem(&buf).unwrap();
|
||||
let page_count = result.page_count;
|
||||
|
||||
let regions = extract_text_in_regions_mem(&buf, &full_page_regions(page_count)).unwrap();
|
||||
assert_eq!(regions.len(), page_count as usize);
|
||||
|
||||
// Each result should have exactly 1 region (we passed one per page)
|
||||
for r in ®ions {
|
||||
assert_eq!(r.regions.len(), 1);
|
||||
}
|
||||
|
||||
// First page should have non-empty text
|
||||
let first = ®ions[0].regions[0];
|
||||
assert!(!first.text.trim().is_empty(), "First page should have text");
|
||||
assert_eq!(regions[0].page, 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_regions_mem_identity_h_needs_ocr() {
|
||||
let buf = std::fs::read("tests/fixtures/shinagawa_identity_h.pdf").unwrap();
|
||||
let regions =
|
||||
extract_text_in_regions_mem(&buf, &[(0, vec![[0.0, 0.0, 1200.0, 1200.0]])]).unwrap();
|
||||
assert_eq!(regions.len(), 1);
|
||||
assert!(
|
||||
regions[0].regions[0].needs_ocr,
|
||||
"Identity-H font without ToUnicode should trigger needs_ocr"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_regions_mem_multiple_regions_per_page() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
let regions = extract_text_in_regions_mem(
|
||||
&buf,
|
||||
&[(
|
||||
0,
|
||||
vec![
|
||||
[0.0, 0.0, 300.0, 100.0], // small top-left
|
||||
[0.0, 0.0, 1200.0, 1200.0], // full page
|
||||
],
|
||||
)],
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(regions.len(), 1);
|
||||
assert_eq!(regions[0].regions.len(), 2);
|
||||
|
||||
let small_len = regions[0].regions[0].text.len();
|
||||
let full_len = regions[0].regions[1].text.len();
|
||||
assert!(
|
||||
full_len >= small_len,
|
||||
"Full-page region ({full_len}) should have at least as much text as small region ({small_len})"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_regions_mem_nonexistent_page() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
let regions =
|
||||
extract_text_in_regions_mem(&buf, &[(9999, vec![[0.0, 0.0, 1200.0, 1200.0]])]).unwrap();
|
||||
assert_eq!(regions.len(), 1);
|
||||
assert!(
|
||||
regions[0].regions[0].needs_ocr,
|
||||
"Nonexistent page should trigger needs_ocr"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_regions_mem_empty_region() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
let regions = extract_text_in_regions_mem(&buf, &[(0, vec![[0.0, 0.0, 0.0, 0.0]])]).unwrap();
|
||||
assert_eq!(regions.len(), 1);
|
||||
assert!(
|
||||
regions[0].regions[0].needs_ocr,
|
||||
"Zero-area region should trigger needs_ocr"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_regions_mem_not_a_pdf() {
|
||||
let result = extract_text_in_regions_mem(b"not a pdf", &[(0, vec![[0.0, 0.0, 100.0, 100.0]])]);
|
||||
assert!(result.is_err(), "Non-PDF input should return an error");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_regions_mem_rotated_page_not_false_empty() {
|
||||
let buf = std::fs::read("tests/fixtures/tnagriculture_06_12.pdf").unwrap();
|
||||
let regions =
|
||||
extract_text_in_regions_mem(&buf, &[(0, vec![[0.0, 0.0, 1200.0, 1200.0]])]).unwrap();
|
||||
assert_eq!(regions.len(), 1);
|
||||
assert_eq!(regions[0].regions.len(), 1);
|
||||
let region = ®ions[0].regions[0];
|
||||
assert!(
|
||||
!region.text.trim().is_empty(),
|
||||
"Rotated page full-region extraction should not be empty"
|
||||
);
|
||||
assert!(
|
||||
!region.needs_ocr,
|
||||
"Rotated page with native text should not be flagged for OCR fallback"
|
||||
);
|
||||
assert!(
|
||||
region
|
||||
.text
|
||||
.contains("DISTRICT WISE PRODUCTION OF SPICES AND CONDIMENTS"),
|
||||
"Expected known title from rotated fixture in extracted region text"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_collect_text_in_region_keeps_partial_overlap_items() {
|
||||
let item = make_text_item("EdgeWord", 100.0, 700.0, 12.0, 1);
|
||||
// Region intersects only the left edge of the item. Center x=124 falls
|
||||
// outside x=[95,120], so center-only containment would drop it.
|
||||
let text = pdf_inspector::collect_text_in_region(&[item], 95.0, 80.0, 120.0, 110.0, 800.0);
|
||||
assert!(
|
||||
text.contains("EdgeWord"),
|
||||
"Partially overlapping items should be retained in region extraction"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_collect_text_in_region_uses_rtl_sorting() {
|
||||
let items = vec![
|
||||
make_text_item("بكم", 240.0, 700.0, 12.0, 1),
|
||||
make_text_item("مرحبا", 300.0, 700.0, 12.0, 1),
|
||||
];
|
||||
let text = pdf_inspector::collect_text_in_region(&items, 0.0, 0.0, 600.0, 800.0, 800.0);
|
||||
assert_eq!(
|
||||
text, "مرحبا بكم",
|
||||
"Region path should reuse RTL-aware line sorting"
|
||||
);
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
// Fast vs normal extraction comparison
|
||||
// =========================================================================
|
||||
|
||||
/// For each text-based fixture PDF, compare `extract_text_in_regions_mem` (fast path)
|
||||
/// against `process_pdf_mem` (normal path). If the fast path claims needs_ocr=false
|
||||
/// for a page, verify the extracted text has meaningful overlap with the normal
|
||||
/// markdown output — catching silent quality regressions.
|
||||
#[test]
|
||||
fn test_extract_regions_fast_vs_normal_comparison() {
|
||||
let fixtures = [
|
||||
"tests/fixtures/nexo-price-en.pdf",
|
||||
"tests/fixtures/td9264.pdf",
|
||||
"tests/fixtures/p1244-1996.pdf",
|
||||
"tests/fixtures/real-estate-pricing.pdf",
|
||||
"tests/fixtures/2013-app2.pdf",
|
||||
"tests/fixtures/firecrawl_docs_tagged.pdf",
|
||||
"tests/fixtures/thermo-freon12.pdf",
|
||||
];
|
||||
|
||||
for fixture in &fixtures {
|
||||
let buf = std::fs::read(fixture).unwrap();
|
||||
let normal = process_pdf_mem(&buf).unwrap();
|
||||
let normal_md = normal.markdown.as_deref().unwrap_or("");
|
||||
let page_count = normal.page_count;
|
||||
let ocr_pages: HashSet<u32> = normal.pages_needing_ocr.iter().copied().collect();
|
||||
|
||||
let regions = extract_text_in_regions_mem(&buf, &full_page_regions(page_count)).unwrap();
|
||||
|
||||
assert_eq!(
|
||||
regions.len(),
|
||||
page_count as usize,
|
||||
"{fixture}: result count should match page count"
|
||||
);
|
||||
|
||||
for pr in ®ions {
|
||||
let region = &pr.regions[0];
|
||||
if !region.needs_ocr && !region.text.trim().is_empty() {
|
||||
// Fast path claims this text is trustworthy.
|
||||
// Check that its words appear in the normal markdown output.
|
||||
let overlap = word_overlap_ratio(®ion.text, normal_md);
|
||||
assert!(
|
||||
overlap >= 0.3,
|
||||
"{fixture} page {}: fast path says needs_ocr=false but only {:.0}% word \
|
||||
overlap with normal extraction (threshold 30%). \
|
||||
Fast text sample: {:?}",
|
||||
pr.page,
|
||||
overlap * 100.0,
|
||||
®ion.text[..region.text.len().min(200)],
|
||||
);
|
||||
}
|
||||
|
||||
// If fast path flags needs_ocr but normal path didn't, that's overly
|
||||
// conservative but not a bug — just worth knowing.
|
||||
if region.needs_ocr && !ocr_pages.contains(&(pr.page + 1)) {
|
||||
eprintln!(
|
||||
"INFO: {fixture} page {}: fast path says needs_ocr=true but normal path extracted fine (conservative, not a bug)",
|
||||
pr.page,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
// extract_tables_in_regions_mem tests
|
||||
// =========================================================================
|
||||
|
||||
#[test]
|
||||
fn test_extract_tables_in_regions_table_pdf() {
|
||||
// tnagriculture has a clear table with district names and spice columns
|
||||
let buf = std::fs::read("tests/fixtures/tnagriculture_06_12.pdf").unwrap();
|
||||
let results =
|
||||
extract_tables_in_regions_mem(&buf, &[(0, vec![[0.0, 0.0, 1200.0, 1200.0]])]).unwrap();
|
||||
|
||||
assert_eq!(results.len(), 1);
|
||||
assert_eq!(results[0].regions.len(), 1);
|
||||
|
||||
let region = &results[0].regions[0];
|
||||
// Should detect a table with pipe-delimited markdown
|
||||
if !region.needs_ocr {
|
||||
assert!(
|
||||
region.text.contains('|'),
|
||||
"Table output should contain pipe delimiters"
|
||||
);
|
||||
// Should have separator row
|
||||
assert!(
|
||||
region.text.lines().any(|l| l.contains("---")),
|
||||
"Table output should contain separator row"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_tables_in_regions_non_table_region() {
|
||||
// Use a small region that likely won't contain enough items for a table
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
let results =
|
||||
extract_tables_in_regions_mem(&buf, &[(0, vec![[0.0, 0.0, 50.0, 50.0]])]).unwrap();
|
||||
|
||||
assert_eq!(results.len(), 1);
|
||||
assert_eq!(results[0].regions.len(), 1);
|
||||
|
||||
let region = &results[0].regions[0];
|
||||
// Small region with few items should fall back to needs_ocr
|
||||
assert!(
|
||||
region.needs_ocr,
|
||||
"Non-table region should set needs_ocr = true"
|
||||
);
|
||||
assert!(
|
||||
region.text.is_empty(),
|
||||
"Non-table region should have empty text"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_tables_in_regions_empty_region() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
let results = extract_tables_in_regions_mem(&buf, &[(0, vec![[0.0, 0.0, 0.0, 0.0]])]).unwrap();
|
||||
|
||||
assert_eq!(results.len(), 1);
|
||||
let region = &results[0].regions[0];
|
||||
assert!(region.needs_ocr);
|
||||
assert!(region.text.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_tables_in_regions_identity_h_needs_ocr() {
|
||||
let buf = std::fs::read("tests/fixtures/shinagawa_identity_h.pdf").unwrap();
|
||||
let results =
|
||||
extract_tables_in_regions_mem(&buf, &[(0, vec![[0.0, 0.0, 1200.0, 1200.0]])]).unwrap();
|
||||
|
||||
assert_eq!(results.len(), 1);
|
||||
let region = &results[0].regions[0];
|
||||
assert!(region.needs_ocr, "Identity-H font should trigger needs_ocr");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_tables_in_regions_not_a_pdf() {
|
||||
let result =
|
||||
extract_tables_in_regions_mem(b"not a pdf", &[(0, vec![[0.0, 0.0, 100.0, 100.0]])]);
|
||||
assert!(result.is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_tables_in_regions_nonexistent_page() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
let results =
|
||||
extract_tables_in_regions_mem(&buf, &[(9999, vec![[0.0, 0.0, 1200.0, 1200.0]])]).unwrap();
|
||||
|
||||
assert_eq!(results.len(), 1);
|
||||
let region = &results[0].regions[0];
|
||||
assert!(region.needs_ocr);
|
||||
assert!(region.text.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_bits_pilani_page4_table_detection() {
|
||||
// Page 4 (0-indexed 3) has a table with multi-line wrapped headers and
|
||||
// numeric data columns. The heuristic detector previously failed because:
|
||||
// 1. Header items at different X positions than data created extra column
|
||||
// clusters (6 cols instead of 4)
|
||||
// 2. Spanning super-header row ("First Degree | First Degree") produced
|
||||
// duplicate header cells that looks_like_partial_table_ex rejected
|
||||
let buf = std::fs::read("tests/fixtures/bits_pilani_feedback.pdf").unwrap();
|
||||
let results =
|
||||
extract_tables_in_regions_mem(&buf, &[(3, vec![[0.0, 0.0, 612.0, 792.0]])]).unwrap();
|
||||
assert_eq!(results.len(), 1);
|
||||
let region = &results[0].regions[0];
|
||||
assert!(
|
||||
!region.needs_ocr,
|
||||
"Page 4 table should be detected, got needs_ocr=true"
|
||||
);
|
||||
assert!(
|
||||
region.text.contains("BIO"),
|
||||
"Should contain department name BIO"
|
||||
);
|
||||
assert!(region.text.contains("8.23"), "Should contain numeric data");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_bits_pilani_page8_table_detection() {
|
||||
// Page 8 (0-indexed 7) has a numbered-row table that already worked.
|
||||
// Verify it still works after changes.
|
||||
let buf = std::fs::read("tests/fixtures/bits_pilani_feedback.pdf").unwrap();
|
||||
let results =
|
||||
extract_tables_in_regions_mem(&buf, &[(7, vec![[0.0, 0.0, 612.0, 792.0]])]).unwrap();
|
||||
assert_eq!(results.len(), 1);
|
||||
let region = &results[0].regions[0];
|
||||
assert!(!region.needs_ocr, "Page 8 table should still be detected");
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
// extract_pages_markdown_mem tests
|
||||
// =========================================================================
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_basic() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, &[0, 1]).unwrap();
|
||||
|
||||
assert_eq!(result.pages.len(), 2);
|
||||
assert_eq!(result.pages[0].page, 0);
|
||||
assert_eq!(result.pages[1].page, 1);
|
||||
// Text-based PDF should produce non-empty markdown
|
||||
assert!(!result.pages[0].markdown.is_empty());
|
||||
assert!(!result.pages[0].needs_ocr);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_page_ordering() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
// Request pages in non-sequential order
|
||||
let result = extract_pages_markdown_mem(&buf, &[1, 0]).unwrap();
|
||||
|
||||
assert_eq!(result.pages.len(), 2);
|
||||
// Results should match input order, not document order
|
||||
assert_eq!(result.pages[0].page, 1);
|
||||
assert_eq!(result.pages[1].page, 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_out_of_range() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, &[9999]).unwrap();
|
||||
|
||||
assert_eq!(result.pages.len(), 1);
|
||||
assert_eq!(result.pages[0].page, 9999);
|
||||
assert!(result.pages[0].markdown.is_empty());
|
||||
assert!(result.pages[0].needs_ocr);
|
||||
assert!(result.pages_needing_ocr.contains(&10000)); // 1-indexed
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_empty_pages_list() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, &[]).unwrap();
|
||||
assert!(result.pages.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_single_page() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, &[0]).unwrap();
|
||||
|
||||
assert_eq!(result.pages.len(), 1);
|
||||
assert_eq!(result.pages[0].page, 0);
|
||||
assert!(!result.pages[0].markdown.is_empty());
|
||||
assert!(!result.pages[0].needs_ocr);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_invalid_buffer() {
|
||||
let result = extract_pages_markdown_mem(b"not a pdf", &[0]);
|
||||
assert!(result.is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_gid_pages_need_ocr() {
|
||||
// shinagawa_identity_h.pdf has GID-encoded fonts
|
||||
let buf = std::fs::read("tests/fixtures/shinagawa_identity_h.pdf").unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, &[0]).unwrap();
|
||||
|
||||
assert_eq!(result.pages.len(), 1);
|
||||
assert!(result.pages[0].needs_ocr);
|
||||
assert!(result.pages_needing_ocr.contains(&1)); // 1-indexed
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_classification_with_tables() {
|
||||
// nexo-price-en.pdf is known to have tables
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
let page_count = process_pdf_mem(&buf).unwrap().page_count;
|
||||
let page_indices: Vec<u32> = (0..page_count).collect();
|
||||
let result = extract_pages_markdown_mem(&buf, &page_indices).unwrap();
|
||||
|
||||
assert!(
|
||||
!result.pages_with_tables.is_empty(),
|
||||
"nexo-price-en.pdf should have pages with tables"
|
||||
);
|
||||
assert!(result.is_complex);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_simple_pdf_no_complexity() {
|
||||
// bare_name_struct.pdf is a simple document with a heading and code block
|
||||
let buf = std::fs::read("tests/fixtures/bare_name_struct.pdf").unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, &[0]).unwrap();
|
||||
|
||||
assert!(result.pages_with_tables.is_empty());
|
||||
assert!(result.pages_with_columns.is_empty());
|
||||
assert!(!result.is_complex);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_classification_matches_process_pdf() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
let full = process_pdf_mem(&buf).unwrap();
|
||||
let page_count = full.page_count;
|
||||
let page_indices: Vec<u32> = (0..page_count).collect();
|
||||
let result = extract_pages_markdown_mem(&buf, &page_indices).unwrap();
|
||||
|
||||
assert_eq!(
|
||||
result.pages_with_tables, full.layout.pages_with_tables,
|
||||
"pages_with_tables should match process_pdf"
|
||||
);
|
||||
assert_eq!(
|
||||
result.pages_with_columns, full.layout.pages_with_columns,
|
||||
"pages_with_columns should match process_pdf"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_consistency_with_process_pdf() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
|
||||
// Get full process_pdf output
|
||||
let full = process_pdf_mem(&buf).unwrap();
|
||||
let full_md = full.markdown.unwrap_or_default();
|
||||
|
||||
// Get per-page output for all pages
|
||||
let page_count = full.page_count;
|
||||
let page_indices: Vec<u32> = (0..page_count).collect();
|
||||
let result = extract_pages_markdown_mem(&buf, &page_indices).unwrap();
|
||||
|
||||
// Concatenated per-page markdown should contain substantial overlap with
|
||||
// the full output (exact match not expected due to header/footer stripping
|
||||
// and cross-page paragraph merging differences)
|
||||
let concat: String = result
|
||||
.pages
|
||||
.iter()
|
||||
.map(|p| p.markdown.as_str())
|
||||
.collect::<Vec<_>>()
|
||||
.join("\n");
|
||||
|
||||
// Both should be non-empty for a text-based PDF
|
||||
assert!(!full_md.is_empty());
|
||||
assert!(!concat.is_empty());
|
||||
|
||||
// The per-page version should contain at least 50% of the full content's
|
||||
// length (accounting for header/footer stripping differences)
|
||||
assert!(
|
||||
concat.len() * 2 >= full_md.len(),
|
||||
"per-page concat ({} chars) is too short vs full ({} chars)",
|
||||
concat.len(),
|
||||
full_md.len()
|
||||
);
|
||||
}
|
||||
|
||||
@@ -8,7 +8,9 @@ Department of the Treasury **Internal Revenue Service**
|
||||
|
||||
# and Report to Employer
|
||||
|
||||
**This publication contains:** **Form 4070A, Employee’s Daily Record of** Tips **Form 4070, Employee’s Report of Tips to** Employer
|
||||
### This publication contains:
|
||||
|
||||
**Form 4070A, Employee’s Daily Record of** Tips **Form 4070, Employee’s Report of Tips to** Employer
|
||||
|
||||
For the period
|
||||
|
||||
@@ -74,7 +76,7 @@ forms simpler, we would be happy to hear from you. You can write to the Tax Form
|
||||
|
||||
**Unreported Tips.—If you received tips of $20 or** more for any month while working for one employer but did not report them to your employer, you must figure and pay social security and Medicare taxes on the unreported tips when you file your tax return. If you have unreported tips, you must use Form 1040 and Form 4137, Social Security and Medicare Tax on Unreported Tip Income, to report them. You may not use Form 1040A or 1040EZ. Employees subject to the Railroad Retirement Tax Act cannot use Form 4137 to pay railroad retirement tax on unreported tips. To get railroad retirement credit, you must report tips to your employer. If you do not report tips to your employer as required, you may be charged a penalty of 50% of the social security and Medicare taxes (or railroad retirement tax) due on the unreported tips unless there was reasonable cause for not reporting them. **Additional Information.—Get Pub. 531, Reporting** Tip Income, and Form 4137 for more information on tips. If you are an employee of certain large food or beverage establishments, see Pub. 531 for tip allocation rules. **Recordkeeping.—If you do not keep a daily** record of tips, you must keep other reliable proof of the tip income you received. This proof includes copies of restaurant bills and credit card charges that show amounts customers added as tips. Keep your tip income records for as long as the information on them may be needed in the administration of any Internal Revenue law.
|
||||
|
||||
**Instructions (continued)**
|
||||
### Instructions (continued)
|
||||
|
||||
Use this space to total your tips for the year
|
||||
|
||||
|
||||
@@ -6,9 +6,7 @@
|
||||
|
||||
8 4 Z E L L / L U R I E R E A L E S T A T E C E N T E R
|
||||
|
||||
**Table I: Cap rate correlations**
|
||||
|
||||
**Cap Rate Correlation With:*** **BBB Corp** **10-Year Bond Yield S&P Dividend** **Treasury (10-15 yr) Yield** Multifamily 0.187 0.771 0.068 Industrial-0.221 0.748-0.307 CBD Office-0.449 0.694-0.458 Retail-0.181 0.649-02.58
|
||||
**Table I: Cap rate correlations** **Cap Rate Correlation With:*** **BBB Corp** **10-Year Bond Yield S&P Dividend** **Treasury (10-15 yr) Yield** Multifamily 0.187 0.771 0.068 Industrial-0.221 0.748-0.307 CBD Office-0.449 0.694-0.458 Retail-0.181 0.649-02.58
|
||||
|
||||
* Based on 25 years of data for the 10-yrT & S&P DivYld; and 14 years for BBB.
|
||||
**Figure 1:** NCREIF cap rates vs. 10-yearTreasury
|
||||
@@ -34,9 +32,7 @@ R E V I E W 8 5
|
||||
|
||||
1982 1986 1990 1994 1998 2002 2006
|
||||
|
||||
**Table II: Correlationsofspreadsbypropertytype**
|
||||
|
||||
**Correlation of Cap Rate Spreads Over Treasury** **Multifamily Industrial CBD Office**
|
||||
**Table II: Correlationsofspreadsbypropertytype** **Correlation of Cap Rate Spreads Over Treasury** **Multifamily Industrial CBD Office**
|
||||
|
||||
||Multifamily|Industrial|CBD Office|
|
||||
|---|---|---|---|
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
**Technical Information**
|
||||
##### Technical Information
|
||||
|
||||
## l T-12 SI
|
||||
|
||||
DuPont Fluorochemicals
|
||||
##### DuPont Fluorochemicals
|
||||
|
||||
#### Thermodynamic Properties
|
||||
|
||||
@@ -20,25 +20,22 @@ Tables of the thermodynamic **Units** properties of R-12 have been developed and
|
||||
|
||||
S.A., Lemmon, E.W., and Peskin, Vf = Fluid (liquid) specific volume
|
||||
A.P., NIST Standard Reference in cubic meters per kilogram Database 23, NIST thermodynamic and transport properties of Vg = Vapour (gas) specific volume refrigerants and refrigerant in cubic meters per kilogram mixtures – REFPROP version 6.01, Standard Reference Data Program, df and dg = Fluid and Vapour National Institute of Standards and (respectively) densities in Technology, 1998). kilograms per cubic meter
|
||||
H = Enthalpy (kJ/kg)
|
||||
##### H = Enthalpy (kJ/kg)
|
||||
|
||||
S = Entropy (kJ/kg.K)
|
||||
##### S = Entropy (kJ/kg.K)
|
||||
|
||||
**Physical Properties**
|
||||
##### Physical Properties
|
||||
|
||||
Chemical Formula CCl2F2
|
||||
|Chemical Formula|CCl2F2|
|
||||
|---|---|
|
||||
|Molecular mass|120.91|
|
||||
|Boiling Point At one atmosphere|-29.75°C|
|
||||
|Critical Temperature|111.97°C|
|
||||
|Critical Pressure|4136 kPa|
|
||||
|Critical Density|565.0 kg/m|
|
||||
|Critical Volume|0.0018 m|
|
||||
|
||||
Molecular mass 120.91
|
||||
|
||||
Boiling Point-29.75°C At one atmosphere
|
||||
|
||||
Critical Temperature 111.97°C
|
||||
|
||||
Critical Pressure 4136 kPa
|
||||
|
||||
3 Critical Density 565.0 kg/m
|
||||
|
||||
Critical Volume 0.0018 m /kg
|
||||
/kg
|
||||
|
||||
l
|
||||
|
||||
|
||||
@@ -0,0 +1,331 @@
|
||||
"""Tests for the pdf_inspector Python bindings."""
|
||||
|
||||
import os
|
||||
import pytest
|
||||
import pdf_inspector
|
||||
|
||||
FIXTURES_DIR = os.path.join(os.path.dirname(__file__), "fixtures")
|
||||
|
||||
|
||||
def fixture_path(name: str) -> str:
|
||||
return os.path.join(FIXTURES_DIR, name)
|
||||
|
||||
|
||||
def fixture_bytes(name: str) -> bytes:
|
||||
with open(fixture_path(name), "rb") as f:
|
||||
return f.read()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# process_pdf
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestProcessPdf:
|
||||
def test_basic(self):
|
||||
result = pdf_inspector.process_pdf(fixture_path("thermo-freon12.pdf"))
|
||||
assert result.pdf_type == "text_based"
|
||||
assert result.page_count == 3
|
||||
assert result.confidence > 0.0
|
||||
assert result.markdown is not None
|
||||
assert len(result.markdown) > 0
|
||||
|
||||
def test_result_repr(self):
|
||||
result = pdf_inspector.process_pdf(fixture_path("thermo-freon12.pdf"))
|
||||
r = repr(result)
|
||||
assert "PdfResult" in r
|
||||
assert "text_based" in r
|
||||
|
||||
def test_with_pages(self):
|
||||
result = pdf_inspector.process_pdf(
|
||||
fixture_path("thermo-freon12.pdf"), pages=[1]
|
||||
)
|
||||
assert result.page_count == 3 # total pages in doc
|
||||
assert result.markdown is not None
|
||||
|
||||
def test_result_fields(self):
|
||||
result = pdf_inspector.process_pdf(fixture_path("thermo-freon12.pdf"))
|
||||
# All fields should be accessible
|
||||
assert isinstance(result.pdf_type, str)
|
||||
assert isinstance(result.page_count, int)
|
||||
assert isinstance(result.processing_time_ms, int)
|
||||
assert isinstance(result.pages_needing_ocr, list)
|
||||
assert isinstance(result.confidence, float)
|
||||
assert isinstance(result.is_complex_layout, bool)
|
||||
assert isinstance(result.pages_with_tables, list)
|
||||
assert isinstance(result.pages_with_columns, list)
|
||||
assert isinstance(result.has_encoding_issues, bool)
|
||||
# title can be None or str
|
||||
assert result.title is None or isinstance(result.title, str)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# process_pdf_bytes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestProcessPdfBytes:
|
||||
def test_basic(self):
|
||||
data = fixture_bytes("thermo-freon12.pdf")
|
||||
result = pdf_inspector.process_pdf_bytes(data)
|
||||
assert result.pdf_type == "text_based"
|
||||
assert result.markdown is not None
|
||||
|
||||
def test_with_pages(self):
|
||||
data = fixture_bytes("thermo-freon12.pdf")
|
||||
result = pdf_inspector.process_pdf_bytes(data, pages=[1, 2])
|
||||
assert result.markdown is not None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# detect_pdf / detect_pdf_bytes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestDetectPdf:
|
||||
def test_detect_file(self):
|
||||
result = pdf_inspector.detect_pdf(fixture_path("thermo-freon12.pdf"))
|
||||
assert result.pdf_type == "text_based"
|
||||
assert result.markdown is None # detect only — no markdown
|
||||
assert result.page_count == 3
|
||||
|
||||
def test_detect_bytes(self):
|
||||
data = fixture_bytes("thermo-freon12.pdf")
|
||||
result = pdf_inspector.detect_pdf_bytes(data)
|
||||
assert result.pdf_type == "text_based"
|
||||
assert result.markdown is None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# classify_pdf / classify_pdf_bytes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestClassifyPdf:
|
||||
def test_classify_file(self):
|
||||
result = pdf_inspector.classify_pdf(fixture_path("thermo-freon12.pdf"))
|
||||
assert result.pdf_type == "text_based"
|
||||
assert result.page_count == 3
|
||||
assert result.confidence > 0.0
|
||||
assert isinstance(result.pages_needing_ocr, list)
|
||||
|
||||
def test_classify_bytes(self):
|
||||
data = fixture_bytes("thermo-freon12.pdf")
|
||||
result = pdf_inspector.classify_pdf_bytes(data)
|
||||
assert result.pdf_type == "text_based"
|
||||
assert result.page_count == 3
|
||||
assert result.confidence > 0.0
|
||||
|
||||
def test_classify_repr(self):
|
||||
result = pdf_inspector.classify_pdf(fixture_path("thermo-freon12.pdf"))
|
||||
r = repr(result)
|
||||
assert "PdfClassification" in r
|
||||
assert "text_based" in r
|
||||
|
||||
def test_classify_fields(self):
|
||||
result = pdf_inspector.classify_pdf(fixture_path("thermo-freon12.pdf"))
|
||||
assert isinstance(result.pdf_type, str)
|
||||
assert isinstance(result.page_count, int)
|
||||
assert isinstance(result.pages_needing_ocr, list)
|
||||
assert isinstance(result.confidence, float)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# extract_text / extract_text_bytes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestExtractText:
|
||||
def test_basic(self):
|
||||
text = pdf_inspector.extract_text(fixture_path("thermo-freon12.pdf"))
|
||||
assert isinstance(text, str)
|
||||
assert len(text) > 0
|
||||
|
||||
def test_bytes(self):
|
||||
data = fixture_bytes("thermo-freon12.pdf")
|
||||
text = pdf_inspector.extract_text_bytes(data)
|
||||
assert isinstance(text, str)
|
||||
assert len(text) > 0
|
||||
|
||||
def test_bytes_matches_file(self):
|
||||
text_file = pdf_inspector.extract_text(fixture_path("thermo-freon12.pdf"))
|
||||
text_bytes = pdf_inspector.extract_text_bytes(fixture_bytes("thermo-freon12.pdf"))
|
||||
assert text_file == text_bytes
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# extract_text_with_positions / extract_text_with_positions_bytes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestExtractTextWithPositions:
|
||||
def test_basic(self):
|
||||
items = pdf_inspector.extract_text_with_positions(
|
||||
fixture_path("thermo-freon12.pdf")
|
||||
)
|
||||
assert len(items) > 0
|
||||
item = items[0]
|
||||
assert isinstance(item.text, str)
|
||||
assert isinstance(item.x, float)
|
||||
assert isinstance(item.y, float)
|
||||
assert isinstance(item.width, float)
|
||||
assert isinstance(item.height, float)
|
||||
assert isinstance(item.font, str)
|
||||
assert isinstance(item.font_size, float)
|
||||
assert isinstance(item.page, int)
|
||||
assert isinstance(item.is_bold, bool)
|
||||
assert isinstance(item.is_italic, bool)
|
||||
assert isinstance(item.item_type, str)
|
||||
|
||||
def test_with_pages(self):
|
||||
items = pdf_inspector.extract_text_with_positions(
|
||||
fixture_path("thermo-freon12.pdf"), pages=[1]
|
||||
)
|
||||
assert len(items) > 0
|
||||
assert all(item.page == 1 for item in items)
|
||||
|
||||
def test_repr(self):
|
||||
items = pdf_inspector.extract_text_with_positions(
|
||||
fixture_path("thermo-freon12.pdf")
|
||||
)
|
||||
r = repr(items[0])
|
||||
assert "TextItem" in r
|
||||
|
||||
def test_bytes(self):
|
||||
data = fixture_bytes("thermo-freon12.pdf")
|
||||
items = pdf_inspector.extract_text_with_positions_bytes(data)
|
||||
assert len(items) > 0
|
||||
assert isinstance(items[0].text, str)
|
||||
|
||||
def test_bytes_with_pages(self):
|
||||
data = fixture_bytes("thermo-freon12.pdf")
|
||||
items = pdf_inspector.extract_text_with_positions_bytes(data, pages=[1])
|
||||
assert len(items) > 0
|
||||
assert all(item.page == 1 for item in items)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# extract_text_in_regions / extract_text_in_regions_bytes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestExtractTextInRegions:
|
||||
def test_file(self):
|
||||
results = pdf_inspector.extract_text_in_regions(
|
||||
fixture_path("thermo-freon12.pdf"),
|
||||
[(0, [[0.0, 0.0, 600.0, 100.0]])],
|
||||
)
|
||||
assert len(results) == 1
|
||||
assert results[0].page == 0
|
||||
assert len(results[0].regions) == 1
|
||||
assert isinstance(results[0].regions[0].text, str)
|
||||
assert isinstance(results[0].regions[0].needs_ocr, bool)
|
||||
|
||||
def test_bytes(self):
|
||||
data = fixture_bytes("thermo-freon12.pdf")
|
||||
results = pdf_inspector.extract_text_in_regions_bytes(
|
||||
data,
|
||||
[(0, [[0.0, 0.0, 600.0, 100.0]])],
|
||||
)
|
||||
assert len(results) == 1
|
||||
assert results[0].page == 0
|
||||
assert len(results[0].regions) == 1
|
||||
assert isinstance(results[0].regions[0].text, str)
|
||||
|
||||
def test_repr(self):
|
||||
results = pdf_inspector.extract_text_in_regions(
|
||||
fixture_path("thermo-freon12.pdf"),
|
||||
[(0, [[0.0, 0.0, 600.0, 100.0]])],
|
||||
)
|
||||
r = repr(results[0])
|
||||
assert "PageRegionTexts" in r
|
||||
r2 = repr(results[0].regions[0])
|
||||
assert "RegionText" in r2
|
||||
|
||||
def test_multiple_regions(self):
|
||||
results = pdf_inspector.extract_text_in_regions(
|
||||
fixture_path("thermo-freon12.pdf"),
|
||||
[(0, [[0.0, 0.0, 300.0, 100.0], [300.0, 0.0, 600.0, 100.0]])],
|
||||
)
|
||||
assert len(results) == 1
|
||||
assert len(results[0].regions) == 2
|
||||
|
||||
def test_multiple_pages(self):
|
||||
results = pdf_inspector.extract_text_in_regions(
|
||||
fixture_path("thermo-freon12.pdf"),
|
||||
[
|
||||
(0, [[0.0, 0.0, 600.0, 100.0]]),
|
||||
(1, [[0.0, 0.0, 600.0, 100.0]]),
|
||||
],
|
||||
)
|
||||
assert len(results) == 2
|
||||
assert results[0].page == 0
|
||||
assert results[1].page == 1
|
||||
|
||||
def test_malformed_region_raises_value_error(self):
|
||||
with pytest.raises(ValueError, match="Invalid region"):
|
||||
pdf_inspector.extract_text_in_regions(
|
||||
fixture_path("thermo-freon12.pdf"),
|
||||
[(0, [[0.0, 0.0, 600.0]])],
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Error handling
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestErrors:
|
||||
def test_nonexistent_file(self):
|
||||
with pytest.raises(ValueError):
|
||||
pdf_inspector.process_pdf("/nonexistent/file.pdf")
|
||||
|
||||
def test_not_a_pdf(self):
|
||||
with pytest.raises(ValueError):
|
||||
pdf_inspector.process_pdf_bytes(b"this is not a pdf")
|
||||
|
||||
def test_empty_bytes(self):
|
||||
with pytest.raises(ValueError):
|
||||
pdf_inspector.process_pdf_bytes(b"")
|
||||
|
||||
def test_classify_not_a_pdf(self):
|
||||
with pytest.raises(ValueError):
|
||||
pdf_inspector.classify_pdf_bytes(b"not a pdf")
|
||||
|
||||
def test_classify_nonexistent(self):
|
||||
with pytest.raises((ValueError, OSError)):
|
||||
pdf_inspector.classify_pdf("/nonexistent/file.pdf")
|
||||
|
||||
def test_extract_text_bytes_not_a_pdf(self):
|
||||
with pytest.raises(ValueError):
|
||||
pdf_inspector.extract_text_bytes(b"not a pdf")
|
||||
|
||||
def test_regions_not_a_pdf(self):
|
||||
with pytest.raises(ValueError):
|
||||
pdf_inspector.extract_text_in_regions_bytes(
|
||||
b"not a pdf", [(0, [[0.0, 0.0, 100.0, 100.0]])]
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Multiple fixtures
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestMultipleFixtures:
|
||||
"""Run basic processing on all available test fixtures."""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"filename",
|
||||
[f for f in os.listdir(FIXTURES_DIR) if f.endswith(".pdf")],
|
||||
)
|
||||
def test_process_all_fixtures(self, filename):
|
||||
result = pdf_inspector.process_pdf(fixture_path(filename))
|
||||
assert result.pdf_type in (
|
||||
"text_based",
|
||||
"scanned",
|
||||
"image_based",
|
||||
"mixed",
|
||||
)
|
||||
assert result.page_count > 0
|
||||
assert result.confidence >= 0.0
|
||||
Reference in New Issue
Block a user