Compare commits
@@ -49,13 +49,23 @@ jobs:
|
||||
working-directory: napi
|
||||
run: bunx napi build --platform --release
|
||||
|
||||
- name: Upload artifact
|
||||
- name: Upload native binary
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: bindings-${{ matrix.target }}
|
||||
path: napi/*.node
|
||||
if-no-files-found: error
|
||||
|
||||
- name: Upload generated JS bindings
|
||||
if: matrix.target == 'x86_64-unknown-linux-gnu'
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: js-bindings
|
||||
path: |
|
||||
napi/index.js
|
||||
napi/index.d.ts
|
||||
if-no-files-found: error
|
||||
|
||||
publish:
|
||||
name: Publish to npm
|
||||
needs: build
|
||||
@@ -79,8 +89,9 @@ jobs:
|
||||
- name: Collect binaries and publish
|
||||
working-directory: napi
|
||||
run: |
|
||||
# Copy all .node binaries into the package directory
|
||||
cp artifacts/bindings-*/*.node .
|
||||
cp artifacts/js-bindings/index.js .
|
||||
cp artifacts/js-bindings/index.d.ts .
|
||||
|
||||
echo "=== Package contents ==="
|
||||
ls -la *.node index.js index.d.ts
|
||||
|
||||
@@ -24,6 +24,10 @@ Thumbs.db
|
||||
# Build cache
|
||||
**/*.rs.bk
|
||||
|
||||
# NAPI generated (regenerated by `napi prepublish` during CI)
|
||||
napi/index.js
|
||||
napi/index.d.ts
|
||||
|
||||
# Local samples and scripts
|
||||
samples/
|
||||
scripts/
|
||||
|
||||
+8
-2
@@ -8,7 +8,14 @@ description = "Fast PDF inspection, classification, and text extraction with sma
|
||||
license = "MIT"
|
||||
repository = "https://github.com/firecrawl/pdf-inspector"
|
||||
|
||||
[lib]
|
||||
name = "pdf_inspector"
|
||||
crate-type = ["lib", "cdylib"]
|
||||
|
||||
[dependencies]
|
||||
# Python bindings
|
||||
pyo3 = { version = "0.25", features = ["extension-module"], optional = true }
|
||||
|
||||
# PDF parsing
|
||||
lopdf = { git = "https://github.com/J-F-Liu/lopdf", rev = "052674053814a9f4897af94f0b8e46a545c9b329", features = ["rayon"] }
|
||||
|
||||
@@ -35,6 +42,7 @@ tempfile = "3.3"
|
||||
|
||||
[features]
|
||||
default = []
|
||||
python = ["pyo3"]
|
||||
|
||||
[[bin]]
|
||||
name = "pdf2md"
|
||||
@@ -47,5 +55,3 @@ path = "src/bin/detect_pdf.rs"
|
||||
[[bin]]
|
||||
name = "dump_ops"
|
||||
path = "src/bin/dump_ops.rs"
|
||||
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# pdf-inspector
|
||||
|
||||
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR.
|
||||
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Includes bindings for [Python](docs/python.md) and [Node.js](napi/README.md).
|
||||
|
||||
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
|
||||
|
||||
@@ -12,90 +12,64 @@ Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in
|
||||
- **Table detection** — Dual-mode: rectangle-based detection from PDF drawing ops, plus heuristic detection from text alignment. Handles financial tables, footnotes, and continuation tables across pages.
|
||||
- **CID font support** — ToUnicode CMap decoding for Type0/Identity-H fonts, UTF-16BE, UTF-8, and Latin-1 encodings.
|
||||
- **Multi-column layout** — Automatic detection of newspaper-style columns, sequential reading order, and RTL text support.
|
||||
- **Encoding issue detection** — Automatically flags broken font encodings (garbled text, replacement characters) so callers can fall back to OCR.
|
||||
- **Encoding issue detection** — Automatically flags broken font encodings so callers can fall back to OCR.
|
||||
- **Single document load** — The document is parsed once and shared between detection and extraction, avoiding redundant I/O.
|
||||
- **Lightweight** — Pure Rust, no ML models, no external services. Single dependency on `lopdf` for PDF parsing.
|
||||
|
||||
## Quick start
|
||||
|
||||
### As a library
|
||||
### Python
|
||||
|
||||
Add to your `Cargo.toml`:
|
||||
```bash
|
||||
pip install maturin
|
||||
maturin develop --release
|
||||
```
|
||||
|
||||
```python
|
||||
import pdf_inspector
|
||||
|
||||
result = pdf_inspector.process_pdf("document.pdf")
|
||||
print(result.pdf_type) # "text_based", "scanned", "image_based", "mixed"
|
||||
print(result.markdown) # Markdown string or None
|
||||
```
|
||||
|
||||
> Full API reference: [docs/python.md](docs/python.md)
|
||||
|
||||
### Node.js
|
||||
|
||||
```bash
|
||||
npm install @firecrawl/pdf-inspector-js
|
||||
```
|
||||
|
||||
```javascript
|
||||
import { readFileSync } from 'fs';
|
||||
import { processPdf, classifyPdf } from '@firecrawl/pdf-inspector-js';
|
||||
|
||||
const result = processPdf(readFileSync('document.pdf'));
|
||||
console.log(result.pdfType); // "TextBased", "Scanned", "ImageBased", "Mixed"
|
||||
console.log(result.markdown); // Markdown string or null
|
||||
```
|
||||
|
||||
> Full API reference: [napi/README.md](napi/README.md)
|
||||
|
||||
### Rust
|
||||
|
||||
```toml
|
||||
[dependencies]
|
||||
pdf-inspector = { git = "https://github.com/firecrawl/pdf-inspector" }
|
||||
```
|
||||
|
||||
Detect and extract in one call:
|
||||
|
||||
```rust
|
||||
use pdf_inspector::process_pdf;
|
||||
|
||||
let result = process_pdf("document.pdf")?;
|
||||
|
||||
println!("Type: {:?}", result.pdf_type); // TextBased, Scanned, ImageBased, Mixed
|
||||
println!("Confidence: {:.0}%", result.confidence * 100.0);
|
||||
println!("Pages: {}", result.page_count);
|
||||
|
||||
println!("Type: {:?}", result.pdf_type);
|
||||
if let Some(markdown) = &result.markdown {
|
||||
println!("{}", markdown);
|
||||
}
|
||||
```
|
||||
|
||||
Fast metadata-only detection (no text extraction or markdown generation):
|
||||
|
||||
```rust
|
||||
use pdf_inspector::detect_pdf;
|
||||
|
||||
let info = detect_pdf("document.pdf")?;
|
||||
|
||||
match info.pdf_type {
|
||||
pdf_inspector::PdfType::TextBased => {
|
||||
// Extract locally — fast and free
|
||||
}
|
||||
_ => {
|
||||
// Route to OCR service
|
||||
// info.pages_needing_ocr tells you exactly which pages
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
Customize processing with `PdfOptions`:
|
||||
|
||||
```rust
|
||||
use pdf_inspector::{process_pdf_with_options, PdfOptions, ProcessMode, DetectionConfig, ScanStrategy};
|
||||
|
||||
// Analyze layout without generating markdown
|
||||
let result = process_pdf_with_options(
|
||||
"document.pdf",
|
||||
PdfOptions::new().mode(ProcessMode::Analyze),
|
||||
)?;
|
||||
|
||||
// Full extraction with custom detection strategy
|
||||
let result = process_pdf_with_options(
|
||||
"large.pdf",
|
||||
PdfOptions::new().detection(DetectionConfig {
|
||||
strategy: ScanStrategy::Sample(5),
|
||||
..Default::default()
|
||||
}),
|
||||
)?;
|
||||
|
||||
// Process only specific pages
|
||||
let result = process_pdf_with_options(
|
||||
"document.pdf",
|
||||
PdfOptions::new().pages([1, 3, 5]),
|
||||
)?;
|
||||
```
|
||||
|
||||
Process from a byte buffer (no filesystem needed):
|
||||
|
||||
```rust
|
||||
use pdf_inspector::process_pdf_mem;
|
||||
|
||||
let bytes = std::fs::read("document.pdf")?;
|
||||
let result = process_pdf_mem(&bytes)?;
|
||||
```
|
||||
> Full API reference: [docs/rust-api.md](docs/rust-api.md)
|
||||
|
||||
### CLI
|
||||
|
||||
@@ -120,7 +94,6 @@ cargo run --bin detect-pdf -- document.pdf
|
||||
cargo run --bin detect-pdf -- document.pdf --json
|
||||
|
||||
# Detection + layout analysis (tables, columns)
|
||||
cargo run --bin detect-pdf -- document.pdf --analyze
|
||||
cargo run --bin detect-pdf -- document.pdf --analyze --json
|
||||
```
|
||||
|
||||
@@ -159,6 +132,7 @@ The document is loaded **once** via `load_document_from_path` / `load_document_f
|
||||
```
|
||||
src/
|
||||
lib.rs — Public API, PdfOptions builder, convenience functions
|
||||
python.rs — PyO3 Python bindings
|
||||
types.rs — Shared types: TextItem, TextLine, PdfRect, ItemType
|
||||
text_utils.rs — Character/text helpers (CJK, RTL, ligatures, bold/italic)
|
||||
process_mode.rs — ProcessMode enum (DetectOnly, Analyze, Full)
|
||||
@@ -169,6 +143,7 @@ src/
|
||||
tables/ — Table detection and formatting
|
||||
markdown/ — Markdown conversion and structure detection
|
||||
bin/ — CLI tools (pdf2md, detect_pdf)
|
||||
napi/ — Node.js/Bun bindings (napi-rs)
|
||||
```
|
||||
|
||||
## How classification works
|
||||
@@ -189,50 +164,6 @@ This detects 300+ page PDFs in milliseconds. The result includes `pages_needing_
|
||||
| `Sample(n)` | Sample `n` evenly distributed pages (first, last, middle) | Very large PDFs where speed matters more than precision |
|
||||
| `Pages(vec)` | Only scan specific 1-indexed page numbers | When the caller knows which pages to check |
|
||||
|
||||
## API
|
||||
|
||||
### Processing modes
|
||||
|
||||
| Mode | What it does | Returns |
|
||||
|---|---|---|
|
||||
| `ProcessMode::Full` (default) | Detect + extract + convert to Markdown | Everything populated |
|
||||
| `ProcessMode::Analyze` | Detect + extract + layout analysis (no Markdown) | `markdown` is `None`, `layout` is populated |
|
||||
| `ProcessMode::DetectOnly` | Classification only (fastest) | `markdown` is `None`, `layout` is default |
|
||||
|
||||
### Functions
|
||||
|
||||
| Function | Description |
|
||||
|---|---|
|
||||
| `process_pdf(path)` | Full processing with defaults |
|
||||
| `detect_pdf(path)` | Fast metadata-only detection (no extraction) |
|
||||
| `process_pdf_with_options(path, options)` | Process with custom `PdfOptions` |
|
||||
| `process_pdf_mem(bytes)` | Full processing from a byte buffer |
|
||||
| `detect_pdf_mem(bytes)` | Fast detection from a byte buffer |
|
||||
| `process_pdf_mem_with_options(bytes, options)` | Process from bytes with custom options |
|
||||
| `extract_text(path)` | Plain text extraction |
|
||||
| `extract_text_with_positions(path)` | Text with X/Y coordinates and font info |
|
||||
| `to_markdown(text, options)` | Convert plain text to Markdown |
|
||||
| `to_markdown_from_items(items, options)` | Markdown from pre-extracted `TextItem`s |
|
||||
| `to_markdown_from_items_with_rects(items, options, rects)` | Markdown with rectangle-based table detection |
|
||||
|
||||
Low-level detection functions are also available via the `detector` module (`detect_pdf_type`, `detect_pdf_type_with_config`, etc.) for callers who need `PdfTypeResult` instead of `PdfProcessResult`.
|
||||
|
||||
### Types
|
||||
|
||||
| Type | Description |
|
||||
|---|---|
|
||||
| `PdfOptions` | Builder for processing configuration (mode, detection, markdown, page filter) |
|
||||
| `ProcessMode` | `DetectOnly`, `Analyze`, `Full` |
|
||||
| `PdfType` | `TextBased`, `Scanned`, `ImageBased`, `Mixed` |
|
||||
| `PdfProcessResult` | Full result: pdf_type, markdown, page_count, confidence, layout, has_encoding_issues, timing |
|
||||
| `PdfTypeResult` | Low-level detection result: type, confidence, page count, pages needing OCR |
|
||||
| `DetectionConfig` | Configuration for detection: scan strategy, thresholds |
|
||||
| `ScanStrategy` | `EarlyExit`, `Full`, `Sample(n)`, `Pages(vec)` |
|
||||
| `LayoutComplexity` | Layout analysis: is_complex, pages_with_tables, pages_with_columns |
|
||||
| `TextItem` | Text with position, font info, and page number |
|
||||
| `MarkdownOptions` | Configuration for Markdown formatting (page numbers, etc.) |
|
||||
| `PdfError` | `Io`, `Parse`, `Encrypted`, `InvalidStructure`, `NotAPdf` |
|
||||
|
||||
## Markdown output
|
||||
|
||||
The converter handles:
|
||||
@@ -255,36 +186,6 @@ The converter handles:
|
||||
| Drop caps | Large initial letters merged with following text |
|
||||
| Dot leaders | TOC-style dots collapsed to " ... " |
|
||||
|
||||
## Debugging with RUST_LOG
|
||||
|
||||
Structured logging via `RUST_LOG` replaces the former debug binaries. Set the environment variable to control which sections emit debug output on stderr:
|
||||
|
||||
```bash
|
||||
# Raw PDF content stream operators (replaces dump_ops)
|
||||
RUST_LOG=pdf_inspector::extractor::content_stream=trace cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Font metadata, encodings, ligatures (replaces debug_fonts / debug_ligatures)
|
||||
RUST_LOG=pdf_inspector::extractor::fonts=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# ToUnicode CMap parsing
|
||||
RUST_LOG=pdf_inspector::tounicode=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Text items per page with x/y/width (replaces debug_spaces / debug_pages)
|
||||
RUST_LOG=pdf_inspector::extractor=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Column detection and reading order (replaces debug_order)
|
||||
RUST_LOG=pdf_inspector::extractor::layout=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Y-gap analysis and paragraph thresholds (replaces debug_ygaps)
|
||||
RUST_LOG=pdf_inspector::markdown::analysis=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Table detection
|
||||
RUST_LOG=pdf_inspector::tables=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Everything
|
||||
RUST_LOG=pdf_inspector=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
```
|
||||
|
||||
## Use case: smart PDF routing
|
||||
|
||||
pdf-inspector was built for pipelines that process PDFs at scale. Instead of sending every PDF through OCR:
|
||||
@@ -299,6 +200,10 @@ PDF arrives
|
||||
|
||||
This saves cost and latency for the majority of PDFs that are already text-based (reports, papers, invoices, legal docs).
|
||||
|
||||
## Debugging
|
||||
|
||||
See [docs/debugging.md](docs/debugging.md) for `RUST_LOG` environment variable usage.
|
||||
|
||||
## License
|
||||
|
||||
MIT
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
# Debugging with RUST_LOG
|
||||
|
||||
Structured logging via `RUST_LOG` replaces the former debug binaries. Set the environment variable to control which sections emit debug output on stderr:
|
||||
|
||||
```bash
|
||||
# Raw PDF content stream operators (replaces dump_ops)
|
||||
RUST_LOG=pdf_inspector::extractor::content_stream=trace cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Font metadata, encodings, ligatures (replaces debug_fonts / debug_ligatures)
|
||||
RUST_LOG=pdf_inspector::extractor::fonts=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# ToUnicode CMap parsing
|
||||
RUST_LOG=pdf_inspector::tounicode=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Text items per page with x/y/width (replaces debug_spaces / debug_pages)
|
||||
RUST_LOG=pdf_inspector::extractor=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Column detection and reading order (replaces debug_order)
|
||||
RUST_LOG=pdf_inspector::extractor::layout=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Y-gap analysis and paragraph thresholds (replaces debug_ygaps)
|
||||
RUST_LOG=pdf_inspector::markdown::analysis=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Table detection
|
||||
RUST_LOG=pdf_inspector::tables=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
|
||||
# Everything
|
||||
RUST_LOG=pdf_inspector=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||
```
|
||||
@@ -0,0 +1,74 @@
|
||||
# Python API
|
||||
|
||||
Python bindings via [PyO3](https://pyo3.rs). Requires Rust toolchain for building from source.
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
pip install maturin
|
||||
maturin develop --release
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
```python
|
||||
import pdf_inspector
|
||||
|
||||
# Full processing: detect + extract + convert to Markdown
|
||||
result = pdf_inspector.process_pdf("document.pdf")
|
||||
print(result.pdf_type) # "text_based", "scanned", "image_based", "mixed"
|
||||
print(result.confidence) # 0.0 - 1.0
|
||||
print(result.page_count) # number of pages
|
||||
print(result.markdown) # Markdown string or None
|
||||
|
||||
# Process specific pages only
|
||||
result = pdf_inspector.process_pdf("document.pdf", pages=[1, 3, 5])
|
||||
|
||||
# Process from bytes (no filesystem needed)
|
||||
with open("document.pdf", "rb") as f:
|
||||
result = pdf_inspector.process_pdf_bytes(f.read())
|
||||
|
||||
# Fast detection only (no text extraction)
|
||||
result = pdf_inspector.detect_pdf("document.pdf")
|
||||
if result.pdf_type == "text_based":
|
||||
print("Can extract locally!")
|
||||
else:
|
||||
print(f"Pages needing OCR: {result.pages_needing_ocr}")
|
||||
|
||||
# Plain text extraction
|
||||
text = pdf_inspector.extract_text("document.pdf")
|
||||
|
||||
# Positioned text items with font info
|
||||
items = pdf_inspector.extract_text_with_positions("document.pdf")
|
||||
for item in items[:5]:
|
||||
print(f"'{item.text}' at ({item.x:.0f}, {item.y:.0f}) size={item.font_size}")
|
||||
```
|
||||
|
||||
## API reference
|
||||
|
||||
| Function | Description |
|
||||
|---|---|
|
||||
| `process_pdf(path, pages=None)` | Full processing (detect + extract + markdown) |
|
||||
| `process_pdf_bytes(data, pages=None)` | Full processing from bytes |
|
||||
| `detect_pdf(path)` | Fast detection only (returns PdfResult) |
|
||||
| `detect_pdf_bytes(data)` | Fast detection from bytes |
|
||||
| `classify_pdf(path)` | Lightweight classification (returns PdfClassification) |
|
||||
| `classify_pdf_bytes(data)` | Lightweight classification from bytes |
|
||||
| `extract_text(path)` | Plain text extraction |
|
||||
| `extract_text_bytes(data)` | Plain text extraction from bytes |
|
||||
| `extract_text_with_positions(path, pages=None)` | Text with X/Y coords and font info |
|
||||
| `extract_text_with_positions_bytes(data, pages=None)` | Text with positions from bytes |
|
||||
| `extract_text_in_regions(path, page_regions)` | Extract text in bounding-box regions |
|
||||
| `extract_text_in_regions_bytes(data, page_regions)` | Region extraction from bytes |
|
||||
|
||||
## Types
|
||||
|
||||
**`PdfResult` fields:** `pdf_type`, `markdown`, `page_count`, `processing_time_ms`, `pages_needing_ocr`, `title`, `confidence`, `is_complex_layout`, `pages_with_tables`, `pages_with_columns`, `has_encoding_issues`
|
||||
|
||||
**`PdfClassification` fields:** `pdf_type`, `page_count`, `pages_needing_ocr` (0-indexed), `confidence`
|
||||
|
||||
**`TextItem` fields:** `text`, `x`, `y`, `width`, `height`, `font`, `font_size`, `page`, `is_bold`, `is_italic`, `item_type`
|
||||
|
||||
**`RegionText` fields:** `text`, `needs_ocr`
|
||||
|
||||
**`PageRegionTexts` fields:** `page` (0-indexed), `regions` (list of RegionText)
|
||||
@@ -0,0 +1,122 @@
|
||||
# Rust API
|
||||
|
||||
Add to your `Cargo.toml`:
|
||||
|
||||
```toml
|
||||
[dependencies]
|
||||
pdf-inspector = { git = "https://github.com/firecrawl/pdf-inspector" }
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
Detect and extract in one call:
|
||||
|
||||
```rust
|
||||
use pdf_inspector::process_pdf;
|
||||
|
||||
let result = process_pdf("document.pdf")?;
|
||||
|
||||
println!("Type: {:?}", result.pdf_type); // TextBased, Scanned, ImageBased, Mixed
|
||||
println!("Confidence: {:.0}%", result.confidence * 100.0);
|
||||
println!("Pages: {}", result.page_count);
|
||||
|
||||
if let Some(markdown) = &result.markdown {
|
||||
println!("{}", markdown);
|
||||
}
|
||||
```
|
||||
|
||||
Fast metadata-only detection (no text extraction or markdown generation):
|
||||
|
||||
```rust
|
||||
use pdf_inspector::detect_pdf;
|
||||
|
||||
let info = detect_pdf("document.pdf")?;
|
||||
|
||||
match info.pdf_type {
|
||||
pdf_inspector::PdfType::TextBased => {
|
||||
// Extract locally — fast and free
|
||||
}
|
||||
_ => {
|
||||
// Route to OCR service
|
||||
// info.pages_needing_ocr tells you exactly which pages
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
Customize processing with `PdfOptions`:
|
||||
|
||||
```rust
|
||||
use pdf_inspector::{process_pdf_with_options, PdfOptions, ProcessMode, DetectionConfig, ScanStrategy};
|
||||
|
||||
// Analyze layout without generating markdown
|
||||
let result = process_pdf_with_options(
|
||||
"document.pdf",
|
||||
PdfOptions::new().mode(ProcessMode::Analyze),
|
||||
)?;
|
||||
|
||||
// Full extraction with custom detection strategy
|
||||
let result = process_pdf_with_options(
|
||||
"large.pdf",
|
||||
PdfOptions::new().detection(DetectionConfig {
|
||||
strategy: ScanStrategy::Sample(5),
|
||||
..Default::default()
|
||||
}),
|
||||
)?;
|
||||
|
||||
// Process only specific pages
|
||||
let result = process_pdf_with_options(
|
||||
"document.pdf",
|
||||
PdfOptions::new().pages([1, 3, 5]),
|
||||
)?;
|
||||
```
|
||||
|
||||
Process from a byte buffer (no filesystem needed):
|
||||
|
||||
```rust
|
||||
use pdf_inspector::process_pdf_mem;
|
||||
|
||||
let bytes = std::fs::read("document.pdf")?;
|
||||
let result = process_pdf_mem(&bytes)?;
|
||||
```
|
||||
|
||||
## Processing modes
|
||||
|
||||
| Mode | What it does | Returns |
|
||||
|---|---|---|
|
||||
| `ProcessMode::Full` (default) | Detect + extract + convert to Markdown | Everything populated |
|
||||
| `ProcessMode::Analyze` | Detect + extract + layout analysis (no Markdown) | `markdown` is `None`, `layout` is populated |
|
||||
| `ProcessMode::DetectOnly` | Classification only (fastest) | `markdown` is `None`, `layout` is default |
|
||||
|
||||
## Functions
|
||||
|
||||
| Function | Description |
|
||||
|---|---|
|
||||
| `process_pdf(path)` | Full processing with defaults |
|
||||
| `detect_pdf(path)` | Fast metadata-only detection (no extraction) |
|
||||
| `process_pdf_with_options(path, options)` | Process with custom `PdfOptions` |
|
||||
| `process_pdf_mem(bytes)` | Full processing from a byte buffer |
|
||||
| `detect_pdf_mem(bytes)` | Fast detection from a byte buffer |
|
||||
| `process_pdf_mem_with_options(bytes, options)` | Process from bytes with custom options |
|
||||
| `extract_text(path)` | Plain text extraction |
|
||||
| `extract_text_with_positions(path)` | Text with X/Y coordinates and font info |
|
||||
| `to_markdown(text, options)` | Convert plain text to Markdown |
|
||||
| `to_markdown_from_items(items, options)` | Markdown from pre-extracted `TextItem`s |
|
||||
| `to_markdown_from_items_with_rects(items, options, rects)` | Markdown with rectangle-based table detection |
|
||||
|
||||
Low-level detection functions are also available via the `detector` module (`detect_pdf_type`, `detect_pdf_type_with_config`, etc.) for callers who need `PdfTypeResult` instead of `PdfProcessResult`.
|
||||
|
||||
## Types
|
||||
|
||||
| Type | Description |
|
||||
|---|---|
|
||||
| `PdfOptions` | Builder for processing configuration (mode, detection, markdown, page filter) |
|
||||
| `ProcessMode` | `DetectOnly`, `Analyze`, `Full` |
|
||||
| `PdfType` | `TextBased`, `Scanned`, `ImageBased`, `Mixed` |
|
||||
| `PdfProcessResult` | Full result: pdf_type, markdown, page_count, confidence, layout, has_encoding_issues, timing |
|
||||
| `PdfTypeResult` | Low-level detection result: type, confidence, page count, pages needing OCR |
|
||||
| `DetectionConfig` | Configuration for detection: scan strategy, thresholds |
|
||||
| `ScanStrategy` | `EarlyExit`, `Full`, `Sample(n)`, `Pages(vec)` |
|
||||
| `LayoutComplexity` | Layout analysis: is_complex, pages_with_tables, pages_with_columns |
|
||||
| `TextItem` | Text with position, font info, and page number |
|
||||
| `MarkdownOptions` | Configuration for Markdown formatting (page numbers, etc.) |
|
||||
| `PdfError` | `Io`, `Parse`, `Encrypted`, `InvalidStructure`, `NotAPdf` |
|
||||
@@ -0,0 +1,98 @@
|
||||
"""Basic usage examples for pdf-inspector Python library."""
|
||||
|
||||
import sys
|
||||
import pdf_inspector
|
||||
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 2:
|
||||
print("Usage: python basic_usage.py <path-to-pdf>")
|
||||
sys.exit(1)
|
||||
|
||||
path = sys.argv[1]
|
||||
|
||||
# 1. Full processing: detect + extract + markdown
|
||||
print("=" * 60)
|
||||
print("Full processing")
|
||||
print("=" * 60)
|
||||
result = pdf_inspector.process_pdf(path)
|
||||
print(f"Type: {result.pdf_type}")
|
||||
print(f"Pages: {result.page_count}")
|
||||
print(f"Confidence: {result.confidence:.0%}")
|
||||
print(f"Time: {result.processing_time_ms}ms")
|
||||
print(f"Title: {result.title}")
|
||||
print(f"Complex: {result.is_complex_layout}")
|
||||
print(f"Tables on: {result.pages_with_tables}")
|
||||
print(f"Columns on: {result.pages_with_columns}")
|
||||
print(f"Encoding: {'issues detected' if result.has_encoding_issues else 'ok'}")
|
||||
print(f"OCR needed: {result.pages_needing_ocr or 'none'}")
|
||||
if result.markdown:
|
||||
print(f"\n--- Markdown ({len(result.markdown)} chars) ---")
|
||||
print(result.markdown[:500])
|
||||
if len(result.markdown) > 500:
|
||||
print(f"\n... ({len(result.markdown) - 500} more chars)")
|
||||
|
||||
# 2. Fast detection only
|
||||
print("\n" + "=" * 60)
|
||||
print("Detection only")
|
||||
print("=" * 60)
|
||||
info = pdf_inspector.detect_pdf(path)
|
||||
print(f"Type: {info.pdf_type}")
|
||||
print(f"Confidence: {info.confidence:.0%}")
|
||||
print(f"Time: {info.processing_time_ms}ms")
|
||||
|
||||
# 3. From bytes
|
||||
print("\n" + "=" * 60)
|
||||
print("From bytes")
|
||||
print("=" * 60)
|
||||
with open(path, "rb") as f:
|
||||
data = f.read()
|
||||
result = pdf_inspector.process_pdf_bytes(data)
|
||||
print(f"Type: {result.pdf_type}, Pages: {result.page_count}")
|
||||
|
||||
# 4. Plain text
|
||||
print("\n" + "=" * 60)
|
||||
print("Plain text extraction")
|
||||
print("=" * 60)
|
||||
text = pdf_inspector.extract_text(path)
|
||||
print(text[:300])
|
||||
|
||||
# 5. Positioned items
|
||||
print("\n" + "=" * 60)
|
||||
print("Positioned text items (first 10)")
|
||||
print("=" * 60)
|
||||
items = pdf_inspector.extract_text_with_positions(path, pages=[1])
|
||||
for item in items[:10]:
|
||||
bold = " [B]" if item.is_bold else ""
|
||||
italic = " [I]" if item.is_italic else ""
|
||||
print(
|
||||
f" p{item.page} ({item.x:6.1f}, {item.y:6.1f}) "
|
||||
f"size={item.font_size:5.1f}{bold}{italic} "
|
||||
f"'{item.text}'"
|
||||
)
|
||||
|
||||
# 6. Lightweight classification
|
||||
print("\n" + "=" * 60)
|
||||
print("Lightweight classification")
|
||||
print("=" * 60)
|
||||
cls = pdf_inspector.classify_pdf(path)
|
||||
print(f"Type: {cls.pdf_type}")
|
||||
print(f"Pages: {cls.page_count}")
|
||||
print(f"Confidence: {cls.confidence:.0%}")
|
||||
print(f"OCR pages: {cls.pages_needing_ocr or 'none'} (0-indexed)")
|
||||
|
||||
# 7. Region-based text extraction
|
||||
print("\n" + "=" * 60)
|
||||
print("Region-based text extraction (page 0, top region)")
|
||||
print("=" * 60)
|
||||
regions = pdf_inspector.extract_text_in_regions(
|
||||
path, [(0, [[0.0, 0.0, 600.0, 200.0]])]
|
||||
)
|
||||
for page_result in regions:
|
||||
for i, region in enumerate(page_result.regions):
|
||||
print(f" Region {i}: needs_ocr={region.needs_ocr}")
|
||||
print(f" Text: {region.text[:200]}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
+1
-1
@@ -1,3 +1,3 @@
|
||||
fn main() {
|
||||
napi_build::setup();
|
||||
napi_build::setup();
|
||||
}
|
||||
|
||||
Vendored
-49
@@ -1,49 +0,0 @@
|
||||
/* auto-generated by NAPI-RS */
|
||||
/* eslint-disable */
|
||||
/**
|
||||
* Classify a PDF: detect type (TextBased/Scanned/Mixed/ImageBased),
|
||||
* page count, and which pages need OCR. Takes PDF bytes as Buffer.
|
||||
*/
|
||||
export declare function classifyPdf(buffer: Buffer): PdfClassification
|
||||
|
||||
/**
|
||||
* Extract text within bounding-box regions from a PDF.
|
||||
*
|
||||
* For hybrid OCR: layout model detects regions in rendered images,
|
||||
* this extracts PDF text within those regions — skipping GPU OCR
|
||||
* for text-based pages.
|
||||
*
|
||||
* Each region result includes `needs_ocr` — set when the extracted text
|
||||
* is unreliable (empty, GID-encoded fonts, garbage, encoding issues).
|
||||
*
|
||||
* Coordinates are PDF points with top-left origin.
|
||||
*/
|
||||
export declare function extractTextInRegions(buffer: Buffer, pageRegions: Array<PageRegions>): Array<PageRegionTexts>
|
||||
|
||||
/** A page's regions for text extraction: (page_index_0based, bboxes). */
|
||||
export interface PageRegions {
|
||||
page: number
|
||||
/** Each bbox is [x1, y1, x2, y2] in PDF points, top-left origin. */
|
||||
regions: Array<Array<number>>
|
||||
}
|
||||
|
||||
/** Extracted text for one page's regions. */
|
||||
export interface PageRegionTexts {
|
||||
page: number
|
||||
regions: Array<RegionText>
|
||||
}
|
||||
|
||||
/** Lightweight PDF classification result. */
|
||||
export interface PdfClassification {
|
||||
pdfType: string
|
||||
pageCount: number
|
||||
pagesNeedingOcr: Array<number>
|
||||
confidence: number
|
||||
}
|
||||
|
||||
/** Extracted text for a single region. */
|
||||
export interface RegionText {
|
||||
text: string
|
||||
/** `true` when the text should not be trusted (empty, GID fonts, garbage, encoding issues). */
|
||||
needsOcr: boolean
|
||||
}
|
||||
-580
@@ -1,580 +0,0 @@
|
||||
// prettier-ignore
|
||||
/* eslint-disable */
|
||||
// @ts-nocheck
|
||||
/* auto-generated by NAPI-RS */
|
||||
|
||||
const { readFileSync } = require('node:fs')
|
||||
let nativeBinding = null
|
||||
const loadErrors = []
|
||||
|
||||
const isMusl = () => {
|
||||
let musl = false
|
||||
if (process.platform === 'linux') {
|
||||
musl = isMuslFromFilesystem()
|
||||
if (musl === null) {
|
||||
musl = isMuslFromReport()
|
||||
}
|
||||
if (musl === null) {
|
||||
musl = isMuslFromChildProcess()
|
||||
}
|
||||
}
|
||||
return musl
|
||||
}
|
||||
|
||||
const isFileMusl = (f) => f.includes('libc.musl-') || f.includes('ld-musl-')
|
||||
|
||||
const isMuslFromFilesystem = () => {
|
||||
try {
|
||||
return readFileSync('/usr/bin/ldd', 'utf-8').includes('musl')
|
||||
} catch {
|
||||
return null
|
||||
}
|
||||
}
|
||||
|
||||
const isMuslFromReport = () => {
|
||||
let report = null
|
||||
if (typeof process.report?.getReport === 'function') {
|
||||
process.report.excludeNetwork = true
|
||||
report = process.report.getReport()
|
||||
}
|
||||
if (!report) {
|
||||
return null
|
||||
}
|
||||
if (report.header && report.header.glibcVersionRuntime) {
|
||||
return false
|
||||
}
|
||||
if (Array.isArray(report.sharedObjects)) {
|
||||
if (report.sharedObjects.some(isFileMusl)) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
const isMuslFromChildProcess = () => {
|
||||
try {
|
||||
return require('child_process').execSync('ldd --version', { encoding: 'utf8' }).includes('musl')
|
||||
} catch (e) {
|
||||
// If we reach this case, we don't know if the system is musl or not, so is better to just fallback to false
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
function requireNative() {
|
||||
if (process.env.NAPI_RS_NATIVE_LIBRARY_PATH) {
|
||||
try {
|
||||
return require(process.env.NAPI_RS_NATIVE_LIBRARY_PATH);
|
||||
} catch (err) {
|
||||
loadErrors.push(err)
|
||||
}
|
||||
} else if (process.platform === 'android') {
|
||||
if (process.arch === 'arm64') {
|
||||
try {
|
||||
return require('./pdf-inspector.android-arm64.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-android-arm64')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-android-arm64/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else if (process.arch === 'arm') {
|
||||
try {
|
||||
return require('./pdf-inspector.android-arm-eabi.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-android-arm-eabi')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-android-arm-eabi/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
loadErrors.push(new Error(`Unsupported architecture on Android ${process.arch}`))
|
||||
}
|
||||
} else if (process.platform === 'win32') {
|
||||
if (process.arch === 'x64') {
|
||||
if (process.config?.variables?.shlib_suffix === 'dll.a' || process.config?.variables?.node_target_type === 'shared_library') {
|
||||
try {
|
||||
return require('./pdf-inspector.win32-x64-gnu.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-win32-x64-gnu')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-win32-x64-gnu/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
try {
|
||||
return require('./pdf-inspector.win32-x64-msvc.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-win32-x64-msvc')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-win32-x64-msvc/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
}
|
||||
} else if (process.arch === 'ia32') {
|
||||
try {
|
||||
return require('./pdf-inspector.win32-ia32-msvc.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-win32-ia32-msvc')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-win32-ia32-msvc/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else if (process.arch === 'arm64') {
|
||||
try {
|
||||
return require('./pdf-inspector.win32-arm64-msvc.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-win32-arm64-msvc')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-win32-arm64-msvc/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
loadErrors.push(new Error(`Unsupported architecture on Windows: ${process.arch}`))
|
||||
}
|
||||
} else if (process.platform === 'darwin') {
|
||||
try {
|
||||
return require('./pdf-inspector.darwin-universal.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-darwin-universal')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-darwin-universal/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
if (process.arch === 'x64') {
|
||||
try {
|
||||
return require('./pdf-inspector.darwin-x64.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-darwin-x64')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-darwin-x64/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else if (process.arch === 'arm64') {
|
||||
try {
|
||||
return require('./pdf-inspector.darwin-arm64.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-darwin-arm64')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-darwin-arm64/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
loadErrors.push(new Error(`Unsupported architecture on macOS: ${process.arch}`))
|
||||
}
|
||||
} else if (process.platform === 'freebsd') {
|
||||
if (process.arch === 'x64') {
|
||||
try {
|
||||
return require('./pdf-inspector.freebsd-x64.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-freebsd-x64')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-freebsd-x64/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else if (process.arch === 'arm64') {
|
||||
try {
|
||||
return require('./pdf-inspector.freebsd-arm64.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-freebsd-arm64')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-freebsd-arm64/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
loadErrors.push(new Error(`Unsupported architecture on FreeBSD: ${process.arch}`))
|
||||
}
|
||||
} else if (process.platform === 'linux') {
|
||||
if (process.arch === 'x64') {
|
||||
if (isMusl()) {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-x64-musl.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-x64-musl')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-x64-musl/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-x64-gnu.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-x64-gnu')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-x64-gnu/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
}
|
||||
} else if (process.arch === 'arm64') {
|
||||
if (isMusl()) {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-arm64-musl.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-arm64-musl')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-arm64-musl/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-arm64-gnu.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-arm64-gnu')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-arm64-gnu/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
}
|
||||
} else if (process.arch === 'arm') {
|
||||
if (isMusl()) {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-arm-musleabihf.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-arm-musleabihf')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-arm-musleabihf/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-arm-gnueabihf.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-arm-gnueabihf')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-arm-gnueabihf/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
}
|
||||
} else if (process.arch === 'loong64') {
|
||||
if (isMusl()) {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-loong64-musl.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-loong64-musl')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-loong64-musl/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-loong64-gnu.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-loong64-gnu')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-loong64-gnu/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
}
|
||||
} else if (process.arch === 'riscv64') {
|
||||
if (isMusl()) {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-riscv64-musl.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-riscv64-musl')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-riscv64-musl/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-riscv64-gnu.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-riscv64-gnu')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-riscv64-gnu/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
}
|
||||
} else if (process.arch === 'ppc64') {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-ppc64-gnu.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-ppc64-gnu')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-ppc64-gnu/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else if (process.arch === 's390x') {
|
||||
try {
|
||||
return require('./pdf-inspector.linux-s390x-gnu.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-linux-s390x-gnu')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-linux-s390x-gnu/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
loadErrors.push(new Error(`Unsupported architecture on Linux: ${process.arch}`))
|
||||
}
|
||||
} else if (process.platform === 'openharmony') {
|
||||
if (process.arch === 'arm64') {
|
||||
try {
|
||||
return require('./pdf-inspector.openharmony-arm64.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-openharmony-arm64')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-openharmony-arm64/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else if (process.arch === 'x64') {
|
||||
try {
|
||||
return require('./pdf-inspector.openharmony-x64.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-openharmony-x64')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-openharmony-x64/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else if (process.arch === 'arm') {
|
||||
try {
|
||||
return require('./pdf-inspector.openharmony-arm.node')
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
try {
|
||||
const binding = require('firecrawl-pdf-inspector-openharmony-arm')
|
||||
const bindingPackageVersion = require('firecrawl-pdf-inspector-openharmony-arm/package.json').version
|
||||
if (bindingPackageVersion !== '0.2.0' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') {
|
||||
throw new Error(`Native binding package version mismatch, expected 0.2.0 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`)
|
||||
}
|
||||
return binding
|
||||
} catch (e) {
|
||||
loadErrors.push(e)
|
||||
}
|
||||
} else {
|
||||
loadErrors.push(new Error(`Unsupported architecture on OpenHarmony: ${process.arch}`))
|
||||
}
|
||||
} else {
|
||||
loadErrors.push(new Error(`Unsupported OS: ${process.platform}, architecture: ${process.arch}`))
|
||||
}
|
||||
}
|
||||
|
||||
nativeBinding = requireNative()
|
||||
|
||||
if (!nativeBinding || process.env.NAPI_RS_FORCE_WASI) {
|
||||
let wasiBinding = null
|
||||
let wasiBindingError = null
|
||||
try {
|
||||
wasiBinding = require('./pdf-inspector.wasi.cjs')
|
||||
nativeBinding = wasiBinding
|
||||
} catch (err) {
|
||||
if (process.env.NAPI_RS_FORCE_WASI) {
|
||||
wasiBindingError = err
|
||||
}
|
||||
}
|
||||
if (!nativeBinding || process.env.NAPI_RS_FORCE_WASI) {
|
||||
try {
|
||||
wasiBinding = require('firecrawl-pdf-inspector-wasm32-wasi')
|
||||
nativeBinding = wasiBinding
|
||||
} catch (err) {
|
||||
if (process.env.NAPI_RS_FORCE_WASI) {
|
||||
if (!wasiBindingError) {
|
||||
wasiBindingError = err
|
||||
} else {
|
||||
wasiBindingError.cause = err
|
||||
}
|
||||
loadErrors.push(err)
|
||||
}
|
||||
}
|
||||
}
|
||||
if (process.env.NAPI_RS_FORCE_WASI === 'error' && !wasiBinding) {
|
||||
const error = new Error('WASI binding not found and NAPI_RS_FORCE_WASI is set to error')
|
||||
error.cause = wasiBindingError
|
||||
throw error
|
||||
}
|
||||
}
|
||||
|
||||
if (!nativeBinding) {
|
||||
if (loadErrors.length > 0) {
|
||||
throw new Error(
|
||||
`Cannot find native binding. ` +
|
||||
`npm has a bug related to optional dependencies (https://github.com/npm/cli/issues/4828). ` +
|
||||
'Please try `npm i` again after removing both package-lock.json and node_modules directory.',
|
||||
{
|
||||
cause: loadErrors.reduce((err, cur) => {
|
||||
cur.cause = err
|
||||
return cur
|
||||
}),
|
||||
},
|
||||
)
|
||||
}
|
||||
throw new Error(`Failed to load native binding`)
|
||||
}
|
||||
|
||||
module.exports = nativeBinding
|
||||
module.exports.classifyPdf = nativeBinding.classifyPdf
|
||||
module.exports.extractTextInRegions = nativeBinding.extractTextInRegions
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "firecrawl-pdf-inspector",
|
||||
"version": "0.2.3",
|
||||
"version": "0.3.2",
|
||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||
"main": "index.js",
|
||||
"types": "index.d.ts",
|
||||
|
||||
+248
-69
@@ -2,58 +2,241 @@
|
||||
|
||||
use napi::bindgen_prelude::*;
|
||||
use napi_derive::napi;
|
||||
use std::collections::HashSet;
|
||||
use std::panic;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Result types
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Full PDF processing result with markdown and metadata.
|
||||
#[napi(object)]
|
||||
pub struct PdfResult {
|
||||
pub pdf_type: String,
|
||||
pub markdown: Option<String>,
|
||||
pub page_count: u32,
|
||||
pub processing_time_ms: u32,
|
||||
/// 1-indexed page numbers that need OCR.
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
pub title: Option<String>,
|
||||
pub confidence: f64,
|
||||
pub is_complex_layout: bool,
|
||||
pub pages_with_tables: Vec<u32>,
|
||||
pub pages_with_columns: Vec<u32>,
|
||||
pub has_encoding_issues: bool,
|
||||
}
|
||||
|
||||
/// Lightweight PDF classification result.
|
||||
#[napi(object)]
|
||||
pub struct PdfClassification {
|
||||
pub pdf_type: String,
|
||||
pub page_count: u32,
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
pub confidence: f64,
|
||||
pub pdf_type: String,
|
||||
pub page_count: u32,
|
||||
/// 0-indexed page numbers that need OCR.
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
pub confidence: f64,
|
||||
}
|
||||
|
||||
/// A positioned text item extracted from a PDF.
|
||||
#[napi(object)]
|
||||
pub struct TextItem {
|
||||
pub text: String,
|
||||
pub x: f64,
|
||||
pub y: f64,
|
||||
pub width: f64,
|
||||
pub height: f64,
|
||||
pub font: String,
|
||||
pub font_size: f64,
|
||||
pub page: u32,
|
||||
pub is_bold: bool,
|
||||
pub is_italic: bool,
|
||||
pub item_type: String,
|
||||
}
|
||||
|
||||
/// A page's regions for text extraction: (page_index_0based, bboxes).
|
||||
#[napi(object)]
|
||||
pub struct PageRegions {
|
||||
pub page: u32,
|
||||
/// Each bbox is [x1, y1, x2, y2] in PDF points, top-left origin.
|
||||
pub regions: Vec<Vec<f64>>,
|
||||
pub page: u32,
|
||||
/// Each bbox is [x1, y1, x2, y2] in PDF points, top-left origin.
|
||||
pub regions: Vec<Vec<f64>>,
|
||||
}
|
||||
|
||||
/// Extracted text for a single region.
|
||||
#[napi(object)]
|
||||
pub struct RegionText {
|
||||
pub text: String,
|
||||
/// `true` when the text should not be trusted (empty, GID fonts, garbage, encoding issues).
|
||||
pub needs_ocr: bool,
|
||||
pub text: String,
|
||||
/// `true` when the text should not be trusted (empty, GID fonts, garbage, encoding issues).
|
||||
pub needs_ocr: bool,
|
||||
}
|
||||
|
||||
/// Extracted text for one page's regions.
|
||||
#[napi(object)]
|
||||
pub struct PageRegionTexts {
|
||||
pub page: u32,
|
||||
pub regions: Vec<RegionText>,
|
||||
pub page: u32,
|
||||
pub regions: Vec<RegionText>,
|
||||
}
|
||||
|
||||
/// Classify a PDF: detect type (TextBased/Scanned/Mixed/ImageBased),
|
||||
/// page count, and which pages need OCR. Takes PDF bytes as Buffer.
|
||||
// ---------------------------------------------------------------------------
|
||||
// Helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn pdf_type_string(t: pdf_inspector::PdfType) -> String {
|
||||
match t {
|
||||
pdf_inspector::PdfType::TextBased => "TextBased".to_string(),
|
||||
pdf_inspector::PdfType::Scanned => "Scanned".to_string(),
|
||||
pdf_inspector::PdfType::ImageBased => "ImageBased".to_string(),
|
||||
pdf_inspector::PdfType::Mixed => "Mixed".to_string(),
|
||||
}
|
||||
}
|
||||
|
||||
fn to_napi_result(r: pdf_inspector::PdfProcessResult) -> PdfResult {
|
||||
PdfResult {
|
||||
pdf_type: pdf_type_string(r.pdf_type),
|
||||
markdown: r.markdown,
|
||||
page_count: r.page_count,
|
||||
processing_time_ms: r.processing_time_ms as u32,
|
||||
pages_needing_ocr: r.pages_needing_ocr,
|
||||
title: r.title,
|
||||
confidence: r.confidence as f64,
|
||||
is_complex_layout: r.layout.is_complex,
|
||||
pages_with_tables: r.layout.pages_with_tables,
|
||||
pages_with_columns: r.layout.pages_with_columns,
|
||||
has_encoding_issues: r.has_encoding_issues,
|
||||
}
|
||||
}
|
||||
|
||||
fn item_type_string(t: &pdf_inspector::types::ItemType) -> String {
|
||||
match t {
|
||||
pdf_inspector::types::ItemType::Text => "text".into(),
|
||||
pdf_inspector::types::ItemType::Image => "image".into(),
|
||||
pdf_inspector::types::ItemType::Link(url) => format!("link:{url}"),
|
||||
pdf_inspector::types::ItemType::FormField => "form_field".into(),
|
||||
}
|
||||
}
|
||||
|
||||
fn to_napi_err(e: impl std::fmt::Display, ctx: &str) -> Error {
|
||||
Error::new(Status::GenericFailure, format!("{ctx}: {e}"))
|
||||
}
|
||||
|
||||
/// Run a closure, catching any Rust panic and converting it to a NAPI error.
|
||||
/// Prevents process abort from unwind panics in the native module.
|
||||
fn catch_panic<F, T>(ctx: &str, f: F) -> Result<T>
|
||||
where
|
||||
F: FnOnce() -> Result<T> + panic::UnwindSafe,
|
||||
{
|
||||
match panic::catch_unwind(f) {
|
||||
Ok(result) => result,
|
||||
Err(payload) => {
|
||||
let msg = if let Some(s) = payload.downcast_ref::<&str>() {
|
||||
s.to_string()
|
||||
} else if let Some(s) = payload.downcast_ref::<String>() {
|
||||
s.clone()
|
||||
} else {
|
||||
"unknown panic".to_string()
|
||||
};
|
||||
Err(Error::new(
|
||||
Status::GenericFailure,
|
||||
format!("{ctx}: Rust panic: {msg}"),
|
||||
))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Public NAPI API
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Process a PDF from a Buffer: detect type, extract text, and convert to Markdown.
|
||||
#[napi]
|
||||
pub fn process_pdf(buffer: Buffer, pages: Option<Vec<u32>>) -> Result<PdfResult> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
catch_panic("process_pdf", move || {
|
||||
let mut opts = pdf_inspector::PdfOptions::new();
|
||||
if let Some(p) = pages {
|
||||
opts = opts.pages(p);
|
||||
}
|
||||
let result = pdf_inspector::process_pdf_mem_with_options(&bytes, opts)
|
||||
.map_err(|e| to_napi_err(e, "process_pdf"))?;
|
||||
Ok(to_napi_result(result))
|
||||
})
|
||||
}
|
||||
|
||||
/// Fast detection only — no text extraction or markdown.
|
||||
#[napi]
|
||||
pub fn detect_pdf(buffer: Buffer) -> Result<PdfResult> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
catch_panic("detect_pdf", move || {
|
||||
let result =
|
||||
pdf_inspector::detect_pdf_mem(&bytes).map_err(|e| to_napi_err(e, "detect_pdf"))?;
|
||||
Ok(to_napi_result(result))
|
||||
})
|
||||
}
|
||||
|
||||
/// Lightweight PDF classification — returns type, page count, and OCR pages.
|
||||
/// Faster than detectPdf as it skips building the full PdfResult.
|
||||
/// Pages in pagesNeedingOcr are 0-indexed.
|
||||
#[napi]
|
||||
pub fn classify_pdf(buffer: Buffer) -> Result<PdfClassification> {
|
||||
let result = pdf_inspector::classify_pdf_mem(&buffer).map_err(|e| {
|
||||
Error::new(Status::GenericFailure, format!("classify_pdf failed: {e}"))
|
||||
})?;
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
catch_panic("classify_pdf", move || {
|
||||
let result =
|
||||
pdf_inspector::classify_pdf_mem(&bytes).map_err(|e| to_napi_err(e, "classify_pdf"))?;
|
||||
Ok(PdfClassification {
|
||||
pdf_type: pdf_type_string(result.pdf_type),
|
||||
page_count: result.page_count,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
confidence: result.confidence as f64,
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
Ok(PdfClassification {
|
||||
pdf_type: match result.pdf_type {
|
||||
pdf_inspector::PdfType::TextBased => "TextBased".to_string(),
|
||||
pdf_inspector::PdfType::Scanned => "Scanned".to_string(),
|
||||
pdf_inspector::PdfType::ImageBased => "ImageBased".to_string(),
|
||||
pdf_inspector::PdfType::Mixed => "Mixed".to_string(),
|
||||
},
|
||||
page_count: result.page_count,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
confidence: result.confidence as f64,
|
||||
})
|
||||
/// Extract plain text from a PDF Buffer.
|
||||
#[napi]
|
||||
pub fn extract_text(buffer: Buffer) -> Result<String> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
catch_panic("extract_text", move || {
|
||||
pdf_inspector::extractor::extract_text_mem(&bytes)
|
||||
.map_err(|e| to_napi_err(e, "extract_text"))
|
||||
})
|
||||
}
|
||||
|
||||
/// Extract text with position information from a PDF Buffer.
|
||||
#[napi]
|
||||
pub fn extract_text_with_positions(
|
||||
buffer: Buffer,
|
||||
pages: Option<Vec<u32>>,
|
||||
) -> Result<Vec<TextItem>> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
catch_panic("extract_text_with_positions", move || {
|
||||
let items = match pages {
|
||||
Some(p) => {
|
||||
let page_set: HashSet<u32> = p.into_iter().collect();
|
||||
pdf_inspector::extractor::extract_text_with_positions_mem_pages(
|
||||
&bytes,
|
||||
Some(&page_set),
|
||||
)
|
||||
.map_err(|e| to_napi_err(e, "extract_text_with_positions"))?
|
||||
}
|
||||
None => pdf_inspector::extractor::extract_text_with_positions_mem(&bytes)
|
||||
.map_err(|e| to_napi_err(e, "extract_text_with_positions"))?,
|
||||
};
|
||||
|
||||
Ok(items
|
||||
.into_iter()
|
||||
.map(|item| TextItem {
|
||||
text: item.text,
|
||||
x: item.x as f64,
|
||||
y: item.y as f64,
|
||||
width: item.width as f64,
|
||||
height: item.height as f64,
|
||||
font: item.font,
|
||||
font_size: item.font_size as f64,
|
||||
page: item.page,
|
||||
is_bold: item.is_bold,
|
||||
is_italic: item.is_italic,
|
||||
item_type: item_type_string(&item.item_type),
|
||||
})
|
||||
.collect())
|
||||
})
|
||||
}
|
||||
|
||||
/// Extract text within bounding-box regions from a PDF.
|
||||
@@ -62,55 +245,51 @@ pub fn classify_pdf(buffer: Buffer) -> Result<PdfClassification> {
|
||||
/// this extracts PDF text within those regions — skipping GPU OCR
|
||||
/// for text-based pages.
|
||||
///
|
||||
/// Each region result includes `needs_ocr` — set when the extracted text
|
||||
/// Each region result includes `needsOcr` — set when the extracted text
|
||||
/// is unreliable (empty, GID-encoded fonts, garbage, encoding issues).
|
||||
///
|
||||
/// Coordinates are PDF points with top-left origin.
|
||||
#[napi]
|
||||
pub fn extract_text_in_regions(
|
||||
buffer: Buffer,
|
||||
page_regions: Vec<PageRegions>,
|
||||
buffer: Buffer,
|
||||
page_regions: Vec<PageRegions>,
|
||||
) -> Result<Vec<PageRegionTexts>> {
|
||||
// Convert from napi types to the Rust API's expected format
|
||||
let regions: Vec<(u32, Vec<[f32; 4]>)> = page_regions
|
||||
.iter()
|
||||
.map(|pr| {
|
||||
let bboxes: Vec<[f32; 4]> = pr
|
||||
.regions
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
let regions: Vec<(u32, Vec<[f32; 4]>)> = page_regions
|
||||
.iter()
|
||||
.map(|r| {
|
||||
if r.len() != 4 {
|
||||
[0.0, 0.0, 0.0, 0.0]
|
||||
} else {
|
||||
[r[0] as f32, r[1] as f32, r[2] as f32, r[3] as f32]
|
||||
}
|
||||
.map(|pr| {
|
||||
let bboxes: Vec<[f32; 4]> = pr
|
||||
.regions
|
||||
.iter()
|
||||
.map(|r| {
|
||||
if r.len() != 4 {
|
||||
[0.0, 0.0, 0.0, 0.0]
|
||||
} else {
|
||||
[r[0] as f32, r[1] as f32, r[2] as f32, r[3] as f32]
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
(pr.page, bboxes)
|
||||
})
|
||||
.collect();
|
||||
(pr.page, bboxes)
|
||||
|
||||
catch_panic("extract_text_in_regions", move || {
|
||||
let results = pdf_inspector::extract_text_in_regions_mem(&bytes, ®ions)
|
||||
.map_err(|e| to_napi_err(e, "extract_text_in_regions"))?;
|
||||
|
||||
Ok(results
|
||||
.into_iter()
|
||||
.map(|page_result| PageRegionTexts {
|
||||
page: page_result.page,
|
||||
regions: page_result
|
||||
.regions
|
||||
.into_iter()
|
||||
.map(|r| RegionText {
|
||||
text: r.text,
|
||||
needs_ocr: r.needs_ocr,
|
||||
})
|
||||
.collect(),
|
||||
})
|
||||
.collect())
|
||||
})
|
||||
.collect();
|
||||
|
||||
let results = pdf_inspector::extract_text_in_regions_mem(&buffer, ®ions).map_err(|e| {
|
||||
Error::new(
|
||||
Status::GenericFailure,
|
||||
format!("extract_text_in_regions failed: {e}"),
|
||||
)
|
||||
})?;
|
||||
|
||||
Ok(
|
||||
results
|
||||
.into_iter()
|
||||
.map(|page_result| PageRegionTexts {
|
||||
page: page_result.page,
|
||||
regions: page_result
|
||||
.regions
|
||||
.into_iter()
|
||||
.map(|r| RegionText {
|
||||
text: r.text,
|
||||
needs_ocr: r.needs_ocr,
|
||||
})
|
||||
.collect(),
|
||||
})
|
||||
.collect(),
|
||||
)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,98 @@
|
||||
import { readFileSync } from 'fs';
|
||||
import { strict as assert } from 'assert';
|
||||
import {
|
||||
processPdf,
|
||||
detectPdf,
|
||||
classifyPdf,
|
||||
extractText,
|
||||
extractTextWithPositions,
|
||||
extractTextInRegions,
|
||||
} from './index.js';
|
||||
|
||||
const fixture = readFileSync('../tests/fixtures/thermo-freon12.pdf');
|
||||
|
||||
// --- processPdf ---
|
||||
console.log('Testing processPdf...');
|
||||
const result = processPdf(fixture);
|
||||
assert.equal(result.pdfType, 'TextBased');
|
||||
assert.equal(result.pageCount, 3);
|
||||
assert.ok(result.confidence > 0);
|
||||
assert.ok(result.markdown && result.markdown.length > 0);
|
||||
assert.equal(typeof result.isComplexLayout, 'boolean');
|
||||
assert.ok(Array.isArray(result.pagesWithTables));
|
||||
assert.ok(Array.isArray(result.pagesWithColumns));
|
||||
assert.equal(typeof result.hasEncodingIssues, 'boolean');
|
||||
console.log(' processPdf: OK');
|
||||
|
||||
// processPdf with pages
|
||||
const result2 = processPdf(fixture, [1]);
|
||||
assert.ok(result2.markdown && result2.markdown.length > 0);
|
||||
console.log(' processPdf with pages: OK');
|
||||
|
||||
// --- detectPdf ---
|
||||
console.log('Testing detectPdf...');
|
||||
const detected = detectPdf(fixture);
|
||||
assert.equal(detected.pdfType, 'TextBased');
|
||||
assert.equal(detected.pageCount, 3);
|
||||
assert.equal(detected.markdown, undefined);
|
||||
console.log(' detectPdf: OK');
|
||||
|
||||
// --- classifyPdf ---
|
||||
console.log('Testing classifyPdf...');
|
||||
const classified = classifyPdf(fixture);
|
||||
assert.equal(classified.pdfType, 'TextBased');
|
||||
assert.equal(classified.pageCount, 3);
|
||||
assert.ok(classified.confidence > 0);
|
||||
assert.ok(Array.isArray(classified.pagesNeedingOcr));
|
||||
console.log(' classifyPdf: OK');
|
||||
|
||||
// --- extractText ---
|
||||
console.log('Testing extractText...');
|
||||
const text = extractText(fixture);
|
||||
assert.equal(typeof text, 'string');
|
||||
assert.ok(text.length > 0);
|
||||
console.log(' extractText: OK');
|
||||
|
||||
// --- extractTextWithPositions ---
|
||||
console.log('Testing extractTextWithPositions...');
|
||||
const items = extractTextWithPositions(fixture);
|
||||
assert.ok(items.length > 0);
|
||||
const item = items[0];
|
||||
assert.equal(typeof item.text, 'string');
|
||||
assert.equal(typeof item.x, 'number');
|
||||
assert.equal(typeof item.y, 'number');
|
||||
assert.equal(typeof item.width, 'number');
|
||||
assert.equal(typeof item.height, 'number');
|
||||
assert.equal(typeof item.font, 'string');
|
||||
assert.equal(typeof item.fontSize, 'number');
|
||||
assert.equal(typeof item.page, 'number');
|
||||
assert.equal(typeof item.isBold, 'boolean');
|
||||
assert.equal(typeof item.isItalic, 'boolean');
|
||||
assert.equal(typeof item.itemType, 'string');
|
||||
console.log(' extractTextWithPositions: OK');
|
||||
|
||||
// with pages filter
|
||||
const page1Items = extractTextWithPositions(fixture, [1]);
|
||||
assert.ok(page1Items.length > 0);
|
||||
assert.ok(page1Items.every(i => i.page === 1));
|
||||
console.log(' extractTextWithPositions with pages: OK');
|
||||
|
||||
// --- extractTextInRegions ---
|
||||
console.log('Testing extractTextInRegions...');
|
||||
const regionResults = extractTextInRegions(fixture, [
|
||||
{ page: 0, regions: [[0, 0, 600, 100]] },
|
||||
]);
|
||||
assert.equal(regionResults.length, 1);
|
||||
assert.equal(regionResults[0].page, 0);
|
||||
assert.equal(regionResults[0].regions.length, 1);
|
||||
assert.equal(typeof regionResults[0].regions[0].text, 'string');
|
||||
assert.equal(typeof regionResults[0].regions[0].needsOcr, 'boolean');
|
||||
console.log(' extractTextInRegions: OK');
|
||||
|
||||
// --- Error handling ---
|
||||
console.log('Testing error handling...');
|
||||
assert.throws(() => processPdf(Buffer.from('not a pdf')), /process_pdf/);
|
||||
assert.throws(() => classifyPdf(Buffer.from('')), /classify_pdf/);
|
||||
console.log(' error handling: OK');
|
||||
|
||||
console.log('\nAll NAPI tests passed!');
|
||||
@@ -0,0 +1,117 @@
|
||||
"""Type stubs for pdf_inspector."""
|
||||
|
||||
from typing import Optional
|
||||
|
||||
class PdfResult:
|
||||
"""Result of processing a PDF file."""
|
||||
pdf_type: str
|
||||
"""'text_based', 'scanned', 'image_based', or 'mixed'."""
|
||||
markdown: Optional[str]
|
||||
page_count: int
|
||||
processing_time_ms: int
|
||||
pages_needing_ocr: list[int]
|
||||
title: Optional[str]
|
||||
confidence: float
|
||||
is_complex_layout: bool
|
||||
pages_with_tables: list[int]
|
||||
pages_with_columns: list[int]
|
||||
has_encoding_issues: bool
|
||||
|
||||
class PdfClassification:
|
||||
"""Lightweight PDF classification result."""
|
||||
pdf_type: str
|
||||
"""'text_based', 'scanned', 'image_based', or 'mixed'."""
|
||||
page_count: int
|
||||
pages_needing_ocr: list[int]
|
||||
"""0-indexed page numbers that need OCR."""
|
||||
confidence: float
|
||||
|
||||
class TextItem:
|
||||
"""A positioned text item extracted from a PDF."""
|
||||
text: str
|
||||
x: float
|
||||
y: float
|
||||
width: float
|
||||
height: float
|
||||
font: str
|
||||
font_size: float
|
||||
page: int
|
||||
is_bold: bool
|
||||
is_italic: bool
|
||||
item_type: str
|
||||
|
||||
class RegionText:
|
||||
"""Extracted text for a single region."""
|
||||
text: str
|
||||
needs_ocr: bool
|
||||
"""True when the text should not be trusted."""
|
||||
|
||||
class PageRegionTexts:
|
||||
"""Extracted text for one page's regions."""
|
||||
page: int
|
||||
"""0-indexed page number."""
|
||||
regions: list[RegionText]
|
||||
|
||||
def process_pdf(path: str, pages: Optional[list[int]] = None) -> PdfResult:
|
||||
"""Process a PDF: detect type, extract text, convert to Markdown."""
|
||||
...
|
||||
|
||||
def process_pdf_bytes(data: bytes, pages: Optional[list[int]] = None) -> PdfResult:
|
||||
"""Process a PDF from bytes in memory."""
|
||||
...
|
||||
|
||||
def detect_pdf(path: str) -> PdfResult:
|
||||
"""Fast detection only — no text extraction."""
|
||||
...
|
||||
|
||||
def detect_pdf_bytes(data: bytes) -> PdfResult:
|
||||
"""Fast detection from bytes."""
|
||||
...
|
||||
|
||||
def classify_pdf(path: str) -> PdfClassification:
|
||||
"""Lightweight classification — type, page count, and OCR pages (0-indexed)."""
|
||||
...
|
||||
|
||||
def classify_pdf_bytes(data: bytes) -> PdfClassification:
|
||||
"""Lightweight classification from bytes."""
|
||||
...
|
||||
|
||||
def extract_text(path: str) -> str:
|
||||
"""Extract plain text from a PDF."""
|
||||
...
|
||||
|
||||
def extract_text_bytes(data: bytes) -> str:
|
||||
"""Extract plain text from PDF bytes."""
|
||||
...
|
||||
|
||||
def extract_text_with_positions(path: str, pages: Optional[list[int]] = None) -> list[TextItem]:
|
||||
"""Extract text with position information."""
|
||||
...
|
||||
|
||||
def extract_text_with_positions_bytes(data: bytes, pages: Optional[list[int]] = None) -> list[TextItem]:
|
||||
"""Extract text with position information from bytes."""
|
||||
...
|
||||
|
||||
def extract_text_in_regions(
|
||||
path: str,
|
||||
page_regions: list[tuple[int, list[list[float]]]],
|
||||
) -> list[PageRegionTexts]:
|
||||
"""Extract text within bounding-box regions from a PDF file.
|
||||
|
||||
Args:
|
||||
path: Path to the PDF file.
|
||||
page_regions: List of (page_0indexed, [[x1, y1, x2, y2], ...]) tuples.
|
||||
"""
|
||||
...
|
||||
|
||||
def extract_text_in_regions_bytes(
|
||||
data: bytes,
|
||||
page_regions: list[tuple[int, list[list[float]]]],
|
||||
) -> list[PageRegionTexts]:
|
||||
"""Extract text within bounding-box regions from PDF bytes.
|
||||
|
||||
Args:
|
||||
data: PDF file contents as bytes.
|
||||
page_regions: List of (page_0indexed, [[x1, y1, x2, y2], ...]) tuples.
|
||||
"""
|
||||
...
|
||||
@@ -0,0 +1,21 @@
|
||||
[build-system]
|
||||
requires = ["maturin>=1.0,<2.0"]
|
||||
build-backend = "maturin"
|
||||
|
||||
[project]
|
||||
name = "pdf-inspector"
|
||||
version = "0.1.0"
|
||||
description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection"
|
||||
license = { text = "MIT" }
|
||||
requires-python = ">=3.8"
|
||||
classifiers = [
|
||||
"Programming Language :: Rust",
|
||||
"Programming Language :: Python :: Implementation :: CPython",
|
||||
"Programming Language :: Python :: 3",
|
||||
"License :: OSI Approved :: MIT License",
|
||||
"Operating System :: OS Independent",
|
||||
"Topic :: Text Processing",
|
||||
]
|
||||
|
||||
[tool.maturin]
|
||||
features = ["python"]
|
||||
+14
-22
@@ -218,7 +218,7 @@ fn columns_have_prose(columns: &[ColumnRegion], items: &[&TextItem]) -> bool {
|
||||
|
||||
// Sort by Y descending (top of page = higher Y in PDF coords)
|
||||
let mut sorted: Vec<&TextItem> = col_items;
|
||||
sorted.sort_by(|a, b| b.y.partial_cmp(&a.y).unwrap_or(std::cmp::Ordering::Equal));
|
||||
sorted.sort_by(|a, b| b.y.total_cmp(&a.y));
|
||||
|
||||
// Group into lines by Y-proximity and measure fill + item count
|
||||
let mut full_lines = 0usize;
|
||||
@@ -616,7 +616,7 @@ fn identify_spanning_lines(items: &[TextItem], columns: &[ColumnRegion]) -> Vec<
|
||||
// Build (original_index, y) pairs sorted by Y descending for grouping
|
||||
let mut indexed: Vec<(usize, f32)> =
|
||||
items.iter().enumerate().map(|(i, it)| (i, it.y)).collect();
|
||||
indexed.sort_by(|a, b| b.1.partial_cmp(&a.1).unwrap_or(std::cmp::Ordering::Equal));
|
||||
indexed.sort_by(|a, b| b.1.total_cmp(&a.1));
|
||||
|
||||
// Group by Y-proximity into rough lines (as index sets)
|
||||
let mut groups: Vec<Vec<usize>> = Vec::new();
|
||||
@@ -778,7 +778,7 @@ pub(crate) fn is_newspaper_layout(
|
||||
return 0.0;
|
||||
}
|
||||
let mut ys: Vec<f32> = lines.iter().map(|l| l.y).collect();
|
||||
ys.sort_by(|a, b| a.partial_cmp(b).unwrap());
|
||||
ys.sort_by(|a, b| a.total_cmp(b));
|
||||
let span = ys.last().unwrap() - ys.first().unwrap();
|
||||
span / (lines.len() as f32 - 1.0)
|
||||
};
|
||||
@@ -845,7 +845,7 @@ fn split_column_stragglers(lines: Vec<TextLine>) -> (Vec<TextLine>, Vec<TextLine
|
||||
|
||||
// Median gap = typical line spacing
|
||||
let mut sorted_gaps = gaps.clone();
|
||||
sorted_gaps.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
sorted_gaps.sort_by(|a, b| a.total_cmp(b));
|
||||
let median_gap = sorted_gaps[sorted_gaps.len() / 2];
|
||||
|
||||
// A gap > 3× median (min 30pt) indicates a break between content clusters
|
||||
@@ -1078,9 +1078,8 @@ pub(crate) fn group_into_lines_with_thresholds(
|
||||
}
|
||||
}
|
||||
|
||||
above.sort_by(|a, b| b.y.partial_cmp(&a.y).unwrap_or(std::cmp::Ordering::Equal));
|
||||
below_spanning
|
||||
.sort_by(|a, b| b.y.partial_cmp(&a.y).unwrap_or(std::cmp::Ordering::Equal));
|
||||
above.sort_by(|a, b| b.y.total_cmp(&a.y));
|
||||
below_spanning.sort_by(|a, b| b.y.total_cmp(&a.y));
|
||||
|
||||
all_lines.extend(above);
|
||||
for col in core_columns {
|
||||
@@ -1101,16 +1100,13 @@ pub(crate) fn group_into_lines_with_thresholds(
|
||||
|
||||
// Sort by Y descending (top-first), then by X for same-Y lines
|
||||
all_page_lines.sort_by(|a, b| {
|
||||
b.y.partial_cmp(&a.y)
|
||||
.unwrap_or(std::cmp::Ordering::Equal)
|
||||
.then(
|
||||
a.items
|
||||
.first()
|
||||
.map(|i| i.x)
|
||||
.unwrap_or(0.0)
|
||||
.partial_cmp(&b.items.first().map(|i| i.x).unwrap_or(0.0))
|
||||
.unwrap_or(std::cmp::Ordering::Equal),
|
||||
)
|
||||
b.y.total_cmp(&a.y).then(
|
||||
a.items
|
||||
.first()
|
||||
.map(|i| i.x)
|
||||
.unwrap_or(0.0)
|
||||
.total_cmp(&b.items.first().map(|i| i.x).unwrap_or(0.0)),
|
||||
)
|
||||
});
|
||||
|
||||
// Merge lines at the same Y (within tolerance) into single lines
|
||||
@@ -1185,11 +1181,7 @@ fn group_single_column(items: Vec<TextItem>, adaptive_threshold: f32) -> Vec<Tex
|
||||
let items = if use_y_sorting {
|
||||
// Sort by Y descending (top to bottom in PDF coords)
|
||||
let mut sorted = items;
|
||||
sorted.sort_by(|a, b| {
|
||||
b.y.partial_cmp(&a.y)
|
||||
.unwrap_or(std::cmp::Ordering::Equal)
|
||||
.then(a.x.partial_cmp(&b.x).unwrap_or(std::cmp::Ordering::Equal))
|
||||
});
|
||||
sorted.sort_by(|a, b| b.y.total_cmp(&a.y).then(a.x.total_cmp(&b.x)));
|
||||
sorted
|
||||
} else {
|
||||
items
|
||||
|
||||
@@ -334,17 +334,14 @@ pub(crate) fn merge_text_items(items: Vec<TextItem>) -> Vec<TextItem> {
|
||||
for (_, _, group) in &mut line_groups {
|
||||
let rtl = is_rtl_text(group.iter().map(|i| &i.text));
|
||||
if rtl {
|
||||
group.sort_by(|a, b| b.x.partial_cmp(&a.x).unwrap_or(std::cmp::Ordering::Equal));
|
||||
group.sort_by(|a, b| b.x.total_cmp(&a.x));
|
||||
} else {
|
||||
group.sort_by(|a, b| a.x.partial_cmp(&b.x).unwrap_or(std::cmp::Ordering::Equal));
|
||||
group.sort_by(|a, b| a.x.total_cmp(&b.x));
|
||||
}
|
||||
}
|
||||
|
||||
// Sort groups by page then Y descending (top of page first)
|
||||
line_groups.sort_by(|a, b| {
|
||||
a.0.cmp(&b.0)
|
||||
.then_with(|| b.1.partial_cmp(&a.1).unwrap_or(std::cmp::Ordering::Equal))
|
||||
});
|
||||
line_groups.sort_by(|a, b| a.0.cmp(&b.0).then_with(|| b.1.total_cmp(&a.1)));
|
||||
|
||||
let mut merged = Vec::new();
|
||||
|
||||
@@ -452,7 +449,7 @@ pub(crate) fn merge_subscript_items(items: Vec<TextItem>) -> Vec<TextItem> {
|
||||
|
||||
for (_, _, mut group) in line_groups {
|
||||
// Sort by X position
|
||||
group.sort_by(|a, b| a.x.partial_cmp(&b.x).unwrap_or(std::cmp::Ordering::Equal));
|
||||
group.sort_by(|a, b| a.x.total_cmp(&b.x));
|
||||
|
||||
// Find the dominant (most common) font size in this group
|
||||
let max_fs = group.iter().map(|i| i.font_size).fold(0.0_f32, f32::max);
|
||||
|
||||
+17
-14
@@ -22,6 +22,9 @@
|
||||
//! ).unwrap();
|
||||
//! ```
|
||||
|
||||
#[cfg(feature = "python")]
|
||||
pub mod python;
|
||||
|
||||
pub mod adobe_korea1;
|
||||
pub mod detector;
|
||||
pub mod extractor;
|
||||
@@ -341,11 +344,15 @@ pub fn extract_text_in_regions_mem(
|
||||
) -> Result<Vec<PageRegionResult>, PdfError> {
|
||||
validate_pdf_bytes(buffer)?;
|
||||
let (doc, _page_count) = load_document_from_mem(buffer)?;
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let pages = doc.get_pages();
|
||||
|
||||
// Build a set of pages we need to extract
|
||||
let needed_pages: HashSet<u32> = page_regions.iter().map(|(p, _)| p + 1).collect(); // to 1-indexed
|
||||
// Build a set of pages we need to extract (1-indexed for lopdf)
|
||||
let needed_pages: HashSet<u32> = page_regions.iter().map(|(p, _)| p + 1).collect();
|
||||
|
||||
// Fast mode: skip expensive TrueType font fallback parsing.
|
||||
// Fonts that can't be decoded from ToUnicode alone will produce empty/garbage
|
||||
// text, triggering needs_ocr=true → GPU OCR fallback in the pipeline.
|
||||
let font_cmaps = FontCMaps::from_doc_pages_fast(&doc, Some(&needed_pages));
|
||||
|
||||
// Extract text items for needed pages only
|
||||
let mut items_by_page: HashMap<u32, Vec<TextItem>> = HashMap::new();
|
||||
@@ -399,6 +406,7 @@ pub fn extract_text_in_regions_mem(
|
||||
let needs_ocr = text.trim().is_empty()
|
||||
|| page_has_gid
|
||||
|| is_garbage_text(&text)
|
||||
|| is_cid_garbage(&text)
|
||||
|| detect_encoding_issues(&text);
|
||||
|
||||
page_results.push(RegionText { text, needs_ocr });
|
||||
@@ -448,7 +456,7 @@ fn obj_to_f32(obj: &lopdf::Object) -> Option<f32> {
|
||||
|
||||
/// Collect text items that fall within a region bbox (top-left origin, PDF points)
|
||||
/// and return them as a single string in reading order.
|
||||
fn collect_text_in_region(
|
||||
pub fn collect_text_in_region(
|
||||
items: &[TextItem],
|
||||
rx1: f32,
|
||||
ry1: f32,
|
||||
@@ -474,17 +482,12 @@ fn collect_text_in_region(
|
||||
return String::new();
|
||||
}
|
||||
|
||||
// Sort top→bottom (descending Y in bottom-left coords), then left→right
|
||||
// Sort top→bottom (descending Y in bottom-left coords), then left→right.
|
||||
// Uses strict total_cmp ordering to guarantee transitivity (required by
|
||||
// Rust's sort). The line-grouping phase below handles fuzzy Y matching.
|
||||
matched.sort_by(|a, b| {
|
||||
let line_threshold = a.font_size.max(b.font_size) * 0.5;
|
||||
let y_diff = b.y - a.y; // descending Y = top to bottom
|
||||
if y_diff.abs() < line_threshold {
|
||||
a.x.partial_cmp(&b.x).unwrap_or(std::cmp::Ordering::Equal)
|
||||
} else {
|
||||
y_diff
|
||||
.partial_cmp(&0.0_f32)
|
||||
.unwrap_or(std::cmp::Ordering::Equal)
|
||||
}
|
||||
b.y.total_cmp(&a.y) // descending Y = top to bottom
|
||||
.then(a.x.total_cmp(&b.x)) // ascending X = left to right
|
||||
});
|
||||
|
||||
// Group into lines and join
|
||||
|
||||
@@ -121,7 +121,7 @@ pub(crate) fn compute_paragraph_threshold(lines: &[TextLine], base_size: f32) ->
|
||||
return fallback;
|
||||
}
|
||||
|
||||
gaps.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
gaps.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
let median = gaps[gaps.len() / 2];
|
||||
|
||||
@@ -221,7 +221,7 @@ pub(crate) fn compute_heading_tiers(lines: &[TextLine], base_size: f32) -> Vec<f
|
||||
}
|
||||
|
||||
// Sort descending
|
||||
heading_sizes.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal));
|
||||
heading_sizes.sort_by(|a, b| b.total_cmp(a));
|
||||
|
||||
// Cluster sizes within 0.5pt into same tier (use first value as representative)
|
||||
let mut tiers: Vec<f32> = Vec::new();
|
||||
|
||||
+4
-4
@@ -42,7 +42,7 @@ pub(crate) fn split_side_by_side(items: &[TextItem]) -> Vec<(f32, f32)> {
|
||||
|
||||
// Sort items by left edge
|
||||
let mut xs: Vec<f32> = items.iter().map(|i| i.x).collect();
|
||||
xs.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
xs.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
// Find all candidate gaps: ≥30pt, in the middle 60% of the X range,
|
||||
// with ≥20 items on each side.
|
||||
@@ -118,7 +118,7 @@ pub(crate) fn split_side_by_side(items: &[TextItem]) -> Vec<(f32, f32)> {
|
||||
})
|
||||
.copied()
|
||||
.collect();
|
||||
balanced_positions.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
balanced_positions.sort_by(|a, b| a.total_cmp(b));
|
||||
balanced_positions.dedup_by(|a, b| (*a - *b).abs() < 50.0);
|
||||
if balanced_positions.len() > 1 {
|
||||
return vec![];
|
||||
@@ -209,7 +209,7 @@ fn split_from_hint_regions(items: &[TextItem], rects: &[PdfRect], page: u32) ->
|
||||
|
||||
// Width outlier filter (same as detect_tables_from_rects)
|
||||
let mut widths: Vec<f32> = page_rects.iter().map(|&(_, _, w, _)| w).collect();
|
||||
widths.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
widths.sort_by(|a, b| a.total_cmp(b));
|
||||
let median_width = widths[widths.len() / 2];
|
||||
page_rects.retain(|&(_, _, w, _)| w <= median_width * 10.0);
|
||||
|
||||
@@ -1037,7 +1037,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
}
|
||||
// Sort by Y descending (top to bottom) so left and right
|
||||
// band lines interleave in visual reading order.
|
||||
page_lines.sort_by(|a, b| b.y.partial_cmp(&a.y).unwrap_or(std::cmp::Ordering::Equal));
|
||||
page_lines.sort_by(|a, b| b.y.total_cmp(&a.y));
|
||||
all_lines.extend(page_lines);
|
||||
}
|
||||
all_lines
|
||||
|
||||
@@ -275,7 +275,7 @@ pub(crate) fn strip_repeated_lines(lines: Vec<TextLine>, page_count: u32) -> Vec
|
||||
page_sorted_ys.entry(line.page).or_default().push(line.y);
|
||||
}
|
||||
for ys in page_sorted_ys.values_mut() {
|
||||
ys.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
ys.sort_by(|a, b| a.total_cmp(b));
|
||||
ys.dedup();
|
||||
}
|
||||
|
||||
|
||||
+453
@@ -0,0 +1,453 @@
|
||||
//! PyO3 Python bindings for pdf-inspector.
|
||||
|
||||
use pyo3::exceptions::PyValueError;
|
||||
use pyo3::prelude::*;
|
||||
use std::collections::HashSet;
|
||||
|
||||
use crate::detector::PdfType;
|
||||
use crate::types::ItemType;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Result wrapper
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Result of processing a PDF file.
|
||||
#[pyclass(name = "PdfResult")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyPdfResult {
|
||||
/// The detected PDF type: "text_based", "scanned", "image_based", or "mixed".
|
||||
#[pyo3(get)]
|
||||
pub pdf_type: String,
|
||||
/// Markdown output (None if detect-only or scanned PDF).
|
||||
#[pyo3(get)]
|
||||
pub markdown: Option<String>,
|
||||
/// Total number of pages.
|
||||
#[pyo3(get)]
|
||||
pub page_count: u32,
|
||||
/// Processing time in milliseconds.
|
||||
#[pyo3(get)]
|
||||
pub processing_time_ms: u64,
|
||||
/// 1-indexed page numbers that need OCR.
|
||||
#[pyo3(get)]
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
/// Title from PDF metadata.
|
||||
#[pyo3(get)]
|
||||
pub title: Option<String>,
|
||||
/// Detection confidence (0.0-1.0).
|
||||
#[pyo3(get)]
|
||||
pub confidence: f32,
|
||||
/// Whether the layout is complex (tables/columns detected).
|
||||
#[pyo3(get)]
|
||||
pub is_complex_layout: bool,
|
||||
/// Pages with tables detected.
|
||||
#[pyo3(get)]
|
||||
pub pages_with_tables: Vec<u32>,
|
||||
/// Pages with multi-column layout.
|
||||
#[pyo3(get)]
|
||||
pub pages_with_columns: Vec<u32>,
|
||||
/// Whether encoding issues were detected.
|
||||
#[pyo3(get)]
|
||||
pub has_encoding_issues: bool,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl PyPdfResult {
|
||||
fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"PdfResult(pdf_type='{}', pages={}, confidence={:.2})",
|
||||
self.pdf_type, self.page_count, self.confidence
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Classification wrapper (lightweight)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Lightweight PDF classification result.
|
||||
#[pyclass(name = "PdfClassification")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyPdfClassification {
|
||||
/// The detected PDF type: "text_based", "scanned", "image_based", or "mixed".
|
||||
#[pyo3(get)]
|
||||
pub pdf_type: String,
|
||||
/// Total number of pages.
|
||||
#[pyo3(get)]
|
||||
pub page_count: u32,
|
||||
/// 0-indexed page numbers that need OCR.
|
||||
#[pyo3(get)]
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
/// Detection confidence (0.0-1.0).
|
||||
#[pyo3(get)]
|
||||
pub confidence: f32,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl PyPdfClassification {
|
||||
fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"PdfClassification(pdf_type='{}', pages={}, confidence={:.2})",
|
||||
self.pdf_type, self.page_count, self.confidence
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Region extraction wrappers
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Extracted text for a single region.
|
||||
#[pyclass(name = "RegionText")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyRegionText {
|
||||
/// Extracted text content.
|
||||
#[pyo3(get)]
|
||||
pub text: String,
|
||||
/// True when the text should not be trusted (empty, GID fonts, garbage, encoding issues).
|
||||
#[pyo3(get)]
|
||||
pub needs_ocr: bool,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl PyRegionText {
|
||||
fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"RegionText(text='{}', needs_ocr={})",
|
||||
self.text.chars().take(40).collect::<String>(),
|
||||
self.needs_ocr
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
/// Extracted text for one page's regions.
|
||||
#[pyclass(name = "PageRegionTexts")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyPageRegionTexts {
|
||||
/// 0-indexed page number.
|
||||
#[pyo3(get)]
|
||||
pub page: u32,
|
||||
/// Per-region results, parallel to the input regions.
|
||||
#[pyo3(get)]
|
||||
pub regions: Vec<PyRegionText>,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl PyPageRegionTexts {
|
||||
fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"PageRegionTexts(page={}, regions={})",
|
||||
self.page,
|
||||
self.regions.len()
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Text item wrapper
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// A positioned text item extracted from a PDF.
|
||||
#[pyclass(name = "TextItem")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyTextItem {
|
||||
#[pyo3(get)]
|
||||
pub text: String,
|
||||
#[pyo3(get)]
|
||||
pub x: f32,
|
||||
#[pyo3(get)]
|
||||
pub y: f32,
|
||||
#[pyo3(get)]
|
||||
pub width: f32,
|
||||
#[pyo3(get)]
|
||||
pub height: f32,
|
||||
#[pyo3(get)]
|
||||
pub font: String,
|
||||
#[pyo3(get)]
|
||||
pub font_size: f32,
|
||||
#[pyo3(get)]
|
||||
pub page: u32,
|
||||
#[pyo3(get)]
|
||||
pub is_bold: bool,
|
||||
#[pyo3(get)]
|
||||
pub is_italic: bool,
|
||||
#[pyo3(get)]
|
||||
pub item_type: String,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl PyTextItem {
|
||||
fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"TextItem(text='{}', page={}, x={:.1}, y={:.1})",
|
||||
self.text.chars().take(40).collect::<String>(),
|
||||
self.page,
|
||||
self.x,
|
||||
self.y,
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn pdf_type_str(t: PdfType) -> String {
|
||||
match t {
|
||||
PdfType::TextBased => "text_based".into(),
|
||||
PdfType::Scanned => "scanned".into(),
|
||||
PdfType::ImageBased => "image_based".into(),
|
||||
PdfType::Mixed => "mixed".into(),
|
||||
}
|
||||
}
|
||||
|
||||
fn to_py_result(r: crate::PdfProcessResult) -> PyPdfResult {
|
||||
PyPdfResult {
|
||||
pdf_type: pdf_type_str(r.pdf_type),
|
||||
markdown: r.markdown,
|
||||
page_count: r.page_count,
|
||||
processing_time_ms: r.processing_time_ms,
|
||||
pages_needing_ocr: r.pages_needing_ocr,
|
||||
title: r.title,
|
||||
confidence: r.confidence,
|
||||
is_complex_layout: r.layout.is_complex,
|
||||
pages_with_tables: r.layout.pages_with_tables,
|
||||
pages_with_columns: r.layout.pages_with_columns,
|
||||
has_encoding_issues: r.has_encoding_issues,
|
||||
}
|
||||
}
|
||||
|
||||
fn to_py_err(e: crate::PdfError) -> PyErr {
|
||||
PyValueError::new_err(e.to_string())
|
||||
}
|
||||
|
||||
fn item_type_str(t: &ItemType) -> String {
|
||||
match t {
|
||||
ItemType::Text => "text".into(),
|
||||
ItemType::Image => "image".into(),
|
||||
ItemType::Link(url) => format!("link:{url}"),
|
||||
ItemType::FormField => "form_field".into(),
|
||||
}
|
||||
}
|
||||
|
||||
fn convert_text_items(items: Vec<crate::TextItem>) -> Vec<PyTextItem> {
|
||||
items
|
||||
.into_iter()
|
||||
.map(|item| PyTextItem {
|
||||
text: item.text,
|
||||
x: item.x,
|
||||
y: item.y,
|
||||
width: item.width,
|
||||
height: item.height,
|
||||
font: item.font,
|
||||
font_size: item.font_size,
|
||||
page: item.page,
|
||||
is_bold: item.is_bold,
|
||||
is_italic: item.is_italic,
|
||||
item_type: item_type_str(&item.item_type),
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn parse_page_regions(page_regions: Vec<(u32, Vec<Vec<f64>>)>) -> Vec<(u32, Vec<[f32; 4]>)> {
|
||||
page_regions
|
||||
.into_iter()
|
||||
.map(|(page, regions)| {
|
||||
let bboxes: Vec<[f32; 4]> = regions
|
||||
.iter()
|
||||
.map(|r| {
|
||||
if r.len() != 4 {
|
||||
[0.0, 0.0, 0.0, 0.0]
|
||||
} else {
|
||||
[r[0] as f32, r[1] as f32, r[2] as f32, r[3] as f32]
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
(page, bboxes)
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn convert_region_results(results: Vec<crate::PageRegionResult>) -> Vec<PyPageRegionTexts> {
|
||||
results
|
||||
.into_iter()
|
||||
.map(|page_result| PyPageRegionTexts {
|
||||
page: page_result.page,
|
||||
regions: page_result
|
||||
.regions
|
||||
.into_iter()
|
||||
.map(|r| PyRegionText {
|
||||
text: r.text,
|
||||
needs_ocr: r.needs_ocr,
|
||||
})
|
||||
.collect(),
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Public Python API
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Process a PDF file: detect type, extract text, and convert to Markdown.
|
||||
#[pyfunction]
|
||||
#[pyo3(signature = (path, pages=None))]
|
||||
fn process_pdf(path: &str, pages: Option<Vec<u32>>) -> PyResult<PyPdfResult> {
|
||||
let mut opts = crate::PdfOptions::new();
|
||||
if let Some(p) = pages {
|
||||
opts = opts.pages(p);
|
||||
}
|
||||
let result = crate::process_pdf_with_options(path, opts).map_err(to_py_err)?;
|
||||
Ok(to_py_result(result))
|
||||
}
|
||||
|
||||
/// Process a PDF from bytes in memory.
|
||||
#[pyfunction]
|
||||
#[pyo3(signature = (data, pages=None))]
|
||||
fn process_pdf_bytes(data: &[u8], pages: Option<Vec<u32>>) -> PyResult<PyPdfResult> {
|
||||
let mut opts = crate::PdfOptions::new();
|
||||
if let Some(p) = pages {
|
||||
opts = opts.pages(p);
|
||||
}
|
||||
let result = crate::process_pdf_mem_with_options(data, opts).map_err(to_py_err)?;
|
||||
Ok(to_py_result(result))
|
||||
}
|
||||
|
||||
/// Fast detection only — no text extraction or markdown.
|
||||
#[pyfunction]
|
||||
fn detect_pdf(path: &str) -> PyResult<PyPdfResult> {
|
||||
let result = crate::detect_pdf(path).map_err(to_py_err)?;
|
||||
Ok(to_py_result(result))
|
||||
}
|
||||
|
||||
/// Fast detection from bytes — no text extraction or markdown.
|
||||
#[pyfunction]
|
||||
fn detect_pdf_bytes(data: &[u8]) -> PyResult<PyPdfResult> {
|
||||
let result = crate::detect_pdf_mem(data).map_err(to_py_err)?;
|
||||
Ok(to_py_result(result))
|
||||
}
|
||||
|
||||
/// Lightweight PDF classification — returns type, page count, and OCR pages.
|
||||
/// Faster than detect_pdf as it skips building the full PdfProcessResult.
|
||||
/// Pages in pages_needing_ocr are 0-indexed.
|
||||
#[pyfunction]
|
||||
fn classify_pdf(path: &str) -> PyResult<PyPdfClassification> {
|
||||
let data = std::fs::read(path).map_err(|e| PyValueError::new_err(e.to_string()))?;
|
||||
classify_pdf_bytes(&data)
|
||||
}
|
||||
|
||||
/// Lightweight PDF classification from bytes.
|
||||
/// Pages in pages_needing_ocr are 0-indexed.
|
||||
#[pyfunction]
|
||||
fn classify_pdf_bytes(data: &[u8]) -> PyResult<PyPdfClassification> {
|
||||
let result = crate::classify_pdf_mem(data).map_err(to_py_err)?;
|
||||
Ok(PyPdfClassification {
|
||||
pdf_type: pdf_type_str(result.pdf_type),
|
||||
page_count: result.page_count,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
confidence: result.confidence,
|
||||
})
|
||||
}
|
||||
|
||||
/// Extract plain text from a PDF file.
|
||||
#[pyfunction]
|
||||
fn extract_text(path: &str) -> PyResult<String> {
|
||||
crate::extract_text(path).map_err(to_py_err)
|
||||
}
|
||||
|
||||
/// Extract plain text from PDF bytes.
|
||||
#[pyfunction]
|
||||
fn extract_text_bytes(data: &[u8]) -> PyResult<String> {
|
||||
crate::extractor::extract_text_mem(data).map_err(to_py_err)
|
||||
}
|
||||
|
||||
/// Extract text with position information from a file.
|
||||
#[pyfunction]
|
||||
#[pyo3(signature = (path, pages=None))]
|
||||
fn extract_text_with_positions(path: &str, pages: Option<Vec<u32>>) -> PyResult<Vec<PyTextItem>> {
|
||||
let items = match pages {
|
||||
Some(p) => {
|
||||
let page_set: HashSet<u32> = p.into_iter().collect();
|
||||
crate::extract_text_with_positions_pages(path, Some(&page_set)).map_err(to_py_err)?
|
||||
}
|
||||
None => crate::extract_text_with_positions(path).map_err(to_py_err)?,
|
||||
};
|
||||
Ok(convert_text_items(items))
|
||||
}
|
||||
|
||||
/// Extract text with position information from bytes.
|
||||
#[pyfunction]
|
||||
#[pyo3(signature = (data, pages=None))]
|
||||
fn extract_text_with_positions_bytes(
|
||||
data: &[u8],
|
||||
pages: Option<Vec<u32>>,
|
||||
) -> PyResult<Vec<PyTextItem>> {
|
||||
let items = match pages {
|
||||
Some(p) => {
|
||||
let page_set: HashSet<u32> = p.into_iter().collect();
|
||||
crate::extractor::extract_text_with_positions_mem_pages(data, Some(&page_set))
|
||||
.map_err(to_py_err)?
|
||||
}
|
||||
None => crate::extractor::extract_text_with_positions_mem(data).map_err(to_py_err)?,
|
||||
};
|
||||
Ok(convert_text_items(items))
|
||||
}
|
||||
|
||||
/// Extract text within bounding-box regions from a PDF file.
|
||||
///
|
||||
/// Args:
|
||||
/// path: Path to the PDF file.
|
||||
/// page_regions: List of (page_0indexed, [[x1, y1, x2, y2], ...]) tuples.
|
||||
/// Coordinates are PDF points with top-left origin.
|
||||
///
|
||||
/// Returns:
|
||||
/// List of PageRegionTexts with per-region text and needs_ocr flag.
|
||||
#[pyfunction]
|
||||
fn extract_text_in_regions(
|
||||
path: &str,
|
||||
page_regions: Vec<(u32, Vec<Vec<f64>>)>,
|
||||
) -> PyResult<Vec<PyPageRegionTexts>> {
|
||||
let data = std::fs::read(path).map_err(|e| PyValueError::new_err(e.to_string()))?;
|
||||
extract_text_in_regions_bytes(&data, page_regions)
|
||||
}
|
||||
|
||||
/// Extract text within bounding-box regions from PDF bytes.
|
||||
///
|
||||
/// Args:
|
||||
/// data: PDF file contents as bytes.
|
||||
/// page_regions: List of (page_0indexed, [[x1, y1, x2, y2], ...]) tuples.
|
||||
/// Coordinates are PDF points with top-left origin.
|
||||
///
|
||||
/// Returns:
|
||||
/// List of PageRegionTexts with per-region text and needs_ocr flag.
|
||||
#[pyfunction]
|
||||
fn extract_text_in_regions_bytes(
|
||||
data: &[u8],
|
||||
page_regions: Vec<(u32, Vec<Vec<f64>>)>,
|
||||
) -> PyResult<Vec<PyPageRegionTexts>> {
|
||||
let regions = parse_page_regions(page_regions);
|
||||
let results = crate::extract_text_in_regions_mem(data, ®ions).map_err(to_py_err)?;
|
||||
Ok(convert_region_results(results))
|
||||
}
|
||||
|
||||
/// Python module definition.
|
||||
#[pymodule]
|
||||
fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
|
||||
m.add_class::<PyPdfResult>()?;
|
||||
m.add_class::<PyPdfClassification>()?;
|
||||
m.add_class::<PyTextItem>()?;
|
||||
m.add_class::<PyRegionText>()?;
|
||||
m.add_class::<PyPageRegionTexts>()?;
|
||||
m.add_function(wrap_pyfunction!(process_pdf, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(process_pdf_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(detect_pdf, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(detect_pdf_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(classify_pdf, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(classify_pdf_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text_with_positions, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text_with_positions_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text_in_regions, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text_in_regions_bytes, m)?)?;
|
||||
Ok(())
|
||||
}
|
||||
@@ -48,7 +48,7 @@ pub(crate) fn merge_adjacent_items(items: &[TextItem]) -> (Vec<TextItem>, Vec<Ve
|
||||
}
|
||||
|
||||
// Sort groups by Y descending (top of page first)
|
||||
line_groups.sort_by(|a, b| b.0.partial_cmp(&a.0).unwrap_or(std::cmp::Ordering::Equal));
|
||||
line_groups.sort_by(|a, b| b.0.total_cmp(&a.0));
|
||||
|
||||
let mut merged_items = Vec::new();
|
||||
let mut index_map: Vec<Vec<usize>> = Vec::new();
|
||||
@@ -284,7 +284,7 @@ fn find_table_regions(items: &[(usize, &TextItem)]) -> Vec<(f32, f32)> {
|
||||
}
|
||||
|
||||
let mut y_positions: Vec<f32> = items.iter().map(|(_, i)| i.y).collect();
|
||||
y_positions.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
y_positions.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
// Find clusters of Y positions (table regions)
|
||||
let mut regions = Vec::new();
|
||||
@@ -347,7 +347,7 @@ fn find_table_regions_strict(items: &[(usize, &TextItem)]) -> Vec<(f32, f32, f32
|
||||
let mut qualifying_rows: Vec<(f32, Vec<f32>)> = Vec::new(); // (y, cluster_starts)
|
||||
for (y, x_positions) in &row_groups {
|
||||
let mut sorted_xs = x_positions.clone();
|
||||
sorted_xs.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
sorted_xs.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
if sorted_xs.is_empty() {
|
||||
continue;
|
||||
@@ -379,14 +379,14 @@ fn find_table_regions_strict(items: &[(usize, &TextItem)]) -> Vec<(f32, f32, f32
|
||||
// Step 3: Find contiguous runs of qualifying rows.
|
||||
// Use adaptive gap: median spacing × 3 (handles wrapped cells where
|
||||
// qualifying rows are spaced further apart), with a floor of 25pt.
|
||||
qualifying_rows.sort_by(|a, b| a.0.partial_cmp(&b.0).unwrap_or(std::cmp::Ordering::Equal));
|
||||
qualifying_rows.sort_by(|a, b| a.0.total_cmp(&b.0));
|
||||
|
||||
let max_gap = if qualifying_rows.len() >= 3 {
|
||||
let mut gaps: Vec<f32> = qualifying_rows
|
||||
.windows(2)
|
||||
.map(|w| (w[1].0 - w[0].0).abs())
|
||||
.collect();
|
||||
gaps.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
gaps.sort_by(|a, b| a.total_cmp(b));
|
||||
let median_gap = gaps[gaps.len() / 2];
|
||||
(median_gap * 3.0).max(25.0)
|
||||
} else {
|
||||
@@ -566,11 +566,9 @@ fn detect_table_in_region(items: &[(usize, &TextItem)], mode: TableDetectionMode
|
||||
// Sort by X position (direction-aware)
|
||||
let rtl = is_rtl_text(col_items.iter().map(|i| &i.text));
|
||||
if rtl {
|
||||
col_items
|
||||
.sort_by(|a, b| b.x.partial_cmp(&a.x).unwrap_or(std::cmp::Ordering::Equal));
|
||||
col_items.sort_by(|a, b| b.x.total_cmp(&a.x));
|
||||
} else {
|
||||
col_items
|
||||
.sort_by(|a, b| a.x.partial_cmp(&b.x).unwrap_or(std::cmp::Ordering::Equal));
|
||||
col_items.sort_by(|a, b| a.x.total_cmp(&b.x));
|
||||
}
|
||||
|
||||
// Join items with subscript-aware spacing
|
||||
|
||||
@@ -167,7 +167,7 @@ pub fn detect_tables_from_lines(items: &[TextItem], lines: &[PdfLine], page: u32
|
||||
|
||||
// Row edges need to be in descending order (top of page = higher Y first)
|
||||
let mut row_edges_desc = row_edges;
|
||||
row_edges_desc.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal));
|
||||
row_edges_desc.sort_by(|a, b| b.total_cmp(a));
|
||||
|
||||
log::debug!(
|
||||
"detect_lines p{}: {} row_edges, {} col_edges, table=({:.0},{:.0})-({:.0},{:.0}), spanning_h={}, spanning_v={}",
|
||||
|
||||
+17
-17
@@ -144,7 +144,7 @@ fn split_wide_cluster(
|
||||
|
||||
// Build sorted list of X-intervals (x_left, x_right) from each rect
|
||||
let mut intervals: Vec<(f32, f32)> = rects.iter().map(|&(x, _, w, _)| (x, x + w)).collect();
|
||||
intervals.sort_by(|a, b| a.0.partial_cmp(&b.0).unwrap_or(std::cmp::Ordering::Equal));
|
||||
intervals.sort_by(|a, b| a.0.total_cmp(&b.0));
|
||||
|
||||
// Merge overlapping intervals to find contiguous X-bands
|
||||
let mut merged: Vec<(f32, f32)> = Vec::new();
|
||||
@@ -262,7 +262,7 @@ pub fn detect_tables_from_rects(
|
||||
// background fills stand out clearly.
|
||||
if page_rects.len() >= 6 {
|
||||
let mut widths: Vec<f32> = page_rects.iter().map(|&(_, _, w, _)| w).collect();
|
||||
widths.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
widths.sort_by(|a, b| a.total_cmp(b));
|
||||
let median_width = widths[widths.len() / 2];
|
||||
let width_threshold = median_width * 10.0;
|
||||
let before = page_rects.len();
|
||||
@@ -333,7 +333,7 @@ pub fn detect_tables_from_rects(
|
||||
// cluster they overlap, so grid detection still has their edges.
|
||||
let is_page_bg = {
|
||||
let mut heights: Vec<f32> = page_rects.iter().map(|&(_, _, _, h)| h).collect();
|
||||
heights.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
heights.sort_by(|a, b| a.total_cmp(b));
|
||||
let median_height = heights[heights.len() / 2];
|
||||
let height_threshold = median_height * 20.0;
|
||||
let flags: Vec<bool> = page_rects
|
||||
@@ -621,7 +621,7 @@ fn merge_overlapping_hints(mut hints: Vec<RectHintRegion>) -> Vec<RectHintRegion
|
||||
return hints;
|
||||
}
|
||||
loop {
|
||||
hints.sort_by(|a, b| a.x_left.partial_cmp(&b.x_left).unwrap());
|
||||
hints.sort_by(|a, b| a.x_left.total_cmp(&b.x_left));
|
||||
let mut merged: Vec<RectHintRegion> = Vec::new();
|
||||
let mut any_merged = false;
|
||||
for hint in &hints {
|
||||
@@ -685,7 +685,7 @@ fn extract_hint_region(group_rects: &[(f32, f32, f32, f32)]) -> Option<RectHintR
|
||||
|
||||
// Compute median height to identify cell-sized rects
|
||||
let mut heights: Vec<f32> = group_rects.iter().map(|&(_, _, _, h)| h).collect();
|
||||
heights.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
heights.sort_by(|a, b| a.total_cmp(b));
|
||||
let median_h = heights[heights.len() / 2];
|
||||
|
||||
// Keep only cell-sized rects (height ≤ 4× median)
|
||||
@@ -857,9 +857,9 @@ fn try_build_grid(
|
||||
|
||||
// Sort column edges left-to-right, row edges top-to-bottom (highest Y first for PDF)
|
||||
let mut col_edges = x_edges;
|
||||
col_edges.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
col_edges.sort_by(|a, b| a.total_cmp(b));
|
||||
let mut row_edges = y_edges;
|
||||
row_edges.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal));
|
||||
row_edges.sort_by(|a, b| b.total_cmp(a));
|
||||
|
||||
let num_cols = col_edges.len() - 1;
|
||||
let num_rows = row_edges.len() - 1;
|
||||
@@ -1048,7 +1048,7 @@ fn try_build_grid(
|
||||
/// Deduplicate nearby edge values within a tolerance, returning sorted unique edges.
|
||||
pub(crate) fn snap_edges(values: &[f32], tolerance: f32) -> Vec<f32> {
|
||||
let mut sorted: Vec<f32> = values.to_vec();
|
||||
sorted.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
sorted.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
let mut snapped: Vec<f32> = Vec::new();
|
||||
for &v in &sorted {
|
||||
@@ -1204,7 +1204,7 @@ fn is_row_stripe_pattern(rects: &[(f32, f32, f32, f32)]) -> bool {
|
||||
}
|
||||
|
||||
let mut widths: Vec<f32> = rects.iter().map(|&(_, _, w, _)| w).collect();
|
||||
widths.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
widths.sort_by(|a, b| a.total_cmp(b));
|
||||
let median_width = widths[widths.len() / 2];
|
||||
|
||||
// Must be page-spanning (>200pt)
|
||||
@@ -1252,7 +1252,7 @@ fn detect_row_stripe_table(
|
||||
|
||||
// Sort row edges top-to-bottom (highest Y first for PDF)
|
||||
let mut row_edges = y_edges;
|
||||
row_edges.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal));
|
||||
row_edges.sort_by(|a, b| b.total_cmp(a));
|
||||
|
||||
// Compute the bounding box of the stripe region for filtering items
|
||||
let y_top = row_edges[0];
|
||||
@@ -1476,7 +1476,7 @@ fn detect_row_stripe_table_from_cell_rects(
|
||||
// bounding box to scope items and derive rows from text Y-positions.
|
||||
let row_edges = if y_edges.len() >= 4 {
|
||||
let mut edges = y_edges;
|
||||
edges.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal));
|
||||
edges.sort_by(|a, b| b.total_cmp(a));
|
||||
edges
|
||||
} else {
|
||||
// Fall back: gather items in the rect region and cluster by Y
|
||||
@@ -1508,11 +1508,11 @@ fn detect_row_stripe_table_from_cell_rects(
|
||||
// Cluster Y positions using median font height as threshold
|
||||
let median_h = {
|
||||
let mut hs: Vec<f32> = region_items.iter().map(|i| i.height).collect();
|
||||
hs.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
hs.sort_by(|a, b| a.total_cmp(b));
|
||||
hs[hs.len() / 2]
|
||||
};
|
||||
let mut ys: Vec<f32> = region_items.iter().map(|i| i.y).collect();
|
||||
ys.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal));
|
||||
ys.sort_by(|a, b| b.total_cmp(a));
|
||||
let mut edges = Vec::new();
|
||||
let threshold = median_h * 0.8;
|
||||
let mut cluster_start = ys[0];
|
||||
@@ -1536,7 +1536,7 @@ fn detect_row_stripe_table_from_cell_rects(
|
||||
edges.push(center - median_h * 0.5);
|
||||
let _ = cluster_start; // suppress unused warning
|
||||
edges = snap_edges(&edges, 3.0);
|
||||
edges.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal));
|
||||
edges.sort_by(|a, b| b.total_cmp(a));
|
||||
if edges.len() < 4 {
|
||||
return None;
|
||||
}
|
||||
@@ -1546,7 +1546,7 @@ fn detect_row_stripe_table_from_cell_rects(
|
||||
// Compute bounding box from non-full-page rects
|
||||
let median_h = {
|
||||
let mut heights: Vec<f32> = group_rects.iter().map(|&(_, _, _, h)| h).collect();
|
||||
heights.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
heights.sort_by(|a, b| a.total_cmp(b));
|
||||
heights[heights.len() / 2]
|
||||
};
|
||||
let content_rects: Vec<_> = group_rects
|
||||
@@ -1724,7 +1724,7 @@ fn detect_merged_cluster_table(
|
||||
}
|
||||
|
||||
let mut row_edges = y_edges;
|
||||
row_edges.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal));
|
||||
row_edges.sort_by(|a, b| b.total_cmp(a));
|
||||
|
||||
// Bounding box of all rects
|
||||
let y_top = row_edges[0];
|
||||
@@ -1890,7 +1890,7 @@ fn detect_merged_cluster_table(
|
||||
/// (no need for anti-paragraph safeguards).
|
||||
fn cluster_x_positions(items: &[(usize, &TextItem)], min_threshold: f32) -> Vec<f32> {
|
||||
let mut x_positions: Vec<f32> = items.iter().map(|(_, i)| i.x).collect();
|
||||
x_positions.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
x_positions.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
if x_positions.is_empty() {
|
||||
return vec![];
|
||||
|
||||
+4
-4
@@ -9,7 +9,7 @@ pub(crate) fn find_column_boundaries(
|
||||
mode: TableDetectionMode,
|
||||
) -> Vec<f32> {
|
||||
let mut x_positions: Vec<f32> = items.iter().map(|(_, i)| i.x).collect();
|
||||
x_positions.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
x_positions.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
if x_positions.is_empty() {
|
||||
return vec![];
|
||||
@@ -43,7 +43,7 @@ pub(crate) fn find_column_boundaries(
|
||||
.collect();
|
||||
|
||||
if consec_gaps.len() > 2 {
|
||||
consec_gaps.sort_by(|a, b| a.partial_cmp(b).unwrap());
|
||||
consec_gaps.sort_by(|a, b| a.total_cmp(b));
|
||||
// Find the biggest jump in the sorted gap sequence — natural break
|
||||
// between within-column jitter and between-column spacing.
|
||||
// Require at least 3 values on each side to avoid outlier-dominated
|
||||
@@ -151,7 +151,7 @@ pub(crate) fn find_column_boundaries(
|
||||
/// Find row boundaries by clustering Y positions
|
||||
pub(crate) fn find_row_boundaries(items: &[(usize, &TextItem)]) -> Vec<f32> {
|
||||
let mut y_positions: Vec<f32> = items.iter().map(|(_, i)| i.y).collect();
|
||||
y_positions.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal)); // Descending
|
||||
y_positions.sort_by(|a, b| b.total_cmp(a)); // Descending
|
||||
|
||||
if y_positions.is_empty() {
|
||||
return vec![];
|
||||
@@ -162,7 +162,7 @@ pub(crate) fn find_row_boundaries(items: &[(usize, &TextItem)]) -> Vec<f32> {
|
||||
// inter-row gaps (≥1× font size), preventing row merging in uniform-spaced PDFs.
|
||||
let cluster_threshold = {
|
||||
let mut font_sizes: Vec<f32> = items.iter().map(|(_, i)| i.font_size).collect();
|
||||
font_sizes.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
font_sizes.sort_by(|a, b| a.total_cmp(b));
|
||||
let median_font = font_sizes[font_sizes.len() / 2];
|
||||
(median_font * 0.8).max(4.0)
|
||||
};
|
||||
|
||||
+6
-6
@@ -33,7 +33,7 @@ pub(crate) fn try_build_rect_guided_table(
|
||||
|
||||
// 1. Derive column boundaries from rect X positions (snapped to 2pt tolerance)
|
||||
let mut x_lefts: Vec<f32> = cluster_rects.iter().map(|&(x, _, _, _)| x).collect();
|
||||
x_lefts.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
x_lefts.sort_by(|a, b| a.total_cmp(b));
|
||||
// Snap: deduplicate within 2pt tolerance
|
||||
let mut col_boundaries: Vec<f32> = Vec::new();
|
||||
for x in &x_lefts {
|
||||
@@ -54,7 +54,7 @@ pub(crate) fn try_build_rect_guided_table(
|
||||
// boundaries so every day gets a column.
|
||||
if col_boundaries.len() >= 2 {
|
||||
let mut spacings: Vec<f32> = col_boundaries.windows(2).map(|w| w[1] - w[0]).collect();
|
||||
spacings.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
spacings.sort_by(|a, b| a.total_cmp(b));
|
||||
let median_spacing = spacings[spacings.len() / 2];
|
||||
let threshold = median_spacing * 1.5;
|
||||
|
||||
@@ -87,7 +87,7 @@ pub(crate) fn try_build_rect_guided_table(
|
||||
|
||||
// 3. Derive row boundaries from item Y positions (5pt tolerance)
|
||||
let mut y_values: Vec<f32> = expanded_items.iter().map(|(item, _)| item.y).collect();
|
||||
y_values.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal)); // descending
|
||||
y_values.sort_by(|a, b| b.total_cmp(a)); // descending
|
||||
let mut row_boundaries: Vec<f32> = Vec::new();
|
||||
for y in &y_values {
|
||||
if row_boundaries
|
||||
@@ -296,7 +296,7 @@ pub(crate) fn try_build_table_from_columns(items: &[TextItem], page: u32) -> Opt
|
||||
|
||||
// Find the top-most row with items in multiple columns (likely the header)
|
||||
let mut ys: Vec<f32> = page_items.iter().map(|i| i.y).collect();
|
||||
ys.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal));
|
||||
ys.sort_by(|a, b| b.total_cmp(a));
|
||||
ys.dedup_by(|a, b| (*a - *b).abs() < y_tol);
|
||||
|
||||
for &header_y in ys.iter().take(5) {
|
||||
@@ -318,7 +318,7 @@ pub(crate) fn try_build_table_from_columns(items: &[TextItem], page: u32) -> Opt
|
||||
if col_items.len() >= 2 {
|
||||
// Sort by X and find the split point
|
||||
let mut sorted: Vec<f32> = col_items.iter().map(|i| i.x).collect();
|
||||
sorted.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
sorted.sort_by(|a, b| a.total_cmp(b));
|
||||
// Split at the midpoint between the two items
|
||||
let split_x = (sorted[0]
|
||||
+ col_items.iter().find(|i| i.x == sorted[0]).unwrap().width
|
||||
@@ -411,7 +411,7 @@ pub(crate) fn try_build_table_from_columns(items: &[TextItem], page: u32) -> Opt
|
||||
}
|
||||
}
|
||||
}
|
||||
row_ys.sort_by(|a, b| b.partial_cmp(a).unwrap_or(std::cmp::Ordering::Equal));
|
||||
row_ys.sort_by(|a, b| b.total_cmp(a));
|
||||
|
||||
if row_ys.len() < 3 || row_ys.len() > 40 {
|
||||
return None;
|
||||
|
||||
+4
-4
@@ -68,9 +68,9 @@ where
|
||||
pub(crate) fn sort_line_items(items: &mut [TextItem]) {
|
||||
let rtl = is_rtl_text(items.iter().map(|i| &i.text));
|
||||
if rtl {
|
||||
items.sort_by(|a, b| b.x.partial_cmp(&a.x).unwrap_or(std::cmp::Ordering::Equal));
|
||||
items.sort_by(|a, b| b.x.total_cmp(&a.x));
|
||||
} else {
|
||||
items.sort_by(|a, b| a.x.partial_cmp(&b.x).unwrap_or(std::cmp::Ordering::Equal));
|
||||
items.sort_by(|a, b| a.x.total_cmp(&b.x));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -376,7 +376,7 @@ fn compute_canva_join_threshold(items: &[TextItem]) -> f32 {
|
||||
}
|
||||
|
||||
let mut sorted: Vec<f32> = ratios;
|
||||
sorted.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
sorted.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
if sorted[sorted.len() - 1] < 0.40 || sorted[0] < 0.40 {
|
||||
return DEFAULT;
|
||||
@@ -478,7 +478,7 @@ fn compute_single_char_join_threshold(items: &[TextItem]) -> f32 {
|
||||
return DEFAULT;
|
||||
}
|
||||
|
||||
ratios.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal));
|
||||
ratios.sort_by(|a, b| a.total_cmp(b));
|
||||
|
||||
// If all gaps are tight (max < 0.40), use default — normal PDF
|
||||
let max_ratio = ratios[ratios.len() - 1];
|
||||
|
||||
+81
-11
@@ -1771,15 +1771,49 @@ impl FontCMaps {
|
||||
/// Iterates every page, collects fonts (including Form XObject fonts),
|
||||
/// and parses any `/ToUnicode` streams via lopdf's decompression.
|
||||
pub fn from_doc(doc: &Document) -> Self {
|
||||
Self::from_doc_pages(doc, None)
|
||||
}
|
||||
|
||||
/// Build FontCMaps for specific pages only. Pass `None` for all pages.
|
||||
pub fn from_doc_pages(doc: &Document, page_filter: Option<&HashSet<u32>>) -> Self {
|
||||
Self::from_doc_pages_inner(doc, page_filter, false)
|
||||
}
|
||||
|
||||
/// Build FontCMaps in fast mode: skip expensive TrueType font fallback
|
||||
/// parsing. Fonts that can't be decoded from their ToUnicode CMap alone
|
||||
/// will be missing, causing text extraction to produce empty/garbage text
|
||||
/// which triggers `needs_ocr` fallback. This is ideal for hybrid OCR
|
||||
/// pipelines where GPU OCR is always available as a fallback.
|
||||
pub fn from_doc_pages_fast(doc: &Document, page_filter: Option<&HashSet<u32>>) -> Self {
|
||||
Self::from_doc_pages_inner(doc, page_filter, true)
|
||||
}
|
||||
|
||||
fn from_doc_pages_inner(
|
||||
doc: &Document,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
skip_truetype_fallback: bool,
|
||||
) -> Self {
|
||||
let mut by_obj_num: HashMap<u32, CMapEntry> = HashMap::new();
|
||||
|
||||
for (_page_num, &page_id) in doc.get_pages().iter() {
|
||||
for (page_num, &page_id) in doc.get_pages().iter() {
|
||||
if let Some(filter) = page_filter {
|
||||
if !filter.contains(page_num) {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
// Page-level fonts (includes inherited parent resources)
|
||||
let fonts = doc.get_page_fonts(page_id).unwrap_or_default();
|
||||
Self::collect_cmaps_from_fonts(&fonts, doc, &mut by_obj_num);
|
||||
Self::collect_cmaps_from_fonts_inner(
|
||||
&fonts,
|
||||
doc,
|
||||
&mut by_obj_num,
|
||||
skip_truetype_fallback,
|
||||
);
|
||||
|
||||
// Fonts inside Form XObjects referenced by this page
|
||||
Self::collect_cmaps_from_xobjects(doc, page_id, &mut by_obj_num);
|
||||
if !skip_truetype_fallback {
|
||||
// Fonts inside Form XObjects referenced by this page
|
||||
Self::collect_cmaps_from_xobjects(doc, page_id, &mut by_obj_num);
|
||||
}
|
||||
}
|
||||
|
||||
FontCMaps { by_obj_num }
|
||||
@@ -1792,6 +1826,15 @@ impl FontCMaps {
|
||||
fonts: &std::collections::BTreeMap<Vec<u8>, &lopdf::Dictionary>,
|
||||
doc: &Document,
|
||||
by_obj_num: &mut HashMap<u32, CMapEntry>,
|
||||
) {
|
||||
Self::collect_cmaps_from_fonts_inner(fonts, doc, by_obj_num, false);
|
||||
}
|
||||
|
||||
fn collect_cmaps_from_fonts_inner(
|
||||
fonts: &std::collections::BTreeMap<Vec<u8>, &lopdf::Dictionary>,
|
||||
doc: &Document,
|
||||
by_obj_num: &mut HashMap<u32, CMapEntry>,
|
||||
skip_truetype_fallback: bool,
|
||||
) {
|
||||
// First pass: collect ToUnicode CMaps
|
||||
for font_dict in fonts.values() {
|
||||
@@ -1825,13 +1868,32 @@ impl FontCMaps {
|
||||
);
|
||||
let (mut primary, mut remapped) =
|
||||
try_remap_subset_cmap(cmap, font_dict, doc, obj_num);
|
||||
let mut fallback = build_fallback_tounicode_from_encoding(font_dict, doc)
|
||||
.or_else(|| build_fallback_cmap_for_type0(font_dict, doc))
|
||||
.or_else(|| build_fallback_cmap_for_simple(font_dict, doc));
|
||||
|
||||
// If the ToUnicode map is extremely sparse, prefer the fallback
|
||||
// (often a better mapping for Symbol/Wingdings/Arabic CID fonts).
|
||||
// Only build expensive fallbacks when the primary CMap is sparse.
|
||||
// build_fallback_cmap_for_type0 can take seconds on large embedded
|
||||
// TrueType fonts (decompressing + parsing 100K+ byte font files).
|
||||
// Skip entirely when the primary CMap is sufficient.
|
||||
let primary_entries = primary.char_map.len() + primary.ranges.len();
|
||||
let mut fallback = if primary_entries < 10 && !skip_truetype_fallback {
|
||||
// Try cheap fallback first; only attempt expensive TrueType
|
||||
// parsing if cheap fallbacks don't yield results.
|
||||
let cheap = build_fallback_tounicode_from_encoding(font_dict, doc)
|
||||
.or_else(|| build_fallback_cmap_for_simple(font_dict, doc));
|
||||
if cheap.is_some() {
|
||||
cheap
|
||||
} else {
|
||||
build_fallback_cmap_for_type0(font_dict, doc)
|
||||
}
|
||||
} else if primary_entries < 10 {
|
||||
// Fast mode: only try cheap fallbacks, skip TrueType parsing.
|
||||
// Regions using this font will get needs_ocr=true.
|
||||
build_fallback_tounicode_from_encoding(font_dict, doc)
|
||||
.or_else(|| build_fallback_cmap_for_simple(font_dict, doc))
|
||||
} else {
|
||||
// Primary is rich enough; only try the cheap encoding fallback
|
||||
build_fallback_tounicode_from_encoding(font_dict, doc)
|
||||
};
|
||||
|
||||
if primary_entries < 10 {
|
||||
if let Some(fb) = fallback.take() {
|
||||
debug!(
|
||||
@@ -1852,8 +1914,12 @@ impl FontCMaps {
|
||||
);
|
||||
} else {
|
||||
// ToUnicode present but parse failed; try fallbacks to avoid empty decoding.
|
||||
let fallback = build_fallback_cmap_for_type0(font_dict, doc)
|
||||
.or_else(|| build_fallback_cmap_for_simple(font_dict, doc));
|
||||
let fallback = if skip_truetype_fallback {
|
||||
build_fallback_cmap_for_simple(font_dict, doc)
|
||||
} else {
|
||||
build_fallback_cmap_for_type0(font_dict, doc)
|
||||
.or_else(|| build_fallback_cmap_for_simple(font_dict, doc))
|
||||
};
|
||||
if let Some(fb) = fallback {
|
||||
debug!(
|
||||
"ToUnicode CMap obj={} parse failed; using fallback (entries={})",
|
||||
@@ -1874,6 +1940,10 @@ impl FontCMaps {
|
||||
|
||||
// Second pass: Identity-H/V fonts without ToUnicode
|
||||
// Try: (1) embedded TrueType/OpenType cmap, (2) predefined CID→Unicode mapping
|
||||
// Skip entirely in fast mode — these fonts require expensive TrueType parsing.
|
||||
if skip_truetype_fallback {
|
||||
return;
|
||||
}
|
||||
for font_dict in fonts.values() {
|
||||
if font_dict.get(b"ToUnicode").is_ok() {
|
||||
continue;
|
||||
|
||||
+190
-2
@@ -4,9 +4,11 @@ use pdf_inspector::detector::{DetectionConfig, ScanStrategy};
|
||||
use pdf_inspector::extractor::group_into_lines;
|
||||
use pdf_inspector::types::TextLine;
|
||||
use pdf_inspector::{
|
||||
detect_pdf_type, extract_text, extract_text_with_positions, process_pdf_with_options,
|
||||
to_markdown, MarkdownOptions, PdfError, PdfOptions, PdfType, TextItem,
|
||||
detect_pdf_type, extract_text, extract_text_in_regions_mem, extract_text_with_positions,
|
||||
process_pdf_mem, process_pdf_with_options, to_markdown, MarkdownOptions, PdfError, PdfOptions,
|
||||
PdfType, TextItem,
|
||||
};
|
||||
use std::collections::HashSet;
|
||||
|
||||
// Helper to create test TextItems
|
||||
fn make_text_item(text: &str, x: f32, y: f32, font_size: f32, page: u32) -> TextItem {
|
||||
@@ -1107,3 +1109,189 @@ fn test_rotated_table_layout_correction() {
|
||||
"District data should be in a markdown table row"
|
||||
);
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
// extract_text_in_regions_mem tests
|
||||
// =========================================================================
|
||||
|
||||
/// Build full-page region args for `page_count` pages.
|
||||
/// Uses a generously large bbox (1200x1200) to capture any page size.
|
||||
fn full_page_regions(page_count: u32) -> Vec<(u32, Vec<[f32; 4]>)> {
|
||||
(0..page_count)
|
||||
.map(|p| (p, vec![[0.0, 0.0, 1200.0, 1200.0]]))
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Normalize text for comparison: lowercase, strip non-alphanumeric, split into words.
|
||||
fn normalize_words(text: &str) -> HashSet<String> {
|
||||
text.split(|c: char| !c.is_alphanumeric())
|
||||
.map(|w| w.to_lowercase())
|
||||
.filter(|w| w.len() > 3)
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Fraction of normalized words in `a` that also appear in `b`.
|
||||
fn word_overlap_ratio(a: &str, b: &str) -> f64 {
|
||||
let words_a = normalize_words(a);
|
||||
if words_a.is_empty() {
|
||||
return if normalize_words(b).is_empty() {
|
||||
1.0
|
||||
} else {
|
||||
0.0
|
||||
};
|
||||
}
|
||||
let words_b = normalize_words(b);
|
||||
let overlap = words_a.intersection(&words_b).count();
|
||||
overlap as f64 / words_a.len() as f64
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_regions_mem_basic_text_pdf() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
let result = process_pdf_mem(&buf).unwrap();
|
||||
let page_count = result.page_count;
|
||||
|
||||
let regions = extract_text_in_regions_mem(&buf, &full_page_regions(page_count)).unwrap();
|
||||
assert_eq!(regions.len(), page_count as usize);
|
||||
|
||||
// Each result should have exactly 1 region (we passed one per page)
|
||||
for r in ®ions {
|
||||
assert_eq!(r.regions.len(), 1);
|
||||
}
|
||||
|
||||
// First page should have non-empty text
|
||||
let first = ®ions[0].regions[0];
|
||||
assert!(!first.text.trim().is_empty(), "First page should have text");
|
||||
assert_eq!(regions[0].page, 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_regions_mem_identity_h_needs_ocr() {
|
||||
let buf = std::fs::read("tests/fixtures/shinagawa_identity_h.pdf").unwrap();
|
||||
let regions =
|
||||
extract_text_in_regions_mem(&buf, &[(0, vec![[0.0, 0.0, 1200.0, 1200.0]])]).unwrap();
|
||||
assert_eq!(regions.len(), 1);
|
||||
assert!(
|
||||
regions[0].regions[0].needs_ocr,
|
||||
"Identity-H font without ToUnicode should trigger needs_ocr"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_regions_mem_multiple_regions_per_page() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
let regions = extract_text_in_regions_mem(
|
||||
&buf,
|
||||
&[(
|
||||
0,
|
||||
vec![
|
||||
[0.0, 0.0, 300.0, 100.0], // small top-left
|
||||
[0.0, 0.0, 1200.0, 1200.0], // full page
|
||||
],
|
||||
)],
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(regions.len(), 1);
|
||||
assert_eq!(regions[0].regions.len(), 2);
|
||||
|
||||
let small_len = regions[0].regions[0].text.len();
|
||||
let full_len = regions[0].regions[1].text.len();
|
||||
assert!(
|
||||
full_len >= small_len,
|
||||
"Full-page region ({full_len}) should have at least as much text as small region ({small_len})"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_regions_mem_nonexistent_page() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
let regions =
|
||||
extract_text_in_regions_mem(&buf, &[(9999, vec![[0.0, 0.0, 1200.0, 1200.0]])]).unwrap();
|
||||
assert_eq!(regions.len(), 1);
|
||||
assert!(
|
||||
regions[0].regions[0].needs_ocr,
|
||||
"Nonexistent page should trigger needs_ocr"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_regions_mem_empty_region() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
let regions = extract_text_in_regions_mem(&buf, &[(0, vec![[0.0, 0.0, 0.0, 0.0]])]).unwrap();
|
||||
assert_eq!(regions.len(), 1);
|
||||
assert!(
|
||||
regions[0].regions[0].needs_ocr,
|
||||
"Zero-area region should trigger needs_ocr"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_regions_mem_not_a_pdf() {
|
||||
let result = extract_text_in_regions_mem(b"not a pdf", &[(0, vec![[0.0, 0.0, 100.0, 100.0]])]);
|
||||
assert!(result.is_err(), "Non-PDF input should return an error");
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
// Fast vs normal extraction comparison
|
||||
// =========================================================================
|
||||
|
||||
/// For each text-based fixture PDF, compare `extract_text_in_regions_mem` (fast path)
|
||||
/// against `process_pdf_mem` (normal path). If the fast path claims needs_ocr=false
|
||||
/// for a page, verify the extracted text has meaningful overlap with the normal
|
||||
/// markdown output — catching silent quality regressions.
|
||||
#[test]
|
||||
fn test_extract_regions_fast_vs_normal_comparison() {
|
||||
let fixtures = [
|
||||
"tests/fixtures/nexo-price-en.pdf",
|
||||
"tests/fixtures/td9264.pdf",
|
||||
"tests/fixtures/p1244-1996.pdf",
|
||||
"tests/fixtures/real-estate-pricing.pdf",
|
||||
"tests/fixtures/2013-app2.pdf",
|
||||
"tests/fixtures/firecrawl_docs_tagged.pdf",
|
||||
"tests/fixtures/thermo-freon12.pdf",
|
||||
];
|
||||
|
||||
for fixture in &fixtures {
|
||||
let buf = std::fs::read(fixture).unwrap();
|
||||
let normal = process_pdf_mem(&buf).unwrap();
|
||||
let normal_md = normal.markdown.as_deref().unwrap_or("");
|
||||
let page_count = normal.page_count;
|
||||
let ocr_pages: HashSet<u32> = normal.pages_needing_ocr.iter().copied().collect();
|
||||
|
||||
let regions = extract_text_in_regions_mem(&buf, &full_page_regions(page_count)).unwrap();
|
||||
|
||||
assert_eq!(
|
||||
regions.len(),
|
||||
page_count as usize,
|
||||
"{fixture}: result count should match page count"
|
||||
);
|
||||
|
||||
for pr in ®ions {
|
||||
let region = &pr.regions[0];
|
||||
if !region.needs_ocr && !region.text.trim().is_empty() {
|
||||
// Fast path claims this text is trustworthy.
|
||||
// Check that its words appear in the normal markdown output.
|
||||
let overlap = word_overlap_ratio(®ion.text, normal_md);
|
||||
assert!(
|
||||
overlap >= 0.3,
|
||||
"{fixture} page {}: fast path says needs_ocr=false but only {:.0}% word \
|
||||
overlap with normal extraction (threshold 30%). \
|
||||
Fast text sample: {:?}",
|
||||
pr.page,
|
||||
overlap * 100.0,
|
||||
®ion.text[..region.text.len().min(200)],
|
||||
);
|
||||
}
|
||||
|
||||
// If fast path flags needs_ocr but normal path didn't, that's overly
|
||||
// conservative but not a bug — just worth knowing.
|
||||
if region.needs_ocr && !ocr_pages.contains(&(pr.page + 1)) {
|
||||
eprintln!(
|
||||
"INFO: {fixture} page {}: fast path says needs_ocr=true but normal path extracted fine (conservative, not a bug)",
|
||||
pr.page,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,324 @@
|
||||
"""Tests for the pdf_inspector Python bindings."""
|
||||
|
||||
import os
|
||||
import pytest
|
||||
import pdf_inspector
|
||||
|
||||
FIXTURES_DIR = os.path.join(os.path.dirname(__file__), "fixtures")
|
||||
|
||||
|
||||
def fixture_path(name: str) -> str:
|
||||
return os.path.join(FIXTURES_DIR, name)
|
||||
|
||||
|
||||
def fixture_bytes(name: str) -> bytes:
|
||||
with open(fixture_path(name), "rb") as f:
|
||||
return f.read()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# process_pdf
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestProcessPdf:
|
||||
def test_basic(self):
|
||||
result = pdf_inspector.process_pdf(fixture_path("thermo-freon12.pdf"))
|
||||
assert result.pdf_type == "text_based"
|
||||
assert result.page_count == 3
|
||||
assert result.confidence > 0.0
|
||||
assert result.markdown is not None
|
||||
assert len(result.markdown) > 0
|
||||
|
||||
def test_result_repr(self):
|
||||
result = pdf_inspector.process_pdf(fixture_path("thermo-freon12.pdf"))
|
||||
r = repr(result)
|
||||
assert "PdfResult" in r
|
||||
assert "text_based" in r
|
||||
|
||||
def test_with_pages(self):
|
||||
result = pdf_inspector.process_pdf(
|
||||
fixture_path("thermo-freon12.pdf"), pages=[1]
|
||||
)
|
||||
assert result.page_count == 3 # total pages in doc
|
||||
assert result.markdown is not None
|
||||
|
||||
def test_result_fields(self):
|
||||
result = pdf_inspector.process_pdf(fixture_path("thermo-freon12.pdf"))
|
||||
# All fields should be accessible
|
||||
assert isinstance(result.pdf_type, str)
|
||||
assert isinstance(result.page_count, int)
|
||||
assert isinstance(result.processing_time_ms, int)
|
||||
assert isinstance(result.pages_needing_ocr, list)
|
||||
assert isinstance(result.confidence, float)
|
||||
assert isinstance(result.is_complex_layout, bool)
|
||||
assert isinstance(result.pages_with_tables, list)
|
||||
assert isinstance(result.pages_with_columns, list)
|
||||
assert isinstance(result.has_encoding_issues, bool)
|
||||
# title can be None or str
|
||||
assert result.title is None or isinstance(result.title, str)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# process_pdf_bytes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestProcessPdfBytes:
|
||||
def test_basic(self):
|
||||
data = fixture_bytes("thermo-freon12.pdf")
|
||||
result = pdf_inspector.process_pdf_bytes(data)
|
||||
assert result.pdf_type == "text_based"
|
||||
assert result.markdown is not None
|
||||
|
||||
def test_with_pages(self):
|
||||
data = fixture_bytes("thermo-freon12.pdf")
|
||||
result = pdf_inspector.process_pdf_bytes(data, pages=[1, 2])
|
||||
assert result.markdown is not None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# detect_pdf / detect_pdf_bytes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestDetectPdf:
|
||||
def test_detect_file(self):
|
||||
result = pdf_inspector.detect_pdf(fixture_path("thermo-freon12.pdf"))
|
||||
assert result.pdf_type == "text_based"
|
||||
assert result.markdown is None # detect only — no markdown
|
||||
assert result.page_count == 3
|
||||
|
||||
def test_detect_bytes(self):
|
||||
data = fixture_bytes("thermo-freon12.pdf")
|
||||
result = pdf_inspector.detect_pdf_bytes(data)
|
||||
assert result.pdf_type == "text_based"
|
||||
assert result.markdown is None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# classify_pdf / classify_pdf_bytes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestClassifyPdf:
|
||||
def test_classify_file(self):
|
||||
result = pdf_inspector.classify_pdf(fixture_path("thermo-freon12.pdf"))
|
||||
assert result.pdf_type == "text_based"
|
||||
assert result.page_count == 3
|
||||
assert result.confidence > 0.0
|
||||
assert isinstance(result.pages_needing_ocr, list)
|
||||
|
||||
def test_classify_bytes(self):
|
||||
data = fixture_bytes("thermo-freon12.pdf")
|
||||
result = pdf_inspector.classify_pdf_bytes(data)
|
||||
assert result.pdf_type == "text_based"
|
||||
assert result.page_count == 3
|
||||
assert result.confidence > 0.0
|
||||
|
||||
def test_classify_repr(self):
|
||||
result = pdf_inspector.classify_pdf(fixture_path("thermo-freon12.pdf"))
|
||||
r = repr(result)
|
||||
assert "PdfClassification" in r
|
||||
assert "text_based" in r
|
||||
|
||||
def test_classify_fields(self):
|
||||
result = pdf_inspector.classify_pdf(fixture_path("thermo-freon12.pdf"))
|
||||
assert isinstance(result.pdf_type, str)
|
||||
assert isinstance(result.page_count, int)
|
||||
assert isinstance(result.pages_needing_ocr, list)
|
||||
assert isinstance(result.confidence, float)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# extract_text / extract_text_bytes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestExtractText:
|
||||
def test_basic(self):
|
||||
text = pdf_inspector.extract_text(fixture_path("thermo-freon12.pdf"))
|
||||
assert isinstance(text, str)
|
||||
assert len(text) > 0
|
||||
|
||||
def test_bytes(self):
|
||||
data = fixture_bytes("thermo-freon12.pdf")
|
||||
text = pdf_inspector.extract_text_bytes(data)
|
||||
assert isinstance(text, str)
|
||||
assert len(text) > 0
|
||||
|
||||
def test_bytes_matches_file(self):
|
||||
text_file = pdf_inspector.extract_text(fixture_path("thermo-freon12.pdf"))
|
||||
text_bytes = pdf_inspector.extract_text_bytes(fixture_bytes("thermo-freon12.pdf"))
|
||||
assert text_file == text_bytes
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# extract_text_with_positions / extract_text_with_positions_bytes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestExtractTextWithPositions:
|
||||
def test_basic(self):
|
||||
items = pdf_inspector.extract_text_with_positions(
|
||||
fixture_path("thermo-freon12.pdf")
|
||||
)
|
||||
assert len(items) > 0
|
||||
item = items[0]
|
||||
assert isinstance(item.text, str)
|
||||
assert isinstance(item.x, float)
|
||||
assert isinstance(item.y, float)
|
||||
assert isinstance(item.width, float)
|
||||
assert isinstance(item.height, float)
|
||||
assert isinstance(item.font, str)
|
||||
assert isinstance(item.font_size, float)
|
||||
assert isinstance(item.page, int)
|
||||
assert isinstance(item.is_bold, bool)
|
||||
assert isinstance(item.is_italic, bool)
|
||||
assert isinstance(item.item_type, str)
|
||||
|
||||
def test_with_pages(self):
|
||||
items = pdf_inspector.extract_text_with_positions(
|
||||
fixture_path("thermo-freon12.pdf"), pages=[1]
|
||||
)
|
||||
assert len(items) > 0
|
||||
assert all(item.page == 1 for item in items)
|
||||
|
||||
def test_repr(self):
|
||||
items = pdf_inspector.extract_text_with_positions(
|
||||
fixture_path("thermo-freon12.pdf")
|
||||
)
|
||||
r = repr(items[0])
|
||||
assert "TextItem" in r
|
||||
|
||||
def test_bytes(self):
|
||||
data = fixture_bytes("thermo-freon12.pdf")
|
||||
items = pdf_inspector.extract_text_with_positions_bytes(data)
|
||||
assert len(items) > 0
|
||||
assert isinstance(items[0].text, str)
|
||||
|
||||
def test_bytes_with_pages(self):
|
||||
data = fixture_bytes("thermo-freon12.pdf")
|
||||
items = pdf_inspector.extract_text_with_positions_bytes(data, pages=[1])
|
||||
assert len(items) > 0
|
||||
assert all(item.page == 1 for item in items)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# extract_text_in_regions / extract_text_in_regions_bytes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestExtractTextInRegions:
|
||||
def test_file(self):
|
||||
results = pdf_inspector.extract_text_in_regions(
|
||||
fixture_path("thermo-freon12.pdf"),
|
||||
[(0, [[0.0, 0.0, 600.0, 100.0]])],
|
||||
)
|
||||
assert len(results) == 1
|
||||
assert results[0].page == 0
|
||||
assert len(results[0].regions) == 1
|
||||
assert isinstance(results[0].regions[0].text, str)
|
||||
assert isinstance(results[0].regions[0].needs_ocr, bool)
|
||||
|
||||
def test_bytes(self):
|
||||
data = fixture_bytes("thermo-freon12.pdf")
|
||||
results = pdf_inspector.extract_text_in_regions_bytes(
|
||||
data,
|
||||
[(0, [[0.0, 0.0, 600.0, 100.0]])],
|
||||
)
|
||||
assert len(results) == 1
|
||||
assert results[0].page == 0
|
||||
assert len(results[0].regions) == 1
|
||||
assert isinstance(results[0].regions[0].text, str)
|
||||
|
||||
def test_repr(self):
|
||||
results = pdf_inspector.extract_text_in_regions(
|
||||
fixture_path("thermo-freon12.pdf"),
|
||||
[(0, [[0.0, 0.0, 600.0, 100.0]])],
|
||||
)
|
||||
r = repr(results[0])
|
||||
assert "PageRegionTexts" in r
|
||||
r2 = repr(results[0].regions[0])
|
||||
assert "RegionText" in r2
|
||||
|
||||
def test_multiple_regions(self):
|
||||
results = pdf_inspector.extract_text_in_regions(
|
||||
fixture_path("thermo-freon12.pdf"),
|
||||
[(0, [[0.0, 0.0, 300.0, 100.0], [300.0, 0.0, 600.0, 100.0]])],
|
||||
)
|
||||
assert len(results) == 1
|
||||
assert len(results[0].regions) == 2
|
||||
|
||||
def test_multiple_pages(self):
|
||||
results = pdf_inspector.extract_text_in_regions(
|
||||
fixture_path("thermo-freon12.pdf"),
|
||||
[
|
||||
(0, [[0.0, 0.0, 600.0, 100.0]]),
|
||||
(1, [[0.0, 0.0, 600.0, 100.0]]),
|
||||
],
|
||||
)
|
||||
assert len(results) == 2
|
||||
assert results[0].page == 0
|
||||
assert results[1].page == 1
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Error handling
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestErrors:
|
||||
def test_nonexistent_file(self):
|
||||
with pytest.raises(ValueError):
|
||||
pdf_inspector.process_pdf("/nonexistent/file.pdf")
|
||||
|
||||
def test_not_a_pdf(self):
|
||||
with pytest.raises(ValueError):
|
||||
pdf_inspector.process_pdf_bytes(b"this is not a pdf")
|
||||
|
||||
def test_empty_bytes(self):
|
||||
with pytest.raises(ValueError):
|
||||
pdf_inspector.process_pdf_bytes(b"")
|
||||
|
||||
def test_classify_not_a_pdf(self):
|
||||
with pytest.raises(ValueError):
|
||||
pdf_inspector.classify_pdf_bytes(b"not a pdf")
|
||||
|
||||
def test_classify_nonexistent(self):
|
||||
with pytest.raises((ValueError, OSError)):
|
||||
pdf_inspector.classify_pdf("/nonexistent/file.pdf")
|
||||
|
||||
def test_extract_text_bytes_not_a_pdf(self):
|
||||
with pytest.raises(ValueError):
|
||||
pdf_inspector.extract_text_bytes(b"not a pdf")
|
||||
|
||||
def test_regions_not_a_pdf(self):
|
||||
with pytest.raises(ValueError):
|
||||
pdf_inspector.extract_text_in_regions_bytes(
|
||||
b"not a pdf", [(0, [[0.0, 0.0, 100.0, 100.0]])]
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Multiple fixtures
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestMultipleFixtures:
|
||||
"""Run basic processing on all available test fixtures."""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"filename",
|
||||
[f for f in os.listdir(FIXTURES_DIR) if f.endswith(".pdf")],
|
||||
)
|
||||
def test_process_all_fixtures(self, filename):
|
||||
result = pdf_inspector.process_pdf(fixture_path(filename))
|
||||
assert result.pdf_type in (
|
||||
"text_based",
|
||||
"scanned",
|
||||
"image_based",
|
||||
"mixed",
|
||||
)
|
||||
assert result.page_count > 0
|
||||
assert result.confidence >= 0.0
|
||||
Reference in New Issue
Block a user