Extract axis-aligned rectangles from W/W* clip operators in content streams — many PDFs define table cells as clipping paths instead of stroked rects. Add merged-cluster fallback in detect_tables_from_rects() that merges all cluster rects when per-cluster detection fails or only produces narrow false-positives (≤3 columns). Uses rect Y-edges for rows and text X-clustering for columns. Also updates lopdf to firecrawl fork (fix-leading-whitespace branch). Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
51 lines
945 B
TOML
51 lines
945 B
TOML
[package]
|
|
name = "pdf-inspector"
|
|
version = "0.1.0"
|
|
edition = "2021"
|
|
autobins = false
|
|
authors = ["Firecrawl Team"]
|
|
description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection"
|
|
license = "MIT"
|
|
repository = "https://github.com/firecrawl/pdf-inspector"
|
|
|
|
[dependencies]
|
|
# PDF parsing
|
|
lopdf = { git = "https://github.com/firecrawl/lopdf", branch = "firecrawl/fix-leading-whitespace", features = ["rayon"] }
|
|
|
|
# Error handling
|
|
thiserror = "2.0"
|
|
|
|
# Parallel processing
|
|
rayon = "1.10"
|
|
|
|
# Logging
|
|
log = "0.4"
|
|
env_logger = "0.11"
|
|
|
|
# Text processing
|
|
regex = "1.10"
|
|
once_cell = "1.19"
|
|
|
|
# TrueType font parsing (for Identity-H CID font cmap extraction)
|
|
ttf-parser = "0.25"
|
|
|
|
[dev-dependencies]
|
|
tempfile = "3.3"
|
|
|
|
[features]
|
|
default = []
|
|
|
|
[[bin]]
|
|
name = "pdf2md"
|
|
path = "src/bin/pdf2md.rs"
|
|
|
|
[[bin]]
|
|
name = "detect-pdf"
|
|
path = "src/bin/detect_pdf.rs"
|
|
|
|
[[bin]]
|
|
name = "dump_ops"
|
|
path = "src/bin/dump_ops.rs"
|
|
|
|
|