Files
pdf-inspector/Cargo.toml
T
Abimael MartellandClaude Opus 4.6 5ab92e03e3 feat(tables): detect tables from clip-path rects via merged-cluster fallback
Extract axis-aligned rectangles from W/W* clip operators in content
streams — many PDFs define table cells as clipping paths instead of
stroked rects. Add merged-cluster fallback in detect_tables_from_rects()
that merges all cluster rects when per-cluster detection fails or only
produces narrow false-positives (≤3 columns). Uses rect Y-edges for
rows and text X-clustering for columns.

Also updates lopdf to firecrawl fork (fix-leading-whitespace branch).

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
2026-03-05 17:12:26 -08:00

51 lines
945 B
TOML

[package]
name = "pdf-inspector"
version = "0.1.0"
edition = "2021"
autobins = false
authors = ["Firecrawl Team"]
description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection"
license = "MIT"
repository = "https://github.com/firecrawl/pdf-inspector"
[dependencies]
# PDF parsing
lopdf = { git = "https://github.com/firecrawl/lopdf", branch = "firecrawl/fix-leading-whitespace", features = ["rayon"] }
# Error handling
thiserror = "2.0"
# Parallel processing
rayon = "1.10"
# Logging
log = "0.4"
env_logger = "0.11"
# Text processing
regex = "1.10"
once_cell = "1.19"
# TrueType font parsing (for Identity-H CID font cmap extraction)
ttf-parser = "0.25"
[dev-dependencies]
tempfile = "3.3"
[features]
default = []
[[bin]]
name = "pdf2md"
path = "src/bin/pdf2md.rs"
[[bin]]
name = "detect-pdf"
path = "src/bin/detect_pdf.rs"
[[bin]]
name = "dump_ops"
path = "src/bin/dump_ops.rs"