[package] name = "pdf-inspector" version = "0.1.0" edition = "2021" autobins = false authors = ["Firecrawl Team"] description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection" license = "MIT" repository = "https://github.com/firecrawl/pdf-inspector" [dependencies] # PDF parsing lopdf = { git = "https://github.com/firecrawl/lopdf", branch = "firecrawl/zlib-checksum-encrypted", features = ["rayon"] } # Error handling thiserror = "2.0" # Parallel processing rayon = "1.10" # Logging log = "0.4" env_logger = "0.11" # Text processing regex = "1.10" once_cell = "1.19" # TrueType font parsing (for Identity-H CID font cmap extraction) ttf-parser = "0.25" [dev-dependencies] tempfile = "3.3" [features] default = [] [[bin]] name = "pdf2md" path = "src/bin/pdf2md.rs" [[bin]] name = "detect-pdf" path = "src/bin/detect_pdf.rs" [[bin]] name = "dump_ops" path = "src/bin/dump_ops.rs"