[package] name = "pdf-inspector" version = "0.1.0" edition = "2021" autobins = false authors = ["Firecrawl Team"] description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection" license = "MIT" repository = "https://github.com/firecrawl/pdf-inspector" [lib] name = "pdf_inspector" crate-type = ["lib", "cdylib"] [dependencies] # Python bindings pyo3 = { version = "0.25", features = ["extension-module"], optional = true } # PDF parsing lopdf = { git = "https://github.com/J-F-Liu/lopdf", rev = "052674053814a9f4897af94f0b8e46a545c9b329", features = ["rayon"] } # Error handling thiserror = "2.0" # Parallel processing rayon = "1.10" # Logging log = "0.4" env_logger = "0.11" # Text processing regex = "1.10" once_cell = "1.19" unicode-normalization = "0.1" # TrueType font parsing (for Identity-H CID font cmap extraction) ttf-parser = "0.25" [dev-dependencies] tempfile = "3.3" [features] default = [] python = ["pyo3"] [[bin]] name = "pdf2md" path = "src/bin/pdf2md.rs" [[bin]] name = "detect-pdf" path = "src/bin/detect_pdf.rs" [[bin]] name = "dump_ops" path = "src/bin/dump_ops.rs"