feat(bindings): expose OCR in Node and Python (#405)
* feat(bindings): expose selective OCR * fix(bindings): address review feedback
This commit is contained in:
Generated
+2253
-30
File diff suppressed because it is too large
Load Diff
+1
-1
@@ -7,7 +7,7 @@ edition = "2021"
|
||||
crate-type = ["cdylib"]
|
||||
|
||||
[dependencies]
|
||||
pdf-inspector = { path = ".." }
|
||||
pdf-inspector = { path = "..", features = ["ocr"] }
|
||||
napi = { version = "3.0.0", features = ["serde-json"] }
|
||||
napi-derive = "3.0.0"
|
||||
|
||||
|
||||
+50
-1
@@ -10,7 +10,8 @@ Built by [Firecrawl](https://firecrawl.dev) for hybrid OCR pipelines — extract
|
||||
- **Region-based extraction** — pull text from bounding boxes with per-region quality checks (`needsOcr`).
|
||||
- **Layout-aware** — multi-column reading order, position and font info per text item, RTL support.
|
||||
- **Robust text decoding** — CID/Type0 fonts via ToUnicode CMaps, plus automatic flagging of broken encodings so callers can fall back to OCR.
|
||||
- **Lightweight** — native Rust core via napi-rs, no ML models, no external services; ~5–6 MB platform binary, TypeScript definitions included.
|
||||
- **Selective OCR** — `Auto` routes only pages rejected by native extraction and returns source/model provenance plus hosted-fallback recommendations.
|
||||
- **External artifacts** — the native package embeds no OCR models, PDFium, or ONNX Runtime; clean `Auto` requests never load or download them.
|
||||
|
||||
## Benchmark
|
||||
|
||||
@@ -36,8 +37,40 @@ bun add @firecrawl/pdf-inspector
|
||||
|
||||
Prebuilt binaries for **Linux x64/ARM64** (glibc and musl/Alpine), **macOS ARM64**, and **Windows x64** — npm installs only the one matching your platform. No Rust toolchain needed.
|
||||
|
||||
OCR calls that route work require compatible PDFium and ONNX Runtime shared
|
||||
libraries. Set `PDFIUM_LIB_PATH` and `ORT_DYLIB_PATH` when they are not on the
|
||||
platform library search path. The pinned OCR model set is downloaded and
|
||||
checksum-verified on the first routed page; use `offline: true` with a warm
|
||||
cache or `modelDirectory` to prohibit network access.
|
||||
|
||||
## API
|
||||
|
||||
### `processPdfWithOcr(buffer: Buffer, options?: OcrOptions): Promise<OcrPdfResult>`
|
||||
|
||||
Run native extraction first and OCR only the pages selected by its quality
|
||||
signals. The default mode is `Auto`; `Off` returns the same detailed result
|
||||
shape without external runtime work, and `Force` OCRs every selected page.
|
||||
The work runs on the libuv thread pool and never blocks Node's event loop.
|
||||
|
||||
```typescript
|
||||
import { OcrMode, processPdfWithOcr } from '@firecrawl/pdf-inspector'
|
||||
|
||||
const result = await processPdfWithOcr(pdf, {
|
||||
mode: OcrMode.Auto,
|
||||
pageNumbers: [1, 3], // 1-indexed
|
||||
})
|
||||
|
||||
for (const page of result.pages) {
|
||||
console.log(page.pageNumber, page.provenance.source)
|
||||
}
|
||||
console.log(result.pagesRoutedToOcr)
|
||||
console.log(result.pagesRecommendingHosted)
|
||||
```
|
||||
|
||||
For offline deployments, pass `modelDirectory` and `offline: true`. Other
|
||||
controls include `dpi`, `minimumConfidence`,
|
||||
`hostedRecommendationConfidence`, and `password`.
|
||||
|
||||
### `classifyPdf(buffer: Buffer): PdfClassification`
|
||||
|
||||
Classify a PDF as TextBased, Scanned, Mixed, or ImageBased (~10-50ms). Returns which pages need OCR.
|
||||
@@ -124,6 +157,22 @@ interface RegionText {
|
||||
needsOcr: boolean // true when text is unreliable
|
||||
ocrReason?: string // "suspected_garbled_text" when known
|
||||
}
|
||||
|
||||
interface OcrPdfResult {
|
||||
markdown: string
|
||||
pages: OcrPageResult[] // 1-indexed pages + provenance
|
||||
pageCount: number
|
||||
pagesRecommendedForOcr: number[]
|
||||
pagesRoutedToOcr: number[]
|
||||
pagesRecommendingHosted: number[]
|
||||
ocrReasonsByPage: PageOcrReasons[]
|
||||
pagesWithTables: number[]
|
||||
pagesWithColumns: number[]
|
||||
isComplex: boolean
|
||||
processingTimeMs: number
|
||||
renderTimeMs: number
|
||||
ocrTimeMs: number
|
||||
}
|
||||
```
|
||||
|
||||
## Platforms
|
||||
|
||||
+241
@@ -27,6 +27,26 @@ pub enum ItemType {
|
||||
FormField,
|
||||
}
|
||||
|
||||
/// Selects when OCR runs.
|
||||
#[napi(string_enum)]
|
||||
#[derive(Clone, Copy)]
|
||||
pub enum OcrMode {
|
||||
/// Never run OCR; return the native extraction in the OCR result shape.
|
||||
Off,
|
||||
/// Run OCR only on pages selected by the native quality signals.
|
||||
Auto,
|
||||
/// Run OCR on every selected page.
|
||||
Force,
|
||||
}
|
||||
|
||||
/// How final page content was sourced.
|
||||
#[napi(string_enum)]
|
||||
pub enum PageContentSource {
|
||||
Native,
|
||||
Ocr,
|
||||
Fused,
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Result types
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -128,6 +148,84 @@ pub struct VectorGridDetectionJs {
|
||||
pub cell_bboxes: Vec<Vec<f64>>,
|
||||
}
|
||||
|
||||
/// Options for one-call native extraction with selective OCR.
|
||||
#[napi(object)]
|
||||
#[derive(Clone)]
|
||||
pub struct OcrOptions {
|
||||
/// OCR routing behavior. Defaults to Auto.
|
||||
pub mode: Option<OcrMode>,
|
||||
/// Optional 1-indexed page selection.
|
||||
pub page_numbers: Option<Vec<u32>>,
|
||||
/// Password for an encrypted PDF.
|
||||
pub password: Option<String>,
|
||||
/// Page rasterization resolution. Defaults to 150 DPI.
|
||||
pub dpi: Option<f64>,
|
||||
/// Drop OCR spans below this inclusive 0-1 threshold.
|
||||
pub minimum_confidence: Option<f64>,
|
||||
/// Recommend hosted parsing below this inclusive 0-1 page confidence.
|
||||
pub hosted_recommendation_confidence: Option<f64>,
|
||||
/// Directory containing an offline OCR model set.
|
||||
pub model_directory: Option<String>,
|
||||
/// Disable model downloads and require a model directory or warm cache.
|
||||
pub offline: Option<bool>,
|
||||
}
|
||||
|
||||
/// Exact OCR model identity retained in page provenance.
|
||||
#[napi(object)]
|
||||
pub struct OcrModelIdentity {
|
||||
pub name: String,
|
||||
pub revision: String,
|
||||
}
|
||||
|
||||
/// Per-page OCR processing timings.
|
||||
#[napi(object)]
|
||||
pub struct OcrTimings {
|
||||
pub render_ms: u32,
|
||||
pub ocr_ms: u32,
|
||||
pub assembly_ms: u32,
|
||||
}
|
||||
|
||||
/// Source, model, confidence, and fallback metadata for one page.
|
||||
#[napi(object)]
|
||||
pub struct OcrPageProvenance {
|
||||
/// 1-indexed page number.
|
||||
pub page_number: u32,
|
||||
pub source: PageContentSource,
|
||||
pub ocr_model: Option<OcrModelIdentity>,
|
||||
pub render_dpi: Option<f64>,
|
||||
pub ocr_confidence: Option<f64>,
|
||||
pub timings: OcrTimings,
|
||||
pub warnings: Vec<String>,
|
||||
pub hosted_recommended: bool,
|
||||
}
|
||||
|
||||
/// Final Markdown and provenance for one page.
|
||||
#[napi(object)]
|
||||
pub struct OcrPageResult {
|
||||
/// 1-indexed page number.
|
||||
pub page_number: u32,
|
||||
pub markdown: String,
|
||||
pub provenance: OcrPageProvenance,
|
||||
}
|
||||
|
||||
/// Complete native/OCR Markdown output.
|
||||
#[napi(object)]
|
||||
pub struct OcrPdfResult {
|
||||
pub markdown: String,
|
||||
pub pages: Vec<OcrPageResult>,
|
||||
pub page_count: u32,
|
||||
pub pages_recommended_for_ocr: Vec<u32>,
|
||||
pub pages_routed_to_ocr: Vec<u32>,
|
||||
pub pages_recommending_hosted: Vec<u32>,
|
||||
pub ocr_reasons_by_page: Vec<PageOcrReasons>,
|
||||
pub pages_with_tables: Vec<u32>,
|
||||
pub pages_with_columns: Vec<u32>,
|
||||
pub is_complex: bool,
|
||||
pub processing_time_ms: u32,
|
||||
pub render_time_ms: u32,
|
||||
pub ocr_time_ms: u32,
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -168,6 +266,103 @@ fn to_napi_page_ocr_reasons(reasons: Vec<pdf_inspector::PageOcrReasons>) -> Vec<
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn to_core_ocr_options(options: Option<OcrOptions>) -> pdf_inspector::vision::OcrPdfOptions {
|
||||
let mut result = pdf_inspector::vision::OcrPdfOptions::auto();
|
||||
let Some(options) = options else {
|
||||
return result;
|
||||
};
|
||||
|
||||
if let Some(mode) = options.mode {
|
||||
result.ocr.mode = match mode {
|
||||
OcrMode::Off => pdf_inspector::vision::OcrMode::Off,
|
||||
OcrMode::Auto => pdf_inspector::vision::OcrMode::Auto,
|
||||
OcrMode::Force => pdf_inspector::vision::OcrMode::Force,
|
||||
};
|
||||
}
|
||||
if let Some(pages) = options.page_numbers {
|
||||
result = result.page_numbers(pages);
|
||||
}
|
||||
if let Some(password) = options.password {
|
||||
result = result.password(password);
|
||||
}
|
||||
if let Some(dpi) = options.dpi {
|
||||
result.render.dpi = dpi as f32;
|
||||
}
|
||||
if let Some(minimum_confidence) = options.minimum_confidence {
|
||||
result.ocr.minimum_confidence = minimum_confidence as f32;
|
||||
}
|
||||
if let Some(confidence) = options.hosted_recommendation_confidence {
|
||||
result.hosted_recommendation_confidence = confidence as f32;
|
||||
}
|
||||
if let Some(directory) = options.model_directory {
|
||||
result.ocr.model_directory = Some(directory.into());
|
||||
}
|
||||
if options.offline.unwrap_or(false) {
|
||||
result.ocr.model_downloads = pdf_inspector::vision::ModelDownloadPolicy::Offline;
|
||||
}
|
||||
result
|
||||
}
|
||||
|
||||
fn convert_page_content_source(
|
||||
source: pdf_inspector::vision::PageContentSource,
|
||||
) -> PageContentSource {
|
||||
match source {
|
||||
pdf_inspector::vision::PageContentSource::Native => PageContentSource::Native,
|
||||
pdf_inspector::vision::PageContentSource::Ocr => PageContentSource::Ocr,
|
||||
pdf_inspector::vision::PageContentSource::Fused => PageContentSource::Fused,
|
||||
_ => PageContentSource::Native,
|
||||
}
|
||||
}
|
||||
|
||||
fn timing_ms(value: u64) -> u32 {
|
||||
u32::try_from(value).unwrap_or(u32::MAX)
|
||||
}
|
||||
|
||||
fn to_napi_ocr_result(result: pdf_inspector::vision::OcrPdfResult) -> OcrPdfResult {
|
||||
OcrPdfResult {
|
||||
markdown: result.markdown,
|
||||
pages: result
|
||||
.pages
|
||||
.into_iter()
|
||||
.map(|page| {
|
||||
let provenance = page.provenance;
|
||||
OcrPageResult {
|
||||
page_number: page.page_number,
|
||||
markdown: page.markdown,
|
||||
provenance: OcrPageProvenance {
|
||||
page_number: provenance.page_number,
|
||||
source: convert_page_content_source(provenance.source),
|
||||
ocr_model: provenance.ocr_model.map(|model| OcrModelIdentity {
|
||||
name: model.name,
|
||||
revision: model.revision,
|
||||
}),
|
||||
render_dpi: provenance.render_dpi.map(f64::from),
|
||||
ocr_confidence: provenance.ocr_confidence.map(f64::from),
|
||||
timings: OcrTimings {
|
||||
render_ms: timing_ms(provenance.timings.render_ms),
|
||||
ocr_ms: timing_ms(provenance.timings.ocr_ms),
|
||||
assembly_ms: timing_ms(provenance.timings.assembly_ms),
|
||||
},
|
||||
warnings: provenance.warnings,
|
||||
hosted_recommended: provenance.hosted_recommended,
|
||||
},
|
||||
}
|
||||
})
|
||||
.collect(),
|
||||
page_count: result.page_count,
|
||||
pages_recommended_for_ocr: result.pages_recommended_for_ocr,
|
||||
pages_routed_to_ocr: result.pages_routed_to_ocr,
|
||||
pages_recommending_hosted: result.pages_recommending_hosted,
|
||||
ocr_reasons_by_page: to_napi_page_ocr_reasons(result.ocr_reasons_by_page),
|
||||
pages_with_tables: result.pages_with_tables,
|
||||
pages_with_columns: result.pages_with_columns,
|
||||
is_complex: result.is_complex,
|
||||
processing_time_ms: timing_ms(result.processing_time_ms),
|
||||
render_time_ms: timing_ms(result.render_time_ms),
|
||||
ocr_time_ms: timing_ms(result.ocr_time_ms),
|
||||
}
|
||||
}
|
||||
|
||||
fn convert_item_type(t: &pdf_inspector::types::ItemType) -> (ItemType, Option<String>) {
|
||||
match t {
|
||||
pdf_inspector::types::ItemType::Text => (ItemType::Text, None),
|
||||
@@ -219,6 +414,13 @@ fn process_pdf_impl(bytes: &[u8], pages: Option<Vec<u32>>) -> Result<PdfResult>
|
||||
Ok(to_napi_result(result))
|
||||
}
|
||||
|
||||
fn process_pdf_with_ocr_impl(bytes: &[u8], options: Option<OcrOptions>) -> Result<OcrPdfResult> {
|
||||
let options = to_core_ocr_options(options);
|
||||
let result = pdf_inspector::vision::process_pdf_with_ocr_mem(bytes, options)
|
||||
.map_err(|error| to_napi_err(error, "process_pdf_with_ocr"))?;
|
||||
Ok(to_napi_ocr_result(result))
|
||||
}
|
||||
|
||||
fn classify_pdf_impl(bytes: &[u8]) -> Result<PdfClassification> {
|
||||
let result =
|
||||
pdf_inspector::classify_pdf_mem(bytes).map_err(|e| to_napi_err(e, "classify_pdf"))?;
|
||||
@@ -818,6 +1020,45 @@ pub fn process_pdf_async(buffer: Buffer, pages: Option<Vec<u32>>) -> AsyncTask<P
|
||||
})
|
||||
}
|
||||
|
||||
pub struct ProcessPdfWithOcrTask {
|
||||
bytes: Vec<u8>,
|
||||
options: Option<OcrOptions>,
|
||||
}
|
||||
|
||||
impl Task for ProcessPdfWithOcrTask {
|
||||
type Output = OcrPdfResult;
|
||||
type JsValue = OcrPdfResult;
|
||||
|
||||
fn compute(&mut self) -> Result<Self::Output> {
|
||||
let bytes = std::mem::take(&mut self.bytes);
|
||||
let options = self.options.take();
|
||||
catch_panic(
|
||||
"process_pdf_with_ocr",
|
||||
panic::AssertUnwindSafe(move || process_pdf_with_ocr_impl(&bytes, options)),
|
||||
)
|
||||
}
|
||||
|
||||
fn resolve(&mut self, _env: Env, output: Self::Output) -> Result<Self::JsValue> {
|
||||
Ok(output)
|
||||
}
|
||||
}
|
||||
|
||||
/// Process a PDF with selective OCR on the libuv thread pool.
|
||||
///
|
||||
/// OCR defaults to Auto, which only loads PDFium, ONNX Runtime, and the OCR
|
||||
/// model if native extraction routes at least one page. The input buffer is
|
||||
/// copied before the promise is returned and is safe to reuse immediately.
|
||||
#[napi(ts_return_type = "Promise<OcrPdfResult>")]
|
||||
pub fn process_pdf_with_ocr(
|
||||
buffer: Buffer,
|
||||
options: Option<OcrOptions>,
|
||||
) -> AsyncTask<ProcessPdfWithOcrTask> {
|
||||
AsyncTask::new(ProcessPdfWithOcrTask {
|
||||
bytes: buffer.to_vec(),
|
||||
options,
|
||||
})
|
||||
}
|
||||
|
||||
pub struct ClassifyPdfTask {
|
||||
bytes: Vec<u8>,
|
||||
}
|
||||
|
||||
@@ -3,6 +3,7 @@ import { strict as assert } from 'assert';
|
||||
import {
|
||||
processPdf,
|
||||
processPdfAsync,
|
||||
processPdfWithOcr,
|
||||
detectPdf,
|
||||
classifyPdf,
|
||||
classifyPdfAsync,
|
||||
@@ -220,6 +221,37 @@ const fromMutated = await inFlight;
|
||||
assert.equal(fromMutated.markdown, result.markdown);
|
||||
console.log(' processPdfAsync input copied at call time: OK');
|
||||
|
||||
// --- Selective OCR ---
|
||||
console.log('Testing processPdfWithOcr...');
|
||||
|
||||
// Off exercises the complete result/provenance contract without loading
|
||||
// external PDFium, ONNX Runtime, or model artifacts.
|
||||
const ocrOff = await processPdfWithOcr(fixture, { mode: 'Off' });
|
||||
assert.equal(ocrOff.pageCount, 3);
|
||||
assert.equal(ocrOff.pages.length, 3);
|
||||
assert.deepEqual(ocrOff.pagesRoutedToOcr, []);
|
||||
assert.ok(ocrOff.pages.every(page => page.provenance.source === 'Native'));
|
||||
assert.ok(ocrOff.pages.every(page => page.provenance.ocrModel === undefined));
|
||||
assert.ok(ocrOff.markdown.length > 0);
|
||||
|
||||
// Auto must preserve the lightweight path for clean text PDFs.
|
||||
const ocrAuto = await processPdfWithOcr(fixture);
|
||||
assert.deepEqual(ocrAuto.pagesRoutedToOcr, []);
|
||||
assert.equal(ocrAuto.renderTimeMs, 0);
|
||||
assert.equal(ocrAuto.ocrTimeMs, 0);
|
||||
|
||||
const ocrSelected = await processPdfWithOcr(fixture, {
|
||||
mode: 'Off',
|
||||
pageNumbers: [2],
|
||||
});
|
||||
assert.deepEqual(ocrSelected.pages.map(page => page.pageNumber), [2]);
|
||||
|
||||
await assert.rejects(
|
||||
processPdfWithOcr(fixture, { mode: 'Off', pageNumbers: [0] }),
|
||||
/page 0/,
|
||||
);
|
||||
console.log(' processPdfWithOcr: OK');
|
||||
|
||||
// concurrent async calls all settle
|
||||
const [c1, c2, c3] = await Promise.all([
|
||||
processPdfAsync(fixture),
|
||||
|
||||
Reference in New Issue
Block a user