Compare commits
3
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
92e2d6ec8e | ||
|
|
74a633dfd0 | ||
|
|
c1957bf750 |
@@ -83,6 +83,22 @@ for (const region of result[0].regions) {
|
|||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
|
### Async variants
|
||||||
|
|
||||||
|
`processPdf`, `classifyPdf`, and `extractPagesMarkdown` are synchronous and parse on the calling thread — in Node, that's the event loop. For a one-off call in a script that's fine, but in a server a large document can hold the loop for tens to hundreds of milliseconds.
|
||||||
|
|
||||||
|
`processPdfAsync`, `classifyPdfAsync`, and `extractPagesMarkdownAsync` take the same arguments and produce the same results, but run the parse on the libuv thread pool and return a promise, keeping the event loop free. The input buffer is copied before the call returns, so it's safe to reuse or mutate immediately:
|
||||||
|
|
||||||
|
```typescript
|
||||||
|
import { classifyPdfAsync, extractPagesMarkdownAsync } from '@firecrawl/pdf-inspector'
|
||||||
|
|
||||||
|
const classification = await classifyPdfAsync(pdf)
|
||||||
|
if (classification.pdfType === 'TextBased') {
|
||||||
|
const { pages } = await extractPagesMarkdownAsync(pdf)
|
||||||
|
// ...
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
## Types
|
## Types
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
|
|||||||
+182
-41
@@ -153,9 +153,7 @@ fn to_napi_result(r: pdf_inspector::PdfProcessResult) -> PdfResult {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
fn to_napi_page_ocr_reasons(
|
fn to_napi_page_ocr_reasons(reasons: Vec<pdf_inspector::PageOcrReasons>) -> Vec<PageOcrReasons> {
|
||||||
reasons: Vec<pdf_inspector::PageOcrReasons>,
|
|
||||||
) -> Vec<PageOcrReasons> {
|
|
||||||
reasons
|
reasons
|
||||||
.into_iter()
|
.into_iter()
|
||||||
.map(|reason| PageOcrReasons {
|
.map(|reason| PageOcrReasons {
|
||||||
@@ -202,6 +200,31 @@ where
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Shared implementations (single body behind sync and async entry points)
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
fn process_pdf_impl(bytes: &[u8], pages: Option<Vec<u32>>) -> Result<PdfResult> {
|
||||||
|
let mut opts = pdf_inspector::PdfOptions::new();
|
||||||
|
if let Some(p) = pages {
|
||||||
|
opts = opts.pages(p);
|
||||||
|
}
|
||||||
|
let result = pdf_inspector::process_pdf_mem_with_options(bytes, opts)
|
||||||
|
.map_err(|e| to_napi_err(e, "process_pdf"))?;
|
||||||
|
Ok(to_napi_result(result))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn classify_pdf_impl(bytes: &[u8]) -> Result<PdfClassification> {
|
||||||
|
let result =
|
||||||
|
pdf_inspector::classify_pdf_mem(bytes).map_err(|e| to_napi_err(e, "classify_pdf"))?;
|
||||||
|
Ok(PdfClassification {
|
||||||
|
pdf_type: convert_pdf_type(result.pdf_type),
|
||||||
|
page_count: result.page_count,
|
||||||
|
pages_needing_ocr: result.pages_needing_ocr,
|
||||||
|
confidence: result.confidence as f64,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// Public NAPI API
|
// Public NAPI API
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
@@ -210,15 +233,7 @@ where
|
|||||||
#[napi]
|
#[napi]
|
||||||
pub fn process_pdf(buffer: Buffer, pages: Option<Vec<u32>>) -> Result<PdfResult> {
|
pub fn process_pdf(buffer: Buffer, pages: Option<Vec<u32>>) -> Result<PdfResult> {
|
||||||
let bytes: Vec<u8> = buffer.to_vec();
|
let bytes: Vec<u8> = buffer.to_vec();
|
||||||
catch_panic("process_pdf", move || {
|
catch_panic("process_pdf", move || process_pdf_impl(&bytes, pages))
|
||||||
let mut opts = pdf_inspector::PdfOptions::new();
|
|
||||||
if let Some(p) = pages {
|
|
||||||
opts = opts.pages(p);
|
|
||||||
}
|
|
||||||
let result = pdf_inspector::process_pdf_mem_with_options(&bytes, opts)
|
|
||||||
.map_err(|e| to_napi_err(e, "process_pdf"))?;
|
|
||||||
Ok(to_napi_result(result))
|
|
||||||
})
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Fast detection only — no text extraction or markdown.
|
/// Fast detection only — no text extraction or markdown.
|
||||||
@@ -238,16 +253,7 @@ pub fn detect_pdf(buffer: Buffer) -> Result<PdfResult> {
|
|||||||
#[napi]
|
#[napi]
|
||||||
pub fn classify_pdf(buffer: Buffer) -> Result<PdfClassification> {
|
pub fn classify_pdf(buffer: Buffer) -> Result<PdfClassification> {
|
||||||
let bytes: Vec<u8> = buffer.to_vec();
|
let bytes: Vec<u8> = buffer.to_vec();
|
||||||
catch_panic("classify_pdf", move || {
|
catch_panic("classify_pdf", move || classify_pdf_impl(&bytes))
|
||||||
let result =
|
|
||||||
pdf_inspector::classify_pdf_mem(&bytes).map_err(|e| to_napi_err(e, "classify_pdf"))?;
|
|
||||||
Ok(PdfClassification {
|
|
||||||
pdf_type: convert_pdf_type(result.pdf_type),
|
|
||||||
page_count: result.page_count,
|
|
||||||
pages_needing_ocr: result.pages_needing_ocr,
|
|
||||||
confidence: result.confidence as f64,
|
|
||||||
})
|
|
||||||
})
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Extract plain text from a PDF Buffer.
|
/// Extract plain text from a PDF Buffer.
|
||||||
@@ -633,25 +639,32 @@ pub fn extract_pages_markdown(
|
|||||||
) -> Result<PagesExtractionResult> {
|
) -> Result<PagesExtractionResult> {
|
||||||
let bytes: Vec<u8> = buffer.to_vec();
|
let bytes: Vec<u8> = buffer.to_vec();
|
||||||
catch_panic("extract_pages_markdown", move || {
|
catch_panic("extract_pages_markdown", move || {
|
||||||
let result = pdf_inspector::extract_pages_markdown_mem(&bytes, pages.as_deref())
|
extract_pages_markdown_impl(&bytes, pages.as_deref())
|
||||||
.map_err(|e| to_napi_err(e, "extract_pages_markdown"))?;
|
})
|
||||||
Ok(PagesExtractionResult {
|
}
|
||||||
pages: result
|
|
||||||
.pages
|
fn extract_pages_markdown_impl(
|
||||||
.into_iter()
|
bytes: &[u8],
|
||||||
.map(|r| PageMarkdownResult {
|
pages: Option<&[u32]>,
|
||||||
page: r.page,
|
) -> Result<PagesExtractionResult> {
|
||||||
markdown: r.markdown,
|
let result = pdf_inspector::extract_pages_markdown_mem(bytes, pages)
|
||||||
needs_ocr: r.needs_ocr,
|
.map_err(|e| to_napi_err(e, "extract_pages_markdown"))?;
|
||||||
ocr_reason: r.ocr_reason,
|
Ok(PagesExtractionResult {
|
||||||
})
|
pages: result
|
||||||
.collect(),
|
.pages
|
||||||
pages_with_tables: result.pages_with_tables,
|
.into_iter()
|
||||||
pages_with_columns: result.pages_with_columns,
|
.map(|r| PageMarkdownResult {
|
||||||
pages_needing_ocr: result.pages_needing_ocr,
|
page: r.page,
|
||||||
ocr_reasons_by_page: to_napi_page_ocr_reasons(result.ocr_reasons_by_page),
|
markdown: r.markdown,
|
||||||
is_complex: result.is_complex,
|
needs_ocr: r.needs_ocr,
|
||||||
})
|
ocr_reason: r.ocr_reason,
|
||||||
|
})
|
||||||
|
.collect(),
|
||||||
|
pages_with_tables: result.pages_with_tables,
|
||||||
|
pages_with_columns: result.pages_with_columns,
|
||||||
|
pages_needing_ocr: result.pages_needing_ocr,
|
||||||
|
ocr_reasons_by_page: to_napi_page_ocr_reasons(result.ocr_reasons_by_page),
|
||||||
|
is_complex: result.is_complex,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -692,3 +705,131 @@ fn to_page_region_texts(results: Vec<pdf_inspector::PageRegionResult>) -> Vec<Pa
|
|||||||
})
|
})
|
||||||
.collect()
|
.collect()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Async variants (libuv thread pool via AsyncTask)
|
||||||
|
//
|
||||||
|
// The synchronous exports above parse on the calling thread, which in Node is
|
||||||
|
// the event loop. These `*Async` variants run the same shared implementations
|
||||||
|
// on the libuv thread pool and hand JavaScript a promise, so servers under
|
||||||
|
// concurrent load keep answering requests while a document parses. The sync
|
||||||
|
// exports keep their names, signatures, and behaviour.
|
||||||
|
//
|
||||||
|
// Each factory copies the input Buffer to an owned `Vec<u8>` on the calling
|
||||||
|
// (JS) thread — deliberately. JS execution is single-threaded, so no JS code
|
||||||
|
// can mutate the buffer while the synchronous part of the call copies it.
|
||||||
|
// Holding the napi `Buffer` and reading it from the worker instead would be
|
||||||
|
// zero-copy, but a caller mutating the buffer before the promise settles
|
||||||
|
// would then race the worker's reads — undefined behavior, not a recoverable
|
||||||
|
// error (a known napi-rs soundness hazard with cross-thread Buffer access).
|
||||||
|
// The copy is a one-time memcpy, negligible next to the parse it unblocks.
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
pub struct ProcessPdfTask {
|
||||||
|
bytes: Vec<u8>,
|
||||||
|
pages: Option<Vec<u32>>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Task for ProcessPdfTask {
|
||||||
|
type Output = PdfResult;
|
||||||
|
type JsValue = PdfResult;
|
||||||
|
|
||||||
|
fn compute(&mut self) -> Result<Self::Output> {
|
||||||
|
let bytes = std::mem::take(&mut self.bytes);
|
||||||
|
let pages = self.pages.take();
|
||||||
|
// AssertUnwindSafe: `bytes`/`pages` are moved into the closure and
|
||||||
|
// dropped on unwind — no shared state can be observed broken.
|
||||||
|
catch_panic(
|
||||||
|
"process_pdf",
|
||||||
|
panic::AssertUnwindSafe(move || process_pdf_impl(&bytes, pages)),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn resolve(&mut self, _env: Env, output: Self::Output) -> Result<Self::JsValue> {
|
||||||
|
Ok(output)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Async variant of [`processPdf`]: same result, but the parse runs on the
|
||||||
|
/// libuv thread pool instead of the event loop and the call returns a
|
||||||
|
/// promise. The buffer is copied before the call returns, so it may be
|
||||||
|
/// reused or mutated immediately.
|
||||||
|
// ts_return_type is required: napi-rs emits `Promise<unknown>` for
|
||||||
|
// `AsyncTask<T>` returns without it.
|
||||||
|
#[napi(ts_return_type = "Promise<PdfResult>")]
|
||||||
|
pub fn process_pdf_async(buffer: Buffer, pages: Option<Vec<u32>>) -> AsyncTask<ProcessPdfTask> {
|
||||||
|
AsyncTask::new(ProcessPdfTask {
|
||||||
|
bytes: buffer.to_vec(),
|
||||||
|
pages,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
pub struct ClassifyPdfTask {
|
||||||
|
bytes: Vec<u8>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Task for ClassifyPdfTask {
|
||||||
|
type Output = PdfClassification;
|
||||||
|
type JsValue = PdfClassification;
|
||||||
|
|
||||||
|
fn compute(&mut self) -> Result<Self::Output> {
|
||||||
|
let bytes = std::mem::take(&mut self.bytes);
|
||||||
|
catch_panic(
|
||||||
|
"classify_pdf",
|
||||||
|
panic::AssertUnwindSafe(move || classify_pdf_impl(&bytes)),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn resolve(&mut self, _env: Env, output: Self::Output) -> Result<Self::JsValue> {
|
||||||
|
Ok(output)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Async variant of [`classifyPdf`]: same result, but the classification runs
|
||||||
|
/// on the libuv thread pool instead of the event loop and the call returns a
|
||||||
|
/// promise. The buffer is copied before the call returns, so it may be
|
||||||
|
/// reused or mutated immediately.
|
||||||
|
#[napi(ts_return_type = "Promise<PdfClassification>")]
|
||||||
|
pub fn classify_pdf_async(buffer: Buffer) -> AsyncTask<ClassifyPdfTask> {
|
||||||
|
AsyncTask::new(ClassifyPdfTask {
|
||||||
|
bytes: buffer.to_vec(),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
pub struct ExtractPagesMarkdownTask {
|
||||||
|
bytes: Vec<u8>,
|
||||||
|
pages: Option<Vec<u32>>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Task for ExtractPagesMarkdownTask {
|
||||||
|
type Output = PagesExtractionResult;
|
||||||
|
type JsValue = PagesExtractionResult;
|
||||||
|
|
||||||
|
fn compute(&mut self) -> Result<Self::Output> {
|
||||||
|
let bytes = std::mem::take(&mut self.bytes);
|
||||||
|
let pages = self.pages.take();
|
||||||
|
catch_panic(
|
||||||
|
"extract_pages_markdown",
|
||||||
|
panic::AssertUnwindSafe(move || extract_pages_markdown_impl(&bytes, pages.as_deref())),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn resolve(&mut self, _env: Env, output: Self::Output) -> Result<Self::JsValue> {
|
||||||
|
Ok(output)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Async variant of [`extractPagesMarkdown`]: same result, but the extraction
|
||||||
|
/// runs on the libuv thread pool instead of the event loop and the call
|
||||||
|
/// returns a promise. The buffer is copied before the call returns, so it
|
||||||
|
/// may be reused or mutated immediately.
|
||||||
|
#[napi(ts_return_type = "Promise<PagesExtractionResult>")]
|
||||||
|
pub fn extract_pages_markdown_async(
|
||||||
|
buffer: Buffer,
|
||||||
|
pages: Option<Vec<u32>>,
|
||||||
|
) -> AsyncTask<ExtractPagesMarkdownTask> {
|
||||||
|
AsyncTask::new(ExtractPagesMarkdownTask {
|
||||||
|
bytes: buffer.to_vec(),
|
||||||
|
pages,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|||||||
@@ -2,13 +2,16 @@ import { readFileSync } from 'fs';
|
|||||||
import { strict as assert } from 'assert';
|
import { strict as assert } from 'assert';
|
||||||
import {
|
import {
|
||||||
processPdf,
|
processPdf,
|
||||||
|
processPdfAsync,
|
||||||
detectPdf,
|
detectPdf,
|
||||||
classifyPdf,
|
classifyPdf,
|
||||||
|
classifyPdfAsync,
|
||||||
extractText,
|
extractText,
|
||||||
extractTextWithPositions,
|
extractTextWithPositions,
|
||||||
extractTextInRegions,
|
extractTextInRegions,
|
||||||
detectVectorGridInRegion,
|
detectVectorGridInRegion,
|
||||||
extractPagesMarkdown,
|
extractPagesMarkdown,
|
||||||
|
extractPagesMarkdownAsync,
|
||||||
} from './index.js';
|
} from './index.js';
|
||||||
|
|
||||||
const fixture = readFileSync('../tests/fixtures/thermo-freon12.pdf');
|
const fixture = readFileSync('../tests/fixtures/thermo-freon12.pdf');
|
||||||
@@ -124,10 +127,75 @@ assert.equal(picked.pages[0].page, 2);
|
|||||||
assert.equal(picked.pages[1].page, 0);
|
assert.equal(picked.pages[1].page, 0);
|
||||||
console.log(' extractPagesMarkdown with pages: OK');
|
console.log(' extractPagesMarkdown with pages: OK');
|
||||||
|
|
||||||
|
// --- Async variants ---
|
||||||
|
console.log('Testing async variants...');
|
||||||
|
|
||||||
|
// processPdfAsync returns a promise and matches the sync result
|
||||||
|
const asyncResultPromise = processPdfAsync(fixture);
|
||||||
|
assert.ok(asyncResultPromise instanceof Promise);
|
||||||
|
const asyncResult = await asyncResultPromise;
|
||||||
|
assert.equal(asyncResult.pdfType, result.pdfType);
|
||||||
|
assert.equal(asyncResult.pageCount, result.pageCount);
|
||||||
|
assert.equal(asyncResult.markdown, result.markdown);
|
||||||
|
console.log(' processPdfAsync: OK');
|
||||||
|
|
||||||
|
// processPdfAsync with pages
|
||||||
|
const asyncResult2 = await processPdfAsync(fixture, [1]);
|
||||||
|
assert.equal(asyncResult2.markdown, result2.markdown);
|
||||||
|
console.log(' processPdfAsync with pages: OK');
|
||||||
|
|
||||||
|
// classifyPdfAsync matches the sync result
|
||||||
|
const asyncClassified = await classifyPdfAsync(fixture);
|
||||||
|
assert.equal(asyncClassified.pdfType, classified.pdfType);
|
||||||
|
assert.equal(asyncClassified.pageCount, classified.pageCount);
|
||||||
|
assert.equal(asyncClassified.confidence, classified.confidence);
|
||||||
|
assert.deepEqual(asyncClassified.pagesNeedingOcr, classified.pagesNeedingOcr);
|
||||||
|
console.log(' classifyPdfAsync: OK');
|
||||||
|
|
||||||
|
// extractPagesMarkdownAsync matches the sync result
|
||||||
|
const asyncAllPages = await extractPagesMarkdownAsync(fixture);
|
||||||
|
assert.equal(asyncAllPages.pages.length, allPages.pages.length);
|
||||||
|
assert.deepEqual(
|
||||||
|
asyncAllPages.pages.map(p => p.markdown),
|
||||||
|
allPages.pages.map(p => p.markdown),
|
||||||
|
);
|
||||||
|
assert.equal(asyncAllPages.isComplex, allPages.isComplex);
|
||||||
|
console.log(' extractPagesMarkdownAsync: OK');
|
||||||
|
|
||||||
|
// selected pages preserve caller order
|
||||||
|
const asyncPicked = await extractPagesMarkdownAsync(fixture, [2, 0]);
|
||||||
|
assert.equal(asyncPicked.pages.length, 2);
|
||||||
|
assert.equal(asyncPicked.pages[0].page, 2);
|
||||||
|
assert.equal(asyncPicked.pages[1].page, 0);
|
||||||
|
console.log(' extractPagesMarkdownAsync with pages: OK');
|
||||||
|
|
||||||
|
// input buffer is copied at call time: mutating it immediately after the
|
||||||
|
// call must not affect the in-flight parse
|
||||||
|
const scratch = Buffer.from(fixture);
|
||||||
|
const inFlight = processPdfAsync(scratch);
|
||||||
|
scratch.fill(0);
|
||||||
|
const fromMutated = await inFlight;
|
||||||
|
assert.equal(fromMutated.markdown, result.markdown);
|
||||||
|
console.log(' processPdfAsync input copied at call time: OK');
|
||||||
|
|
||||||
|
// concurrent async calls all settle
|
||||||
|
const [c1, c2, c3] = await Promise.all([
|
||||||
|
processPdfAsync(fixture),
|
||||||
|
classifyPdfAsync(fixture),
|
||||||
|
extractPagesMarkdownAsync(fixture),
|
||||||
|
]);
|
||||||
|
assert.equal(c1.pdfType, 'TextBased');
|
||||||
|
assert.equal(c2.pdfType, 'TextBased');
|
||||||
|
assert.equal(c3.pages.length, 3);
|
||||||
|
console.log(' concurrent async calls: OK');
|
||||||
|
|
||||||
// --- Error handling ---
|
// --- Error handling ---
|
||||||
console.log('Testing error handling...');
|
console.log('Testing error handling...');
|
||||||
assert.throws(() => processPdf(Buffer.from('not a pdf')), /process_pdf/);
|
assert.throws(() => processPdf(Buffer.from('not a pdf')), /process_pdf/);
|
||||||
assert.throws(() => classifyPdf(Buffer.from('')), /classify_pdf/);
|
assert.throws(() => classifyPdf(Buffer.from('')), /classify_pdf/);
|
||||||
|
await assert.rejects(processPdfAsync(Buffer.from('not a pdf')), /process_pdf/);
|
||||||
|
await assert.rejects(classifyPdfAsync(Buffer.from('')), /classify_pdf/);
|
||||||
|
await assert.rejects(extractPagesMarkdownAsync(Buffer.from('')), /extract_pages_markdown/);
|
||||||
console.log(' error handling: OK');
|
console.log(' error handling: OK');
|
||||||
|
|
||||||
console.log('\nAll NAPI tests passed!');
|
console.log('\nAll NAPI tests passed!');
|
||||||
|
|||||||
Reference in New Issue
Block a user