From 9868c02f6e184c5213e184e6e18cb7fd92637394 Mon Sep 17 00:00:00 2001 From: Abimael Martell Date: Thu, 2 Apr 2026 12:18:34 -0700 Subject: [PATCH] add npm package description, keywords, and README Co-Authored-By: Claude Opus 4.6 (1M context) --- napi/README.md | 99 +++++++++++++++++++++++++++++++++++++++++++++++ napi/package.json | 18 ++++++++- 2 files changed, 115 insertions(+), 2 deletions(-) create mode 100644 napi/README.md diff --git a/napi/README.md b/napi/README.md new file mode 100644 index 0000000..e1c987a --- /dev/null +++ b/napi/README.md @@ -0,0 +1,99 @@ +# firecrawl-pdf-inspector + +Fast PDF classification and region-based text extraction for Node.js/Bun. Native Rust performance via [napi-rs](https://napi.rs). + +Built by [Firecrawl](https://firecrawl.dev) for hybrid OCR pipelines — extract text from PDF structure where possible, fall back to OCR only when needed. + +## Install + +```bash +npm install firecrawl-pdf-inspector +# or +bun add firecrawl-pdf-inspector +``` + +Prebuilt binaries included for **linux-x64** and **macOS ARM64**. No Rust toolchain needed. + +## API + +### `classifyPdf(buffer: Buffer): PdfClassification` + +Classify a PDF as TextBased, Scanned, Mixed, or ImageBased (~10-50ms). Returns which pages need OCR. + +```typescript +import { classifyPdf } from 'firecrawl-pdf-inspector' +import { readFileSync } from 'fs' + +const pdf = readFileSync('document.pdf') +const result = classifyPdf(pdf) + +console.log(result.pdfType) // "TextBased" | "Scanned" | "Mixed" | "ImageBased" +console.log(result.pageCount) // 42 +console.log(result.pagesNeedingOcr) // [5, 12, 15] (0-indexed) +console.log(result.confidence) // 0.875 +``` + +### `extractTextInRegions(buffer: Buffer, pageRegions: PageRegions[]): PageRegionTexts[]` + +Extract text within bounding-box regions from a PDF. Designed for hybrid OCR pipelines where a layout model detects regions in rendered page images, and this function extracts text from the PDF structure for text-based pages — skipping GPU OCR. + +Each region result includes a `needsOcr` flag that signals unreliable extraction (empty text, GID-encoded fonts, garbage text, encoding issues). + +```typescript +import { extractTextInRegions } from 'firecrawl-pdf-inspector' + +const result = extractTextInRegions(pdf, [ + { + page: 0, // 0-indexed + regions: [ + [0, 0, 300, 400], // [x1, y1, x2, y2] in PDF points, top-left origin + [300, 0, 612, 400], + ] + } +]) + +for (const region of result[0].regions) { + if (region.needsOcr) { + // Unreliable text — send this region to OCR instead + } else { + console.log(region.text) // Extracted text in reading order + } +} +``` + +## Types + +```typescript +interface PdfClassification { + pdfType: string // "TextBased" | "Scanned" | "Mixed" | "ImageBased" + pageCount: number + pagesNeedingOcr: number[] // 0-indexed page numbers + confidence: number // 0.0 - 1.0 +} + +interface PageRegions { + page: number // 0-indexed + regions: number[][] // [[x1, y1, x2, y2], ...] in PDF points, top-left origin +} + +interface PageRegionTexts { + page: number + regions: RegionText[] +} + +interface RegionText { + text: string + needsOcr: boolean // true when text is unreliable +} +``` + +## Platforms + +| Platform | Architecture | Supported | +|----------|-------------|-----------| +| Linux | x64 | Yes | +| macOS | ARM64 | Yes | + +## License + +MIT diff --git a/napi/package.json b/napi/package.json index 1c58a8c..bededb2 100644 --- a/napi/package.json +++ b/napi/package.json @@ -1,18 +1,32 @@ { "name": "firecrawl-pdf-inspector", - "version": "0.2.2", + "version": "0.2.3", + "description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.", "main": "index.js", "types": "index.d.ts", "license": "MIT", + "keywords": [ + "pdf", + "pdf-extraction", + "pdf-parser", + "text-extraction", + "ocr", + "pdf-classification", + "napi", + "rust", + "firecrawl" + ], "files": [ "index.js", "index.d.ts", - "*.node" + "*.node", + "README.md" ], "repository": { "type": "git", "url": "https://github.com/firecrawl/pdf-inspector" }, + "homepage": "https://github.com/firecrawl/pdf-inspector", "publishConfig": { "access": "public" },