unify NAPI and Python binding APIs for consistent surface

Both bindings now expose the same 6 function families: process, detect,
classify, extractText, extractTextWithPositions, and extractTextInRegions.
Bumps PyO3 from 0.22 to 0.25 for Python 3.14 support.

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
Abimael Martell
2026-04-02 11:48:18 -07:00
co-authored by Claude Opus 4.6
parent 5f829f5258
commit 506b2a0c70
11 changed files with 935 additions and 199 deletions
+61
View File
@@ -17,6 +17,15 @@ class PdfResult:
pages_with_columns: list[int]
has_encoding_issues: bool
class PdfClassification:
"""Lightweight PDF classification result."""
pdf_type: str
"""'text_based', 'scanned', 'image_based', or 'mixed'."""
page_count: int
pages_needing_ocr: list[int]
"""0-indexed page numbers that need OCR."""
confidence: float
class TextItem:
"""A positioned text item extracted from a PDF."""
text: str
@@ -31,6 +40,18 @@ class TextItem:
is_italic: bool
item_type: str
class RegionText:
"""Extracted text for a single region."""
text: str
needs_ocr: bool
"""True when the text should not be trusted."""
class PageRegionTexts:
"""Extracted text for one page's regions."""
page: int
"""0-indexed page number."""
regions: list[RegionText]
def process_pdf(path: str, pages: Optional[list[int]] = None) -> PdfResult:
"""Process a PDF: detect type, extract text, convert to Markdown."""
...
@@ -47,10 +68,50 @@ def detect_pdf_bytes(data: bytes) -> PdfResult:
"""Fast detection from bytes."""
...
def classify_pdf(path: str) -> PdfClassification:
"""Lightweight classification — type, page count, and OCR pages (0-indexed)."""
...
def classify_pdf_bytes(data: bytes) -> PdfClassification:
"""Lightweight classification from bytes."""
...
def extract_text(path: str) -> str:
"""Extract plain text from a PDF."""
...
def extract_text_bytes(data: bytes) -> str:
"""Extract plain text from PDF bytes."""
...
def extract_text_with_positions(path: str, pages: Optional[list[int]] = None) -> list[TextItem]:
"""Extract text with position information."""
...
def extract_text_with_positions_bytes(data: bytes, pages: Optional[list[int]] = None) -> list[TextItem]:
"""Extract text with position information from bytes."""
...
def extract_text_in_regions(
path: str,
page_regions: list[tuple[int, list[list[float]]]],
) -> list[PageRegionTexts]:
"""Extract text within bounding-box regions from a PDF file.
Args:
path: Path to the PDF file.
page_regions: List of (page_0indexed, [[x1, y1, x2, y2], ...]) tuples.
"""
...
def extract_text_in_regions_bytes(
data: bytes,
page_regions: list[tuple[int, list[list[float]]]],
) -> list[PageRegionTexts]:
"""Extract text within bounding-box regions from PDF bytes.
Args:
data: PDF file contents as bytes.
page_regions: List of (page_0indexed, [[x1, y1, x2, y2], ...]) tuples.
"""
...