unify NAPI and Python binding APIs for consistent surface
Both bindings now expose the same 6 function families: process, detect, classify, extractText, extractTextWithPositions, and extractTextInRegions. Bumps PyO3 from 0.22 to 0.25 for Python 3.14 support. Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.6
parent
5f829f5258
commit
506b2a0c70
@@ -17,6 +17,15 @@ class PdfResult:
|
||||
pages_with_columns: list[int]
|
||||
has_encoding_issues: bool
|
||||
|
||||
class PdfClassification:
|
||||
"""Lightweight PDF classification result."""
|
||||
pdf_type: str
|
||||
"""'text_based', 'scanned', 'image_based', or 'mixed'."""
|
||||
page_count: int
|
||||
pages_needing_ocr: list[int]
|
||||
"""0-indexed page numbers that need OCR."""
|
||||
confidence: float
|
||||
|
||||
class TextItem:
|
||||
"""A positioned text item extracted from a PDF."""
|
||||
text: str
|
||||
@@ -31,6 +40,18 @@ class TextItem:
|
||||
is_italic: bool
|
||||
item_type: str
|
||||
|
||||
class RegionText:
|
||||
"""Extracted text for a single region."""
|
||||
text: str
|
||||
needs_ocr: bool
|
||||
"""True when the text should not be trusted."""
|
||||
|
||||
class PageRegionTexts:
|
||||
"""Extracted text for one page's regions."""
|
||||
page: int
|
||||
"""0-indexed page number."""
|
||||
regions: list[RegionText]
|
||||
|
||||
def process_pdf(path: str, pages: Optional[list[int]] = None) -> PdfResult:
|
||||
"""Process a PDF: detect type, extract text, convert to Markdown."""
|
||||
...
|
||||
@@ -47,10 +68,50 @@ def detect_pdf_bytes(data: bytes) -> PdfResult:
|
||||
"""Fast detection from bytes."""
|
||||
...
|
||||
|
||||
def classify_pdf(path: str) -> PdfClassification:
|
||||
"""Lightweight classification — type, page count, and OCR pages (0-indexed)."""
|
||||
...
|
||||
|
||||
def classify_pdf_bytes(data: bytes) -> PdfClassification:
|
||||
"""Lightweight classification from bytes."""
|
||||
...
|
||||
|
||||
def extract_text(path: str) -> str:
|
||||
"""Extract plain text from a PDF."""
|
||||
...
|
||||
|
||||
def extract_text_bytes(data: bytes) -> str:
|
||||
"""Extract plain text from PDF bytes."""
|
||||
...
|
||||
|
||||
def extract_text_with_positions(path: str, pages: Optional[list[int]] = None) -> list[TextItem]:
|
||||
"""Extract text with position information."""
|
||||
...
|
||||
|
||||
def extract_text_with_positions_bytes(data: bytes, pages: Optional[list[int]] = None) -> list[TextItem]:
|
||||
"""Extract text with position information from bytes."""
|
||||
...
|
||||
|
||||
def extract_text_in_regions(
|
||||
path: str,
|
||||
page_regions: list[tuple[int, list[list[float]]]],
|
||||
) -> list[PageRegionTexts]:
|
||||
"""Extract text within bounding-box regions from a PDF file.
|
||||
|
||||
Args:
|
||||
path: Path to the PDF file.
|
||||
page_regions: List of (page_0indexed, [[x1, y1, x2, y2], ...]) tuples.
|
||||
"""
|
||||
...
|
||||
|
||||
def extract_text_in_regions_bytes(
|
||||
data: bytes,
|
||||
page_regions: list[tuple[int, list[list[float]]]],
|
||||
) -> list[PageRegionTexts]:
|
||||
"""Extract text within bounding-box regions from PDF bytes.
|
||||
|
||||
Args:
|
||||
data: PDF file contents as bytes.
|
||||
page_regions: List of (page_0indexed, [[x1, y1, x2, y2], ...]) tuples.
|
||||
"""
|
||||
...
|
||||
|
||||
Reference in New Issue
Block a user