feat: expose per-page markdown extraction to Python and Node (#53)
* feat: expose per-page markdown extraction to Python and Node (#49) Implements the feature requested in issue #49: a list-of-pages markdown output from the Python API. Matching the existing project pattern, the feature lives in the Rust core and is surfaced through every binding. - Rust core: `extract_pages_markdown` (path) and `extract_pages_markdown_mem` (bytes) now take `Option<&[u32]>` — `None` returns every page in document order; a slice restricts and preserves caller order. - Python: new `extract_pages_markdown(path, pages=None)` and `extract_pages_markdown_bytes(data, pages=None)` functions plus `PageMarkdown` / `PagesExtractionResult` classes; stub file updated. - Node: `extractPagesMarkdown(buffer, pages?)` — `pages` is now optional. - Tests: 2 new Rust integration tests, 9 new Python tests, 2 new Node assertions. All 372 unit + 107 integration + 53 Python tests pass. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com> * chore: bump version from 1.3.0 to 1.4.0 Minor bump for the new per-page markdown extraction API exposed through the Python and Node bindings. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com> --------- Co-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.7
parent
b4c34ba4b7
commit
4b5ae91f54
@@ -52,6 +52,28 @@ class PageRegionTexts:
|
||||
"""0-indexed page number."""
|
||||
regions: list[RegionText]
|
||||
|
||||
class PageMarkdown:
|
||||
"""Per-page markdown extraction result."""
|
||||
page: int
|
||||
"""0-indexed page number."""
|
||||
markdown: str
|
||||
"""Formatted markdown for this page (empty string when needs_ocr is True)."""
|
||||
needs_ocr: bool
|
||||
"""True when text on this page is unreliable and OCR should be used instead."""
|
||||
|
||||
class PagesExtractionResult:
|
||||
"""Per-page markdown output with document-wide layout classification."""
|
||||
pages: list[PageMarkdown]
|
||||
"""Per-page markdown results, in the order requested."""
|
||||
pages_with_tables: list[int]
|
||||
"""1-indexed pages where tables were detected."""
|
||||
pages_with_columns: list[int]
|
||||
"""1-indexed pages where multi-column layout was detected."""
|
||||
pages_needing_ocr: list[int]
|
||||
"""1-indexed pages that need OCR."""
|
||||
is_complex: bool
|
||||
"""True if any page has tables or multi-column layout."""
|
||||
|
||||
def process_pdf(path: str, pages: Optional[list[int]] = None) -> PdfResult:
|
||||
"""Process a PDF: detect type, extract text, convert to Markdown."""
|
||||
...
|
||||
@@ -115,3 +137,31 @@ def extract_text_in_regions_bytes(
|
||||
page_regions: List of (page_0indexed, [[x1, y1, x2, y2], ...]) tuples.
|
||||
"""
|
||||
...
|
||||
|
||||
def extract_pages_markdown(
|
||||
path: str,
|
||||
pages: Optional[list[int]] = None,
|
||||
) -> PagesExtractionResult:
|
||||
"""Extract formatted markdown for pages of a PDF, with layout classification.
|
||||
|
||||
Args:
|
||||
path: Path to the PDF file.
|
||||
pages: Optional list of 0-indexed pages. When ``None`` (default), every
|
||||
page is returned in document order. Otherwise, output matches the
|
||||
caller-supplied order.
|
||||
|
||||
Returns:
|
||||
PagesExtractionResult with per-page markdown and document-wide layout
|
||||
classification (tables, columns, OCR needs).
|
||||
"""
|
||||
...
|
||||
|
||||
def extract_pages_markdown_bytes(
|
||||
data: bytes,
|
||||
pages: Optional[list[int]] = None,
|
||||
) -> PagesExtractionResult:
|
||||
"""Extract formatted markdown for pages of a PDF from bytes.
|
||||
|
||||
See :func:`extract_pages_markdown` for details.
|
||||
"""
|
||||
...
|
||||
|
||||
Reference in New Issue
Block a user