Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0b0ca64ffb |
@@ -80,6 +80,17 @@ for page in result.pages:
|
|||||||
|
|
||||||
# Restrict to specific 0-indexed pages (preserves caller order)
|
# Restrict to specific 0-indexed pages (preserves caller order)
|
||||||
result = pdf_inspector.extract_pages_markdown("document.pdf", pages=[0, 2])
|
result = pdf_inspector.extract_pages_markdown("document.pdf", pages=[0, 2])
|
||||||
|
|
||||||
|
# Structure-tree elements from tagged PDFs (empty list when untagged).
|
||||||
|
# Pages are 1-indexed to match TextItem.page, so (page, mcid) joins directly
|
||||||
|
# against extract_text_with_positions — e.g. to recover real heading levels:
|
||||||
|
elements = pdf_inspector.extract_structure_elements("tagged.pdf")
|
||||||
|
roles = {(e.page, e.mcid): e.role for e in elements}
|
||||||
|
headings = [
|
||||||
|
item.text
|
||||||
|
for item in pdf_inspector.extract_text_with_positions("tagged.pdf")
|
||||||
|
if item.mcid is not None and roles.get((item.page, item.mcid), "").startswith("H")
|
||||||
|
]
|
||||||
```
|
```
|
||||||
|
|
||||||
## API reference
|
## API reference
|
||||||
@@ -100,6 +111,8 @@ result = pdf_inspector.extract_pages_markdown("document.pdf", pages=[0, 2])
|
|||||||
| `extract_text_in_regions_bytes(data, page_regions)` | Region extraction from bytes |
|
| `extract_text_in_regions_bytes(data, page_regions)` | Region extraction from bytes |
|
||||||
| `extract_pages_markdown(path, pages=None)` | Per-page Markdown + layout metadata (all pages by default) |
|
| `extract_pages_markdown(path, pages=None)` | Per-page Markdown + layout metadata (all pages by default) |
|
||||||
| `extract_pages_markdown_bytes(data, pages=None)` | Per-page Markdown from bytes |
|
| `extract_pages_markdown_bytes(data, pages=None)` | Per-page Markdown from bytes |
|
||||||
|
| `extract_structure_elements(path, pages=None)` | Structure-tree elements from tagged PDFs (page, mcid, role) |
|
||||||
|
| `extract_structure_elements_bytes(data, pages=None)` | Structure-tree elements from bytes |
|
||||||
|
|
||||||
## Types
|
## Types
|
||||||
|
|
||||||
@@ -144,6 +157,12 @@ class TextItem: # extract_text_with_positions
|
|||||||
is_underline: bool
|
is_underline: bool
|
||||||
is_strikeout: bool
|
is_strikeout: bool
|
||||||
item_type: str
|
item_type: str
|
||||||
|
mcid: int | None # marked-content ID for tagged PDFs (None otherwise)
|
||||||
|
|
||||||
|
class StructureElement: # extract_structure_elements
|
||||||
|
page: int # 1-indexed (matches TextItem.page)
|
||||||
|
mcid: int
|
||||||
|
role: str # "H1".."H6", "P", "Table", ... (resolved via /RoleMap)
|
||||||
|
|
||||||
class RegionText: # extract_text_in_regions
|
class RegionText: # extract_text_in_regions
|
||||||
text: str
|
text: str
|
||||||
|
|||||||
+32
-1
@@ -138,6 +138,34 @@ for page in &result.pages {
|
|||||||
println!("Complex layout? {}", result.is_complex);
|
println!("Complex layout? {}", result.is_complex);
|
||||||
```
|
```
|
||||||
|
|
||||||
|
Extract structure-tree elements from tagged PDFs, and join them against
|
||||||
|
`extract_text_with_positions` to attach semantic roles (heading levels,
|
||||||
|
paragraphs, table cells) to extracted text:
|
||||||
|
|
||||||
|
```rust
|
||||||
|
use pdf_inspector::{extract_structure_elements, extract_text_with_positions};
|
||||||
|
use std::collections::HashMap;
|
||||||
|
|
||||||
|
// One entry per marked-content reference, sorted by (page, mcid); empty for
|
||||||
|
// untagged PDFs. Pages are 1-indexed to match `TextItem::page`, so the
|
||||||
|
// (page, mcid) pair is a direct join key.
|
||||||
|
let elements = extract_structure_elements("tagged.pdf", None)?;
|
||||||
|
let roles: HashMap<(u32, i64), &str> = elements
|
||||||
|
.iter()
|
||||||
|
.map(|e| ((e.page, e.mcid), e.role.as_str()))
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
for item in extract_text_with_positions("tagged.pdf")? {
|
||||||
|
if let Some(mcid) = item.mcid {
|
||||||
|
if let Some(role) = roles.get(&(item.page, mcid)) {
|
||||||
|
if role.starts_with('H') {
|
||||||
|
println!("{}: {}", role, item.text);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
## Processing modes
|
## Processing modes
|
||||||
|
|
||||||
| Mode | What it does | Returns |
|
| Mode | What it does | Returns |
|
||||||
@@ -163,6 +191,8 @@ println!("Complex layout? {}", result.is_complex);
|
|||||||
| `to_markdown_from_items_with_rects(items, options, rects)` | Markdown with rectangle-based table detection |
|
| `to_markdown_from_items_with_rects(items, options, rects)` | Markdown with rectangle-based table detection |
|
||||||
| `extract_pages_markdown(path, pages)` | Per-page Markdown + layout metadata (file) |
|
| `extract_pages_markdown(path, pages)` | Per-page Markdown + layout metadata (file) |
|
||||||
| `extract_pages_markdown_mem(bytes, pages)` | Per-page Markdown from bytes |
|
| `extract_pages_markdown_mem(bytes, pages)` | Per-page Markdown from bytes |
|
||||||
|
| `extract_structure_elements(path, pages)` | Structure-tree elements from tagged PDFs (page, mcid, role) |
|
||||||
|
| `extract_structure_elements_mem(bytes, pages)` | Structure-tree elements from bytes |
|
||||||
|
|
||||||
Low-level detection functions are also available via the `detector` module (`detect_pdf_type`, `detect_pdf_type_with_config`, etc.) for callers who need `PdfTypeResult` instead of `PdfProcessResult`.
|
Low-level detection functions are also available via the `detector` module (`detect_pdf_type`, `detect_pdf_type_with_config`, etc.) for callers who need `PdfTypeResult` instead of `PdfProcessResult`.
|
||||||
|
|
||||||
@@ -178,7 +208,8 @@ Low-level detection functions are also available via the `detector` module (`det
|
|||||||
| `DetectionConfig` | Configuration for detection: scan strategy, thresholds |
|
| `DetectionConfig` | Configuration for detection: scan strategy, thresholds |
|
||||||
| `ScanStrategy` | `EarlyExit`, `Full`, `Sample(n)`, `Pages(vec)` |
|
| `ScanStrategy` | `EarlyExit`, `Full`, `Sample(n)`, `Pages(vec)` |
|
||||||
| `LayoutComplexity` | Layout analysis: is_complex, pages_with_tables, pages_with_columns |
|
| `LayoutComplexity` | Layout analysis: is_complex, pages_with_tables, pages_with_columns |
|
||||||
| `TextItem` | Text with position, font info, and page number |
|
| `TextItem` | Text with position, font info, page number, and optional structure-tree `mcid` |
|
||||||
|
| `StructureElement` | Tagged-PDF structure reference: page (1-indexed), mcid, role (`"H1"`..`"H6"`, `"P"`, …) |
|
||||||
| `MarkdownOptions` | Configuration for Markdown formatting (page numbers, etc.) |
|
| `MarkdownOptions` | Configuration for Markdown formatting (page numbers, etc.) |
|
||||||
| `PageMarkdown` | Per-page result: page (0-indexed), markdown, needs_ocr |
|
| `PageMarkdown` | Per-page result: page (0-indexed), markdown, needs_ocr |
|
||||||
| `PagesExtractionResult` | Per-page output + 1-indexed pages_with_tables / pages_with_columns / pages_needing_ocr, is_complex |
|
| `PagesExtractionResult` | Per-page output + 1-indexed pages_with_tables / pages_with_columns / pages_needing_ocr, is_complex |
|
||||||
|
|||||||
Reference in New Issue
Block a user