Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
34991951c6 |
@@ -35,8 +35,3 @@ scripts/
|
||||
# Test output
|
||||
test_output/
|
||||
|
||||
# Python
|
||||
__pycache__/
|
||||
*.pyc
|
||||
.pytest_cache/
|
||||
|
||||
|
||||
@@ -61,8 +61,7 @@ src/
|
||||
|
||||
- **Unit tests**: inline `#[cfg(test)] mod tests` in each module with synthetic data.
|
||||
- **Integration tests**: `tests/integration_tests.rs` with fixture PDFs in `tests/fixtures/`.
|
||||
- **Regression suite**: sibling repo `pdf-evals` with 187+ snapshot PDFs. Run `cargo build --release` then `bench.py test` in that repo before committing.
|
||||
- **Semantic quality**: run `bench.py score` in `pdf-evals` for the semantic verdict (TEDS + MHS + reading order + char/word + list preservation, composited). Character-level diff alone misclassifies structural improvements (e.g., column-detection rewrites) as regressions — `score` is the tie-breaker. See `pdf-evals/CLAUDE.md` "Semantic scoring".
|
||||
- **Regression suite**: sibling repo `pdf-evals` with 179+ snapshot PDFs. Run `cargo build --release` then `bench.py test` in that repo before committing.
|
||||
|
||||
## Debugging
|
||||
|
||||
|
||||
@@ -42,14 +42,6 @@ text = pdf_inspector.extract_text("document.pdf")
|
||||
items = pdf_inspector.extract_text_with_positions("document.pdf")
|
||||
for item in items[:5]:
|
||||
print(f"'{item.text}' at ({item.x:.0f}, {item.y:.0f}) size={item.font_size}")
|
||||
|
||||
# Per-page markdown (one Markdown string per page, plus layout metadata)
|
||||
result = pdf_inspector.extract_pages_markdown("document.pdf")
|
||||
for page in result.pages:
|
||||
print(f"Page {page.page}: {len(page.markdown)} chars, needs_ocr={page.needs_ocr}")
|
||||
|
||||
# Restrict to specific 0-indexed pages (preserves caller order)
|
||||
result = pdf_inspector.extract_pages_markdown("document.pdf", pages=[0, 2])
|
||||
```
|
||||
|
||||
## API reference
|
||||
@@ -68,8 +60,6 @@ result = pdf_inspector.extract_pages_markdown("document.pdf", pages=[0, 2])
|
||||
| `extract_text_with_positions_bytes(data, pages=None)` | Text with positions from bytes |
|
||||
| `extract_text_in_regions(path, page_regions)` | Extract text in bounding-box regions |
|
||||
| `extract_text_in_regions_bytes(data, page_regions)` | Region extraction from bytes |
|
||||
| `extract_pages_markdown(path, pages=None)` | Per-page Markdown + layout metadata (all pages by default) |
|
||||
| `extract_pages_markdown_bytes(data, pages=None)` | Per-page Markdown from bytes |
|
||||
|
||||
## Types
|
||||
|
||||
@@ -82,7 +72,3 @@ result = pdf_inspector.extract_pages_markdown("document.pdf", pages=[0, 2])
|
||||
**`RegionText` fields:** `text`, `needs_ocr`
|
||||
|
||||
**`PageRegionTexts` fields:** `page` (0-indexed), `regions` (list of RegionText)
|
||||
|
||||
**`PageMarkdown` fields:** `page` (0-indexed), `markdown`, `needs_ocr`
|
||||
|
||||
**`PagesExtractionResult` fields:** `pages` (list of PageMarkdown), `pages_with_tables` (1-indexed), `pages_with_columns` (1-indexed), `pages_needing_ocr` (1-indexed), `is_complex`
|
||||
|
||||
@@ -79,27 +79,6 @@ let bytes = std::fs::read("document.pdf")?;
|
||||
let result = process_pdf_mem(&bytes)?;
|
||||
```
|
||||
|
||||
Extract per-page Markdown (one string per page, plus document-wide layout
|
||||
metadata):
|
||||
|
||||
```rust
|
||||
use pdf_inspector::extract_pages_markdown;
|
||||
|
||||
// Pass `None` for every page in document order, or a slice of 0-indexed
|
||||
// pages to restrict the output (caller-supplied order is preserved).
|
||||
let result = extract_pages_markdown("document.pdf", None)?;
|
||||
|
||||
for page in &result.pages {
|
||||
if page.needs_ocr {
|
||||
// Route this page to OCR
|
||||
} else {
|
||||
println!("Page {}: {}", page.page, page.markdown);
|
||||
}
|
||||
}
|
||||
|
||||
println!("Complex layout? {}", result.is_complex);
|
||||
```
|
||||
|
||||
## Processing modes
|
||||
|
||||
| Mode | What it does | Returns |
|
||||
@@ -123,8 +102,6 @@ println!("Complex layout? {}", result.is_complex);
|
||||
| `to_markdown(text, options)` | Convert plain text to Markdown |
|
||||
| `to_markdown_from_items(items, options)` | Markdown from pre-extracted `TextItem`s |
|
||||
| `to_markdown_from_items_with_rects(items, options, rects)` | Markdown with rectangle-based table detection |
|
||||
| `extract_pages_markdown(path, pages)` | Per-page Markdown + layout metadata (file) |
|
||||
| `extract_pages_markdown_mem(bytes, pages)` | Per-page Markdown from bytes |
|
||||
|
||||
Low-level detection functions are also available via the `detector` module (`detect_pdf_type`, `detect_pdf_type_with_config`, etc.) for callers who need `PdfTypeResult` instead of `PdfProcessResult`.
|
||||
|
||||
@@ -142,6 +119,4 @@ Low-level detection functions are also available via the `detector` module (`det
|
||||
| `LayoutComplexity` | Layout analysis: is_complex, pages_with_tables, pages_with_columns |
|
||||
| `TextItem` | Text with position, font info, and page number |
|
||||
| `MarkdownOptions` | Configuration for Markdown formatting (page numbers, etc.) |
|
||||
| `PageMarkdown` | Per-page result: page (0-indexed), markdown, needs_ocr |
|
||||
| `PagesExtractionResult` | Per-page output + 1-indexed pages_with_tables / pages_with_columns / pages_needing_ocr, is_complex |
|
||||
| `PdfError` | `Io`, `Parse`, `Encrypted`, `InvalidStructure`, `NotAPdf` |
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@firecrawl/pdf-inspector",
|
||||
"version": "1.4.0",
|
||||
"version": "1.3.0",
|
||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||
"main": "index.js",
|
||||
"types": "index.d.ts",
|
||||
|
||||
+5
-10
@@ -343,26 +343,21 @@ pub struct PagesExtractionResult {
|
||||
pub is_complex: bool,
|
||||
}
|
||||
|
||||
/// Extract formatted markdown for pages of a PDF, with layout classification
|
||||
/// metadata.
|
||||
/// Extract formatted markdown for specific pages of a PDF, with layout
|
||||
/// classification metadata.
|
||||
///
|
||||
/// Returns per-page markdown and classification data (tables, columns,
|
||||
/// OCR needs) from a single parse. Font statistics are computed from the
|
||||
/// full document so header detection is consistent across pages.
|
||||
///
|
||||
/// Omit `pages` (or pass `undefined`) to return every page in document
|
||||
/// order. Pass an array of 0-indexed page numbers to restrict output to
|
||||
/// those pages, in caller-supplied order.
|
||||
#[napi]
|
||||
pub fn extract_pages_markdown(
|
||||
buffer: Buffer,
|
||||
pages: Option<Vec<u32>>,
|
||||
pages: Vec<u32>,
|
||||
) -> Result<PagesExtractionResult> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
catch_panic("extract_pages_markdown", move || {
|
||||
let result =
|
||||
pdf_inspector::extract_pages_markdown_mem(&bytes, pages.as_deref())
|
||||
.map_err(|e| to_napi_err(e, "extract_pages_markdown"))?;
|
||||
let result = pdf_inspector::extract_pages_markdown_mem(&bytes, &pages)
|
||||
.map_err(|e| to_napi_err(e, "extract_pages_markdown"))?;
|
||||
Ok(PagesExtractionResult {
|
||||
pages: result
|
||||
.pages
|
||||
|
||||
@@ -7,7 +7,6 @@ import {
|
||||
extractText,
|
||||
extractTextWithPositions,
|
||||
extractTextInRegions,
|
||||
extractPagesMarkdown,
|
||||
} from './index.js';
|
||||
|
||||
const fixture = readFileSync('../tests/fixtures/thermo-freon12.pdf');
|
||||
@@ -90,28 +89,6 @@ assert.equal(typeof regionResults[0].regions[0].text, 'string');
|
||||
assert.equal(typeof regionResults[0].regions[0].needsOcr, 'boolean');
|
||||
console.log(' extractTextInRegions: OK');
|
||||
|
||||
// --- extractPagesMarkdown ---
|
||||
console.log('Testing extractPagesMarkdown...');
|
||||
|
||||
// omit pages → every page in document order
|
||||
const allPages = extractPagesMarkdown(fixture);
|
||||
assert.equal(allPages.pages.length, 3);
|
||||
assert.deepEqual(allPages.pages.map(p => p.page), [0, 1, 2]);
|
||||
assert.ok(typeof allPages.pages[0].markdown === 'string');
|
||||
assert.equal(typeof allPages.pages[0].needsOcr, 'boolean');
|
||||
assert.ok(Array.isArray(allPages.pagesWithTables));
|
||||
assert.ok(Array.isArray(allPages.pagesWithColumns));
|
||||
assert.ok(Array.isArray(allPages.pagesNeedingOcr));
|
||||
assert.equal(typeof allPages.isComplex, 'boolean');
|
||||
console.log(' extractPagesMarkdown (no pages arg): OK');
|
||||
|
||||
// selected pages preserve caller order
|
||||
const picked = extractPagesMarkdown(fixture, [2, 0]);
|
||||
assert.equal(picked.pages.length, 2);
|
||||
assert.equal(picked.pages[0].page, 2);
|
||||
assert.equal(picked.pages[1].page, 0);
|
||||
console.log(' extractPagesMarkdown with pages: OK');
|
||||
|
||||
// --- Error handling ---
|
||||
console.log('Testing error handling...');
|
||||
assert.throws(() => processPdf(Buffer.from('not a pdf')), /process_pdf/);
|
||||
|
||||
@@ -52,28 +52,6 @@ class PageRegionTexts:
|
||||
"""0-indexed page number."""
|
||||
regions: list[RegionText]
|
||||
|
||||
class PageMarkdown:
|
||||
"""Per-page markdown extraction result."""
|
||||
page: int
|
||||
"""0-indexed page number."""
|
||||
markdown: str
|
||||
"""Formatted markdown for this page (empty string when needs_ocr is True)."""
|
||||
needs_ocr: bool
|
||||
"""True when text on this page is unreliable and OCR should be used instead."""
|
||||
|
||||
class PagesExtractionResult:
|
||||
"""Per-page markdown output with document-wide layout classification."""
|
||||
pages: list[PageMarkdown]
|
||||
"""Per-page markdown results, in the order requested."""
|
||||
pages_with_tables: list[int]
|
||||
"""1-indexed pages where tables were detected."""
|
||||
pages_with_columns: list[int]
|
||||
"""1-indexed pages where multi-column layout was detected."""
|
||||
pages_needing_ocr: list[int]
|
||||
"""1-indexed pages that need OCR."""
|
||||
is_complex: bool
|
||||
"""True if any page has tables or multi-column layout."""
|
||||
|
||||
def process_pdf(path: str, pages: Optional[list[int]] = None) -> PdfResult:
|
||||
"""Process a PDF: detect type, extract text, convert to Markdown."""
|
||||
...
|
||||
@@ -137,31 +115,3 @@ def extract_text_in_regions_bytes(
|
||||
page_regions: List of (page_0indexed, [[x1, y1, x2, y2], ...]) tuples.
|
||||
"""
|
||||
...
|
||||
|
||||
def extract_pages_markdown(
|
||||
path: str,
|
||||
pages: Optional[list[int]] = None,
|
||||
) -> PagesExtractionResult:
|
||||
"""Extract formatted markdown for pages of a PDF, with layout classification.
|
||||
|
||||
Args:
|
||||
path: Path to the PDF file.
|
||||
pages: Optional list of 0-indexed pages. When ``None`` (default), every
|
||||
page is returned in document order. Otherwise, output matches the
|
||||
caller-supplied order.
|
||||
|
||||
Returns:
|
||||
PagesExtractionResult with per-page markdown and document-wide layout
|
||||
classification (tables, columns, OCR needs).
|
||||
"""
|
||||
...
|
||||
|
||||
def extract_pages_markdown_bytes(
|
||||
data: bytes,
|
||||
pages: Optional[list[int]] = None,
|
||||
) -> PagesExtractionResult:
|
||||
"""Extract formatted markdown for pages of a PDF from bytes.
|
||||
|
||||
See :func:`extract_pages_markdown` for details.
|
||||
"""
|
||||
...
|
||||
|
||||
@@ -626,33 +626,6 @@ fn find_relative_valleys(
|
||||
valleys
|
||||
}
|
||||
|
||||
/// Detect whether a side of a gutter consists predominantly of list-marker
|
||||
/// glyphs (•, ●, ○, ◦, ▪, ▫, ◆, ◇). A column of bullets on the left margin
|
||||
/// creates a spurious histogram valley between the bullet and the content.
|
||||
/// Treating it as a real column splits each list item's text across two
|
||||
/// "columns," so we reject these candidates.
|
||||
fn is_list_marker_column(items: &[&&TextItem]) -> bool {
|
||||
const LIST_MARKERS: &[char] = &['•', '●', '○', '◦', '▪', '▫', '◆', '◇', '■', '□'];
|
||||
if items.is_empty() {
|
||||
return false;
|
||||
}
|
||||
let marker_count = items
|
||||
.iter()
|
||||
.filter(|i| {
|
||||
let t = i.text.trim();
|
||||
let mut chars = t.chars();
|
||||
match (chars.next(), chars.next()) {
|
||||
(Some(c), None) => LIST_MARKERS.contains(&c),
|
||||
_ => false,
|
||||
}
|
||||
})
|
||||
.count();
|
||||
// Require ≥80% of items on this side to be standalone markers. A handful
|
||||
// of non-marker items (stray page numbers, footnote refs) shouldn't
|
||||
// defeat the check.
|
||||
marker_count as f32 / items.len() as f32 >= 0.8
|
||||
}
|
||||
|
||||
/// Validate valley candidates with vertical consistency checks and build column regions.
|
||||
///
|
||||
/// When `center_assign` is true, items are assigned to columns based on their
|
||||
@@ -721,19 +694,6 @@ fn validate_and_build_columns(
|
||||
continue;
|
||||
}
|
||||
|
||||
// Reject valleys where the smaller side is just a column of list
|
||||
// markers (bullets aligned at the left margin). This is a common
|
||||
// pattern in PDFs where ● starts each list item: histogram detection
|
||||
// sees the gap between bullet and content as a gutter.
|
||||
let smaller_items: &[&&TextItem] = if left_items.len() <= right_items.len() {
|
||||
&left_items
|
||||
} else {
|
||||
&right_items
|
||||
};
|
||||
if is_list_marker_column(smaller_items) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Check vertical overlap
|
||||
if y_range > 0.0 {
|
||||
let left_y_min = left_items.iter().map(|i| i.y).fold(f32::INFINITY, f32::min);
|
||||
@@ -1820,63 +1780,6 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bullet_marker_column_not_detected_as_column() {
|
||||
// Pattern: every line is `● <content>`, with ● at x=90 and content
|
||||
// starting at x=104. Histogram detection sees a gutter between them
|
||||
// and would split the page into a "bullet column" and "content column",
|
||||
// scrambling every list item.
|
||||
let mut items = Vec::new();
|
||||
for i in 0..15 {
|
||||
let y = 750.0 - i as f32 * 30.0;
|
||||
items.push(make_item(1, 90.0, y, "●"));
|
||||
items.push(make_item(
|
||||
1,
|
||||
104.0,
|
||||
y,
|
||||
"FullContentLineTextHere________________",
|
||||
));
|
||||
}
|
||||
// Pad with content to satisfy min item count for column detection.
|
||||
for i in 0..15 {
|
||||
let y = 300.0 - i as f32 * 14.0;
|
||||
items.push(make_item(1, 72.0, y, "FootnoteText_____________________"));
|
||||
}
|
||||
|
||||
let cols = detect_columns(&items, 1, false);
|
||||
assert_eq!(
|
||||
cols.len(),
|
||||
1,
|
||||
"Bullet markers aligned at left margin should not be treated as their own column"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn is_list_marker_column_detects_bullets() {
|
||||
let items = vec![
|
||||
make_item(1, 90.0, 100.0, "●"),
|
||||
make_item(1, 90.0, 114.0, "●"),
|
||||
make_item(1, 90.0, 128.0, "●"),
|
||||
make_item(1, 90.0, 142.0, "●"),
|
||||
];
|
||||
let refs: Vec<&TextItem> = items.iter().collect();
|
||||
let wrapped: Vec<&&TextItem> = refs.iter().collect();
|
||||
assert!(is_list_marker_column(&wrapped));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn is_list_marker_column_rejects_prose() {
|
||||
let items = vec![
|
||||
make_item(1, 30.0, 100.0, "Regular prose line"),
|
||||
make_item(1, 30.0, 114.0, "Another sentence"),
|
||||
make_item(1, 30.0, 128.0, "Third line"),
|
||||
make_item(1, 30.0, 142.0, "Fourth line"),
|
||||
];
|
||||
let refs: Vec<&TextItem> = items.iter().collect();
|
||||
let wrapped: Vec<&&TextItem> = refs.iter().collect();
|
||||
assert!(!is_list_marker_column(&wrapped));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn premask_narrow_line_not_masked() {
|
||||
// Items that form a line spanning only ~40% of column width → not masked
|
||||
|
||||
+4
-32
@@ -336,17 +336,13 @@ pub struct PagesExtractionResult {
|
||||
pub is_complex: bool,
|
||||
}
|
||||
|
||||
/// Extract formatted markdown for pages of a PDF, with layout
|
||||
/// Extract formatted markdown for specific pages of a PDF, with layout
|
||||
/// classification metadata.
|
||||
///
|
||||
/// Unlike [`process_pdf_mem`] which returns one concatenated markdown string,
|
||||
/// this returns per-page markdown so callers can mix direct extraction
|
||||
/// (for simple text pages) with GPU OCR (for complex/scanned pages).
|
||||
///
|
||||
/// When `pages` is `None`, every page (0-indexed, in document order) is
|
||||
/// returned. When `Some(&[...])`, only the listed 0-indexed pages are
|
||||
/// returned, in the caller's order.
|
||||
///
|
||||
/// Font statistics are computed from the full document so header
|
||||
/// detection thresholds are consistent regardless of which pages are
|
||||
/// requested. Per-page `needs_ocr` is set when the page has GID-encoded
|
||||
@@ -356,7 +352,7 @@ pub struct PagesExtractionResult {
|
||||
/// at near-zero cost since the items/rects/lines are already in memory.
|
||||
pub fn extract_pages_markdown_mem(
|
||||
buffer: &[u8],
|
||||
pages: Option<&[u32]>,
|
||||
pages: &[u32],
|
||||
) -> Result<PagesExtractionResult, PdfError> {
|
||||
validate_pdf_bytes(buffer)?;
|
||||
let (doc, page_count) = load_document_from_mem(buffer)?;
|
||||
@@ -372,20 +368,10 @@ pub fn extract_pages_markdown_mem(
|
||||
// Compute font stats from full document (cross-page consistency).
|
||||
let font_stats = markdown::analysis::calculate_font_stats_from_items(&all_items);
|
||||
|
||||
// When caller doesn't specify pages, return every page in document order.
|
||||
let all_pages: Vec<u32>;
|
||||
let pages_slice: &[u32] = match pages {
|
||||
Some(p) => p,
|
||||
None => {
|
||||
all_pages = (0..page_count).collect();
|
||||
&all_pages
|
||||
}
|
||||
};
|
||||
|
||||
let mut results = Vec::with_capacity(pages_slice.len());
|
||||
let mut results = Vec::with_capacity(pages.len());
|
||||
let mut pages_needing_ocr = Vec::new();
|
||||
|
||||
for &page_0idx in pages_slice {
|
||||
for &page_0idx in pages {
|
||||
// Out-of-range pages → empty + needs_ocr
|
||||
if page_0idx >= page_count {
|
||||
pages_needing_ocr.push(page_0idx + 1);
|
||||
@@ -458,20 +444,6 @@ pub fn extract_pages_markdown_mem(
|
||||
})
|
||||
}
|
||||
|
||||
/// Path-based wrapper for [`extract_pages_markdown_mem`].
|
||||
///
|
||||
/// Reads the PDF from disk and extracts per-page markdown. Pass `None` for
|
||||
/// `pages` to return every page in document order, or `Some(&[...])` to
|
||||
/// restrict to specific 0-indexed pages (in caller-supplied order).
|
||||
pub fn extract_pages_markdown<P: AsRef<Path>>(
|
||||
path: P,
|
||||
pages: Option<&[u32]>,
|
||||
) -> Result<PagesExtractionResult, PdfError> {
|
||||
validate_pdf_file(&path)?;
|
||||
let buffer = std::fs::read(path.as_ref())?;
|
||||
extract_pages_markdown_mem(&buffer, pages)
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
// Region-based text extraction (for hybrid OCR pipelines)
|
||||
// =========================================================================
|
||||
|
||||
@@ -64,22 +64,6 @@ pub(crate) fn is_caption_line(text: &str) -> bool {
|
||||
false
|
||||
}
|
||||
|
||||
/// Check if text starts with an unambiguous bullet marker (●, •, ○, ◦).
|
||||
///
|
||||
/// Narrower than [`is_list_item`]: it excludes numbered/lettered patterns
|
||||
/// like `1.` or `a)`, which legitimately appear as section headings in many
|
||||
/// documents. Used by the heading classifier to reject bullet lines without
|
||||
/// also demoting numbered headings.
|
||||
pub(crate) fn starts_with_bullet_marker(text: &str) -> bool {
|
||||
let trimmed = text.trim_start();
|
||||
trimmed.starts_with("• ")
|
||||
|| trimmed.starts_with("● ")
|
||||
|| trimmed.starts_with("○ ")
|
||||
|| trimmed.starts_with("◦ ")
|
||||
|| trimmed.starts_with("- ")
|
||||
|| trimmed.starts_with("* ")
|
||||
}
|
||||
|
||||
/// Check if text looks like a list item
|
||||
pub(crate) fn is_list_item(text: &str) -> bool {
|
||||
let trimmed = text.trim_start();
|
||||
@@ -131,16 +115,6 @@ pub(crate) fn format_list_item(text: &str) -> String {
|
||||
if let Some(rest) = trimmed.strip_prefix(*bullet) {
|
||||
return format!("- {}", rest.trim_start());
|
||||
}
|
||||
// Bullet inside a leading bold/italic run (e.g. "**● Label:** rest").
|
||||
// The run wraps both the marker and the following label because both
|
||||
// use a bold font in the PDF.
|
||||
for wrapper in ["**", "*"] {
|
||||
if let Some(after_open) = trimmed.strip_prefix(wrapper) {
|
||||
if let Some(rest) = after_open.strip_prefix(*bullet) {
|
||||
return format!("- {}{}", wrapper, rest.trim_start());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if trimmed.starts_with("- ") || trimmed.starts_with("* ") {
|
||||
@@ -224,42 +198,3 @@ pub(crate) fn is_monospace_font(font_name: &str) -> bool {
|
||||
|
||||
patterns.iter().any(|p| lower.contains(p))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn format_list_item_plain_bullet() {
|
||||
assert_eq!(format_list_item("● Item"), "- Item");
|
||||
assert_eq!(format_list_item("• Item"), "- Item");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn format_list_item_bullet_inside_bold() {
|
||||
// PDF that uses bold font for both the marker and the label produces
|
||||
// a single bold run like "**● Label:** rest"; the bullet must still
|
||||
// be stripped and the bold wrapper preserved on the label.
|
||||
assert_eq!(
|
||||
format_list_item("**● Fraud: Willing cooperation;**"),
|
||||
"- **Fraud: Willing cooperation;**"
|
||||
);
|
||||
assert_eq!(
|
||||
format_list_item("**● Label:** rest of line"),
|
||||
"- **Label:** rest of line"
|
||||
);
|
||||
assert_eq!(format_list_item("*● Italic:* rest"), "- *Italic:* rest");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn format_list_item_already_dash() {
|
||||
assert_eq!(format_list_item("- existing"), "- existing");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn is_list_item_with_bullet_space() {
|
||||
assert!(is_list_item("● Item"));
|
||||
assert!(is_list_item("• Item"));
|
||||
assert!(is_list_item("- Item"));
|
||||
}
|
||||
}
|
||||
|
||||
+1
-90
@@ -9,9 +9,7 @@ use super::analysis::{
|
||||
bold_heading_level, calculate_font_stats, compute_heading_tiers, compute_paragraph_threshold,
|
||||
detect_header_level, font_size_rarity, has_dot_leaders,
|
||||
};
|
||||
use super::classify::{
|
||||
format_list_item, is_caption_line, is_list_item, is_monospace_font, starts_with_bullet_marker,
|
||||
};
|
||||
use super::classify::{format_list_item, is_caption_line, is_list_item, is_monospace_font};
|
||||
use super::postprocess::clean_markdown;
|
||||
use super::preprocess::{merge_drop_caps, merge_heading_lines};
|
||||
use super::MarkdownOptions;
|
||||
@@ -586,30 +584,9 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
.as_ref()
|
||||
.and_then(struct_role_heading_level)
|
||||
.filter(|level| !overused_heading_levels.contains(level));
|
||||
|
||||
// Protect wrapped list items: when inside a list, a visually-continuing
|
||||
// line (same indent, line-wrap spacing) must not be reclassified as a
|
||||
// heading by the font heuristic — PDFs often bold the lead phrase of a
|
||||
// list item across multiple wrap lines, and an all-bold middle line
|
||||
// would otherwise split one item into a heading + stray body text.
|
||||
// We gate on the document's paragraph threshold so genuine section
|
||||
// headings that follow a numbered paragraph (y_gap > para_threshold)
|
||||
// remain detectable.
|
||||
let looks_like_list_continuation = in_list
|
||||
&& match (last_list_x, line.items.first().map(|i| i.x)) {
|
||||
(Some(list_x), Some(curr_x)) => {
|
||||
let x_ok = curr_x >= list_x - 5.0 && curr_x <= list_x + 50.0;
|
||||
let y_ok = y_gap >= 0.0 && y_gap <= para_threshold;
|
||||
x_ok && y_ok && !is_list_item(plain_trimmed)
|
||||
}
|
||||
_ => false,
|
||||
};
|
||||
|
||||
let heuristic_heading = if options.detect_headers
|
||||
&& !looks_like_list_continuation
|
||||
&& plain_trimmed.len() > 3
|
||||
&& plain_trimmed.split_whitespace().count() <= 15
|
||||
&& !starts_with_bullet_marker(plain_trimmed)
|
||||
{
|
||||
let line_font_size = line.items.first().map(|i| i.font_size).unwrap_or(base_size);
|
||||
detect_header_level(line_font_size, base_size, &heading_tiers).or_else(|| {
|
||||
@@ -1473,70 +1450,4 @@ mod tests {
|
||||
overused
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_wrapped_bold_lead_in_list_item_not_heading() {
|
||||
// Regression: numbered-list items whose bold "lead" phrase wraps onto
|
||||
// a second line (e.g. definitions in system cards) must not have the
|
||||
// wrapped line reclassified as a heading. The middle line is
|
||||
// all_bold + standalone (in_paragraph=false while in_list), which
|
||||
// previously tripped the rarity heuristic and emitted #### in the
|
||||
// middle of the item, splitting the body into stray bullets.
|
||||
let make = |text: &str, x: f32, y: f32, bold: bool| {
|
||||
let mut item = make_item(text, 1, None);
|
||||
item.x = x;
|
||||
item.y = y;
|
||||
item.is_bold = bold;
|
||||
item
|
||||
};
|
||||
|
||||
let lines = vec![
|
||||
// "1. **bold lead phrase start**"
|
||||
make_line(vec![
|
||||
make("1. ", 72.0, 700.0, false),
|
||||
make(
|
||||
"Chemical and biological weapons threat model 1 (CB-1): Non-novel",
|
||||
90.0,
|
||||
700.0,
|
||||
true,
|
||||
),
|
||||
]),
|
||||
// wrapped continuation of the bold lead — all_bold, same indent
|
||||
make_line(vec![make(
|
||||
"chemical/biological weapons production capabilities: A model has CB-1",
|
||||
90.0,
|
||||
686.0,
|
||||
true,
|
||||
)]),
|
||||
// body text of the same list item
|
||||
make_line(vec![make(
|
||||
"capabilities if it has the ability to significantly help individuals.",
|
||||
90.0,
|
||||
672.0,
|
||||
false,
|
||||
)]),
|
||||
];
|
||||
|
||||
let md = to_markdown_from_lines_with_tables_and_images(
|
||||
lines,
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
None,
|
||||
);
|
||||
|
||||
assert!(
|
||||
!md.contains("#### "),
|
||||
"wrapped bold lead must not become a heading: {md}"
|
||||
);
|
||||
assert!(
|
||||
md.lines().filter(|l| l.starts_with("- ")).count() == 0,
|
||||
"continuation body must not become a stray bullet: {md}"
|
||||
);
|
||||
assert!(
|
||||
md.contains("1. ") && md.contains("A model has CB-1"),
|
||||
"numbered list item should remain intact: {md}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
-121
@@ -146,67 +146,6 @@ impl PyPageRegionTexts {
|
||||
// Text item wrapper
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Per-page markdown extraction result.
|
||||
#[pyclass(name = "PageMarkdown")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyPageMarkdown {
|
||||
/// 0-indexed page number.
|
||||
#[pyo3(get)]
|
||||
pub page: u32,
|
||||
/// Formatted markdown for this page.
|
||||
#[pyo3(get)]
|
||||
pub markdown: String,
|
||||
/// True when text on this page is unreliable (GID-encoded fonts,
|
||||
/// encoding issues, garbage text, or empty extraction).
|
||||
#[pyo3(get)]
|
||||
pub needs_ocr: bool,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl PyPageMarkdown {
|
||||
fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"PageMarkdown(page={}, markdown='{}', needs_ocr={})",
|
||||
self.page,
|
||||
self.markdown.chars().take(40).collect::<String>(),
|
||||
self.needs_ocr
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
/// Combined per-page markdown extraction and layout classification result.
|
||||
#[pyclass(name = "PagesExtractionResult")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyPagesExtractionResult {
|
||||
/// Per-page markdown results, in the order requested.
|
||||
#[pyo3(get)]
|
||||
pub pages: Vec<PyPageMarkdown>,
|
||||
/// 1-indexed pages where tables were detected.
|
||||
#[pyo3(get)]
|
||||
pub pages_with_tables: Vec<u32>,
|
||||
/// 1-indexed pages where multi-column layout was detected.
|
||||
#[pyo3(get)]
|
||||
pub pages_with_columns: Vec<u32>,
|
||||
/// 1-indexed pages that need OCR (scanned/image-based or unreliable text).
|
||||
#[pyo3(get)]
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
/// True if any page has tables or columns.
|
||||
#[pyo3(get)]
|
||||
pub is_complex: bool,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl PyPagesExtractionResult {
|
||||
fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"PagesExtractionResult(pages={}, pages_with_tables={:?}, is_complex={})",
|
||||
self.pages.len(),
|
||||
self.pages_with_tables,
|
||||
self.is_complex
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
/// A positioned text item extracted from a PDF.
|
||||
#[pyclass(name = "TextItem")]
|
||||
#[derive(Clone)]
|
||||
@@ -341,24 +280,6 @@ fn parse_page_regions(
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn to_py_pages_result(r: crate::PagesExtractionResult) -> PyPagesExtractionResult {
|
||||
PyPagesExtractionResult {
|
||||
pages: r
|
||||
.pages
|
||||
.into_iter()
|
||||
.map(|p| PyPageMarkdown {
|
||||
page: p.page,
|
||||
markdown: p.markdown,
|
||||
needs_ocr: p.needs_ocr,
|
||||
})
|
||||
.collect(),
|
||||
pages_with_tables: r.pages_with_tables,
|
||||
pages_with_columns: r.pages_with_columns,
|
||||
pages_needing_ocr: r.pages_needing_ocr,
|
||||
is_complex: r.is_complex,
|
||||
}
|
||||
}
|
||||
|
||||
fn convert_region_results(results: Vec<crate::PageRegionResult>) -> Vec<PyPageRegionTexts> {
|
||||
results
|
||||
.into_iter()
|
||||
@@ -521,44 +442,6 @@ fn extract_text_in_regions_bytes(
|
||||
Ok(convert_region_results(results))
|
||||
}
|
||||
|
||||
/// Extract formatted markdown for pages of a PDF file, with layout
|
||||
/// classification metadata.
|
||||
///
|
||||
/// Returns per-page markdown and classification data (tables, columns,
|
||||
/// OCR needs) from a single parse. Font statistics are computed from the
|
||||
/// full document so header detection is consistent across pages.
|
||||
///
|
||||
/// Args:
|
||||
/// path: Path to the PDF file.
|
||||
/// pages: Optional list of 0-indexed pages. When None (default), every
|
||||
/// page is returned in document order. When provided, output
|
||||
/// matches the caller-supplied order.
|
||||
///
|
||||
/// Returns:
|
||||
/// PagesExtractionResult with per-page markdown and classification data.
|
||||
#[pyfunction]
|
||||
#[pyo3(signature = (path, pages=None))]
|
||||
fn extract_pages_markdown(
|
||||
path: &str,
|
||||
pages: Option<Vec<u32>>,
|
||||
) -> PyResult<PyPagesExtractionResult> {
|
||||
let result = crate::extract_pages_markdown(path, pages.as_deref()).map_err(to_py_err)?;
|
||||
Ok(to_py_pages_result(result))
|
||||
}
|
||||
|
||||
/// Extract formatted markdown for pages of a PDF from bytes.
|
||||
///
|
||||
/// See [`extract_pages_markdown`] for details.
|
||||
#[pyfunction]
|
||||
#[pyo3(signature = (data, pages=None))]
|
||||
fn extract_pages_markdown_bytes(
|
||||
data: &[u8],
|
||||
pages: Option<Vec<u32>>,
|
||||
) -> PyResult<PyPagesExtractionResult> {
|
||||
let result = crate::extract_pages_markdown_mem(data, pages.as_deref()).map_err(to_py_err)?;
|
||||
Ok(to_py_pages_result(result))
|
||||
}
|
||||
|
||||
/// Python module definition.
|
||||
#[pymodule]
|
||||
fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
|
||||
@@ -567,8 +450,6 @@ fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
|
||||
m.add_class::<PyTextItem>()?;
|
||||
m.add_class::<PyRegionText>()?;
|
||||
m.add_class::<PyPageRegionTexts>()?;
|
||||
m.add_class::<PyPageMarkdown>()?;
|
||||
m.add_class::<PyPagesExtractionResult>()?;
|
||||
m.add_function(wrap_pyfunction!(process_pdf, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(process_pdf_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(detect_pdf, m)?)?;
|
||||
@@ -581,7 +462,5 @@ fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
|
||||
m.add_function(wrap_pyfunction!(extract_text_with_positions_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text_in_regions, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_text_in_regions_bytes, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_pages_markdown, m)?)?;
|
||||
m.add_function(wrap_pyfunction!(extract_pages_markdown_bytes, m)?)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -1156,18 +1156,11 @@ fn propagate_merged_cells(
|
||||
continue;
|
||||
}
|
||||
|
||||
// Find first and last grid rows that the rect spans.
|
||||
//
|
||||
// A merged-cell rect CONTAINS the rows it spans — its bottom sits
|
||||
// at or below the row bottom and its top sits at or above the row
|
||||
// top (within tol). A pure overlap check (any overlap within tol)
|
||||
// gives false positives at shared row boundaries: a rect whose
|
||||
// top equals row N's bottom would be considered to span row N
|
||||
// even though it lies entirely below, cascading body text from
|
||||
// unrelated rows into a single header cell.
|
||||
let spans = |r: usize| ry <= row_edges[r + 1] + tol && (ry + rh) >= row_edges[r] - tol;
|
||||
let first_row = (0..num_rows).find(|&r| spans(r));
|
||||
let last_row = (0..num_rows).rfind(|&r| spans(r));
|
||||
// Find first and last grid rows that the rect spans
|
||||
let first_row = (0..num_rows)
|
||||
.find(|&r| ry <= row_edges[r] + tol && (ry + rh) >= row_edges[r + 1] - tol);
|
||||
let last_row = (0..num_rows)
|
||||
.rfind(|&r| ry <= row_edges[r] + tol && (ry + rh) >= row_edges[r + 1] - tol);
|
||||
|
||||
let (first, last) = match (first_row, last_row) {
|
||||
(Some(f), Some(l)) if l > f => (f, l),
|
||||
@@ -2357,27 +2350,6 @@ mod tests {
|
||||
assert_eq!(cells[1][0], "B");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_propagate_merged_cells_rect_tangent_to_row_boundary() {
|
||||
// Regression: a rect whose top exactly equals a row's bottom lies
|
||||
// entirely outside that row, so it must not be considered to span
|
||||
// it. With the old overlap-based predicate this cascaded into body
|
||||
// text from unrelated rows being merged into a single header cell
|
||||
// (mythos system card CB task-based evaluations table).
|
||||
//
|
||||
// Layout: two rows 0..80 and 80..160 (bottom → top in PDF coords),
|
||||
// rect occupies only the lower row (y=0..80). Its top equals the
|
||||
// upper row's bottom; it must not span the upper row.
|
||||
let col_edges = vec![0.0, 50.0];
|
||||
let row_edges = vec![160.0, 80.0, 0.0]; // top → bot
|
||||
let mut cells = vec![vec!["Upper".to_string()], vec!["Lower".to_string()]];
|
||||
let group_rects = vec![(0.0, 0.0, 50.0, 80.0)]; // rect at y=0..80
|
||||
let skip = vec![false];
|
||||
propagate_merged_cells(&mut cells, &col_edges, &row_edges, &group_rects, &skip);
|
||||
assert_eq!(cells[0][0], "Upper", "upper row must not be merged");
|
||||
assert_eq!(cells[1][0], "Lower", "lower row must not be touched");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_propagate_merged_cells_empty_cells_preserved() {
|
||||
let col_edges = vec![0.0, 50.0];
|
||||
|
||||
+15
-53
@@ -4,10 +4,10 @@ use pdf_inspector::detector::{DetectionConfig, ScanStrategy};
|
||||
use pdf_inspector::extractor::group_into_lines;
|
||||
use pdf_inspector::types::TextLine;
|
||||
use pdf_inspector::{
|
||||
detect_pdf_type, extract_pages_markdown, extract_pages_markdown_mem,
|
||||
extract_tables_in_regions_mem, extract_text, extract_text_in_regions_mem,
|
||||
extract_text_with_positions, process_pdf_mem, process_pdf_with_options, to_markdown,
|
||||
MarkdownOptions, PdfError, PdfOptions, PdfType, TextItem,
|
||||
detect_pdf_type, extract_pages_markdown_mem, extract_tables_in_regions_mem, extract_text,
|
||||
extract_text_in_regions_mem, extract_text_with_positions, process_pdf_mem,
|
||||
process_pdf_with_options, to_markdown, MarkdownOptions, PdfError, PdfOptions, PdfType,
|
||||
TextItem,
|
||||
};
|
||||
use std::collections::HashSet;
|
||||
|
||||
@@ -1481,7 +1481,7 @@ fn test_bits_pilani_page8_table_detection() {
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_basic() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, Some(&[0, 1])).unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, &[0, 1]).unwrap();
|
||||
|
||||
assert_eq!(result.pages.len(), 2);
|
||||
assert_eq!(result.pages[0].page, 0);
|
||||
@@ -1495,7 +1495,7 @@ fn test_extract_pages_markdown_basic() {
|
||||
fn test_extract_pages_markdown_page_ordering() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
// Request pages in non-sequential order
|
||||
let result = extract_pages_markdown_mem(&buf, Some(&[1, 0])).unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, &[1, 0]).unwrap();
|
||||
|
||||
assert_eq!(result.pages.len(), 2);
|
||||
// Results should match input order, not document order
|
||||
@@ -1506,7 +1506,7 @@ fn test_extract_pages_markdown_page_ordering() {
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_out_of_range() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, Some(&[9999])).unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, &[9999]).unwrap();
|
||||
|
||||
assert_eq!(result.pages.len(), 1);
|
||||
assert_eq!(result.pages[0].page, 9999);
|
||||
@@ -1518,14 +1518,14 @@ fn test_extract_pages_markdown_out_of_range() {
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_empty_pages_list() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, Some(&[])).unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, &[]).unwrap();
|
||||
assert!(result.pages.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_single_page() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, Some(&[0])).unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, &[0]).unwrap();
|
||||
|
||||
assert_eq!(result.pages.len(), 1);
|
||||
assert_eq!(result.pages[0].page, 0);
|
||||
@@ -1535,7 +1535,7 @@ fn test_extract_pages_markdown_single_page() {
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_invalid_buffer() {
|
||||
let result = extract_pages_markdown_mem(b"not a pdf", Some(&[0]));
|
||||
let result = extract_pages_markdown_mem(b"not a pdf", &[0]);
|
||||
assert!(result.is_err());
|
||||
}
|
||||
|
||||
@@ -1543,7 +1543,7 @@ fn test_extract_pages_markdown_invalid_buffer() {
|
||||
fn test_extract_pages_markdown_gid_pages_need_ocr() {
|
||||
// shinagawa_identity_h.pdf has GID-encoded fonts
|
||||
let buf = std::fs::read("tests/fixtures/shinagawa_identity_h.pdf").unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, Some(&[0])).unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, &[0]).unwrap();
|
||||
|
||||
assert_eq!(result.pages.len(), 1);
|
||||
assert!(result.pages[0].needs_ocr);
|
||||
@@ -1556,7 +1556,7 @@ fn test_extract_pages_markdown_classification_with_tables() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
let page_count = process_pdf_mem(&buf).unwrap().page_count;
|
||||
let page_indices: Vec<u32> = (0..page_count).collect();
|
||||
let result = extract_pages_markdown_mem(&buf, Some(&page_indices)).unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, &page_indices).unwrap();
|
||||
|
||||
assert!(
|
||||
!result.pages_with_tables.is_empty(),
|
||||
@@ -1569,7 +1569,7 @@ fn test_extract_pages_markdown_classification_with_tables() {
|
||||
fn test_extract_pages_markdown_simple_pdf_no_complexity() {
|
||||
// bare_name_struct.pdf is a simple document with a heading and code block
|
||||
let buf = std::fs::read("tests/fixtures/bare_name_struct.pdf").unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, Some(&[0])).unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, &[0]).unwrap();
|
||||
|
||||
assert!(result.pages_with_tables.is_empty());
|
||||
assert!(result.pages_with_columns.is_empty());
|
||||
@@ -1582,7 +1582,7 @@ fn test_extract_pages_markdown_classification_matches_process_pdf() {
|
||||
let full = process_pdf_mem(&buf).unwrap();
|
||||
let page_count = full.page_count;
|
||||
let page_indices: Vec<u32> = (0..page_count).collect();
|
||||
let result = extract_pages_markdown_mem(&buf, Some(&page_indices)).unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, &page_indices).unwrap();
|
||||
|
||||
assert_eq!(
|
||||
result.pages_with_tables, full.layout.pages_with_tables,
|
||||
@@ -1605,7 +1605,7 @@ fn test_extract_pages_markdown_consistency_with_process_pdf() {
|
||||
// Get per-page output for all pages
|
||||
let page_count = full.page_count;
|
||||
let page_indices: Vec<u32> = (0..page_count).collect();
|
||||
let result = extract_pages_markdown_mem(&buf, Some(&page_indices)).unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, &page_indices).unwrap();
|
||||
|
||||
// Concatenated per-page markdown should contain substantial overlap with
|
||||
// the full output (exact match not expected due to header/footer stripping
|
||||
@@ -1630,41 +1630,3 @@ fn test_extract_pages_markdown_consistency_with_process_pdf() {
|
||||
full_md.len()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_none_returns_all_pages() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
let page_count = process_pdf_mem(&buf).unwrap().page_count;
|
||||
|
||||
let result = extract_pages_markdown_mem(&buf, None).unwrap();
|
||||
|
||||
assert_eq!(result.pages.len() as u32, page_count);
|
||||
for (i, page) in result.pages.iter().enumerate() {
|
||||
assert_eq!(page.page, i as u32, "pages should be in document order");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_path_api() {
|
||||
let path = "tests/fixtures/nexo-price-en.pdf";
|
||||
let buf = std::fs::read(path).unwrap();
|
||||
|
||||
let via_path = extract_pages_markdown(path, Some(&[0])).unwrap();
|
||||
let via_mem = extract_pages_markdown_mem(&buf, Some(&[0])).unwrap();
|
||||
|
||||
assert_eq!(via_path.pages.len(), via_mem.pages.len());
|
||||
assert_eq!(via_path.pages[0].markdown, via_mem.pages[0].markdown);
|
||||
assert_eq!(via_path.pages[0].needs_ocr, via_mem.pages[0].needs_ocr);
|
||||
assert_eq!(via_path.is_complex, via_mem.is_complex);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_path_none_returns_all_pages() {
|
||||
let path = "tests/fixtures/nexo-price-en.pdf";
|
||||
let page_count = process_pdf_mem(&std::fs::read(path).unwrap())
|
||||
.unwrap()
|
||||
.page_count;
|
||||
|
||||
let result = extract_pages_markdown(path, None).unwrap();
|
||||
assert_eq!(result.pages.len() as u32, page_count);
|
||||
}
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
본 가격표는 국내 거주 중인 외국인을 위한 한국어 가격표의 비공식 번역본입니다. ※ The post-tax benefit sales price is provided for your reference only, reflecting the current tax benefits and eco-friendly vehicle individual consumption tax reductions. 본 가격표와 한국어 가격표의 내용이 상이한 경우 한국어 가격표의 내용이 우선하므로, 반드시 한국어 가격표의 내용을 확인하십시오. The final sales price may vary depending on the addition of optional items and whether the eco-friendly vehicle criteria are met, so please be sure to check the quotation. This price list is an unofficial translation of the Korean price list for the convenience of foreign residents in South Korea. ※ Please check the Korean price list for information on colors, details, and fuel consumption for each model. If the price list differs from the Korean price list, please check the contents of the Korean price list first. ※ All optional item prices are listed based on pre-tax reduction amounts. The actual sales price, which reflects the total individual consumption tax reduction including optional items, may differ depending on applicable tax benefits. ※ The items (specifications, colors, etc.) and prices listed in this pricing table are subject to change without prior notice depending on the holding of new car launch events, improvements made in automobile performance, introduction of related laws and regulations, and changes in company circumstances. The all-new NEXO Release Date: June 10, 2025 / (Unit: KRW)
|
||||
|
||||
|Classification|Selling price before tax benefit Supply value(surtax)|Selling price after tax benefit|Standard equipment|Options (before tax benefit)|
|
||||
|Classification Exclusive Exclusive|Selling price before tax benefit Supply value(surtax) 80,509,000 73,190,000(7,319,000)|Selling price after tax benefit 76,435,000|Standard equipment • Powertrain/Performance: Fuel cell system(150kW drive motor, lithium-ion battery, and reducer), Regenerative braking system, Column-Type Shift By Wire(vibration warning), Drive mode select • Safety: 9 airbag system(1st-row advanced/center side airbags, 1st/2nd-row side airbags, and rollover-resistant curtain airbags), Multi-Collision Brake System, Active hood system(for pedestrian protection), Safety unlock function, Artificial engine sound(for pedestrian protection), Child seat fastening system (2 in 2nd-row), Fire extinguisher for vehicles, Pedal Misapplication Safety Assist • Smart Safety Technology: Forward Collision-avoidance Assist(vehicles/ pedestrians/two-wheeled vehicles/junction turning/front oncoming), Smart Cruise Control with Stop & Go, Lane Keeping Assist, Lane Following Assist 2, Blind-spot Collision Warning(driving), Blind-spot Collision-avoidance Assist(forward exit), Rear Cross-traffic Collision-avoidance Assist, Safety Exit Assist, Driver Attention Warning, High Beam Assist, Advanced Rear Occupant Alert, Intelligent Speed Limit Assist, Hands-On Detection, Highway Driving Assist, Navigation-based Smart Cruise Control(safety speed zone/curve control), Vibration warning steering wheel • Exterior: Full LED headlamps(projection type), LED turn signal lamps (front and rear), LED Daytime Running Lights, LED positioning lights, LED rear combination lamps, LED third brake lights, 18-inch alloy wheels & tires, Solar glass(windshield), Double-glazed soundproof glass(windshield, and 1st/ 2nd-row doors), Outside mirror(heating, power-folding, power adjustment,|Options (before tax benefit) ▶ Hi-pass(e hi-pass) [200,000]|
|
||||
|---|---|---|---|---|
|
||||
|Exclusive|80,509,000 73,190,000(7,319,000) with 3.5% individual consumption tax applied 79,287,000|76,435,000 with 3.5% individual consumption tax applied 76,435,000|• Powertrain/Performance: Fuel cell system(150kW drive motor, lithium-ion battery, and reducer), Regenerative braking system, Column-Type Shift By Wire(vibration warning), Drive mode select • Safety: 9 airbag system(1st-row advanced/center side airbags, 1st/2nd-row side airbags, and rollover-resistant curtain airbags), Multi-Collision Brake System, Active hood system(for pedestrian protection), Safety unlock function, Artificial engine sound(for pedestrian protection), Child seat fastening system (2 in 2nd-row), Fire extinguisher for vehicles, Pedal Misapplication Safety Assist • Smart Safety Technology: Forward Collision-avoidance Assist(vehicles/ pedestrians/two-wheeled vehicles/junction turning/front oncoming), Smart Cruise Control with Stop & Go, Lane Keeping Assist, Lane Following Assist 2, Blind-spot Collision Warning(driving), Blind-spot Collision-avoidance Assist(forward exit), Rear Cross-traffic Collision-avoidance Assist, Safety Exit Assist, Driver Attention Warning, High Beam Assist, Advanced Rear Occupant Alert, Intelligent Speed Limit Assist, Hands-On Detection, Highway Driving Assist, Navigation-based Smart Cruise Control(safety speed zone/curve control), Vibration warning steering wheel • Exterior: Full LED headlamps(projection type), LED turn signal lamps (front and rear), LED Daytime Running Lights, LED positioning lights, LED rear combination lamps, LED third brake lights, 18-inch alloy wheels & tires, Solar glass(windshield), Double-glazed soundproof glass(windshield, and 1st/ 2nd-row doors), Outside mirror(heating, power-folding, power adjustment, and LED turn signal lamps), Auto flush door handles, Black door garnish • Interior: Panoramic curved display, 12.3-inch color LCD cluster, Leather- upholstered steering wheel(with heating, two-tone color, and Interactive Pixel Lights), LED interior lamp (map lamp, personal lamp, sun visor lamp, and luggage lamp), Metallic door scuff plate • Seat: Synthetic leather seats, 1st-row manual seats, Heated 1st-row seats, 2nd-row 60/40-split folding seats(reclining) • Convenience: Proximity key with push-button start, Smart key remote start, Electronic Parking Brake(with automatic vehicle hold), Paddle shift(regenerative control), Dual-zone full automatic air conditioning(with high-performance antibacterial combination filter, auto defog system, fine dust sensor, air cleaning mode, and after-blow function), 2nd-row seat air vent, Auto light control system, USB Type-C Ports(1×27W switchable charging/data port in 1st-row, and 2×100W charging ports in both 1st and 2nd-row), ECM room mirror(frameless), Rain sensor, Power windows with pinch protection(1st/2nd-row), Power outlet (1 in 1st-row), Parking Distance Warning-Forward/Reverse, Rear View Monitor, Wireless phone charger(single), Walk-away lock, Route planner, Hyundai AI Assistant • Infotainment: 12.3-inch navigation(Bluelink, phone projection, Bluetooth hands-free, and In-car Payment), Audio system(6 speakers), Over-The-Air navigation updates|▶ Hi-pass(e hi-pass) [200,000]|
|
||||
|Exclusive Special|83,500,000 75,909,091(7,590,909) with 3.5% individual consumption tax applied 82,232,000|79,275,000 with 3.5% individual consumption tax applied 79,275,000|▶ Standard equipment of Exclusive plus • Smart Safety Technology: Forward Collision-avoidance Assist(intersection crossing/changing lanes in oncoming traffic/approaching from either side/ evasive steering assist), Highway Driving Assist 2, Navigation-based Smart Cruise Control(access road) • Exterior: Roof rack • Interior: Metallic pedal, Driving mode-dependent ambient mood lighting(crash pad, 1st/2nd-row door trim) • Seat: Synthetic leather seats(patch applied), Power-adjustable driver's seat(8-way, lumbar support, and Integrated Memory System(driver's seat and outside mirror connected)), Power-adjustable front passenger’s seat(8-way), Ventilated 1st-row seats, Heated 2nd-row seats • Convenience: Hi-pass(e hi-pass), In-car fingerprint authentication system(personalization, startup, payment, and etc.), Smart power tailgate|▶ [600,000] Built-in Cam 2 Plus, Augmented reality navigation ▶ [850,000] Indoor/outdoor V2L ▶ [950,000] Parking Assist ▶ [1,150,000] Audio by BANG & OLUFSEN sound system ▶ [250,000] 19-inch alloy wheels & tires|
|
||||
|Prestige|87,893,000 79,902,727(7,990,273) with 3.5% individual consumption tax applied 86,559,000|83,445,000 with 3.5% individual consumption tax applied 83,445,000|▶ Standard equipment of Exclusive Special plus • Smart Safety Technology: Remote Smart Parking Assist 2, Parking Collison- avoidance Assist(front/side/rear) • Exterior: Intelligent Front-Lighting System(IFS), Dynamic welcome/escort lighting(1 type), Sequential turn signals(front and rear), Ambient lighting auto flush door handles, Two-tone door garnish, Glossy black rear diffuser • Interior: Recycled PET suede interior materials(headlining/sunvisor), Fabric upholstered crash pad • Seat: BIO-processed natural leather seats(metal patch applied, embossed design punching), Passenger's seat walk-in device, 1st-row relaxation comfort seats(leg rest included), Ventilated 2nd-row seats • Convenience: Parking Distance Warning-Side, Head-Up Display, Digital key 2, Wireless phone charger(dual), Surround View Monitor, Blind-spot View Monitor, LED reverse light guide • Infotainment: Audio by BANG & OLUFSEN sound system(14 speakers, including external amp), Active road noise control, Active Sound Design|▶ [600,000] Built-in Cam 2 Plus, Augmented reality navigation ▶ [850,000] Indoor/outdoor V2L ▶ [900,000] Vision roof ▶ [1,380,000] Digital side mirror ▶ [750,000] Camera package ▶ [250,000] 19-inch alloy wheels & tires|
|
||||
|Special|with 3.5% individual consumption tax applied 79,287,000 83,500,000 75,909,091(7,590,909) with 3.5% individual consumption tax applied 82,232,000|with 3.5% individual consumption tax applied 76,435,000 79,275,000 with 3.5% individual consumption tax applied 79,275,000|and LED turn signal lamps), Auto flush door handles, Black door garnish • Interior: Panoramic curved display, 12.3-inch color LCD cluster, Leather- upholstered steering wheel(with heating, two-tone color, and Interactive Pixel Lights), LED interior lamp (map lamp, personal lamp, sun visor lamp, and luggage lamp), Metallic door scuff plate • Seat: Synthetic leather seats, 1st-row manual seats, Heated 1st-row seats, 2nd-row 60/40-split folding seats(reclining) • Convenience: Proximity key with push-button start, Smart key remote start, Electronic Parking Brake(with automatic vehicle hold), Paddle shift(regenerative control), Dual-zone full automatic air conditioning(with high-performance antibacterial combination filter, auto defog system, fine dust sensor, air cleaning mode, and after-blow function), 2nd-row seat air vent, Auto light control system, USB Type-C Ports(1×27W switchable charging/data port in 1st-row, and 2×100W charging ports in both 1st and 2nd-row), ECM room mirror(frameless), Rain sensor, Power windows with pinch protection(1st/2nd-row), Power outlet (1 in 1st-row), Parking Distance Warning-Forward/Reverse, Rear View Monitor, Wireless phone charger(single), Walk-away lock, Route planner, Hyundai AI Assistant • Infotainment: 12.3-inch navigation(Bluelink, phone projection, Bluetooth hands-free, and In-car Payment), Audio system(6 speakers), Over-The-Air navigation updates ▶ Standard equipment of Exclusive plus • Smart Safety Technology: Forward Collision-avoidance Assist(intersection crossing/changing lanes in oncoming traffic/approaching from either side/ evasive steering assist), Highway Driving Assist 2, Navigation-based Smart Cruise Control(access road) • Exterior: Roof rack • Interior: Metallic pedal, Driving mode-dependent ambient mood lighting(crash pad, 1st/2nd-row door trim) • Seat: Synthetic leather seats(patch applied), Power-adjustable driver's|▶ [600,000] Built-in Cam 2 Plus, Augmented reality navigation ▶ [850,000] Indoor/outdoor V2L ▶ [950,000] Parking Assist ▶ [1,150,000] Audio by BANG & OLUFSEN|
|
||||
|Prestige|87,893,000 79,902,727(7,990,273) with 3.5% individual consumption tax applied 86,559,000|83,445,000 with 3.5% individual consumption tax applied 83,445,000|seat(8-way, lumbar support, and Integrated Memory System(driver's seat and outside mirror connected)), Power-adjustable front passenger’s seat(8-way), Ventilated 1st-row seats, Heated 2nd-row seats • Convenience: Hi-pass(e hi-pass), In-car fingerprint authentication system(personalization, startup, payment, and etc.), Smart power tailgate ▶ Standard equipment of Exclusive Special plus • Smart Safety Technology: Remote Smart Parking Assist 2, Parking Collison- avoidance Assist(front/side/rear) • Exterior: Intelligent Front-Lighting System(IFS), Dynamic welcome/escort lighting(1 type), Sequential turn signals(front and rear), Ambient lighting auto flush door handles, Two-tone door garnish, Glossy black rear diffuser • Interior: Recycled PET suede interior materials(headlining/sunvisor), Fabric upholstered crash pad • Seat: BIO-processed natural leather seats(metal patch applied, embossed design punching), Passenger's seat walk-in device, 1st-row relaxation comfort seats(leg rest included), Ventilated 2nd-row seats • Convenience: Parking Distance Warning-Side, Head-Up Display, Digital key 2, Wireless phone charger(dual), Surround View Monitor, Blind-spot View Monitor, LED reverse light guide • Infotainment: Audio by BANG & OLUFSEN sound system(14 speakers, including external amp), Active road noise control, Active Sound Design|sound system ▶ [250,000] 19-inch alloy wheels & tires ▶ [600,000] Built-in Cam 2 Plus, Augmented reality navigation ▶ [850,000] Indoor/outdoor V2L ▶ [900,000] Vision roof ▶ [1,380,000] Digital side mirror ▶ [750,000] Camera package ▶ [250,000] 19-inch alloy wheels & tires|
|
||||
|
||||
**Classification Details** **Indoor/outdoor V2L** Indoor V2L, Outdoor V2L(connectorless type) **Parking Assist** Surround View Monitor, Blind-spot View Monitor, Parking Distance Warning-Side, Parking Collison-avoidance Assist-Rear **Audio by BANG & OLUFSEN** Audio by BANG & OLUFSEN sound system(14 speakers, including external amp.), Active road noise control, Active Sound Design **sound system** **Camera package** Digital center mirror(with camera sensor cleaning system), Driver monitoring system THE ALL-NEW NEXO /// ECO-FRIENDLY CAR
|
||||
|
||||
|
||||
+32
-20
@@ -13,10 +13,8 @@ IDENTIFICATION NUMBER OF CORPORATION] ELECTS TO BE TREATED AS A COMPONENT MEMBER
|
||||
(v) Election-- (A) Election filed. An election filed under paragraph (d)(2)(iv) of
|
||||
this section is irrevocable and effective until paragraph (d)(2)(ii) or (iii) of §1.1563-3 applies or until a change in the stock ownership of the corporation results in
|
||||
|
||||
|termination of membership in the controlled group in which such corporation has||
|
||||
|termination of membership in the controlled group in which such corporation has been included. (B) Election not filed.|In the event no election is filed in accordance with the|
|
||||
|---|---|
|
||||
|been included.||
|
||||
|(B) Election not filed.|In the event no election is filed in accordance with the|
|
||||
|provisions of paragraph (d)(2)(iv) of this section, then the Internal Revenue Service||
|
||||
|will determine the group in which such corporation is to be included. Such||
|
||||
|determination will be binding for all subsequent years unless the corporation files a||
|
||||
@@ -103,32 +101,45 @@ section and paragraph
|
||||
(c)(4), and (c)(5) of this section, and paragraph
|
||||
(c)(2) of §1.382-8T
|
||||
|
||||
||§1.382-8(g), Example|
|
||||
|---|---|
|
||||
||The first sentence of §1.382-8(g), Example §1.382-8(g), Example §1.382-8(g), Example|
|
||||
|
||||
(2)(c)
|
||||
(2)(e)
|
||||
(3)(b)
|
||||
(3)(c)(1)(B)
|
||||
|
||||
||The second sentence of|
|
||||
|---|---|
|
||||
||§1.382-8(g), Example The second sentence of §1.382-8(g), Example The first sentence of §1.1502-32(b)(4)(v)(A) The first sentence of §1.1502-32(b)(4)(v)(B)|
|
||||
|
||||
(4)(c)
|
||||
(5)(c)
|
||||
|
||||
|The fifth sentence of|paragraph (c) of this|paragraphs (c)(1), (c)(3),|
|
||||
|---|---|---|
|
||||
|§1.382-8(f)|section|(c)(4), and (c)(5) of this section, and paragraph (c)(2) of §1.382-8T|
|
||||
|§1.382-8(g), Example|paragraph (c) of this|paragraphs (c)(1), (c)(3), section, and paragraph (c)(2) of §1.382-8T|
|
||||
|The second sentence of|paragraph (c) of this|paragraphs (c)(1), (c)(3),|
|
||||
|§1.382-8(g), Example|section paragraph (c)(2) of this section paragraph (c)(2) of this section paragraph (c)(2) of this section paragraphs (c)(1) and (2) of this section paragraph (c)(2) of this section paragraph (c)(2) of this section paragraph (b)(4)(iv) of this section paragraph (b)(4)(iv) of this section|(c)(4), and (c)(5) of this (c)(2) of §1.382-8T paragraph (c)(2) of §1.382-8T paragraph (c)(2) of §1.382-8T paragraph (c)(2) of §1.382-8T paragraph (c)(1) of this section and paragraph (c)(2) of §1.382-8T paragraph (c)(2) of §1.382-8T paragraph (c)(2) of §1.382-8T paragraph (b)(4)(iv) of §1.1502-32T paragraph (b)(4)(iv) of §1.1502-32T|
|
||||
|§1.382-8(g), Example|section|(c)(4), and (c)(5) of this|
|
||||
|
||||
(1)(b)(2) section (c)(4), and (c)(5) of this
|
||||
(1)(c) section, and paragraph
|
||||
|
||||
(c)(2) of §1.382-8T
|
||||
|
||||
|§1.382-8(g), Example|paragraph (c)(2) of this|paragraph (c)(2) of|
|
||||
|---|---|---|
|
||||
|The first sentence of|paragraph (c)(2) of this|paragraph (c)(2) of|
|
||||
|§1.382-8(g), Example|section|§1.382-8T|
|
||||
|§1.382-8(g), Example|paragraph (c)(2) of this|paragraph (c)(2) of|
|
||||
|§1.382-8(g), Example|paragraphs (c)(1) and (2)|paragraph (c)(1) of this|
|
||||
|
||||
(2)(c) section §1.382-8T
|
||||
(2)(e)
|
||||
(3)(b) section §1.382-8T
|
||||
(3)(c)(1)(B) of this section section and paragraph
|
||||
|
||||
(c)(2) of §1.382-8T
|
||||
|
||||
|The second sentence of|paragraph (c)(2) of this|paragraph (c)(2) of|
|
||||
|---|---|---|
|
||||
|§1.382-8(g), Example|section|§1.382-8T|
|
||||
|The second sentence of|paragraph (c)(2) of this|paragraph (c)(2) of|
|
||||
|§1.382-8(g), Example|section|§1.382-8T|
|
||||
|The first sentence of|paragraph (b)(4)(iv) of|paragraph (b)(4)(iv) of|
|
||||
|§1.1502-32(b)(4)(v)(A)|this section|§1.1502-32T|
|
||||
|The first sentence of|paragraph (b)(4)(iv) of|paragraph (b)(4)(iv) of|
|
||||
|§1.1502-32(b)(4)(v)(B)|this section|§1.1502-32T|
|
||||
|
||||
(4)(c)
|
||||
(5)(c)
|
||||
|
||||
|§1.1502-35(c)(4)(ii)(B)|§1.1502-76(b)(2)(ii)(D)|§1.1502-76T(b)(2)(ii)(D)|
|
||||
|---|---|---|
|
||||
|§1.1502-76(b)(2)(ii)(A)(2)|paragraph (b)(2)(ii)(D) of this section|paragraph (b)(2)(ii)(D) of §1.1502-76T|
|
||||
@@ -182,3 +193,4 @@ section and paragraph
|
||||
Deputy Commissioner for Services and Enforcement.
|
||||
|
||||
Approved: May 19, 2006 Eric Solomon Acting Deputy Assistant Secretary of the Treasury (Tax Policy).
|
||||
|
||||
|
||||
@@ -270,78 +270,6 @@ class TestExtractTextInRegions:
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# extract_pages_markdown / extract_pages_markdown_bytes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class TestExtractPagesMarkdown:
|
||||
def test_default_returns_all_pages(self):
|
||||
result = pdf_inspector.extract_pages_markdown(
|
||||
fixture_path("thermo-freon12.pdf")
|
||||
)
|
||||
assert len(result.pages) == 3
|
||||
assert [p.page for p in result.pages] == [0, 1, 2]
|
||||
assert all(isinstance(p.markdown, str) for p in result.pages)
|
||||
|
||||
def test_bytes_default_returns_all_pages(self):
|
||||
data = fixture_bytes("thermo-freon12.pdf")
|
||||
result = pdf_inspector.extract_pages_markdown_bytes(data)
|
||||
assert len(result.pages) == 3
|
||||
|
||||
def test_selected_pages_preserve_order(self):
|
||||
result = pdf_inspector.extract_pages_markdown(
|
||||
fixture_path("thermo-freon12.pdf"), pages=[2, 0]
|
||||
)
|
||||
assert [p.page for p in result.pages] == [2, 0]
|
||||
|
||||
def test_bytes_selected_pages_preserve_order(self):
|
||||
data = fixture_bytes("thermo-freon12.pdf")
|
||||
result = pdf_inspector.extract_pages_markdown_bytes(data, pages=[1])
|
||||
assert len(result.pages) == 1
|
||||
assert result.pages[0].page == 1
|
||||
|
||||
def test_page_fields(self):
|
||||
result = pdf_inspector.extract_pages_markdown(
|
||||
fixture_path("thermo-freon12.pdf"), pages=[0]
|
||||
)
|
||||
page = result.pages[0]
|
||||
assert isinstance(page.page, int)
|
||||
assert isinstance(page.markdown, str)
|
||||
assert isinstance(page.needs_ocr, bool)
|
||||
assert not page.needs_ocr # text-based fixture
|
||||
assert len(page.markdown) > 0
|
||||
|
||||
def test_result_fields(self):
|
||||
result = pdf_inspector.extract_pages_markdown(
|
||||
fixture_path("thermo-freon12.pdf")
|
||||
)
|
||||
assert isinstance(result.pages, list)
|
||||
assert isinstance(result.pages_with_tables, list)
|
||||
assert isinstance(result.pages_with_columns, list)
|
||||
assert isinstance(result.pages_needing_ocr, list)
|
||||
assert isinstance(result.is_complex, bool)
|
||||
|
||||
def test_out_of_range_page_marks_needs_ocr(self):
|
||||
result = pdf_inspector.extract_pages_markdown(
|
||||
fixture_path("thermo-freon12.pdf"), pages=[9999]
|
||||
)
|
||||
assert len(result.pages) == 1
|
||||
assert result.pages[0].needs_ocr
|
||||
assert result.pages[0].markdown == ""
|
||||
|
||||
def test_repr(self):
|
||||
result = pdf_inspector.extract_pages_markdown(
|
||||
fixture_path("thermo-freon12.pdf"), pages=[0]
|
||||
)
|
||||
assert "PagesExtractionResult" in repr(result)
|
||||
assert "PageMarkdown" in repr(result.pages[0])
|
||||
|
||||
def test_not_a_pdf(self):
|
||||
with pytest.raises(ValueError):
|
||||
pdf_inspector.extract_pages_markdown_bytes(b"not a pdf")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Error handling
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
Reference in New Issue
Block a user