Compare commits
10
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
051253329f | ||
|
|
2c24e4c62c | ||
|
|
993d2a865d | ||
|
|
ec6e54afb8 | ||
|
|
89dd20d02c | ||
|
|
3d33ff3dbd | ||
|
|
75e9b09593 | ||
|
|
f4aab3b36f | ||
|
|
f65b25906c | ||
|
|
9947485a92 |
+1
-1
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "pdf-inspector"
|
||||
version = "1.14.0"
|
||||
version = "1.14.1"
|
||||
edition = "2021"
|
||||
autobins = false
|
||||
authors = ["Firecrawl Team"]
|
||||
|
||||
@@ -80,6 +80,17 @@ for page in result.pages:
|
||||
|
||||
# Restrict to specific 0-indexed pages (preserves caller order)
|
||||
result = pdf_inspector.extract_pages_markdown("document.pdf", pages=[0, 2])
|
||||
|
||||
# Structure-tree elements from tagged PDFs (empty list when untagged).
|
||||
# Pages are 1-indexed to match TextItem.page, so (page, mcid) joins directly
|
||||
# against extract_text_with_positions — e.g. to recover real heading levels:
|
||||
elements = pdf_inspector.extract_structure_elements("tagged.pdf")
|
||||
roles = {(e.page, e.mcid): e.role for e in elements}
|
||||
headings = [
|
||||
item.text
|
||||
for item in pdf_inspector.extract_text_with_positions("tagged.pdf")
|
||||
if item.mcid is not None and roles.get((item.page, item.mcid), "").startswith("H")
|
||||
]
|
||||
```
|
||||
|
||||
## API reference
|
||||
@@ -100,6 +111,8 @@ result = pdf_inspector.extract_pages_markdown("document.pdf", pages=[0, 2])
|
||||
| `extract_text_in_regions_bytes(data, page_regions)` | Region extraction from bytes |
|
||||
| `extract_pages_markdown(path, pages=None)` | Per-page Markdown + layout metadata (all pages by default) |
|
||||
| `extract_pages_markdown_bytes(data, pages=None)` | Per-page Markdown from bytes |
|
||||
| `extract_structure_elements(path, pages=None)` | Structure-tree elements from tagged PDFs (page, mcid, role) |
|
||||
| `extract_structure_elements_bytes(data, pages=None)` | Structure-tree elements from bytes |
|
||||
|
||||
## Types
|
||||
|
||||
@@ -144,6 +157,12 @@ class TextItem: # extract_text_with_positions
|
||||
is_underline: bool
|
||||
is_strikeout: bool
|
||||
item_type: str
|
||||
mcid: int | None # marked-content ID for tagged PDFs (None otherwise)
|
||||
|
||||
class StructureElement: # extract_structure_elements
|
||||
page: int # 1-indexed (matches TextItem.page)
|
||||
mcid: int
|
||||
role: str # "H1".."H6", "P", "Table", ... (resolved via /RoleMap)
|
||||
|
||||
class RegionText: # extract_text_in_regions
|
||||
text: str
|
||||
|
||||
+32
-1
@@ -138,6 +138,34 @@ for page in &result.pages {
|
||||
println!("Complex layout? {}", result.is_complex);
|
||||
```
|
||||
|
||||
Extract structure-tree elements from tagged PDFs, and join them against
|
||||
`extract_text_with_positions` to attach semantic roles (heading levels,
|
||||
paragraphs, table cells) to extracted text:
|
||||
|
||||
```rust
|
||||
use pdf_inspector::{extract_structure_elements, extract_text_with_positions};
|
||||
use std::collections::HashMap;
|
||||
|
||||
// One entry per marked-content reference, sorted by (page, mcid); empty for
|
||||
// untagged PDFs. Pages are 1-indexed to match `TextItem::page`, so the
|
||||
// (page, mcid) pair is a direct join key.
|
||||
let elements = extract_structure_elements("tagged.pdf", None)?;
|
||||
let roles: HashMap<(u32, i64), &str> = elements
|
||||
.iter()
|
||||
.map(|e| ((e.page, e.mcid), e.role.as_str()))
|
||||
.collect();
|
||||
|
||||
for item in extract_text_with_positions("tagged.pdf")? {
|
||||
if let Some(mcid) = item.mcid {
|
||||
if let Some(role) = roles.get(&(item.page, mcid)) {
|
||||
if role.starts_with('H') {
|
||||
println!("{}: {}", role, item.text);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
## Processing modes
|
||||
|
||||
| Mode | What it does | Returns |
|
||||
@@ -163,6 +191,8 @@ println!("Complex layout? {}", result.is_complex);
|
||||
| `to_markdown_from_items_with_rects(items, options, rects)` | Markdown with rectangle-based table detection |
|
||||
| `extract_pages_markdown(path, pages)` | Per-page Markdown + layout metadata (file) |
|
||||
| `extract_pages_markdown_mem(bytes, pages)` | Per-page Markdown from bytes |
|
||||
| `extract_structure_elements(path, pages)` | Structure-tree elements from tagged PDFs (page, mcid, role) |
|
||||
| `extract_structure_elements_mem(bytes, pages)` | Structure-tree elements from bytes |
|
||||
|
||||
Low-level detection functions are also available via the `detector` module (`detect_pdf_type`, `detect_pdf_type_with_config`, etc.) for callers who need `PdfTypeResult` instead of `PdfProcessResult`.
|
||||
|
||||
@@ -178,7 +208,8 @@ Low-level detection functions are also available via the `detector` module (`det
|
||||
| `DetectionConfig` | Configuration for detection: scan strategy, thresholds |
|
||||
| `ScanStrategy` | `EarlyExit`, `Full`, `Sample(n)`, `Pages(vec)` |
|
||||
| `LayoutComplexity` | Layout analysis: is_complex, pages_with_tables, pages_with_columns |
|
||||
| `TextItem` | Text with position, font info, and page number |
|
||||
| `TextItem` | Text with position, font info, page number, and optional structure-tree `mcid` |
|
||||
| `StructureElement` | Tagged-PDF structure reference: page (1-indexed), mcid, role (`"H1"`..`"H6"`, `"P"`, …) |
|
||||
| `MarkdownOptions` | Configuration for Markdown formatting (page numbers, etc.) |
|
||||
| `PageMarkdown` | Per-page result: page (0-indexed), markdown, needs_ocr |
|
||||
| `PagesExtractionResult` | Per-page output + 1-indexed pages_with_tables / pages_with_columns / pages_needing_ocr, is_complex |
|
||||
|
||||
Generated
+2
-2
@@ -851,7 +851,7 @@ checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe"
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector"
|
||||
version = "1.14.0"
|
||||
version = "1.14.1"
|
||||
dependencies = [
|
||||
"env_logger",
|
||||
"include_dir",
|
||||
@@ -867,7 +867,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector-napi"
|
||||
version = "1.14.0"
|
||||
version = "1.14.1"
|
||||
dependencies = [
|
||||
"napi",
|
||||
"napi-build",
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "pdf-inspector-napi"
|
||||
version = "1.14.0"
|
||||
version = "1.14.1"
|
||||
edition = "2021"
|
||||
|
||||
[lib]
|
||||
|
||||
+6
-6
@@ -8,12 +8,12 @@
|
||||
"@napi-rs/cli": "^3.4.1",
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.14.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.14.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.14.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.14.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-musl": "1.14.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.14.0",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.14.1",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.14.1",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.14.1",
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.14.1",
|
||||
"@firecrawl/pdf-inspector-linux-x64-musl": "1.14.1",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.14.1",
|
||||
},
|
||||
},
|
||||
},
|
||||
|
||||
+7
-7
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@firecrawl/pdf-inspector",
|
||||
"version": "1.14.0",
|
||||
"version": "1.14.1",
|
||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||
"main": "index.js",
|
||||
"types": "index.d.ts",
|
||||
@@ -52,11 +52,11 @@
|
||||
"@napi-rs/cli": "^3.4.1"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.14.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-musl": "1.14.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.14.0",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.14.0",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.14.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.14.0"
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.14.1",
|
||||
"@firecrawl/pdf-inspector-linux-x64-musl": "1.14.1",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-gnu": "1.14.1",
|
||||
"@firecrawl/pdf-inspector-linux-arm64-musl": "1.14.1",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.14.1",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.14.1"
|
||||
}
|
||||
}
|
||||
|
||||
+1
-1
@@ -6,7 +6,7 @@ build-backend = "maturin"
|
||||
name = "pdf-inspector"
|
||||
# Keep package versions in sync with `python3 scripts/version.py <version>`.
|
||||
# CI publishes automatically when the synchronized change lands on main.
|
||||
version = "1.14.0"
|
||||
version = "1.14.1"
|
||||
description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection"
|
||||
readme = "docs/python.md"
|
||||
license = { text = "MIT" }
|
||||
|
||||
+1
-1
@@ -975,7 +975,7 @@ result = pdf_inspector.<span class="fn">process_pdf</span>(<span class="str">"do
|
||||
<script>
|
||||
(() => {
|
||||
const MAX_FILE_SIZE = 25 * 1024 * 1024;
|
||||
const WASM_MODULE_URL = "https://cdn.jsdelivr.net/npm/@firecrawl/pdf-inspector-wasm@1.14.0/pdf_inspector_wasm.js";
|
||||
const WASM_MODULE_URL = "https://cdn.jsdelivr.net/npm/@firecrawl/pdf-inspector-wasm@1.14.1/pdf_inspector_wasm.js";
|
||||
const input = document.querySelector("#pdf-input");
|
||||
const dropZone = document.querySelector("#drop-zone");
|
||||
const filePanel = document.querySelector("#demo-file");
|
||||
|
||||
@@ -19,7 +19,7 @@ use super::fonts::{
|
||||
CMapDecisionCache, FontStyleCache,
|
||||
};
|
||||
use super::underline::UnderlineLine;
|
||||
use super::xobjects::{extract_form_xobject_text, get_page_xobjects, XObjectType};
|
||||
use super::xobjects::{extract_form_xobject_text, get_page_xobjects, FormWalkBudget, XObjectType};
|
||||
use super::{get_number, image_bbox_from_ctm, multiply_matrices};
|
||||
|
||||
/// Strip PDF comments (% to end of line) from content stream bytes.
|
||||
@@ -137,8 +137,11 @@ fn rise_adjusted(tm: &[f32; 6], rise: f32) -> [f32; 6] {
|
||||
]
|
||||
}
|
||||
|
||||
/// Returns `(page_extraction, has_gid_fonts)` where `has_gid_fonts` indicates
|
||||
/// the page uses fonts with unresolvable gid-encoded glyphs.
|
||||
/// Returns `(page_extraction, has_gid_fonts, coords_rotated, skipped_invisible)`
|
||||
/// where `has_gid_fonts` indicates the page uses fonts with unresolvable
|
||||
/// gid-encoded glyphs and `skipped_invisible` reports that invisible (Tr 3)
|
||||
/// text was present but suppressed — callers can use it to decide whether an
|
||||
/// `include_invisible` retry could recover anything at all.
|
||||
pub(crate) fn extract_page_text_items(
|
||||
doc: &Document,
|
||||
page_id: ObjectId,
|
||||
@@ -146,7 +149,8 @@ pub(crate) fn extract_page_text_items(
|
||||
font_cmaps: &FontCMaps,
|
||||
include_invisible: bool,
|
||||
style_cache: &mut FontStyleCache,
|
||||
) -> Result<(PageExtraction, bool, bool), PdfError> {
|
||||
form_budget: &mut FormWalkBudget,
|
||||
) -> Result<(PageExtraction, bool, bool, bool), PdfError> {
|
||||
use lopdf::content::Content;
|
||||
|
||||
let mut items = Vec::new();
|
||||
@@ -262,12 +266,15 @@ pub(crate) fn extract_page_text_items(
|
||||
content.operations.len(),
|
||||
MAX_OPERATIONS
|
||||
);
|
||||
return Ok(((Vec::new(), Vec::new(), Vec::new()), false, false));
|
||||
return Ok(((Vec::new(), Vec::new(), Vec::new()), false, false, false));
|
||||
}
|
||||
|
||||
// Graphics state tracking
|
||||
let mut ctm = [1.0f32, 0.0, 0.0, 1.0, 0.0, 0.0]; // Current Transformation Matrix
|
||||
let mut text_rendering_mode: i32 = 0; // 0=fill, 1=stroke, 2=fill+stroke, 3=invisible
|
||||
// Invisible (Tr 3) text was present but suppressed — reported to callers
|
||||
// so an include_invisible retry is attempted only when it can recover.
|
||||
let mut skipped_invisible = false;
|
||||
let mut line_width: f32 = 1.0;
|
||||
#[derive(Clone)]
|
||||
struct SavedGraphicsState {
|
||||
@@ -494,6 +501,14 @@ pub(crate) fn extract_page_text_items(
|
||||
// For Mixed/template PDFs, include_invisible=true extracts
|
||||
// the OCR text layer that sits behind scanned images.
|
||||
if text_rendering_mode == 3 && !include_invisible {
|
||||
if op
|
||||
.operands
|
||||
.first()
|
||||
.and_then(get_operand_bytes)
|
||||
.is_some_and(|raw| !raw.is_empty())
|
||||
{
|
||||
skipped_invisible = true;
|
||||
}
|
||||
if let Some(w_ts) = w_ts_opt {
|
||||
text_matrix[4] += w_ts * text_matrix[0];
|
||||
text_matrix[5] += w_ts * text_matrix[1];
|
||||
@@ -565,6 +580,16 @@ pub(crate) fn extract_page_text_items(
|
||||
if in_text_block && !op.operands.is_empty() {
|
||||
if let Ok(array) = op.operands[0].as_array() {
|
||||
let font_info = font_widths.get(¤t_font);
|
||||
// Numeric-only TJ arrays (pure kerning) show no
|
||||
// text — they must not trigger the invisible retry.
|
||||
if text_rendering_mode == 3
|
||||
&& !include_invisible
|
||||
&& array
|
||||
.iter()
|
||||
.any(|el| get_operand_bytes(el).is_some_and(|raw| !raw.is_empty()))
|
||||
{
|
||||
skipped_invisible = true;
|
||||
}
|
||||
let is_invisible = (text_rendering_mode == 3 && !include_invisible)
|
||||
|| suppress_glyph_extraction;
|
||||
// Capture first-glyph position for ActualText
|
||||
@@ -770,6 +795,16 @@ pub(crate) fn extract_page_text_items(
|
||||
)
|
||||
})
|
||||
});
|
||||
if text_rendering_mode == 3
|
||||
&& !include_invisible
|
||||
&& op
|
||||
.operands
|
||||
.first()
|
||||
.and_then(get_operand_bytes)
|
||||
.is_some_and(|raw| !raw.is_empty())
|
||||
{
|
||||
skipped_invisible = true;
|
||||
}
|
||||
if !((text_rendering_mode == 3 && !include_invisible)
|
||||
|| suppress_glyph_extraction
|
||||
|| op.operands.is_empty())
|
||||
@@ -882,6 +917,7 @@ pub(crate) fn extract_page_text_items(
|
||||
&ctm,
|
||||
&mut cmap_decisions,
|
||||
style_cache,
|
||||
form_budget,
|
||||
);
|
||||
items.extend(form_items);
|
||||
}
|
||||
@@ -1251,6 +1287,12 @@ pub(crate) fn extract_page_text_items(
|
||||
}
|
||||
}
|
||||
|
||||
if form_budget.was_truncated() {
|
||||
log::warn!(
|
||||
"page {page_num}: Form XObject expansion truncated (invocation or operation budget reached); nested form text may be incomplete"
|
||||
);
|
||||
}
|
||||
|
||||
// Underline detection reads only painted ink: `re` rects confirmed by
|
||||
// a paint operator plus filled-subpath rects — never clip-only rects,
|
||||
// which draw nothing.
|
||||
@@ -1300,7 +1342,12 @@ pub(crate) fn extract_page_text_items(
|
||||
|
||||
let items = super::merge_text_items(items);
|
||||
let items = super::merge_subscript_items(items);
|
||||
Ok(((items, rects, lines), has_gid_fonts, coords_rotated))
|
||||
Ok((
|
||||
(items, rects, lines),
|
||||
has_gid_fonts,
|
||||
coords_rotated,
|
||||
skipped_invisible,
|
||||
))
|
||||
}
|
||||
|
||||
/// Counts of text operators with horizontal vs rotated combined matrices.
|
||||
@@ -1498,13 +1545,14 @@ mod tests {
|
||||
|
||||
let (doc, page_id) = simple_doc_with_content(content);
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let ((items, _, _), _, _) = extract_page_text_items(
|
||||
let ((items, _, _), _, _, _) = extract_page_text_items(
|
||||
&doc,
|
||||
page_id,
|
||||
1,
|
||||
&font_cmaps,
|
||||
false,
|
||||
&mut FontStyleCache::new(),
|
||||
&mut FormWalkBudget::new(),
|
||||
)
|
||||
.unwrap();
|
||||
items
|
||||
@@ -1734,9 +1782,10 @@ BT /F1 12 Tf 0 1 -1 0 240 100 Tm (WORLD) Tj ET
|
||||
&font_cmaps,
|
||||
false,
|
||||
&mut FontStyleCache::new(),
|
||||
&mut FormWalkBudget::new(),
|
||||
)
|
||||
.unwrap();
|
||||
let ((items, rects, lines), _has_gid, _coords_rotated) = result;
|
||||
let ((items, rects, lines), _has_gid, _coords_rotated, _skipped_invisible) = result;
|
||||
assert!(items.is_empty());
|
||||
assert!(rects.is_empty());
|
||||
assert!(lines.is_empty());
|
||||
@@ -1817,13 +1866,14 @@ BT 30 700 Tm <41> Tj ET";
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let ((items, _, _), _, _) = extract_page_text_items(
|
||||
let ((items, _, _), _, _, _) = extract_page_text_items(
|
||||
&doc,
|
||||
page_id,
|
||||
1,
|
||||
&font_cmaps,
|
||||
false,
|
||||
&mut FontStyleCache::new(),
|
||||
&mut FormWalkBudget::new(),
|
||||
)
|
||||
.unwrap();
|
||||
let text = items
|
||||
|
||||
+107
-16
@@ -482,7 +482,11 @@ pub(crate) fn parse_cid_w_array(
|
||||
widths: &mut HashMap<u16, u16>,
|
||||
) {
|
||||
let mut i = 0;
|
||||
let mut assigned = 0usize;
|
||||
while i < w_array.len() {
|
||||
if assigned >= crate::tounicode::MAX_CID_W_EXPANSION {
|
||||
return;
|
||||
}
|
||||
let start_cid = match &w_array[i] {
|
||||
Object::Integer(n) => *n as u16,
|
||||
Object::Real(n) => *n as u16,
|
||||
@@ -501,12 +505,14 @@ pub(crate) fn parse_cid_w_array(
|
||||
Object::Array(arr) => {
|
||||
// [c [w1 w2 ...]] — consecutive widths starting at c
|
||||
for (j, w_obj) in arr.iter().enumerate() {
|
||||
let w = match w_obj {
|
||||
Object::Integer(n) => *n as u16,
|
||||
Object::Real(n) => *n as u16,
|
||||
_ => continue,
|
||||
};
|
||||
widths.insert(start_cid + j as u16, w);
|
||||
if !assign_cid_width(
|
||||
widths,
|
||||
start_cid.wrapping_add(j as u16),
|
||||
w_obj,
|
||||
&mut assigned,
|
||||
) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
i += 1;
|
||||
}
|
||||
@@ -514,12 +520,14 @@ pub(crate) fn parse_cid_w_array(
|
||||
// Could be a reference to an array
|
||||
if let Ok(Object::Array(arr)) = doc.get_object(*r) {
|
||||
for (j, w_obj) in arr.iter().enumerate() {
|
||||
let w = match w_obj {
|
||||
Object::Integer(n) => *n as u16,
|
||||
Object::Real(n) => *n as u16,
|
||||
_ => continue,
|
||||
};
|
||||
widths.insert(start_cid + j as u16, w);
|
||||
if !assign_cid_width(
|
||||
widths,
|
||||
start_cid.wrapping_add(j as u16),
|
||||
w_obj,
|
||||
&mut assigned,
|
||||
) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
i += 1;
|
||||
} else {
|
||||
@@ -542,8 +550,8 @@ pub(crate) fn parse_cid_w_array(
|
||||
continue;
|
||||
}
|
||||
};
|
||||
for cid in start_cid..=end {
|
||||
widths.insert(cid, w);
|
||||
if !assign_cid_width_range(widths, start_cid, end, w, &mut assigned) {
|
||||
return;
|
||||
}
|
||||
i += 1;
|
||||
}
|
||||
@@ -561,8 +569,8 @@ pub(crate) fn parse_cid_w_array(
|
||||
continue;
|
||||
}
|
||||
};
|
||||
for cid in start_cid..=end {
|
||||
widths.insert(cid, w);
|
||||
if !assign_cid_width_range(widths, start_cid, end, w, &mut assigned) {
|
||||
return;
|
||||
}
|
||||
i += 1;
|
||||
}
|
||||
@@ -573,6 +581,45 @@ pub(crate) fn parse_cid_w_array(
|
||||
}
|
||||
}
|
||||
|
||||
fn assign_cid_width(
|
||||
widths: &mut HashMap<u16, u16>,
|
||||
cid: u16,
|
||||
w_obj: &Object,
|
||||
assigned: &mut usize,
|
||||
) -> bool {
|
||||
let w = match w_obj {
|
||||
Object::Integer(n) => *n as u16,
|
||||
Object::Real(n) => *n as u16,
|
||||
_ => return true,
|
||||
};
|
||||
if *assigned >= crate::tounicode::MAX_CID_W_EXPANSION {
|
||||
return false;
|
||||
}
|
||||
widths.insert(cid, w);
|
||||
*assigned += 1;
|
||||
true
|
||||
}
|
||||
|
||||
fn assign_cid_width_range(
|
||||
widths: &mut HashMap<u16, u16>,
|
||||
start: u16,
|
||||
end: u16,
|
||||
w: u16,
|
||||
assigned: &mut usize,
|
||||
) -> bool {
|
||||
if start > end {
|
||||
return true;
|
||||
}
|
||||
for cid in start..=end {
|
||||
if *assigned >= crate::tounicode::MAX_CID_W_EXPANSION {
|
||||
return false;
|
||||
}
|
||||
widths.insert(cid, w);
|
||||
*assigned += 1;
|
||||
}
|
||||
true
|
||||
}
|
||||
|
||||
/// Compute the width of a string in text space units,
|
||||
/// given raw bytes and font width info.
|
||||
/// Returns width in text space units (font_units * units_scale * font_size).
|
||||
@@ -2313,4 +2360,48 @@ end",
|
||||
// invalid CMap result — so it must not clear the gid flag.
|
||||
assert!(gid_flagged(Some("<01> <FFFD>\n<02> <FFFD>")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_cid_w_array_range_and_consecutive() {
|
||||
use super::parse_cid_w_array;
|
||||
use lopdf::{Document, Object};
|
||||
use std::collections::HashMap;
|
||||
|
||||
let doc = Document::new();
|
||||
let mut widths = HashMap::new();
|
||||
let w = vec![
|
||||
Object::Integer(10),
|
||||
Object::Integer(12),
|
||||
Object::Integer(500),
|
||||
Object::Integer(20),
|
||||
Object::Array(vec![Object::Integer(100), Object::Integer(200)]),
|
||||
];
|
||||
parse_cid_w_array(&doc, &w, &mut widths);
|
||||
assert_eq!(widths.get(&10), Some(&500));
|
||||
assert_eq!(widths.get(&11), Some(&500));
|
||||
assert_eq!(widths.get(&12), Some(&500));
|
||||
assert_eq!(widths.get(&20), Some(&100));
|
||||
assert_eq!(widths.get(&21), Some(&200));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_cid_w_array_repeated_full_ranges_stay_bounded() {
|
||||
use super::parse_cid_w_array;
|
||||
use crate::tounicode::MAX_CID_W_EXPANSION;
|
||||
use lopdf::{Document, Object};
|
||||
use std::collections::HashMap;
|
||||
|
||||
let doc = Document::new();
|
||||
let mut widths = HashMap::new();
|
||||
let mut w = Vec::new();
|
||||
for _ in 0..5_000 {
|
||||
w.push(Object::Integer(0));
|
||||
w.push(Object::Integer(65535));
|
||||
w.push(Object::Integer(500));
|
||||
}
|
||||
parse_cid_w_array(&doc, &w, &mut widths);
|
||||
assert!(widths.len() <= MAX_CID_W_EXPANSION);
|
||||
assert_eq!(widths.get(&0), Some(&500));
|
||||
assert_eq!(widths.get(&65535), Some(&500));
|
||||
}
|
||||
}
|
||||
|
||||
+252
-14
@@ -37,6 +37,7 @@ pub(crate) use layout::group_prefiltered_items_into_lines_with_thresholds_and_re
|
||||
pub(crate) use layout::is_newspaper_layout;
|
||||
pub(crate) use layout::ColumnRegion;
|
||||
pub use layout::{group_into_lines, group_into_lines_preserving_all_text};
|
||||
pub(crate) use xobjects::FormWalkBudget;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Public API
|
||||
@@ -282,18 +283,22 @@ fn extract_positioned_text_impl(
|
||||
font_cmaps,
|
||||
include_invisible,
|
||||
&mut style_cache,
|
||||
&mut FormWalkBudget::new(),
|
||||
);
|
||||
let ((mut items, mut rects, mut lines), has_gid_fonts, coords_rotated) = match page_result {
|
||||
Ok(extraction) => extraction,
|
||||
Err(error) if required_pages.is_some_and(|required| !required.contains(page_num)) => {
|
||||
debug!(
|
||||
"page {}: skipping context-only extraction error: {}",
|
||||
page_num, error
|
||||
);
|
||||
continue;
|
||||
}
|
||||
Err(error) => return Err(error),
|
||||
};
|
||||
let ((mut items, mut rects, mut lines), has_gid_fonts, coords_rotated, _skipped_invisible) =
|
||||
match page_result {
|
||||
Ok(extraction) => extraction,
|
||||
Err(error)
|
||||
if required_pages.is_some_and(|required| !required.contains(page_num)) =>
|
||||
{
|
||||
debug!(
|
||||
"page {}: skipping context-only extraction error: {}",
|
||||
page_num, error
|
||||
);
|
||||
continue;
|
||||
}
|
||||
Err(error) => return Err(error),
|
||||
};
|
||||
// Clip to the visible page box: single-page extracts and imposed
|
||||
// spreads keep neighboring pages' content in the stream, positioned
|
||||
// outside the CropBox. Extracting it interleaves invisible text into
|
||||
@@ -884,6 +889,92 @@ fn tracked_run_space_floor(group: &[&TextItem], start: usize) -> Option<(usize,
|
||||
Some((end, floor * fs))
|
||||
}
|
||||
|
||||
/// Fractional font-size band within which `merge_text_items` treats two runs as
|
||||
/// the same size. Shared with `is_small_caps_continuation`, which exists only to
|
||||
/// rescue junctions this band would otherwise break.
|
||||
const MERGE_FONT_SIZE_BAND: f32 = 0.20;
|
||||
|
||||
/// Detect a small-caps continuation: typesetters render small caps as a
|
||||
/// full-size capital immediately followed by shrunken capitals in the same
|
||||
/// font (`(R) Tj` at 9.98pt, then `(OLANDO) Tj` at 6.74pt). Those runs are one
|
||||
/// word, but the font-size band in `merge_text_items` would split them,
|
||||
/// leaving "R" and "OLANDO" as separate items — which then read as separate
|
||||
/// table columns, since column boundaries cluster on item start positions.
|
||||
///
|
||||
/// Gated tightly so it cannot absorb the other reasons a smaller run follows a
|
||||
/// larger one:
|
||||
/// - runs the size band already accepts — excluded by requiring the junction
|
||||
/// to *cross* the band, so within-band pairs keep the normal word-spacing
|
||||
/// logic instead of having their space suppressed
|
||||
/// - superscripts / footnote markers — excluded by requiring an uppercase
|
||||
/// *letter* on both sides, so digits never qualify
|
||||
/// - drop caps — excluded because the body text that follows is mixed case
|
||||
/// - adjacent table cells or separate words — excluded by requiring the runs
|
||||
/// to be visually contiguous (essentially no gap)
|
||||
fn is_small_caps_continuation(
|
||||
text_so_far: &str,
|
||||
first: &TextItem,
|
||||
next: &TextItem,
|
||||
gap: f32,
|
||||
) -> bool {
|
||||
// Must shrink. Real small caps sit near 0.7-0.8 of the full cap height;
|
||||
// anything smaller is a superscript or a different run entirely.
|
||||
if first.font_size <= 0.0 || next.font_size >= first.font_size {
|
||||
return false;
|
||||
}
|
||||
// Only rescue junctions the size band would have broken. Within-band pairs
|
||||
// merge on their own, and suppressing their space would swallow real word
|
||||
// gaps between two similarly-sized uppercase words.
|
||||
if (next.font_size - first.font_size).abs() <= first.font_size * MERGE_FONT_SIZE_BAND {
|
||||
return false;
|
||||
}
|
||||
if next.font_size / first.font_size < 0.55 {
|
||||
return false;
|
||||
}
|
||||
// Visually contiguous: the capital and its small caps touch. A real word
|
||||
// space or a column gap disqualifies.
|
||||
if !(-first.font_size * 0.2..=first.font_size * 0.15).contains(&gap) {
|
||||
return false;
|
||||
}
|
||||
// The continuation must be all-uppercase letters (digits and lowercase
|
||||
// both disqualify), and must contain at least one letter.
|
||||
let mut saw_letter = false;
|
||||
for ch in next.text.chars() {
|
||||
if ch.is_alphabetic() {
|
||||
saw_letter = true;
|
||||
if !ch.is_uppercase() {
|
||||
return false;
|
||||
}
|
||||
} else if ch.is_numeric() {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if !saw_letter {
|
||||
return false;
|
||||
}
|
||||
// What we are continuing must itself end in a capital. Check the actual
|
||||
// trailing character rather than skipping back to the nearest letter: after
|
||||
// "ANGELA M. MAZZARELLI1" the run to continue is the footnote marker, not
|
||||
// the "I" before it.
|
||||
let trimmed = text_so_far.trim_end();
|
||||
if trimmed.chars().last().is_some_and(|c| c.is_numeric()) {
|
||||
// One legitimate exception: an ordinal suffix set as a smaller run,
|
||||
// e.g. "JULY 4" + "TH". Only the four English suffixes qualify —
|
||||
// anything else after a digit is a footnote marker or numeric suffix.
|
||||
return matches!(trimmed_suffix(next), "TH" | "ST" | "ND" | "RD");
|
||||
}
|
||||
trimmed
|
||||
.chars()
|
||||
.rev()
|
||||
.find(|c| c.is_alphabetic())
|
||||
.is_some_and(|c| c.is_uppercase())
|
||||
}
|
||||
|
||||
/// The continuation run's text, trimmed — used to spot ordinal suffixes.
|
||||
fn trimmed_suffix(next: &TextItem) -> &str {
|
||||
next.text.trim()
|
||||
}
|
||||
|
||||
pub(crate) fn merge_text_items(items: Vec<TextItem>) -> Vec<TextItem> {
|
||||
if items.is_empty() {
|
||||
return items;
|
||||
@@ -942,8 +1033,16 @@ pub(crate) fn merge_text_items(items: Vec<TextItem>) -> Vec<TextItem> {
|
||||
let mut j = i + 1;
|
||||
while j < group.len() {
|
||||
let next = group[j];
|
||||
// Must be similar font size (within 20%)
|
||||
if (next.font_size - first.font_size).abs() > first.font_size * 0.20 {
|
||||
// A small-caps junction is mid-word: it both survives the
|
||||
// font-size band below and must never take a space.
|
||||
let small_caps_join =
|
||||
is_small_caps_continuation(&text, first, next, next.x - end_x);
|
||||
// Must be similar font size, except for genuine small-caps
|
||||
// runs, where the shrunken capitals are the same word as the
|
||||
// full-size initial (see helper).
|
||||
if (next.font_size - first.font_size).abs() > first.font_size * MERGE_FONT_SIZE_BAND
|
||||
&& !small_caps_join
|
||||
{
|
||||
break;
|
||||
}
|
||||
// Never merge across style boundaries: the merged item
|
||||
@@ -997,7 +1096,7 @@ pub(crate) fn merge_text_items(items: Vec<TextItem>) -> Vec<TextItem> {
|
||||
Some((run_end, floor)) if j <= run_end => floor,
|
||||
_ => threshold,
|
||||
};
|
||||
if needs_bullet_space || gap > effective_threshold {
|
||||
if !small_caps_join && (needs_bullet_space || gap > effective_threshold) {
|
||||
text.push(' ');
|
||||
}
|
||||
text.push_str(&next.text);
|
||||
@@ -3001,6 +3100,145 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
/// Small caps as typesetters emit them: a full-size capital at 9.98pt
|
||||
/// immediately followed by shrunken capitals at 6.74pt, touching.
|
||||
/// Modelled on `199AD3d.pdf` p.5 ("ROLANDO T. ACOSTA, P.J.").
|
||||
#[test]
|
||||
fn small_caps_run_merges_into_one_word() {
|
||||
let items = vec![
|
||||
make_item_fs("R", 144.36, 581.84, 7.20, 9.98),
|
||||
make_item_fs("OLANDO", 151.56, 581.84, 30.56, 6.74),
|
||||
make_item_fs("T. A", 185.45, 581.84, 17.58, 9.98),
|
||||
make_item_fs("COSTA", 203.94, 581.84, 23.15, 6.74),
|
||||
make_item_fs(", P.J.", 227.09, 581.84, 22.56, 9.98),
|
||||
];
|
||||
let merged = merge_text_items(items);
|
||||
assert_eq!(merged.len(), 1, "got {:?}", merged);
|
||||
assert_eq!(merged[0].text, "ROLANDO T. ACOSTA, P.J.");
|
||||
}
|
||||
|
||||
/// The full two-column row: both names must merge independently and the
|
||||
/// 72pt column gap between them must survive as an item boundary.
|
||||
#[test]
|
||||
fn small_caps_merge_does_not_swallow_a_second_column() {
|
||||
let items = vec![
|
||||
// Column 1: "ROLANDO T. ACOSTA, P.J." ending at x=249.65
|
||||
make_item_fs("R", 144.36, 581.84, 7.20, 9.98),
|
||||
make_item_fs("OLANDO", 151.56, 581.84, 30.56, 6.74),
|
||||
make_item_fs("T. A", 185.45, 581.84, 17.58, 9.98),
|
||||
make_item_fs("COSTA", 203.94, 581.84, 23.15, 6.74),
|
||||
make_item_fs(", P.J.", 227.09, 581.84, 22.56, 9.98),
|
||||
// Column 2 starts at x=321.96 — a 72pt gap.
|
||||
make_item_fs("A", 321.96, 581.84, 7.20, 9.98),
|
||||
make_item_fs("NIL", 329.17, 581.84, 12.72, 6.74),
|
||||
make_item_fs("C. S", 345.04, 581.84, 19.59, 9.98),
|
||||
make_item_fs("INGH", 364.62, 581.84, 19.08, 6.74),
|
||||
];
|
||||
let merged = merge_text_items(items);
|
||||
let texts: Vec<&str> = merged.iter().map(|i| i.text.as_str()).collect();
|
||||
assert_eq!(
|
||||
texts,
|
||||
vec!["ROLANDO T. ACOSTA, P.J.", "ANIL C. SINGH"],
|
||||
"column gap should keep the two names apart"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn small_caps_merge_keeps_word_space_between_same_size_capitals() {
|
||||
// Two uppercase words at sizes the merge band already accepts (9.98 and
|
||||
// 9.0, a 10% drop) separated by a real word gap. The small-caps path
|
||||
// must not claim this junction and swallow the space.
|
||||
let items = vec![
|
||||
make_item_fs("SEE", 100.0, 500.0, 18.0, 9.98),
|
||||
make_item_fs("ALSO", 119.2, 500.0, 24.0, 9.0),
|
||||
];
|
||||
let merged = merge_text_items(items);
|
||||
assert_eq!(merged.len(), 1, "got {:?}", merged);
|
||||
assert_eq!(merged[0].text, "SEE ALSO");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn trailing_digit_is_not_a_capital_awaiting_small_caps() {
|
||||
// "...MAZZARELLI1" ends in a footnote marker; the backward search for an
|
||||
// uppercase letter must not skip the digit and glue the next run.
|
||||
assert!(!is_small_caps_continuation(
|
||||
"ANGELA M. MAZZARELLI1",
|
||||
&make_item_fs("ANGELA", 100.0, 500.0, 40.0, 9.98),
|
||||
&make_item_fs("SHULMAN", 140.0, 500.0, 30.0, 6.74),
|
||||
0.0,
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ordinal_suffix_after_a_digit_still_merges() {
|
||||
// "TUESDAY, JULY 4" + "TH" is one word in the source; the digit guard
|
||||
// must not block the four English ordinal suffixes.
|
||||
for suffix in ["TH", "ST", "ND", "RD"] {
|
||||
assert!(
|
||||
is_small_caps_continuation(
|
||||
"TUESDAY, JULY 4",
|
||||
&make_item_fs("JULY", 100.0, 500.0, 30.0, 12.0),
|
||||
&make_item_fs(suffix, 130.0, 500.0, 8.0, 8.0),
|
||||
0.0,
|
||||
),
|
||||
"{suffix} should merge after a digit"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn superscript_footnote_marker_is_not_a_small_caps_continuation() {
|
||||
// A digit must never qualify — otherwise footnote markers get glued on
|
||||
// without the superscript handling.
|
||||
assert!(!is_small_caps_continuation(
|
||||
"MAZZARELLI",
|
||||
&make_item_fs("MAZZARELLI", 100.0, 500.0, 50.0, 9.98),
|
||||
&make_item_fs("1", 150.0, 503.0, 3.0, 6.74),
|
||||
0.0,
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn drop_cap_is_not_a_small_caps_continuation() {
|
||||
// Mixed-case body text after a large initial is a drop cap, not small
|
||||
// caps.
|
||||
assert!(!is_small_caps_continuation(
|
||||
"T",
|
||||
&make_item_fs("T", 100.0, 500.0, 20.0, 30.0),
|
||||
&make_item_fs("he court held", 120.0, 500.0, 60.0, 10.0),
|
||||
0.0,
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn separate_word_is_not_a_small_caps_continuation() {
|
||||
// A real word space disqualifies even when both runs are uppercase.
|
||||
let first = make_item_fs("SEE", 100.0, 500.0, 20.0, 9.98);
|
||||
let next = make_item_fs("ALSO", 128.0, 500.0, 25.0, 6.74);
|
||||
assert!(!is_small_caps_continuation("SEE", &first, &next, 8.0));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn lowercase_continuation_is_not_small_caps() {
|
||||
assert!(!is_small_caps_continuation(
|
||||
"SMALL",
|
||||
&make_item_fs("SMALL", 100.0, 500.0, 30.0, 9.98),
|
||||
&make_item_fs("caps", 130.0, 500.0, 20.0, 6.74),
|
||||
0.0,
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn too_small_a_ratio_is_not_small_caps() {
|
||||
// 0.4 ratio is a superscript/sub-run, outside the small-caps band.
|
||||
assert!(!is_small_caps_continuation(
|
||||
"A",
|
||||
&make_item_fs("A", 100.0, 500.0, 7.0, 10.0),
|
||||
&make_item_fs("BC", 107.0, 500.0, 8.0, 4.0),
|
||||
0.0,
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_merge_subscript_items_chemical_formula() {
|
||||
// NH₃: "NH" at fs=8 followed by subscript "3" at fs=4.7
|
||||
|
||||
+594
-17
@@ -16,6 +16,76 @@ use super::{get_number, image_bbox_from_ctm, multiply_matrices};
|
||||
|
||||
const MAX_FORM_XOBJECT_DEPTH: u8 = 5;
|
||||
|
||||
/// Upper bound on Form XObject invocations during a single page extraction.
|
||||
/// Depth alone is not enough: an acyclic DAG where each form invokes the next
|
||||
/// N times expands to N^depth work before the depth cap is reached.
|
||||
const MAX_FORM_XOBJECT_INVOCATIONS: usize = 10_000;
|
||||
|
||||
/// Upper bound on content-stream operations walked across all Form XObject
|
||||
/// expansions for a page. Nested forms are decoded independently of the
|
||||
/// page-level operation cap, so this keeps total form work in the same
|
||||
/// ballpark as that page cap.
|
||||
const MAX_FORM_XOBJECT_OPERATIONS: usize = 1_000_000;
|
||||
|
||||
/// Shared budget for Form XObject expansion on a page. Bounds both nested DAG
|
||||
/// expansion and repeated sibling `/Do` invocations of the same form.
|
||||
pub(crate) struct FormWalkBudget {
|
||||
invocations: usize,
|
||||
operations: usize,
|
||||
max_invocations: usize,
|
||||
max_operations: usize,
|
||||
truncated: bool,
|
||||
}
|
||||
|
||||
impl FormWalkBudget {
|
||||
pub(crate) fn new() -> Self {
|
||||
Self::with_limits(MAX_FORM_XOBJECT_INVOCATIONS, MAX_FORM_XOBJECT_OPERATIONS)
|
||||
}
|
||||
|
||||
fn with_limits(max_invocations: usize, max_operations: usize) -> Self {
|
||||
Self {
|
||||
invocations: 0,
|
||||
operations: 0,
|
||||
max_invocations,
|
||||
max_operations,
|
||||
truncated: false,
|
||||
}
|
||||
}
|
||||
|
||||
fn exhausted(&mut self) -> bool {
|
||||
if self.invocations >= self.max_invocations || self.operations >= self.max_operations {
|
||||
self.truncated = true;
|
||||
true
|
||||
} else {
|
||||
false
|
||||
}
|
||||
}
|
||||
|
||||
fn charge_invocation(&mut self) -> bool {
|
||||
if self.exhausted() {
|
||||
return false;
|
||||
}
|
||||
self.invocations += 1;
|
||||
true
|
||||
}
|
||||
|
||||
/// Charge one walked content-stream operator. Independent of the
|
||||
/// invocation cap so a form that was already admitted can finish its
|
||||
/// stream (up to the operation cap).
|
||||
fn charge_operation(&mut self) -> bool {
|
||||
if self.operations >= self.max_operations {
|
||||
self.truncated = true;
|
||||
return false;
|
||||
}
|
||||
self.operations += 1;
|
||||
true
|
||||
}
|
||||
|
||||
pub(crate) fn was_truncated(&self) -> bool {
|
||||
self.truncated
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) enum XObjectType {
|
||||
Image,
|
||||
Form(ObjectId),
|
||||
@@ -109,6 +179,7 @@ fn collect_xobjects_from_dict(
|
||||
}
|
||||
|
||||
/// Extract text items from a Form XObject
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub(crate) fn extract_form_xobject_text(
|
||||
doc: &Document,
|
||||
form_id: ObjectId,
|
||||
@@ -117,6 +188,7 @@ pub(crate) fn extract_form_xobject_text(
|
||||
parent_ctm: &[f32; 6],
|
||||
cmap_decisions: &mut CMapDecisionCache,
|
||||
style_cache: &mut FontStyleCache,
|
||||
budget: &mut FormWalkBudget,
|
||||
) -> Vec<TextItem> {
|
||||
extract_form_xobject_text_inner(
|
||||
doc,
|
||||
@@ -127,6 +199,7 @@ pub(crate) fn extract_form_xobject_text(
|
||||
cmap_decisions,
|
||||
style_cache,
|
||||
0,
|
||||
budget,
|
||||
)
|
||||
}
|
||||
|
||||
@@ -140,11 +213,16 @@ fn extract_form_xobject_text_inner(
|
||||
cmap_decisions: &mut CMapDecisionCache,
|
||||
style_cache: &mut FontStyleCache,
|
||||
depth: u8,
|
||||
budget: &mut FormWalkBudget,
|
||||
) -> Vec<TextItem> {
|
||||
use lopdf::content::Content;
|
||||
|
||||
let mut items = Vec::new();
|
||||
|
||||
if !budget.charge_invocation() {
|
||||
return items;
|
||||
}
|
||||
|
||||
// Get the Form XObject stream
|
||||
let Ok(Object::Stream(stream)) = doc.get_object(form_id) else {
|
||||
return items;
|
||||
@@ -246,19 +324,55 @@ fn extract_form_xobject_text_inner(
|
||||
let mut current_font = String::new();
|
||||
let mut current_font_size: f32 = 12.0;
|
||||
let mut text_matrix = [1.0f32, 0.0, 0.0, 1.0, 0.0, 0.0];
|
||||
// Text line matrix (TLM) — Td/TD/T* move relative to the start of the
|
||||
// current line, not to the position left by the last show operator.
|
||||
let mut line_matrix = [1.0f32, 0.0, 0.0, 1.0, 0.0, 0.0];
|
||||
let mut text_leading: f32 = 0.0; // TL parameter (text-space units)
|
||||
let mut char_spacing: f32 = 0.0; // Tc parameter
|
||||
let mut word_spacing: f32 = 0.0; // Tw parameter
|
||||
let mut in_text_block = false;
|
||||
let mut fill_is_white = false;
|
||||
let mut ctm = base_ctm;
|
||||
let mut ctm_stack: Vec<[f32; 6]> = Vec::new();
|
||||
|
||||
// Text state (Tc/Tw/TL/Tf) and the fill colour are part of the graphics
|
||||
// state and must be saved/restored by q/Q alongside the CTM.
|
||||
#[derive(Clone)]
|
||||
struct GraphicsState {
|
||||
ctm: [f32; 6],
|
||||
char_spacing: f32,
|
||||
word_spacing: f32,
|
||||
text_leading: f32,
|
||||
current_font: String,
|
||||
current_font_size: f32,
|
||||
fill_is_white: bool,
|
||||
}
|
||||
let mut ctm_stack: Vec<GraphicsState> = Vec::new();
|
||||
|
||||
for op in &content.operations {
|
||||
if !budget.charge_operation() {
|
||||
break;
|
||||
}
|
||||
match op.operator.as_str() {
|
||||
"q" => {
|
||||
ctm_stack.push(ctm);
|
||||
ctm_stack.push(GraphicsState {
|
||||
ctm,
|
||||
char_spacing,
|
||||
word_spacing,
|
||||
text_leading,
|
||||
current_font: current_font.clone(),
|
||||
current_font_size,
|
||||
fill_is_white,
|
||||
});
|
||||
}
|
||||
"Q" => {
|
||||
if let Some(saved) = ctm_stack.pop() {
|
||||
ctm = saved;
|
||||
ctm = saved.ctm;
|
||||
char_spacing = saved.char_spacing;
|
||||
word_spacing = saved.word_spacing;
|
||||
text_leading = saved.text_leading;
|
||||
current_font = saved.current_font;
|
||||
current_font_size = saved.current_font_size;
|
||||
fill_is_white = saved.fill_is_white;
|
||||
}
|
||||
}
|
||||
"cm" => {
|
||||
@@ -276,7 +390,7 @@ fn extract_form_xobject_text_inner(
|
||||
let xobj_name = String::from_utf8_lossy(name).to_string();
|
||||
match form_xobjects.get(&xobj_name) {
|
||||
Some(XObjectType::Form(nested_id)) => {
|
||||
if depth < MAX_FORM_XOBJECT_DEPTH {
|
||||
if depth < MAX_FORM_XOBJECT_DEPTH && !budget.exhausted() {
|
||||
let nested_items = extract_form_xobject_text_inner(
|
||||
doc,
|
||||
*nested_id,
|
||||
@@ -286,6 +400,7 @@ fn extract_form_xobject_text_inner(
|
||||
cmap_decisions,
|
||||
style_cache,
|
||||
depth + 1,
|
||||
budget,
|
||||
);
|
||||
items.extend(nested_items);
|
||||
}
|
||||
@@ -321,6 +436,7 @@ fn extract_form_xobject_text_inner(
|
||||
"BT" => {
|
||||
in_text_block = true;
|
||||
text_matrix = [1.0, 0.0, 0.0, 1.0, 0.0, 0.0];
|
||||
line_matrix = text_matrix;
|
||||
}
|
||||
"ET" => {
|
||||
in_text_block = false;
|
||||
@@ -333,12 +449,33 @@ fn extract_form_xobject_text_inner(
|
||||
current_font_size = get_number(&op.operands[1]).unwrap_or(12.0);
|
||||
}
|
||||
}
|
||||
"TL" => {
|
||||
// Set text leading (used by T*, ', and ")
|
||||
if let Some(tl) = op.operands.first().and_then(get_number) {
|
||||
text_leading = tl;
|
||||
}
|
||||
}
|
||||
"Tc" => {
|
||||
if let Some(tc) = op.operands.first().and_then(get_number) {
|
||||
char_spacing = tc;
|
||||
}
|
||||
}
|
||||
"Tw" => {
|
||||
if let Some(tw) = op.operands.first().and_then(get_number) {
|
||||
word_spacing = tw;
|
||||
}
|
||||
}
|
||||
"Td" | "TD" => {
|
||||
// Move text position: TLM = T(tx,ty) x TLM; Tm = TLM
|
||||
if op.operands.len() >= 2 {
|
||||
let tx = get_number(&op.operands[0]).unwrap_or(0.0);
|
||||
let ty = get_number(&op.operands[1]).unwrap_or(0.0);
|
||||
text_matrix[4] += tx * text_matrix[0] + ty * text_matrix[2];
|
||||
text_matrix[5] += tx * text_matrix[1] + ty * text_matrix[3];
|
||||
line_matrix[4] += tx * line_matrix[0] + ty * line_matrix[2];
|
||||
line_matrix[5] += tx * line_matrix[1] + ty * line_matrix[3];
|
||||
text_matrix = line_matrix;
|
||||
if op.operator == "TD" {
|
||||
text_leading = -ty;
|
||||
}
|
||||
}
|
||||
}
|
||||
"Tm" => {
|
||||
@@ -347,8 +484,20 @@ fn extract_form_xobject_text_inner(
|
||||
text_matrix[i] =
|
||||
get_number(operand).unwrap_or(if i == 0 || i == 3 { 1.0 } else { 0.0 });
|
||||
}
|
||||
line_matrix = text_matrix;
|
||||
}
|
||||
}
|
||||
"T*" => {
|
||||
// Move to start of next line: equivalent to `0 -TL Td`
|
||||
let tl = if text_leading != 0.0 {
|
||||
text_leading
|
||||
} else {
|
||||
current_font_size * 1.2
|
||||
};
|
||||
line_matrix[4] += (-tl) * line_matrix[2];
|
||||
line_matrix[5] += (-tl) * line_matrix[3];
|
||||
text_matrix = line_matrix;
|
||||
}
|
||||
"g" => {
|
||||
if let Some(gray) = op.operands.first().and_then(get_number) {
|
||||
fill_is_white = gray > 0.95;
|
||||
@@ -384,17 +533,33 @@ fn extract_form_xobject_text_inner(
|
||||
_ => fill_is_white = false,
|
||||
}
|
||||
}
|
||||
"Tj" => {
|
||||
if in_text_block && !op.operands.is_empty() {
|
||||
"Tj" | "'" | "\"" => {
|
||||
// `'` = move to next line then show; `"` = set word/char spacing,
|
||||
// move to next line, then show (string is the last operand).
|
||||
if op.operator != "Tj" {
|
||||
if op.operator == "\"" && op.operands.len() >= 3 {
|
||||
word_spacing = get_number(&op.operands[0]).unwrap_or(word_spacing);
|
||||
char_spacing = get_number(&op.operands[1]).unwrap_or(char_spacing);
|
||||
}
|
||||
let tl = if text_leading != 0.0 {
|
||||
text_leading
|
||||
} else {
|
||||
current_font_size * 1.2
|
||||
};
|
||||
line_matrix[4] += (-tl) * line_matrix[2];
|
||||
line_matrix[5] += (-tl) * line_matrix[3];
|
||||
text_matrix = line_matrix;
|
||||
}
|
||||
if let (true, Some(show_operand)) = (in_text_block, op.operands.last()) {
|
||||
if fill_is_white {
|
||||
if let Some(font_info) = font_widths.get(¤t_font) {
|
||||
if let Some(raw_bytes) = get_operand_bytes(&op.operands[0]) {
|
||||
if let Some(raw_bytes) = get_operand_bytes(show_operand) {
|
||||
let w_ts = compute_string_width_ts(
|
||||
raw_bytes,
|
||||
font_info,
|
||||
current_font_size,
|
||||
0.0,
|
||||
0.0,
|
||||
char_spacing,
|
||||
word_spacing,
|
||||
);
|
||||
text_matrix[4] += w_ts * text_matrix[0];
|
||||
text_matrix[5] += w_ts * text_matrix[1];
|
||||
@@ -403,7 +568,7 @@ fn extract_form_xobject_text_inner(
|
||||
continue;
|
||||
}
|
||||
if let Some(text) = extract_text_from_operand(
|
||||
&op.operands[0],
|
||||
show_operand,
|
||||
¤t_font,
|
||||
font_base_names.get(¤t_font).map(|s| s.as_str()),
|
||||
font_cmaps,
|
||||
@@ -419,13 +584,13 @@ fn extract_form_xobject_text_inner(
|
||||
* type3_scales.get(¤t_font).copied().unwrap_or(1.0);
|
||||
let (x, y) = (combined[4], combined[5]);
|
||||
let width = if let Some(font_info) = font_widths.get(¤t_font) {
|
||||
if let Some(raw_bytes) = get_operand_bytes(&op.operands[0]) {
|
||||
if let Some(raw_bytes) = get_operand_bytes(show_operand) {
|
||||
let w_ts = compute_string_width_ts(
|
||||
raw_bytes,
|
||||
font_info,
|
||||
current_font_size,
|
||||
0.0,
|
||||
0.0,
|
||||
char_spacing,
|
||||
word_spacing,
|
||||
);
|
||||
text_matrix[4] += w_ts * text_matrix[0];
|
||||
text_matrix[5] += w_ts * text_matrix[1];
|
||||
@@ -548,8 +713,8 @@ fn extract_form_xobject_text_inner(
|
||||
raw_bytes,
|
||||
fi,
|
||||
current_font_size,
|
||||
0.0,
|
||||
0.0,
|
||||
char_spacing,
|
||||
word_spacing,
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -683,3 +848,415 @@ pub(crate) fn get_form_fonts<'a>(
|
||||
|
||||
fonts
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::extractor::content_stream::extract_page_text_items;
|
||||
use lopdf::{dictionary, Dictionary, Stream};
|
||||
|
||||
/// Build an acyclic Form XObject DAG: `levels` form objects, each non-leaf
|
||||
/// invoking the next form `branches` times. The leaf draws a single `(X)`.
|
||||
/// Returns `(doc, root_form_id)`.
|
||||
fn form_dag(branches: usize, levels: usize) -> (Document, ObjectId) {
|
||||
assert!(levels >= 2);
|
||||
let mut doc = Document::new();
|
||||
let font_id = doc.add_object(dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "Type1",
|
||||
"BaseFont" => "Helvetica",
|
||||
});
|
||||
let ids: Vec<ObjectId> = (0..levels).map(|_| doc.new_object_id()).collect();
|
||||
for level in 0..levels {
|
||||
let stream = if level + 1 == levels {
|
||||
Stream::new(
|
||||
dictionary! {
|
||||
"Type" => "XObject",
|
||||
"Subtype" => "Form",
|
||||
"BBox" => vec![0.into(), 0.into(), 100.into(), 100.into()],
|
||||
"Resources" => dictionary! {
|
||||
"Font" => dictionary! {
|
||||
"F1" => Object::Reference(font_id),
|
||||
},
|
||||
},
|
||||
},
|
||||
b"BT /F1 10 Tf 10 10 Td (X) Tj ET\n".to_vec(),
|
||||
)
|
||||
} else {
|
||||
let next_name = format!("Fm{}", level + 1);
|
||||
let content = format!("/{next_name} Do\n").repeat(branches);
|
||||
let mut xobjects = Dictionary::new();
|
||||
xobjects.set(next_name, Object::Reference(ids[level + 1]));
|
||||
let mut resources = Dictionary::new();
|
||||
resources.set("XObject", Object::Dictionary(xobjects));
|
||||
let mut dict = dictionary! {
|
||||
"Type" => "XObject",
|
||||
"Subtype" => "Form",
|
||||
"BBox" => vec![0.into(), 0.into(), 100.into(), 100.into()],
|
||||
};
|
||||
dict.set("Resources", Object::Dictionary(resources));
|
||||
Stream::new(dict, content.into_bytes())
|
||||
};
|
||||
doc.set_object(ids[level], Object::Stream(stream));
|
||||
}
|
||||
(doc, ids[0])
|
||||
}
|
||||
|
||||
fn page_invoking_form(mut doc: Document, form_id: ObjectId) -> (Document, ObjectId) {
|
||||
let content_id = doc.add_object(Object::Stream(Stream::new(
|
||||
dictionary! {},
|
||||
b"/Fm0 Do\n".to_vec(),
|
||||
)));
|
||||
let page_id = doc.add_object(dictionary! {
|
||||
"Type" => "Page",
|
||||
"Contents" => Object::Reference(content_id),
|
||||
"Resources" => dictionary! {
|
||||
"XObject" => dictionary! {
|
||||
"Fm0" => Object::Reference(form_id),
|
||||
},
|
||||
},
|
||||
"MediaBox" => vec![0.into(), 0.into(), 612.into(), 792.into()],
|
||||
});
|
||||
let pages_id = doc.add_object(dictionary! {
|
||||
"Type" => "Pages",
|
||||
"Count" => Object::Integer(1),
|
||||
"Kids" => vec![Object::Reference(page_id)],
|
||||
});
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"Pages" => Object::Reference(pages_id),
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
(doc, page_id)
|
||||
}
|
||||
|
||||
fn extract_form(
|
||||
doc: &Document,
|
||||
form_id: ObjectId,
|
||||
budget: &mut FormWalkBudget,
|
||||
) -> Vec<TextItem> {
|
||||
extract_form_xobject_text(
|
||||
doc,
|
||||
form_id,
|
||||
1,
|
||||
&FontCMaps::from_doc(doc),
|
||||
&[1.0, 0.0, 0.0, 1.0, 0.0, 0.0],
|
||||
&mut CMapDecisionCache::new(),
|
||||
&mut FontStyleCache::new(),
|
||||
budget,
|
||||
)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn nested_form_still_extracts_leaf_text() {
|
||||
let (doc, root) = form_dag(1, 3);
|
||||
let items = extract_form(&doc, root, &mut FormWalkBudget::new());
|
||||
assert_eq!(items.len(), 1);
|
||||
assert_eq!(items[0].text, "X");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn acyclic_form_dag_within_budget_keeps_all_leaves() {
|
||||
// 4 sibling invocations across 4 nested levels → 4^4 leaf drawings.
|
||||
// Default budgets are far above 256, so legitimate nesting is intact.
|
||||
let (doc, root) = form_dag(4, 5);
|
||||
let items = extract_form(&doc, root, &mut FormWalkBudget::new());
|
||||
assert_eq!(items.len(), 4usize.pow(4));
|
||||
assert!(items.iter().all(|item| item.text == "X"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn acyclic_form_dag_stops_at_invocation_budget() {
|
||||
// Same DAG as above would draw 256 leaves; a tiny invocation cap must
|
||||
// stop expansion rather than walking the full tree.
|
||||
let (doc, root) = form_dag(4, 5);
|
||||
let mut budget = FormWalkBudget::with_limits(20, MAX_FORM_XOBJECT_OPERATIONS);
|
||||
let items = extract_form(&doc, root, &mut budget);
|
||||
assert!(
|
||||
items.len() < 4usize.pow(4),
|
||||
"invocation budget must truncate DAG expansion; got {} items",
|
||||
items.len()
|
||||
);
|
||||
assert!(budget.was_truncated());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn form_operations_stop_at_budget() {
|
||||
let mut doc = Document::new();
|
||||
let font_id = doc.add_object(dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "Type1",
|
||||
"BaseFont" => "Helvetica",
|
||||
});
|
||||
let mut content = b"q Q\n".repeat(50);
|
||||
content.extend_from_slice(b"BT /F1 10 Tf 10 10 Td (X) Tj ET\n");
|
||||
let form_id = doc.add_object(Object::Stream(Stream::new(
|
||||
dictionary! {
|
||||
"Type" => "XObject",
|
||||
"Subtype" => "Form",
|
||||
"BBox" => vec![0.into(), 0.into(), 100.into(), 100.into()],
|
||||
"Resources" => dictionary! {
|
||||
"Font" => dictionary! {
|
||||
"F1" => Object::Reference(font_id),
|
||||
},
|
||||
},
|
||||
},
|
||||
content,
|
||||
)));
|
||||
|
||||
let mut budget = FormWalkBudget::with_limits(MAX_FORM_XOBJECT_INVOCATIONS, 10);
|
||||
let items = extract_form(&doc, form_id, &mut budget);
|
||||
assert!(
|
||||
items.is_empty(),
|
||||
"operation budget must stop before the trailing text show"
|
||||
);
|
||||
assert!(budget.was_truncated());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_level_form_dag_stays_within_production_budget() {
|
||||
// A page-level `/Do` of an 8-wide, 6-level Form DAG would expand to
|
||||
// 8^5 = 32_768 leaf drawings without a budget. The production
|
||||
// invocation cap must keep extraction bounded.
|
||||
let (doc, root) = form_dag(8, 6);
|
||||
let (doc, page_id) = page_invoking_form(doc, root);
|
||||
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let ((items, _, _), _, _, _) = extract_page_text_items(
|
||||
&doc,
|
||||
page_id,
|
||||
1,
|
||||
&font_cmaps,
|
||||
false,
|
||||
&mut FontStyleCache::new(),
|
||||
&mut FormWalkBudget::new(),
|
||||
)
|
||||
.unwrap();
|
||||
assert!(
|
||||
items.len() <= MAX_FORM_XOBJECT_INVOCATIONS,
|
||||
"page-level Form expansion must stay within the invocation cap; got {}",
|
||||
items.len()
|
||||
);
|
||||
assert!(
|
||||
!items.is_empty(),
|
||||
"budget must still allow some nested form text through"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn shared_form_budget_spans_two_extraction_passes() {
|
||||
// The invisible-layer retry calls extract_page_text_items twice for
|
||||
// the same page; both passes must share one budget.
|
||||
let (doc, root) = form_dag(1, 2);
|
||||
let (doc, page_id) = page_invoking_form(doc, root);
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
// Root + leaf = 2 invocations on the first pass.
|
||||
let mut budget = FormWalkBudget::with_limits(2, MAX_FORM_XOBJECT_OPERATIONS);
|
||||
let ((first, _, _), _, _, _) = extract_page_text_items(
|
||||
&doc,
|
||||
page_id,
|
||||
1,
|
||||
&font_cmaps,
|
||||
false,
|
||||
&mut FontStyleCache::new(),
|
||||
&mut budget,
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(first.iter().filter(|item| item.text == "X").count(), 1);
|
||||
assert!(!budget.was_truncated());
|
||||
|
||||
let ((second, _, _), _, _, _) = extract_page_text_items(
|
||||
&doc,
|
||||
page_id,
|
||||
1,
|
||||
&font_cmaps,
|
||||
true,
|
||||
&mut FontStyleCache::new(),
|
||||
&mut budget,
|
||||
)
|
||||
.unwrap();
|
||||
assert!(
|
||||
second.iter().all(|item| item.text != "X"),
|
||||
"second pass must not get a fresh invocation budget"
|
||||
);
|
||||
assert!(budget.was_truncated());
|
||||
}
|
||||
|
||||
/// Build a document whose page draws *all* of its content through a single
|
||||
/// Form XObject — the shape emitted by print-to-PDF producers like PDFlib,
|
||||
/// where the page stream itself is only `q /X1 Do Q`.
|
||||
fn doc_with_form_content(form_content: &[u8]) -> (Document, ObjectId) {
|
||||
let mut doc = Document::new();
|
||||
let widths: Vec<Object> = (0..=255).map(|_| 600.into()).collect();
|
||||
let font_id = doc.add_object(dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "Type1",
|
||||
"BaseFont" => "Helvetica",
|
||||
"FirstChar" => 0,
|
||||
"LastChar" => 255,
|
||||
"Widths" => Object::Array(widths),
|
||||
});
|
||||
let form_id = doc.add_object(Object::Stream(Stream::new(
|
||||
dictionary! {
|
||||
"Type" => "XObject",
|
||||
"Subtype" => "Form",
|
||||
"BBox" => vec![0.into(), 0.into(), 612.into(), 792.into()],
|
||||
"Resources" => dictionary! {
|
||||
"Font" => dictionary! { "F1" => Object::Reference(font_id) },
|
||||
},
|
||||
},
|
||||
form_content.to_vec(),
|
||||
)));
|
||||
let content_id = doc.add_object(Object::Stream(Stream::new(
|
||||
dictionary! {},
|
||||
b"q /X1 Do Q".to_vec(),
|
||||
)));
|
||||
let page_id = doc.add_object(dictionary! {
|
||||
"Type" => "Page",
|
||||
"Contents" => Object::Reference(content_id),
|
||||
"Resources" => dictionary! {
|
||||
"XObject" => dictionary! { "X1" => Object::Reference(form_id) },
|
||||
},
|
||||
"MediaBox" => vec![0.into(), 0.into(), 612.into(), 792.into()],
|
||||
});
|
||||
let pages_id = doc.add_object(dictionary! {
|
||||
"Type" => "Pages",
|
||||
"Count" => Object::Integer(1),
|
||||
"Kids" => vec![Object::Reference(page_id)],
|
||||
});
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"Pages" => Object::Reference(pages_id),
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
(doc, page_id)
|
||||
}
|
||||
|
||||
fn form_items(form_content: &[u8]) -> Vec<TextItem> {
|
||||
let (doc, page_id) = doc_with_form_content(form_content);
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let ((items, _, _), _, _, _) = extract_page_text_items(
|
||||
&doc,
|
||||
page_id,
|
||||
1,
|
||||
&font_cmaps,
|
||||
false,
|
||||
&mut FontStyleCache::new(),
|
||||
&mut FormWalkBudget::new(),
|
||||
)
|
||||
.unwrap();
|
||||
items
|
||||
}
|
||||
|
||||
fn find<'a>(items: &'a [TextItem], text: &str) -> &'a TextItem {
|
||||
items
|
||||
.iter()
|
||||
.find(|item| item.text == text)
|
||||
.unwrap_or_else(|| {
|
||||
let found: Vec<&String> = items.iter().map(|i| &i.text).collect();
|
||||
panic!("no item {text:?} in {found:?}")
|
||||
})
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn t_star_inside_form_moves_to_next_line() {
|
||||
// T* was previously unhandled inside Form XObjects, so every line after
|
||||
// the first piled onto the preceding baseline and drifted right.
|
||||
let items =
|
||||
form_items(b"BT /F1 12 Tf 12 TL 1 0 0 1 100 700 Tm (first) Tj T* (second) Tj ET");
|
||||
|
||||
let first = find(&items, "first");
|
||||
let second = find(&items, "second");
|
||||
assert!((first.y - 700.0).abs() < 0.1, "first y = {}", first.y);
|
||||
assert!((second.y - 688.0).abs() < 0.1, "second y = {}", second.y);
|
||||
assert!((second.x - 100.0).abs() < 0.1, "second x = {}", second.x);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn td_inside_form_is_relative_to_line_start_not_shown_text() {
|
||||
// Td moves relative to the text *line* matrix. Applying it to the
|
||||
// matrix already advanced by Tj marched each line off the right edge.
|
||||
let items = form_items(b"BT /F1 12 Tf 1 0 0 1 100 700 Tm (AAAAA) Tj 0 -12 Td (B) Tj ET");
|
||||
|
||||
let b = find(&items, "B");
|
||||
assert!((b.x - 100.0).abs() < 0.1, "B x = {} (expected 100)", b.x);
|
||||
assert!((b.y - 688.0).abs() < 0.1, "B y = {}", b.y);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn td_inside_form_sets_leading_for_later_t_star() {
|
||||
// `TD` sets the leading to -ty as a side effect; a following T* must
|
||||
// reuse it.
|
||||
let items = form_items(
|
||||
b"BT /F1 12 Tf 1 0 0 1 100 700 Tm (one) Tj 0 -15 TD (two) Tj T* (three) Tj ET",
|
||||
);
|
||||
|
||||
assert!((find(&items, "two").y - 685.0).abs() < 0.1);
|
||||
let three = find(&items, "three");
|
||||
assert!((three.y - 670.0).abs() < 0.1, "three y = {}", three.y);
|
||||
assert!((three.x - 100.0).abs() < 0.1, "three x = {}", three.x);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn quote_operator_inside_form_moves_to_next_line() {
|
||||
let items = form_items(b"BT /F1 12 Tf 12 TL 1 0 0 1 100 700 Tm (first) Tj (second) ' ET");
|
||||
|
||||
let second = find(&items, "second");
|
||||
assert!((second.y - 688.0).abs() < 0.1, "second y = {}", second.y);
|
||||
assert!((second.x - 100.0).abs() < 0.1, "second x = {}", second.x);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn double_quote_operator_inside_form_sets_spacing_and_moves() {
|
||||
// `aw ac (string) "` — set word spacing and char spacing, then T* and show.
|
||||
let items =
|
||||
form_items(b"BT /F1 12 Tf 12 TL 1 0 0 1 100 700 Tm (first) Tj 0 0 (second) \" ET");
|
||||
|
||||
let second = find(&items, "second");
|
||||
assert!((second.y - 688.0).abs() < 0.1, "second y = {}", second.y);
|
||||
assert!((second.x - 100.0).abs() < 0.1, "second x = {}", second.x);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn char_spacing_inside_form_widens_advance() {
|
||||
// Tc was hardcoded to 0 in the form parser, so advance widths drifted.
|
||||
// 2 glyphs x 600/1000 x 12pt = 14.4, plus 2 x Tc(2.0) = 18.4.
|
||||
let items = form_items(b"BT /F1 12 Tf 1 0 0 1 100 700 Tm 2 Tc (AB) Tj ET");
|
||||
|
||||
let ab = find(&items, "AB");
|
||||
assert!((ab.width - 18.4).abs() < 0.1, "AB width = {}", ab.width);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn q_restores_fill_colour_inside_form() {
|
||||
// A white fill set inside q/Q must not leak past the Q — otherwise the
|
||||
// following black text is treated as invisible and dropped entirely.
|
||||
let items = form_items(
|
||||
b"BT /F1 12 Tf 12 TL 1 0 0 1 100 700 Tm q 1 g (hidden) Tj Q T* (visible) Tj ET",
|
||||
);
|
||||
|
||||
assert!(
|
||||
items.iter().any(|item| item.text == "visible"),
|
||||
"text after Q was dropped: {:?}",
|
||||
items.iter().map(|i| &i.text).collect::<Vec<_>>()
|
||||
);
|
||||
assert!(
|
||||
!items.iter().any(|item| item.text == "hidden"),
|
||||
"white-filled text should still be suppressed"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn q_restores_text_state_inside_form() {
|
||||
// Tc/TL live in the graphics state; `Q` must roll them back.
|
||||
let items =
|
||||
form_items(b"BT /F1 12 Tf 12 TL 1 0 0 1 100 700 Tm q 30 TL (a) Tj Q T* (b) Tj ET");
|
||||
|
||||
let b = find(&items, "b");
|
||||
assert!(
|
||||
(b.y - 688.0).abs() < 0.1,
|
||||
"b y = {} (leading should restore to 12)",
|
||||
b.y
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
+99
-24
@@ -757,6 +757,23 @@ pub struct PageRegionResult {
|
||||
pub regions: Vec<RegionText>,
|
||||
}
|
||||
|
||||
/// Minimum alphanumeric mass an invisible (Tr 3) text layer must carry for
|
||||
/// the OCR-layer fallback in [`extract_text_in_regions_mem`] to adopt it. A
|
||||
/// real OCR layer carries far more; a stray watermark or artifact does not.
|
||||
const OCR_LAYER_MIN_ALNUM: usize = 40;
|
||||
|
||||
/// Alphanumeric mass of extracted items, ignoring raster placeholders.
|
||||
/// `[Image: ...]` items (ItemType::Image) are synthesized for image
|
||||
/// XObjects — they mark that pixels exist, not that text was read, so they
|
||||
/// must not count as coverage.
|
||||
fn non_placeholder_alnum(items: &[TextItem]) -> usize {
|
||||
items
|
||||
.iter()
|
||||
.filter(|it| !matches!(it.item_type, types::ItemType::Image))
|
||||
.map(|it| it.text.chars().filter(|c| c.is_alphanumeric()).count())
|
||||
.sum()
|
||||
}
|
||||
|
||||
/// Extract text within bounding-box regions from a PDF in memory.
|
||||
///
|
||||
/// This is designed for hybrid OCR pipelines: a layout model detects regions
|
||||
@@ -809,8 +826,11 @@ pub fn extract_text_in_regions_mem(
|
||||
let height = get_page_height(&doc, page_id).unwrap_or(792.0);
|
||||
page_heights.insert(*page_num, height);
|
||||
|
||||
// Extract text items for this page
|
||||
let ((mut items, _rects, _lines), has_gid, coords_rotated) =
|
||||
// Extract text items for this page. The Form XObject budget is shared
|
||||
// with the invisible-layer retry below so one page cannot consume two
|
||||
// full expansion budgets.
|
||||
let mut form_budget = extractor::FormWalkBudget::new();
|
||||
let ((mut items, _rects, _lines), mut has_gid, mut coords_rotated, skipped_invisible) =
|
||||
extractor::content_stream::extract_page_text_items(
|
||||
&doc,
|
||||
page_id,
|
||||
@@ -818,7 +838,54 @@ pub fn extract_text_in_regions_mem(
|
||||
&font_cmaps,
|
||||
false,
|
||||
&mut style_cache,
|
||||
&mut form_budget,
|
||||
)?;
|
||||
// OCR-layer fallback: scanned pages often carry their text as an
|
||||
// invisible (Tr 3) layer behind the page raster. The visible-only
|
||||
// pass sees nothing there but `[Image: ...]` placeholders, so every
|
||||
// region on the page reports needs_ocr even though the exact text is
|
||||
// embedded in the PDF — and this extractor then disagrees with the
|
||||
// markdown path, which already retries Mixed PDFs with the invisible
|
||||
// layer included. Retry page-scoped, and only when (a) the first
|
||||
// pass actually SKIPPED invisible text — blank pages and image-only
|
||||
// scans without an OCR layer must not pay a second content-stream
|
||||
// parse (review catch) — and (b) the page has NO visible text item
|
||||
// at all (punctuation counts, whitespace-only artifacts don't): an
|
||||
// invisible OCR layer transcribes the raster, so any visible glyph
|
||||
// has an invisible twin there and adoption would duplicate it
|
||||
// (review catches — strict gate, no fuzzy dedupe). Adopt the retry
|
||||
// only when it contributes real, non-garbage text.
|
||||
let has_visible_text = items.iter().any(|it| {
|
||||
!matches!(it.item_type, types::ItemType::Image) && !it.text.trim().is_empty()
|
||||
});
|
||||
if skipped_invisible && !has_visible_text {
|
||||
if let Ok(((inv_items, _inv_rects, _inv_lines), inv_gid, inv_rotated, _)) =
|
||||
extractor::content_stream::extract_page_text_items(
|
||||
&doc,
|
||||
page_id,
|
||||
*page_num,
|
||||
&font_cmaps,
|
||||
true,
|
||||
&mut style_cache,
|
||||
&mut form_budget,
|
||||
)
|
||||
{
|
||||
let inv_alnum = non_placeholder_alnum(&inv_items);
|
||||
// Judge the WHOLE recovered layer, not a prefix — a broken
|
||||
// OCR layer can hide its garbage past any fixed sample size
|
||||
// (review catch).
|
||||
let sample: String = inv_items
|
||||
.iter()
|
||||
.filter(|it| !matches!(it.item_type, types::ItemType::Image))
|
||||
.map(|it| it.text.as_str())
|
||||
.collect();
|
||||
if inv_alnum >= OCR_LAYER_MIN_ALNUM && !is_garbage_text(&sample) {
|
||||
items = inv_items;
|
||||
has_gid = inv_gid;
|
||||
coords_rotated = inv_rotated;
|
||||
}
|
||||
}
|
||||
}
|
||||
let threshold = text_utils::fix_letterspaced_items(&mut items);
|
||||
if threshold > 0.10 {
|
||||
page_thresholds.insert(*page_num, threshold);
|
||||
@@ -975,7 +1042,7 @@ pub fn extract_tables_in_regions_mem(
|
||||
let height = get_page_height(&doc, page_id).unwrap_or(792.0);
|
||||
page_heights.insert(*page_num, height);
|
||||
|
||||
let ((mut items, rects, lines), has_gid, coords_rotated) =
|
||||
let ((mut items, rects, lines), has_gid, coords_rotated, _skipped_invisible) =
|
||||
extractor::content_stream::extract_page_text_items(
|
||||
&doc,
|
||||
page_id,
|
||||
@@ -983,6 +1050,7 @@ pub fn extract_tables_in_regions_mem(
|
||||
&font_cmaps,
|
||||
false,
|
||||
&mut style_cache,
|
||||
&mut extractor::FormWalkBudget::new(),
|
||||
)?;
|
||||
let threshold = text_utils::fix_letterspaced_items(&mut items);
|
||||
if threshold > 0.10 {
|
||||
@@ -1286,7 +1354,7 @@ pub fn detect_vector_grid_in_region_mem(
|
||||
let needed_pages = HashSet::from([page_1idx]);
|
||||
let font_cmaps = FontCMaps::from_doc_pages_fast(&doc, Some(&needed_pages));
|
||||
let page_h = get_page_height(&doc, page_id).unwrap_or(792.0);
|
||||
let ((mut items, rects, lines), _has_gid, coords_rotated) =
|
||||
let ((mut items, rects, lines), _has_gid, coords_rotated, _skipped_invisible) =
|
||||
extractor::content_stream::extract_page_text_items(
|
||||
&doc,
|
||||
page_id,
|
||||
@@ -1294,6 +1362,7 @@ pub fn detect_vector_grid_in_region_mem(
|
||||
&font_cmaps,
|
||||
false,
|
||||
&mut extractor::FontStyleCache::new(),
|
||||
&mut extractor::FormWalkBudget::new(),
|
||||
)?;
|
||||
text_utils::fix_letterspaced_items(&mut items);
|
||||
|
||||
@@ -1480,15 +1549,17 @@ mod vector_grid_tests {
|
||||
let &page_id = pages.get(&1).unwrap();
|
||||
let needed: HashSet<u32> = HashSet::from([1]);
|
||||
let cmaps = FontCMaps::from_doc_pages_fast(&doc, Some(&needed));
|
||||
let ((items, rects, _lines), _has_gid, _rotated) = extract_page_text_items(
|
||||
&doc,
|
||||
page_id,
|
||||
1,
|
||||
&cmaps,
|
||||
false,
|
||||
&mut crate::extractor::FontStyleCache::new(),
|
||||
)
|
||||
.unwrap();
|
||||
let ((items, rects, _lines), _has_gid, _rotated, _skipped_invisible) =
|
||||
extract_page_text_items(
|
||||
&doc,
|
||||
page_id,
|
||||
1,
|
||||
&cmaps,
|
||||
false,
|
||||
&mut crate::extractor::FontStyleCache::new(),
|
||||
&mut crate::extractor::FormWalkBudget::new(),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let (rect_tables, _) = detect_tables_from_rects(&items, &rects, 1);
|
||||
assert_eq!(rect_tables.len(), 1, "expected one rect-detected table");
|
||||
@@ -1522,15 +1593,17 @@ mod vector_grid_tests {
|
||||
let &page_id = pages.get(&page_num).unwrap();
|
||||
let needed: HashSet<u32> = HashSet::from([page_num]);
|
||||
let cmaps = FontCMaps::from_doc_pages_fast(&doc, Some(&needed));
|
||||
let ((items, rects, _lines), _has_gid, _rotated) = extract_page_text_items(
|
||||
&doc,
|
||||
page_id,
|
||||
page_num,
|
||||
&cmaps,
|
||||
false,
|
||||
&mut crate::extractor::FontStyleCache::new(),
|
||||
)
|
||||
.unwrap();
|
||||
let ((items, rects, _lines), _has_gid, _rotated, _skipped_invisible) =
|
||||
extract_page_text_items(
|
||||
&doc,
|
||||
page_id,
|
||||
page_num,
|
||||
&cmaps,
|
||||
false,
|
||||
&mut crate::extractor::FontStyleCache::new(),
|
||||
&mut crate::extractor::FormWalkBudget::new(),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let (rect_tables, _) = detect_tables_from_rects(&items, &rects, page_num);
|
||||
rect_tables
|
||||
@@ -2258,7 +2331,7 @@ pub fn extract_tables_with_structure_cells_mem(
|
||||
let height = get_page_height(&doc, page_id).unwrap_or(792.0);
|
||||
page_heights.insert(*page_num, height);
|
||||
|
||||
let ((mut items, _rects, _lines), _has_gid, coords_rotated) =
|
||||
let ((mut items, _rects, _lines), _has_gid, coords_rotated, _skipped_invisible) =
|
||||
extractor::content_stream::extract_page_text_items(
|
||||
&doc,
|
||||
page_id,
|
||||
@@ -2266,6 +2339,7 @@ pub fn extract_tables_with_structure_cells_mem(
|
||||
&font_cmaps,
|
||||
false,
|
||||
&mut style_cache,
|
||||
&mut extractor::FormWalkBudget::new(),
|
||||
)?;
|
||||
let threshold = text_utils::fix_letterspaced_items(&mut items);
|
||||
if threshold > 0.10 {
|
||||
@@ -3060,7 +3134,7 @@ fn detect_tsr_quality_issue(
|
||||
let mut needed: HashSet<u32> = HashSet::new();
|
||||
needed.insert(page_1idx);
|
||||
let font_cmaps = FontCMaps::from_doc_pages_fast(&doc, Some(&needed));
|
||||
let ((mut items, _rects, _lines), _has_gid, coords_rotated) =
|
||||
let ((mut items, _rects, _lines), _has_gid, coords_rotated, _skipped_invisible) =
|
||||
extractor::content_stream::extract_page_text_items(
|
||||
&doc,
|
||||
page_id,
|
||||
@@ -3068,6 +3142,7 @@ fn detect_tsr_quality_issue(
|
||||
&font_cmaps,
|
||||
false,
|
||||
&mut extractor::FontStyleCache::new(),
|
||||
&mut extractor::FormWalkBudget::new(),
|
||||
)?;
|
||||
let adaptive_threshold = text_utils::fix_letterspaced_items(&mut items);
|
||||
let coords = if coords_rotated {
|
||||
|
||||
@@ -453,6 +453,110 @@ fn merged_retry_skips_body_font(detected_columns: bool, has_chart_regions: bool)
|
||||
detected_columns && !has_chart_regions
|
||||
}
|
||||
|
||||
/// Identity of a piece of page furniture: the same trimmed text drawn at the
|
||||
/// same position (quantized to 0.5pt) — page numbers excluded by construction
|
||||
/// because their text differs per page.
|
||||
type FurnitureKey = (String, i32, i32);
|
||||
|
||||
fn furniture_key(item: &TextItem) -> FurnitureKey {
|
||||
(
|
||||
item.text.trim().to_string(),
|
||||
(item.x * 2.0).round() as i32,
|
||||
(item.y * 2.0).round() as i32,
|
||||
)
|
||||
}
|
||||
|
||||
/// Minimum distinct pages an identical (text, position) must appear on before
|
||||
/// it counts as a running header/footer rather than coincidence.
|
||||
const RUNNING_FURNITURE_MIN_PAGES: usize = 3;
|
||||
|
||||
/// Fraction of each page's vertical content extent, at the top and at the
|
||||
/// bottom, where running furniture may live. Repetition alone is not enough:
|
||||
/// a form template repeated per record carries identical labels at identical
|
||||
/// mid-page coordinates on every page, and those are real table cells. What
|
||||
/// makes a header/footer is repetition *at the page edge*.
|
||||
const RUNNING_FURNITURE_BAND: f32 = 0.2;
|
||||
|
||||
/// Collect the keys of items that repeat verbatim at the same position on at
|
||||
/// least [`RUNNING_FURNITURE_MIN_PAGES`] distinct pages, restricted to the
|
||||
/// top/bottom [`RUNNING_FURNITURE_BAND`] of each page's content extent —
|
||||
/// running headers and footers. Single- and two-page documents produce an
|
||||
/// empty set.
|
||||
fn running_furniture_keys(items: &[TextItem]) -> HashSet<FurnitureKey> {
|
||||
// Vertical content extent per page, so the edge bands adapt to the
|
||||
// document's real margins instead of assuming a media box.
|
||||
let mut page_extent: HashMap<u32, (f32, f32)> = HashMap::new();
|
||||
for item in items {
|
||||
if item.text.trim().is_empty() {
|
||||
continue;
|
||||
}
|
||||
let entry = page_extent.entry(item.page).or_insert((item.y, item.y));
|
||||
entry.0 = entry.0.min(item.y);
|
||||
entry.1 = entry.1.max(item.y);
|
||||
}
|
||||
|
||||
let mut pages_by_key: HashMap<FurnitureKey, HashSet<u32>> = HashMap::new();
|
||||
for item in items {
|
||||
if item.text.trim().is_empty() {
|
||||
continue;
|
||||
}
|
||||
let Some(&(min_y, max_y)) = page_extent.get(&item.page) else {
|
||||
continue;
|
||||
};
|
||||
// A page whose text has no vertical span gives no evidence of where
|
||||
// its edges are — without this guard, a zero band would classify its
|
||||
// every item as edge furniture.
|
||||
let extent = max_y - min_y;
|
||||
if extent <= 0.0 {
|
||||
continue;
|
||||
}
|
||||
let band = extent * RUNNING_FURNITURE_BAND;
|
||||
if item.y > min_y + band && item.y < max_y - band {
|
||||
continue; // mid-page: never furniture, however often it repeats
|
||||
}
|
||||
pages_by_key
|
||||
.entry(furniture_key(item))
|
||||
.or_default()
|
||||
.insert(item.page);
|
||||
}
|
||||
pages_by_key
|
||||
.into_iter()
|
||||
.filter(|(_, pages)| pages.len() >= RUNNING_FURNITURE_MIN_PAGES)
|
||||
.map(|(key, _)| key)
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Reject a heuristic table whose items are almost entirely running
|
||||
/// headers/footers. A wrapped document title repeated at the bottom of every
|
||||
/// page aligns well enough to read as a grid, but it is page furniture, not
|
||||
/// data — vetoing the table lets the text flow as prose instead. Real tables
|
||||
/// carry per-page content, so even a repeated *header row* stays under the
|
||||
/// threshold once its body rows differ.
|
||||
fn is_running_furniture_table(
|
||||
detection_items: &[TextItem],
|
||||
table: &crate::tables::Table,
|
||||
running: &HashSet<FurnitureKey>,
|
||||
) -> bool {
|
||||
if running.is_empty() {
|
||||
return false;
|
||||
}
|
||||
let mut total = 0usize;
|
||||
let mut furniture = 0usize;
|
||||
for &idx in &table.item_indices {
|
||||
let Some(item) = detection_items.get(idx) else {
|
||||
continue;
|
||||
};
|
||||
if item.text.trim().is_empty() {
|
||||
continue;
|
||||
}
|
||||
total += 1;
|
||||
if running.contains(&furniture_key(item)) {
|
||||
furniture += 1;
|
||||
}
|
||||
}
|
||||
total > 0 && (furniture as f32) >= (total as f32) * 0.8
|
||||
}
|
||||
|
||||
/// Reject a heuristic table only when its cells are overwhelmingly parallel
|
||||
/// prose fragments. This is deliberately narrower than disabling body-font
|
||||
/// detection for the whole page: numeric, compact, headed, and otherwise
|
||||
@@ -1188,6 +1292,12 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
let mut table_items: HashSet<usize> = HashSet::new();
|
||||
let mut page_tables: HashMap<u32, Vec<PositionedMarkdown>> = HashMap::new();
|
||||
|
||||
// Running headers/footers repeat verbatim at the same position on many
|
||||
// pages. When such a block wraps a long title over aligned lines, the
|
||||
// heuristic detector reads it as a table. Knowing which items are page
|
||||
// furniture is a document-wide question, so answer it once here.
|
||||
let running_furniture = running_furniture_keys(&text_items);
|
||||
|
||||
// Pre-group items by page with their global indices (O(n) instead of O(pages*n))
|
||||
let mut page_groups: HashMap<u32, Vec<(usize, &TextItem)>> = HashMap::new();
|
||||
for (global_idx, item) in text_items.iter().enumerate() {
|
||||
@@ -1517,6 +1627,15 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
);
|
||||
continue;
|
||||
}
|
||||
if is_running_furniture_table(subset_items, &table, &running_furniture) {
|
||||
log::debug!(
|
||||
"page {}: rejected {}x{} running header/footer table hypothesis",
|
||||
page,
|
||||
table.rows.len(),
|
||||
table.columns.len()
|
||||
);
|
||||
continue;
|
||||
}
|
||||
for &idx in &table.item_indices {
|
||||
if let Some(&band_idx) = index_map.get(idx) {
|
||||
if let Some(&page_idx) = band_index_map.get(band_idx) {
|
||||
@@ -1703,6 +1822,15 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
);
|
||||
continue;
|
||||
}
|
||||
if is_running_furniture_table(&chart_free, table, &running_furniture) {
|
||||
log::debug!(
|
||||
"page {}: rejected {}x{} merged-band running header/footer table hypothesis",
|
||||
page,
|
||||
table.rows.len(),
|
||||
table.columns.len()
|
||||
);
|
||||
continue;
|
||||
}
|
||||
for &idx in &table.item_indices {
|
||||
if let Some(&page_idx) = chart_free_map
|
||||
.get(idx)
|
||||
@@ -2058,6 +2186,174 @@ mod tests {
|
||||
assert!(md.contains("- Second item"));
|
||||
}
|
||||
|
||||
fn furniture_item(text: &str, x: f32, y: f32, page: u32) -> TextItem {
|
||||
let mut it = make_item(x, y, page);
|
||||
it.text = text.into();
|
||||
it
|
||||
}
|
||||
|
||||
/// Items repeating verbatim at the same position on 3+ pages are running
|
||||
/// furniture; the same text on fewer pages, or at different positions, is
|
||||
/// not.
|
||||
#[test]
|
||||
fn running_furniture_requires_three_pages_at_same_position() {
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=3 {
|
||||
// Body content so each page has a real vertical extent.
|
||||
items.push(furniture_item("body", 85.0, 700.0, page));
|
||||
items.push(furniture_item("TITULAR DEL", 85.0, 68.0, page));
|
||||
}
|
||||
// Same text but only two pages.
|
||||
for page in 1..=2 {
|
||||
items.push(furniture_item("SECRETARÍA", 200.0, 68.0, page));
|
||||
}
|
||||
// Same text on three pages but at drifting positions.
|
||||
for (page, x) in [(1, 300.0), (2, 320.0), (3, 340.0)] {
|
||||
items.push(furniture_item("MÉXICO", x, 68.0, page));
|
||||
}
|
||||
|
||||
let running = running_furniture_keys(&items);
|
||||
assert!(running.contains(&furniture_key(&furniture_item(
|
||||
"TITULAR DEL",
|
||||
85.0,
|
||||
68.0,
|
||||
1
|
||||
))));
|
||||
assert!(!running.contains(&furniture_key(&furniture_item(
|
||||
"SECRETARÍA",
|
||||
200.0,
|
||||
68.0,
|
||||
1
|
||||
))));
|
||||
assert!(!running.contains(&furniture_key(&furniture_item("MÉXICO", 300.0, 68.0, 1))));
|
||||
}
|
||||
|
||||
/// A table made of running-footer items is vetoed; a table whose body rows
|
||||
/// carry per-page content is kept even when its header row repeats.
|
||||
#[test]
|
||||
fn running_furniture_table_veto() {
|
||||
// The footer block, present identically on pages 1-3.
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=3 {
|
||||
items.push(furniture_item("PROPOSICIÓN CON PUNTO", 85.0, 78.5, page));
|
||||
items.push(furniture_item("EL SENADO", 286.6, 78.5, page));
|
||||
items.push(furniture_item("TITULAR DEL", 85.0, 68.0, page));
|
||||
items.push(furniture_item("A TRAVÉS DE LA", 243.4, 68.0, page));
|
||||
}
|
||||
// A real table on page 1: repeated header row, per-page data rows.
|
||||
let header = [
|
||||
furniture_item("Year", 85.0, 500.0, 1),
|
||||
furniture_item("Total", 200.0, 500.0, 1),
|
||||
];
|
||||
let data = [
|
||||
furniture_item("2023", 85.0, 488.0, 1),
|
||||
furniture_item("1,204", 200.0, 488.0, 1),
|
||||
furniture_item("2024", 85.0, 476.0, 1),
|
||||
furniture_item("1,377", 200.0, 476.0, 1),
|
||||
];
|
||||
// Header repeats on every page (like a continued table's header).
|
||||
for page in 2..=3 {
|
||||
items.push(furniture_item("Year", 85.0, 500.0, page));
|
||||
items.push(furniture_item("Total", 200.0, 500.0, page));
|
||||
}
|
||||
items.extend(header.iter().cloned());
|
||||
items.extend(data.iter().cloned());
|
||||
|
||||
let running = running_furniture_keys(&items);
|
||||
|
||||
let table_of = |detection_items: &[TextItem]| crate::tables::Table {
|
||||
columns: vec![],
|
||||
rows: vec![],
|
||||
cells: vec![],
|
||||
item_indices: (0..detection_items.len()).collect(),
|
||||
kind: crate::tables::TableKind::Data,
|
||||
};
|
||||
|
||||
// Footer-only candidate: every item is furniture -> vetoed.
|
||||
let footer_items: Vec<TextItem> = (1..=1)
|
||||
.flat_map(|page| {
|
||||
vec![
|
||||
furniture_item("PROPOSICIÓN CON PUNTO", 85.0, 78.5, page),
|
||||
furniture_item("EL SENADO", 286.6, 78.5, page),
|
||||
furniture_item("TITULAR DEL", 85.0, 68.0, page),
|
||||
furniture_item("A TRAVÉS DE LA", 243.4, 68.0, page),
|
||||
]
|
||||
})
|
||||
.collect();
|
||||
assert!(is_running_furniture_table(
|
||||
&footer_items,
|
||||
&table_of(&footer_items),
|
||||
&running
|
||||
));
|
||||
|
||||
// Real table: header row repeats across pages, body rows do not ->
|
||||
// 2 furniture of 6 items (33%) stays under the 80% threshold.
|
||||
let real_items: Vec<TextItem> =
|
||||
header.iter().cloned().chain(data.iter().cloned()).collect();
|
||||
assert!(!is_running_furniture_table(
|
||||
&real_items,
|
||||
&table_of(&real_items),
|
||||
&running
|
||||
));
|
||||
}
|
||||
|
||||
/// A form template repeated per record carries identical labels at
|
||||
/// identical mid-page coordinates on every page — those are real table
|
||||
/// cells, not furniture. Only the page-edge bands qualify.
|
||||
#[test]
|
||||
fn mid_page_repetition_is_not_furniture() {
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=4 {
|
||||
// Content spanning the page: y 60 (bottom) to 740 (top).
|
||||
items.push(furniture_item("body top", 85.0, 740.0, page));
|
||||
items.push(furniture_item("body bottom", 85.0, 60.0, page));
|
||||
// Form labels repeated dead centre on every page.
|
||||
items.push(furniture_item("Name of creditor", 85.0, 400.0, page));
|
||||
items.push(furniture_item("Amount of claim", 300.0, 400.0, page));
|
||||
// A genuine footer inside the bottom band.
|
||||
items.push(furniture_item("FORM 78 — page footer", 85.0, 70.0, page));
|
||||
}
|
||||
|
||||
let running = running_furniture_keys(&items);
|
||||
assert!(
|
||||
!running.contains(&furniture_key(&furniture_item(
|
||||
"Name of creditor",
|
||||
85.0,
|
||||
400.0,
|
||||
1
|
||||
))),
|
||||
"mid-page form labels must not be furniture"
|
||||
);
|
||||
assert!(running.contains(&furniture_key(&furniture_item(
|
||||
"FORM 78 — page footer",
|
||||
85.0,
|
||||
70.0,
|
||||
1
|
||||
))));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn running_furniture_empty_on_short_documents() {
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=2 {
|
||||
items.push(furniture_item("body", 85.0, 700.0, page));
|
||||
items.push(furniture_item("FOOTER", 85.0, 68.0, page));
|
||||
}
|
||||
assert!(running_furniture_keys(&items).is_empty());
|
||||
}
|
||||
|
||||
/// A page whose text has no vertical span (a single line) gives no
|
||||
/// evidence of where its edges are; its items never become furniture.
|
||||
#[test]
|
||||
fn zero_span_page_contributes_no_furniture() {
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=4 {
|
||||
items.push(furniture_item("ROW LABEL", 85.0, 400.0, page));
|
||||
items.push(furniture_item("ROW VALUE", 300.0, 400.0, page));
|
||||
}
|
||||
assert!(running_furniture_keys(&items).is_empty());
|
||||
}
|
||||
|
||||
fn make_item(x: f32, y: f32, page: u32) -> TextItem {
|
||||
TextItem {
|
||||
text: "A".into(),
|
||||
|
||||
+73
-12
@@ -1832,6 +1832,11 @@ fn merge_cmaps(mut base: ToUnicodeCMap, overlay: ToUnicodeCMap) -> ToUnicodeCMap
|
||||
base
|
||||
}
|
||||
|
||||
/// Upper bound on CID `/W` range expansion. The CID domain is 16-bit, so more
|
||||
/// than 65,536 unique keys cannot exist; repeating full-width ranges must not
|
||||
/// re-expand the same domain.
|
||||
pub(crate) const MAX_CID_W_EXPANSION: usize = 65_536;
|
||||
|
||||
/// Check if a CIDFont's /W (widths) array contains CID values that look like
|
||||
/// Unicode codepoints rather than low-value GIDs.
|
||||
///
|
||||
@@ -1843,20 +1848,23 @@ pub(crate) fn cid_values_look_like_unicode(cid_font_dict: &lopdf::Dictionary) ->
|
||||
_ => return false,
|
||||
};
|
||||
|
||||
// The /W array format: [cid [w1 w2 ...]] or [cid_start cid_end w]
|
||||
// We extract all CID values (the first element of each group).
|
||||
let mut cids: Vec<u16> = Vec::new();
|
||||
// The /W array format: [cid [w1 w2 ...]] or [cid_start cid_end w].
|
||||
// Collect unique CIDs only: repeating a full-width range must not grow a
|
||||
// temporary vector (or the sort) with the range length on every copy.
|
||||
let mut seen = HashSet::new();
|
||||
let mut i = 0;
|
||||
while i < w_arr.len() {
|
||||
while i < w_arr.len() && seen.len() < MAX_CID_W_EXPANSION {
|
||||
if let Ok(cid) = w_arr[i].as_i64() {
|
||||
cids.push(cid as u16);
|
||||
// Skip the width data
|
||||
let start = cid as u16;
|
||||
if i + 1 < w_arr.len() {
|
||||
match &w_arr[i + 1] {
|
||||
Object::Array(widths) => {
|
||||
// [cid [w1 w2 ...]] — CIDs are cid, cid+1, ..., cid+len-1
|
||||
for j in 1..widths.len() {
|
||||
cids.push((cid as u16).wrapping_add(j as u16));
|
||||
for j in 0..widths.len() {
|
||||
if seen.len() >= MAX_CID_W_EXPANSION {
|
||||
break;
|
||||
}
|
||||
seen.insert(start.wrapping_add(j as u16));
|
||||
}
|
||||
i += 2;
|
||||
}
|
||||
@@ -1864,9 +1872,7 @@ pub(crate) fn cid_values_look_like_unicode(cid_font_dict: &lopdf::Dictionary) ->
|
||||
// [cid_start cid_end w] — range of CIDs
|
||||
if i + 2 < w_arr.len() {
|
||||
if let Ok(cid_end) = w_arr[i + 1].as_i64() {
|
||||
for c in (cid as u16)..=(cid_end as u16) {
|
||||
cids.push(c);
|
||||
}
|
||||
record_unique_cid_range(start, cid_end as u16, &mut seen);
|
||||
}
|
||||
i += 3;
|
||||
} else {
|
||||
@@ -1875,6 +1881,7 @@ pub(crate) fn cid_values_look_like_unicode(cid_font_dict: &lopdf::Dictionary) ->
|
||||
}
|
||||
}
|
||||
} else {
|
||||
seen.insert(start);
|
||||
i += 1;
|
||||
}
|
||||
} else {
|
||||
@@ -1882,10 +1889,11 @@ pub(crate) fn cid_values_look_like_unicode(cid_font_dict: &lopdf::Dictionary) ->
|
||||
}
|
||||
}
|
||||
|
||||
if cids.is_empty() {
|
||||
if seen.is_empty() {
|
||||
return false;
|
||||
}
|
||||
|
||||
let mut cids: Vec<u16> = seen.into_iter().collect();
|
||||
cids.sort_unstable();
|
||||
let median = cids[cids.len() / 2];
|
||||
// Unicode text CIDs are typically >= 0x20 (space) with letters at 0x41+.
|
||||
@@ -1894,6 +1902,18 @@ pub(crate) fn cid_values_look_like_unicode(cid_font_dict: &lopdf::Dictionary) ->
|
||||
median >= 0x41
|
||||
}
|
||||
|
||||
fn record_unique_cid_range(start: u16, end: u16, seen: &mut HashSet<u16>) {
|
||||
if start > end {
|
||||
return;
|
||||
}
|
||||
for cid in start..=end {
|
||||
if seen.len() >= MAX_CID_W_EXPANSION {
|
||||
return;
|
||||
}
|
||||
seen.insert(cid);
|
||||
}
|
||||
}
|
||||
|
||||
/// Build a ToUnicodeCMap from predefined CID→Unicode mapping based on CIDSystemInfo.
|
||||
///
|
||||
/// Supports Adobe-Korea1 (Korean) character collection. Can be extended for
|
||||
@@ -3298,4 +3318,45 @@ endbfrange
|
||||
"An indirect /Subtype naming CIDFontType2 must still reach the remap"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cid_values_look_like_unicode_letter_range() {
|
||||
let mut dict = lopdf::Dictionary::new();
|
||||
dict.set(
|
||||
"W",
|
||||
Object::Array(vec![
|
||||
Object::Integer(0x41),
|
||||
Object::Integer(0x5A),
|
||||
Object::Integer(500),
|
||||
]),
|
||||
);
|
||||
assert!(cid_values_look_like_unicode(&dict));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cid_values_look_like_unicode_low_gids() {
|
||||
let mut dict = lopdf::Dictionary::new();
|
||||
dict.set(
|
||||
"W",
|
||||
Object::Array(vec![
|
||||
Object::Integer(0),
|
||||
Object::Array(vec![Object::Integer(500); 10]),
|
||||
]),
|
||||
);
|
||||
assert!(!cid_values_look_like_unicode(&dict));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cid_values_look_like_unicode_repeated_full_ranges_stay_bounded() {
|
||||
// Repeating `[0 65535 w]` must not materialize 65,536 CIDs per copy.
|
||||
let mut w = Vec::new();
|
||||
for _ in 0..5_000 {
|
||||
w.push(Object::Integer(0));
|
||||
w.push(Object::Integer(65535));
|
||||
w.push(Object::Integer(500));
|
||||
}
|
||||
let mut dict = lopdf::Dictionary::new();
|
||||
dict.set("W", Object::Array(w));
|
||||
assert!(cid_values_look_like_unicode(&dict));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1654,6 +1654,282 @@ fn test_extract_regions_mem_basic_text_pdf() {
|
||||
assert_eq!(regions[0].page, 0);
|
||||
}
|
||||
|
||||
/// Build a synthetic "scanned page" PDF: a full-page image XObject with a
|
||||
/// text layer drawn in the given render mode (3 = invisible OCR overlay,
|
||||
/// 0 = normal visible fill). `visible_extra` optionally adds a normally
|
||||
/// rendered line so double-layer behavior can be tested; `layer_lines`
|
||||
/// overrides the layer content (default: three pangram lines);
|
||||
/// `quote_ops` shows every layer line via the `'` operator instead of Tj
|
||||
/// (both are standard show-text encodings for OCR layers).
|
||||
fn make_pdf_with_custom_text_layer(
|
||||
text_render_mode: i32,
|
||||
visible_extra: Option<&str>,
|
||||
layer_lines: Option<&[&str]>,
|
||||
quote_ops: bool,
|
||||
) -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
fn add_object(pdf: &mut Vec<u8>, offsets: &mut Vec<usize>, id: usize, body: &str) {
|
||||
offsets.push(pdf.len());
|
||||
pdf.extend_from_slice(format!("{id} 0 obj\n").as_bytes());
|
||||
pdf.extend_from_slice(body.as_bytes());
|
||||
pdf.extend_from_slice(b"\nendobj\n");
|
||||
}
|
||||
fn add_stream_object(
|
||||
pdf: &mut Vec<u8>,
|
||||
offsets: &mut Vec<usize>,
|
||||
id: usize,
|
||||
dict: &str,
|
||||
stream_bytes: &[u8],
|
||||
) {
|
||||
offsets.push(pdf.len());
|
||||
pdf.extend_from_slice(format!("{id} 0 obj\n").as_bytes());
|
||||
pdf.extend_from_slice(
|
||||
format!("<< {} /Length {} >>\nstream\n", dict, stream_bytes.len()).as_bytes(),
|
||||
);
|
||||
pdf.extend_from_slice(stream_bytes);
|
||||
pdf.extend_from_slice(b"\nendstream\nendobj\n");
|
||||
}
|
||||
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
1,
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
2,
|
||||
"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
3,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] \
|
||||
/Resources << /Font << /F1 5 0 R >> /XObject << /Im0 6 0 R >> >> \
|
||||
/Contents 4 0 R >>",
|
||||
);
|
||||
// Full-page raster, then the text layer in the requested render mode —
|
||||
// several lines so the OCR-layer gate's alnum floor (40) is well cleared.
|
||||
let mut content = String::from("q 612 0 0 792 0 0 cm /Im0 Do Q\n");
|
||||
let default_layer = [
|
||||
"The quick brown fox jumps over the lazy dog",
|
||||
"Pack my box with five dozen liquor jugs tonight",
|
||||
"Sphinx of black quartz judge my vow carefully",
|
||||
];
|
||||
let layer: &[&str] = layer_lines.unwrap_or(&default_layer);
|
||||
if quote_ops {
|
||||
// Every line shown via `'` (move-to-next-line + show) — nothing on
|
||||
// this layer goes through Tj, pinning the `'` suppression path.
|
||||
content.push_str(&format!(
|
||||
"BT /F1 12 Tf {text_render_mode} Tr 16 TL 72 716 Td "
|
||||
));
|
||||
for line in layer {
|
||||
content.push_str(&format!("({line}) ' "));
|
||||
}
|
||||
} else {
|
||||
content.push_str(&format!("BT /F1 12 Tf {text_render_mode} Tr 72 700 Td "));
|
||||
for (i, line) in layer.iter().enumerate() {
|
||||
if i > 0 {
|
||||
content.push_str("0 -16 Td ");
|
||||
}
|
||||
content.push_str(&format!("({line}) Tj "));
|
||||
}
|
||||
}
|
||||
content.push_str("ET\n");
|
||||
if let Some(extra) = visible_extra {
|
||||
content.push_str(&format!("BT /F1 12 Tf 0 Tr 72 500 Td ({extra}) Tj ET\n"));
|
||||
}
|
||||
add_stream_object(&mut pdf, &mut offsets, 4, "", content.as_bytes());
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
5,
|
||||
"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
|
||||
);
|
||||
let image_pixel = [128u8];
|
||||
add_stream_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
6,
|
||||
"/Type /XObject /Subtype /Image /Width 1 /Height 1 \
|
||||
/ColorSpace /DeviceGray /BitsPerComponent 8",
|
||||
&image_pixel,
|
||||
);
|
||||
|
||||
let xref_start = pdf.len();
|
||||
pdf.extend_from_slice(format!("xref\n0 {}\n", offsets.len()).as_bytes());
|
||||
pdf.extend_from_slice(b"0000000000 65535 f \n");
|
||||
for offset in offsets.iter().skip(1) {
|
||||
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
|
||||
}
|
||||
pdf.extend_from_slice(
|
||||
format!(
|
||||
"trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{}\n%%EOF",
|
||||
offsets.len(),
|
||||
xref_start
|
||||
)
|
||||
.as_bytes(),
|
||||
);
|
||||
pdf
|
||||
}
|
||||
|
||||
fn make_pdf_with_text_layer(text_render_mode: i32, visible_extra: Option<&str>) -> Vec<u8> {
|
||||
make_pdf_with_custom_text_layer(text_render_mode, visible_extra, None, false)
|
||||
}
|
||||
|
||||
/// A scanned page whose only text is an invisible (Tr 3) OCR layer behind
|
||||
/// the raster must serve that layer from the region extractor instead of
|
||||
/// reporting the region as needs_ocr — the exact text is already in the PDF.
|
||||
#[test]
|
||||
fn test_extract_regions_mem_recovers_invisible_ocr_layer() {
|
||||
let buf = make_pdf_with_text_layer(3, None);
|
||||
let regions = extract_text_in_regions_mem(&buf, &full_page_regions(1)).unwrap();
|
||||
assert_eq!(regions.len(), 1);
|
||||
let region = ®ions[0].regions[0];
|
||||
assert!(
|
||||
region.text.contains("quick brown fox"),
|
||||
"invisible OCR layer should be served as region text, got: {:?}",
|
||||
region.text
|
||||
);
|
||||
assert!(
|
||||
!region.needs_ocr,
|
||||
"recovered OCR layer must not fall back to GPU OCR"
|
||||
);
|
||||
}
|
||||
|
||||
/// ANY visible text on the page — even a single short line — must block the
|
||||
/// invisible-layer adoption entirely: the invisible pass returns visible
|
||||
/// items too, so adopting it alongside visible text would duplicate the
|
||||
/// visible words. Strict zero-visible gate, no fuzzy dedupe.
|
||||
#[test]
|
||||
fn test_extract_regions_mem_visible_text_blocks_invisible_layer() {
|
||||
let buf = make_pdf_with_text_layer(3, Some("Folio 142"));
|
||||
let regions = extract_text_in_regions_mem(&buf, &full_page_regions(1)).unwrap();
|
||||
let region = ®ions[0].regions[0];
|
||||
assert!(
|
||||
region.text.contains("Folio 142"),
|
||||
"visible text should be extracted, got: {:?}",
|
||||
region.text
|
||||
);
|
||||
assert!(
|
||||
!region.text.contains("quick brown fox"),
|
||||
"invisible layer must not be adopted when any visible text exists, got: {:?}",
|
||||
region.text
|
||||
);
|
||||
assert_eq!(
|
||||
region.text.matches("Folio 142").count(),
|
||||
1,
|
||||
"visible text must appear exactly once, got: {:?}",
|
||||
region.text
|
||||
);
|
||||
}
|
||||
|
||||
/// An invisible OCR layer shown entirely via the `'` show-text operator
|
||||
/// (move-to-next-line + show) must also be recovered — the skipped_invisible
|
||||
/// signal has to fire on every show-text path, not just Tj/TJ.
|
||||
#[test]
|
||||
fn test_extract_regions_mem_recovers_quote_operator_layer() {
|
||||
let buf = make_pdf_with_custom_text_layer(3, None, None, true);
|
||||
let regions = extract_text_in_regions_mem(&buf, &full_page_regions(1)).unwrap();
|
||||
let region = ®ions[0].regions[0];
|
||||
assert!(
|
||||
region.text.contains("quick brown fox"),
|
||||
"'-operator OCR layer should be recovered, got: {:?}",
|
||||
region.text
|
||||
);
|
||||
assert!(!region.needs_ocr);
|
||||
}
|
||||
|
||||
/// An invisible layer below the 40-alnum floor (a stray watermark line)
|
||||
/// must NOT be adopted — the region keeps its needs_ocr fallback.
|
||||
#[test]
|
||||
fn test_extract_regions_mem_tiny_invisible_layer_not_adopted() {
|
||||
let buf = make_pdf_with_custom_text_layer(3, None, Some(&["Scanned by ACME"]), false);
|
||||
let regions = extract_text_in_regions_mem(&buf, &full_page_regions(1)).unwrap();
|
||||
let region = ®ions[0].regions[0];
|
||||
assert!(
|
||||
!region.text.contains("Scanned by ACME"),
|
||||
"below-floor invisible layer must not be adopted, got: {:?}",
|
||||
region.text
|
||||
);
|
||||
// Only the raster placeholder remains — needs_ocr stays whatever main
|
||||
// reports for placeholder-only regions (false today; downstream
|
||||
// pipelines route placeholder-only text to OCR themselves, and this PR
|
||||
// deliberately does not change that contract).
|
||||
assert!(
|
||||
region.text.trim().starts_with("[Image:"),
|
||||
"region should hold only the raster placeholder, got: {:?}",
|
||||
region.text
|
||||
);
|
||||
}
|
||||
|
||||
/// An invisible layer that clears the alnum floor but is mostly symbol
|
||||
/// garbage (a broken OCR run) must be rejected by the garbage gate.
|
||||
#[test]
|
||||
fn test_extract_regions_mem_garbage_invisible_layer_not_adopted() {
|
||||
// Each line: 5 alphanumerics among 15 symbol chars. Ten lines clear the
|
||||
// 40-alnum floor (50 alnum) while staying well under the half-alnum
|
||||
// ratio is_garbage_text requires.
|
||||
let garbage_lines: Vec<&str> = vec!["a@@b%%c&&d==e~~"; 10];
|
||||
let buf = make_pdf_with_custom_text_layer(3, None, Some(&garbage_lines), false);
|
||||
let regions = extract_text_in_regions_mem(&buf, &full_page_regions(1)).unwrap();
|
||||
let region = ®ions[0].regions[0];
|
||||
assert!(
|
||||
!region.text.contains("a@@b"),
|
||||
"garbage invisible layer must not be adopted, got: {:?}",
|
||||
region.text
|
||||
);
|
||||
assert!(
|
||||
region.text.trim().starts_with("[Image:"),
|
||||
"region should hold only the raster placeholder, got: {:?}",
|
||||
region.text
|
||||
);
|
||||
}
|
||||
|
||||
/// Punctuation-only visible text (zero alphanumerics) must ALSO block
|
||||
/// adoption — the gate is item-presence, not alphanumeric mass. (Real-world
|
||||
/// rationale: an invisible OCR layer transcribes the raster, so visible
|
||||
/// glyphs typically have invisible twins there; this fixture's layers are
|
||||
/// disjoint, so it pins the gate itself, not the duplication scenario.)
|
||||
#[test]
|
||||
fn test_extract_regions_mem_punctuation_visible_blocks_invisible_layer() {
|
||||
let buf = make_pdf_with_text_layer(3, Some("... --- ..."));
|
||||
let regions = extract_text_in_regions_mem(&buf, &full_page_regions(1)).unwrap();
|
||||
let region = ®ions[0].regions[0];
|
||||
assert!(
|
||||
!region.text.contains("quick brown fox"),
|
||||
"invisible layer must not be adopted over punctuation-only visible text, got: {:?}",
|
||||
region.text
|
||||
);
|
||||
assert_eq!(
|
||||
region.text.matches("... --- ...").count(),
|
||||
1,
|
||||
"visible punctuation must be preserved exactly once, got: {:?}",
|
||||
region.text
|
||||
);
|
||||
}
|
||||
|
||||
/// Regression guard: a normal visible-text page (render mode 0) is served
|
||||
/// once and only once — if the fallback ever mis-fired here and merged a
|
||||
/// second pass, the phrase would duplicate.
|
||||
#[test]
|
||||
fn test_extract_regions_mem_visible_layer_unchanged() {
|
||||
let buf = make_pdf_with_text_layer(0, None);
|
||||
let regions = extract_text_in_regions_mem(&buf, &full_page_regions(1)).unwrap();
|
||||
let region = ®ions[0].regions[0];
|
||||
assert_eq!(
|
||||
region.text.matches("quick brown fox").count(),
|
||||
1,
|
||||
"visible text must appear exactly once, got: {:?}",
|
||||
region.text
|
||||
);
|
||||
assert!(!region.needs_ocr);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_regions_mem_identity_h_needs_ocr() {
|
||||
let buf = std::fs::read("tests/fixtures/shinagawa_identity_h.pdf").unwrap();
|
||||
|
||||
Generated
+2
-2
@@ -724,7 +724,7 @@ checksum = "d6790f58c7ff633d8771f42965289203411a5e5c68388703c06e14f24770b41e"
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector"
|
||||
version = "1.14.0"
|
||||
version = "1.14.1"
|
||||
dependencies = [
|
||||
"env_logger",
|
||||
"include_dir",
|
||||
@@ -740,7 +740,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector-wasm"
|
||||
version = "1.14.0"
|
||||
version = "1.14.1"
|
||||
dependencies = [
|
||||
"console_error_panic_hook",
|
||||
"js-sys",
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "pdf-inspector-wasm"
|
||||
version = "1.14.0"
|
||||
version = "1.14.1"
|
||||
edition = "2021"
|
||||
authors = ["Firecrawl Team"]
|
||||
description = "Browser WebAssembly bindings for pdf-inspector"
|
||||
|
||||
Reference in New Issue
Block a user