Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8bbfd4906f | ||
|
|
3b1fb02333 |
@@ -385,6 +385,7 @@ pub(crate) fn extract_page_text_items(
|
|||||||
&font_encodings,
|
&font_encodings,
|
||||||
&encoding_cache,
|
&encoding_cache,
|
||||||
&mut cmap_decisions,
|
&mut cmap_decisions,
|
||||||
|
&font_widths,
|
||||||
) {
|
) {
|
||||||
let combined = multiply_matrices(&text_matrix, &ctm);
|
let combined = multiply_matrices(&text_matrix, &ctm);
|
||||||
let rendered_size = effective_font_size(current_font_size, &combined);
|
let rendered_size = effective_font_size(current_font_size, &combined);
|
||||||
@@ -533,6 +534,7 @@ pub(crate) fn extract_page_text_items(
|
|||||||
&font_encodings,
|
&font_encodings,
|
||||||
&encoding_cache,
|
&encoding_cache,
|
||||||
&mut cmap_decisions,
|
&mut cmap_decisions,
|
||||||
|
&font_widths,
|
||||||
) {
|
) {
|
||||||
current_text.push_str(&text);
|
current_text.push_str(&text);
|
||||||
}
|
}
|
||||||
@@ -620,6 +622,7 @@ pub(crate) fn extract_page_text_items(
|
|||||||
&font_encodings,
|
&font_encodings,
|
||||||
&encoding_cache,
|
&encoding_cache,
|
||||||
&mut cmap_decisions,
|
&mut cmap_decisions,
|
||||||
|
&font_widths,
|
||||||
) {
|
) {
|
||||||
if !text.trim().is_empty() {
|
if !text.trim().is_empty() {
|
||||||
let combined = multiply_matrices(&text_matrix, &ctm);
|
let combined = multiply_matrices(&text_matrix, &ctm);
|
||||||
|
|||||||
+122
-1
@@ -720,7 +720,11 @@ pub(crate) fn extract_text_from_operand(
|
|||||||
font_encodings: &PageFontEncodings,
|
font_encodings: &PageFontEncodings,
|
||||||
encoding_cache: &HashMap<String, Encoding<'_>>,
|
encoding_cache: &HashMap<String, Encoding<'_>>,
|
||||||
cmap_decisions: &mut CMapDecisionCache,
|
cmap_decisions: &mut CMapDecisionCache,
|
||||||
|
font_widths: &PageFontWidths,
|
||||||
) -> Option<String> {
|
) -> Option<String> {
|
||||||
|
let is_type0_cid_font = font_widths
|
||||||
|
.get(current_font)
|
||||||
|
.is_some_and(|info| info.is_cid);
|
||||||
let result = (|| -> Option<String> {
|
let result = (|| -> Option<String> {
|
||||||
if let Object::String(bytes, _) = obj {
|
if let Object::String(bytes, _) = obj {
|
||||||
let mut decode_with_entry = |entry: &crate::tounicode::CMapEntry| -> Option<String> {
|
let mut decode_with_entry = |entry: &crate::tounicode::CMapEntry| -> Option<String> {
|
||||||
@@ -962,7 +966,31 @@ pub(crate) fn extract_text_from_operand(
|
|||||||
return Some(symbol_text);
|
return Some(symbol_text);
|
||||||
}
|
}
|
||||||
|
|
||||||
// Latin-1 fallback
|
// Latin-1 fallback. Safe ONLY for fonts that use single-byte
|
||||||
|
// encodings — for these, an unmapped byte is a valid character
|
||||||
|
// code in Latin-1/WinAnsi space. CID fonts (Type0 / Identity-H)
|
||||||
|
// emit multi-byte CIDs that aren't characters; per-byte Latin-1
|
||||||
|
// produces mojibake (e.g. 2-byte CID 0xCDD9 → "ÍÙ" for the
|
||||||
|
// production scrape_id 019de78c-... samples).
|
||||||
|
//
|
||||||
|
// For a CID font (has_cmap is set OR a /ToUnicode reference
|
||||||
|
// exists) with any non-ASCII bytes, emit a single U+FFFD per
|
||||||
|
// CID instead. This both replaces the mojibake with a proper
|
||||||
|
// "decode failed" marker AND keeps `detect_encoding_issues`
|
||||||
|
// tripping so the page is flagged for OCR — the existing
|
||||||
|
// garbage-detection path that the high-Latin-1 mojibake used
|
||||||
|
// to satisfy by accident.
|
||||||
|
if is_type0_cid_font && bytes.iter().any(|&b| b > 0x7F) {
|
||||||
|
// 2-byte CIDs (Identity-H) are by far the common case; for
|
||||||
|
// an odd byte count we still emit at least one marker so
|
||||||
|
// detection downstream fires.
|
||||||
|
let cid_count = (bytes.len() / 2).max(1);
|
||||||
|
return Some("\u{FFFD}".repeat(cid_count));
|
||||||
|
}
|
||||||
|
// Pure ASCII bytes round-trip safely (Latin-1 == ASCII for
|
||||||
|
// 0x00..=0x7F), and non-CID (Type1 / TrueType / Type3) fonts
|
||||||
|
// use single-byte encodings where Latin-1 fallback is the
|
||||||
|
// canonical interpretation.
|
||||||
Some(bytes.iter().map(|&b| b as char).collect())
|
Some(bytes.iter().map(|&b| b as char).collect())
|
||||||
} else {
|
} else {
|
||||||
None
|
None
|
||||||
@@ -1213,4 +1241,97 @@ mod tests {
|
|||||||
let bad = "###!!!@@@$$$";
|
let bad = "###!!!@@@$$$";
|
||||||
assert!(score_text(good) > score_text(bad));
|
assert!(score_text(good) > score_text(bad));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn cid_font_with_unparseable_cmap_does_not_emit_latin1_mojibake() {
|
||||||
|
// Type0/CID font (font_widths reports `is_cid=true`) where the
|
||||||
|
// ToUnicode CMap couldn't be parsed (FontCMaps doesn't have the
|
||||||
|
// obj_num). Bytes are a 2-byte CID stream containing high bytes
|
||||||
|
// that aren't valid UTF-8 — exactly the case in the production
|
||||||
|
// samples (Identity-H text where the ToUnicode CMap was missing
|
||||||
|
// or malformed, scrape_id 019de78c-..., e.g. "Í Ù Z)¿").
|
||||||
|
//
|
||||||
|
// Without the guard, the function falls through to the byte-by-byte
|
||||||
|
// Latin-1 fallback and produces "ÍÙ" (U+00CD U+00D9). The correct
|
||||||
|
// behavior is to emit U+FFFD per CID so downstream
|
||||||
|
// `detect_encoding_issues` flags the page for OCR.
|
||||||
|
let bytes = vec![0xCD_u8, 0xD9, 0xCD, 0xD9];
|
||||||
|
let obj = Object::String(bytes, lopdf::StringFormat::Hexadecimal);
|
||||||
|
|
||||||
|
let font_cmaps = FontCMaps::default();
|
||||||
|
let mut font_tounicode_refs: HashMap<String, u32> = HashMap::new();
|
||||||
|
font_tounicode_refs.insert("F0".to_string(), 999);
|
||||||
|
let inline_cmaps = HashMap::new();
|
||||||
|
let font_encodings: PageFontEncodings = HashMap::new();
|
||||||
|
let encoding_cache: HashMap<String, Encoding<'_>> = HashMap::new();
|
||||||
|
let mut decisions = CMapDecisionCache::new();
|
||||||
|
let mut font_widths: PageFontWidths = HashMap::new();
|
||||||
|
font_widths.insert("F0".to_string(), make_font_info(&[], 1000, true));
|
||||||
|
|
||||||
|
let result = extract_text_from_operand(
|
||||||
|
&obj,
|
||||||
|
"F0",
|
||||||
|
None,
|
||||||
|
&font_cmaps,
|
||||||
|
&font_tounicode_refs,
|
||||||
|
&inline_cmaps,
|
||||||
|
&font_encodings,
|
||||||
|
&encoding_cache,
|
||||||
|
&mut decisions,
|
||||||
|
&font_widths,
|
||||||
|
);
|
||||||
|
|
||||||
|
let text = result.expect("CID font fallback should still emit a marker");
|
||||||
|
assert!(
|
||||||
|
!text.contains('\u{00CD}') && !text.contains('\u{00D9}'),
|
||||||
|
"CID font with unparseable CMap leaked Latin-1 mojibake: {text:?}"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
text.contains('\u{FFFD}'),
|
||||||
|
"CID font with unparseable CMap should emit U+FFFD so detect_encoding_issues fires: {text:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn simple_font_latin1_fallback_passes_high_bytes_through() {
|
||||||
|
// A Type1/TrueType simple font (is_cid=false) with a `/ToUnicode`
|
||||||
|
// reference but no usable CMap and no `/Differences` map.
|
||||||
|
// Per-byte Latin-1 IS the canonical interpretation here — these
|
||||||
|
// bytes are character codes, not CIDs. The CID guard must NOT
|
||||||
|
// strip them. Reproduces the false positive that an earlier
|
||||||
|
// version of the guard introduced for fonts in PDFs like
|
||||||
|
// pdf-evals/Navigating-Artificial-Intelligence-..., where bytes
|
||||||
|
// like 0xB6 are legitimate Latin-1 character codes.
|
||||||
|
let bytes = vec![0x24_u8, 0x47, 0xB6, 0x56]; // "$G¶V"
|
||||||
|
let obj = Object::String(bytes, lopdf::StringFormat::Hexadecimal);
|
||||||
|
|
||||||
|
let font_cmaps = FontCMaps::default();
|
||||||
|
let mut font_tounicode_refs: HashMap<String, u32> = HashMap::new();
|
||||||
|
font_tounicode_refs.insert("F1".to_string(), 999);
|
||||||
|
let inline_cmaps = HashMap::new();
|
||||||
|
let font_encodings: PageFontEncodings = HashMap::new();
|
||||||
|
let encoding_cache: HashMap<String, Encoding<'_>> = HashMap::new();
|
||||||
|
let mut decisions = CMapDecisionCache::new();
|
||||||
|
let mut font_widths: PageFontWidths = HashMap::new();
|
||||||
|
font_widths.insert("F1".to_string(), make_font_info(&[], 1000, false));
|
||||||
|
|
||||||
|
let text = extract_text_from_operand(
|
||||||
|
&obj,
|
||||||
|
"F1",
|
||||||
|
None,
|
||||||
|
&font_cmaps,
|
||||||
|
&font_tounicode_refs,
|
||||||
|
&inline_cmaps,
|
||||||
|
&font_encodings,
|
||||||
|
&encoding_cache,
|
||||||
|
&mut decisions,
|
||||||
|
&font_widths,
|
||||||
|
)
|
||||||
|
.expect("simple font should round-trip Latin-1 bytes");
|
||||||
|
assert_eq!(text, "$G\u{00B6}V");
|
||||||
|
assert!(
|
||||||
|
!text.contains('\u{FFFD}'),
|
||||||
|
"simple font fallback must not stamp FFFD over legitimate bytes: {text:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -373,6 +373,7 @@ fn extract_form_xobject_text_inner(
|
|||||||
&font_encodings,
|
&font_encodings,
|
||||||
&encoding_cache,
|
&encoding_cache,
|
||||||
cmap_decisions,
|
cmap_decisions,
|
||||||
|
&font_widths,
|
||||||
) {
|
) {
|
||||||
let combined = multiply_matrices(&text_matrix, &ctm);
|
let combined = multiply_matrices(&text_matrix, &ctm);
|
||||||
let rendered_size = effective_font_size(current_font_size, &combined);
|
let rendered_size = effective_font_size(current_font_size, &combined);
|
||||||
@@ -517,6 +518,7 @@ fn extract_form_xobject_text_inner(
|
|||||||
&font_encodings,
|
&font_encodings,
|
||||||
&encoding_cache,
|
&encoding_cache,
|
||||||
cmap_decisions,
|
cmap_decisions,
|
||||||
|
&font_widths,
|
||||||
) {
|
) {
|
||||||
current_text.push_str(&text);
|
current_text.push_str(&text);
|
||||||
}
|
}
|
||||||
|
|||||||
+130
-45
@@ -1556,26 +1556,32 @@ pub fn extract_tables_with_structure_cells_mem(
|
|||||||
|
|
||||||
normalize_cell_bands(&mut cells);
|
normalize_cell_bands(&mut cells);
|
||||||
|
|
||||||
// Stage 1: exclusive per-item assignment. For each PDF text item,
|
// Stage 1: exclusive per-token assignment. Each PDF text item is
|
||||||
// find the cell(s) whose (band-clamped) bbox satisfies the strict
|
// first split into whitespace-separated tokens with estimated x
|
||||||
// membership rule (`tsr_region_contains_item`: center inside OR
|
// positions (see `split_item_into_token_subitems`). For each token
|
||||||
// >=60% overlap on both axes). If multiple cells qualify, assign
|
// we find the cell(s) whose (band-clamped) bbox satisfies the
|
||||||
// the item to the cell whose center is geometrically closest. If
|
// strict membership rule (`tsr_region_contains_item`: center
|
||||||
// exactly one qualifies, assign to that. If none, the item is an
|
// inside OR >=60% overlap on both axes). If multiple cells
|
||||||
// orphan and stage 2 below tries to recover it.
|
// qualify, the closest-center wins. Tokens that don't land in any
|
||||||
|
// cell are eligible for stage-2 orphan recovery.
|
||||||
//
|
//
|
||||||
// The exclusivity (one item → one cell) prevents the cell-overlap
|
// Per-token (rather than per-item) routing is what prevents the
|
||||||
// bug where SLANet emits cells whose y-extents overlap between
|
// dense-grid collapse bug: a row rendered as one wide Tj
|
||||||
// rows: under the previous "for each cell, gather items" approach,
|
// ("Marshall Islands 0.9 0.9 0.9") produces one TextItem whose
|
||||||
// an item whose center fell in two cells' overlap got duplicated
|
// CENTER lies in only one cell, so per-item routing parks the
|
||||||
// into both. Closest-center disambiguation routes it to the
|
// whole row in that single cell and leaves the rest of the row
|
||||||
// correct row.
|
// empty. Per-token routing distributes the words to whichever
|
||||||
let mut claimed: std::collections::HashSet<usize> = std::collections::HashSet::new();
|
// cells their (estimated) positions fall into. Single-token items
|
||||||
let mut item_to_cell: std::collections::HashMap<usize, usize> =
|
// collapse to a one-element token list and behave exactly as
|
||||||
std::collections::HashMap::new();
|
// before.
|
||||||
|
//
|
||||||
|
// The token-level exclusivity (one token → one cell) still
|
||||||
|
// prevents the cell-overlap bug where SLANet emits cells whose
|
||||||
|
// y-extents overlap between rows. Closest-center disambiguation
|
||||||
|
// routes each token to the correct row.
|
||||||
|
|
||||||
// Pre-compute each cell's bounds + center (in PDF-pt-flipped space)
|
// Pre-compute each cell's bounds + center (in PDF-pt-flipped space)
|
||||||
// so we don't redo the work per item.
|
// so we don't redo the work per token.
|
||||||
let cell_meta: Vec<Option<(RegionBounds, f32, f32)>> = cells
|
let cell_meta: Vec<Option<(RegionBounds, f32, f32)>> = cells
|
||||||
.iter()
|
.iter()
|
||||||
.map(|cell| {
|
.map(|cell| {
|
||||||
@@ -1590,44 +1596,53 @@ pub fn extract_tables_with_structure_cells_mem(
|
|||||||
})
|
})
|
||||||
.collect();
|
.collect();
|
||||||
|
|
||||||
for (item_idx, item) in items.iter().enumerate() {
|
let mut per_cell_items: Vec<Vec<TextItem>> = vec![Vec::new(); cells.len()];
|
||||||
let item_w = text_utils::effective_width(item);
|
// Tokens of an item that did NOT land in any cell during stage 1.
|
||||||
let item_cx = item.x + item_w * 0.5;
|
// These are the orphan candidates handed to `tsr_assign_orphan_items`.
|
||||||
let item_cy = item.y + item.height * 0.5;
|
// Using token-grain orphan candidates (rather than the original wide
|
||||||
let mut best: Option<(usize, f32)> = None;
|
// item) lets stage 2 recover individual words that fell just outside
|
||||||
for (cell_idx, meta) in cell_meta.iter().enumerate() {
|
// their cell's clamped band, without re-attributing already-claimed
|
||||||
let Some((bounds, ccx, ccy)) = meta else {
|
// tokens.
|
||||||
continue;
|
let mut orphan_token_subitems: Vec<TextItem> = Vec::new();
|
||||||
};
|
|
||||||
if !tsr_region_contains_item(item, *bounds) {
|
for item in items.iter() {
|
||||||
continue;
|
let token_subitems = split_item_into_token_subitems(item);
|
||||||
|
for token_item in token_subitems {
|
||||||
|
let token_w = text_utils::effective_width(&token_item);
|
||||||
|
let token_cx = token_item.x + token_w * 0.5;
|
||||||
|
let token_cy = token_item.y + token_item.height * 0.5;
|
||||||
|
let mut best: Option<(usize, f32)> = None;
|
||||||
|
for (cell_idx, meta) in cell_meta.iter().enumerate() {
|
||||||
|
let Some((bounds, ccx, ccy)) = meta else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
if !tsr_region_contains_item(&token_item, *bounds) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let dx = token_cx - ccx;
|
||||||
|
let dy = token_cy - ccy;
|
||||||
|
let dist_sq = dx * dx + dy * dy;
|
||||||
|
if best.is_none_or(|(_, d)| dist_sq < d) {
|
||||||
|
best = Some((cell_idx, dist_sq));
|
||||||
|
}
|
||||||
}
|
}
|
||||||
let dx = item_cx - ccx;
|
if let Some((ci, _)) = best {
|
||||||
let dy = item_cy - ccy;
|
per_cell_items[ci].push(token_item);
|
||||||
let dist_sq = dx * dx + dy * dy;
|
} else {
|
||||||
if best.is_none_or(|(_, d)| dist_sq < d) {
|
orphan_token_subitems.push(token_item);
|
||||||
best = Some((cell_idx, dist_sq));
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if let Some((ci, _)) = best {
|
|
||||||
claimed.insert(item_idx);
|
|
||||||
item_to_cell.insert(item_idx, ci);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// Build per-cell text from the assigned items. Markdown cells must
|
// Build per-cell text from the assigned tokens. Markdown cells must
|
||||||
// be one line — collapse line breaks from the line-grouping pass.
|
// be one line — collapse line breaks from the line-grouping pass.
|
||||||
let mut per_cell_items: Vec<Vec<TextItem>> = vec![Vec::new(); cells.len()];
|
|
||||||
for (&item_idx, &cell_idx) in &item_to_cell {
|
|
||||||
per_cell_items[cell_idx].push(items[item_idx].clone());
|
|
||||||
}
|
|
||||||
for (cell_idx, matched) in per_cell_items.into_iter().enumerate() {
|
for (cell_idx, matched) in per_cell_items.into_iter().enumerate() {
|
||||||
cells[cell_idx].text = collect_text_from_matched_items(matched, adaptive_threshold)
|
cells[cell_idx].text = collect_text_from_matched_items(matched, adaptive_threshold)
|
||||||
.replace(['\n', '\r'], " ");
|
.replace(['\n', '\r'], " ");
|
||||||
}
|
}
|
||||||
|
|
||||||
// Stage 2: orphan assignment — text items that didn't land in any
|
// Stage 2: orphan assignment — tokens that didn't land in any cell
|
||||||
// cell during stage 1 get assigned to their nearest *empty* cell,
|
// during stage 1 get assigned to their nearest *empty* cell,
|
||||||
// clamped by a plausibility cap derived from cell geometry.
|
// clamped by a plausibility cap derived from cell geometry.
|
||||||
//
|
//
|
||||||
// This recovers two failure modes left by `normalize_cell_bands`:
|
// This recovers two failure modes left by `normalize_cell_bands`:
|
||||||
@@ -1639,7 +1654,13 @@ pub fn extract_tables_with_structure_cells_mem(
|
|||||||
// Empty-cell-only is the safety net: a cell already filled by stage 1
|
// Empty-cell-only is the safety net: a cell already filled by stage 1
|
||||||
// is never overwritten or augmented, so the cell-bleed case PR #62
|
// is never overwritten or augmented, so the cell-bleed case PR #62
|
||||||
// closed cannot regress.
|
// closed cannot regress.
|
||||||
tsr_assign_orphan_items(items, &mut cells, &claimed, page_h, coords);
|
tsr_assign_orphan_items(
|
||||||
|
&orphan_token_subitems,
|
||||||
|
&mut cells,
|
||||||
|
&std::collections::HashSet::new(),
|
||||||
|
page_h,
|
||||||
|
coords,
|
||||||
|
);
|
||||||
|
|
||||||
results.push(cells);
|
results.push(cells);
|
||||||
}
|
}
|
||||||
@@ -1647,6 +1668,70 @@ pub fn extract_tables_with_structure_cells_mem(
|
|||||||
Ok(results)
|
Ok(results)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Split a `TextItem` into one virtual sub-item per whitespace-separated
|
||||||
|
/// token, with each token's `x` / `width` estimated from the original item's
|
||||||
|
/// effective width and the token's character offset.
|
||||||
|
///
|
||||||
|
/// PDFs often render an entire row's content as a single Tj — e.g.
|
||||||
|
/// "Marshall Islands 0.9 0.9 0.9" — producing one wide TextItem whose
|
||||||
|
/// center sits in only one of the model-emitted cells. Per-item routing
|
||||||
|
/// then parks the whole row in that one cell. Splitting on whitespace
|
||||||
|
/// gives each word its own approximate position so per-cell routing can
|
||||||
|
/// distribute the words to whichever cells their estimated centers fall
|
||||||
|
/// into.
|
||||||
|
///
|
||||||
|
/// The character-width estimate is `effective_width / char_count`.
|
||||||
|
/// `effective_width` returns the explicit `item.width` when known and
|
||||||
|
/// otherwise falls back to `char_count * font_size * 0.5`. Either way the
|
||||||
|
/// estimate is uniform across the item — fine for routing, since we only
|
||||||
|
/// need to know which cell each token's center lands in, not its exact
|
||||||
|
/// position. Single-token items collapse to a one-element vector
|
||||||
|
/// equivalent to the input item, making this a no-op for the common case.
|
||||||
|
fn split_item_into_token_subitems(item: &TextItem) -> Vec<TextItem> {
|
||||||
|
let total_chars = item.text.chars().count();
|
||||||
|
if total_chars == 0 {
|
||||||
|
return Vec::new();
|
||||||
|
}
|
||||||
|
let item_w = text_utils::effective_width(item);
|
||||||
|
let char_w = item_w / total_chars as f32;
|
||||||
|
|
||||||
|
let mut tokens: Vec<TextItem> = Vec::new();
|
||||||
|
let mut current_token = String::new();
|
||||||
|
let mut current_start_idx: Option<usize> = None;
|
||||||
|
|
||||||
|
let push_token =
|
||||||
|
|tokens: &mut Vec<TextItem>, text: String, start_idx: usize, end_idx: usize| {
|
||||||
|
if text.is_empty() {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let mut sub = item.clone();
|
||||||
|
sub.text = text;
|
||||||
|
sub.x = item.x + start_idx as f32 * char_w;
|
||||||
|
sub.width = (end_idx - start_idx) as f32 * char_w;
|
||||||
|
tokens.push(sub);
|
||||||
|
};
|
||||||
|
|
||||||
|
for (idx, ch) in item.text.chars().enumerate() {
|
||||||
|
if ch.is_whitespace() {
|
||||||
|
if let Some(start_idx) = current_start_idx.take() {
|
||||||
|
let text = std::mem::take(&mut current_token);
|
||||||
|
push_token(&mut tokens, text, start_idx, idx);
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
if current_start_idx.is_none() {
|
||||||
|
current_start_idx = Some(idx);
|
||||||
|
}
|
||||||
|
current_token.push(ch);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if let Some(start_idx) = current_start_idx {
|
||||||
|
let text = std::mem::take(&mut current_token);
|
||||||
|
push_token(&mut tokens, text, start_idx, total_chars);
|
||||||
|
}
|
||||||
|
|
||||||
|
tokens
|
||||||
|
}
|
||||||
|
|
||||||
/// Compute plausibility caps for the orphan-assignment pass. Returns
|
/// Compute plausibility caps for the orphan-assignment pass. Returns
|
||||||
/// `(cap_x, cap_y)` — the maximum x/y distance from a text item's center
|
/// `(cap_x, cap_y)` — the maximum x/y distance from a text item's center
|
||||||
/// to a candidate empty cell's bbox before the candidate is rejected.
|
/// to a candidate empty cell's bbox before the candidate is rejected.
|
||||||
|
|||||||
+366
-2
@@ -1189,9 +1189,38 @@ fn test_firecrawl_tagged_pdf_struct_tree() {
|
|||||||
#[test]
|
#[test]
|
||||||
fn test_identity_h_no_tounicode_suppresses_garbage() {
|
fn test_identity_h_no_tounicode_suppresses_garbage() {
|
||||||
// shinagawa_identity_h.pdf uses YuGothic with Identity-H encoding and no
|
// shinagawa_identity_h.pdf uses YuGothic with Identity-H encoding and no
|
||||||
// ToUnicode CMap. The raw CID values look like random Latin characters.
|
// usable ToUnicode CMap. The raw CID bytes (e.g. 0x08 0x37, 0x0E 0x0F)
|
||||||
// We should suppress the garbage and flag the page for OCR.
|
// contain non-ASCII high bytes and previously fell through to the
|
||||||
|
// per-byte Latin-1 fallback, producing high-Latin-1 mojibake that
|
||||||
|
// `is_cid_garbage` flagged. The Type0/CID guard in
|
||||||
|
// `extract_text_from_operand` now emits one U+FFFD per CID instead of
|
||||||
|
// mojibake; `detect_encoding_issues` trips on that and suppresses the
|
||||||
|
// markdown / flags the page for OCR — so we still pass this test, but
|
||||||
|
// via the deliberate marker path rather than by accident.
|
||||||
let buf = std::fs::read("tests/fixtures/shinagawa_identity_h.pdf").unwrap();
|
let buf = std::fs::read("tests/fixtures/shinagawa_identity_h.pdf").unwrap();
|
||||||
|
|
||||||
|
// Pre-suppression check: the raw text items must contain the U+FFFD
|
||||||
|
// markers that prove the Type0/CID fallback fired. This pins the
|
||||||
|
// mechanism so a future regression that re-enables Latin-1 mojibake
|
||||||
|
// would fail loudly here, not just silently change the suppression
|
||||||
|
// chain to one that depends on `is_cid_garbage` + high-Latin-1 chars.
|
||||||
|
let items = pdf_inspector::extractor::extract_text_with_positions_mem(&buf).unwrap();
|
||||||
|
let combined: String = items.iter().map(|i| i.text.as_str()).collect();
|
||||||
|
assert!(
|
||||||
|
combined.contains('\u{FFFD}'),
|
||||||
|
"Type0/CID font with unparseable ToUnicode CMap should emit U+FFFD per CID; \
|
||||||
|
got {} chars: {:?}",
|
||||||
|
combined.len(),
|
||||||
|
&combined[..combined.len().min(100)]
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
!combined
|
||||||
|
.chars()
|
||||||
|
.any(|c| ('\u{0080}'..='\u{00FF}').contains(&c)),
|
||||||
|
"Latin-1 mojibake (high bytes) must not leak from Type0/CID fallback; got: {:?}",
|
||||||
|
&combined[..combined.len().min(100)]
|
||||||
|
);
|
||||||
|
|
||||||
let result = pdf_inspector::process_pdf_mem(&buf).unwrap();
|
let result = pdf_inspector::process_pdf_mem(&buf).unwrap();
|
||||||
|
|
||||||
// Page 1 should be flagged for OCR
|
// Page 1 should be flagged for OCR
|
||||||
@@ -3000,3 +3029,338 @@ fn test_extract_pages_markdown_path_none_returns_all_pages() {
|
|||||||
let result = extract_pages_markdown(path, None).unwrap();
|
let result = extract_pages_markdown(path, None).unwrap();
|
||||||
assert_eq!(result.pages.len() as u32, page_count);
|
assert_eq!(result.pages.len() as u32, page_count);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ============================================================================
|
||||||
|
// PROBE: investigate dense-cell text-assignment failure mode (failure mode 2)
|
||||||
|
// ============================================================================
|
||||||
|
|
||||||
|
fn synthetic_wide_row_pdf() -> Vec<u8> {
|
||||||
|
use lopdf::content::{Content, Operation};
|
||||||
|
use lopdf::{dictionary, Document, Object, Stream};
|
||||||
|
|
||||||
|
let mut doc = Document::with_version("1.5");
|
||||||
|
let pages_id = doc.new_object_id();
|
||||||
|
let page_id = doc.new_object_id();
|
||||||
|
let font_id = doc.new_object_id();
|
||||||
|
let content_id = doc.new_object_id();
|
||||||
|
|
||||||
|
doc.objects.insert(
|
||||||
|
font_id,
|
||||||
|
dictionary! {
|
||||||
|
"Type" => "Font",
|
||||||
|
"Subtype" => "Type1",
|
||||||
|
"BaseFont" => "Helvetica",
|
||||||
|
}
|
||||||
|
.into(),
|
||||||
|
);
|
||||||
|
|
||||||
|
let operations = vec![
|
||||||
|
Operation::new("BT", vec![]),
|
||||||
|
Operation::new("Tf", vec!["F1".into(), 10.into()]),
|
||||||
|
Operation::new("Td", vec![20.into(), 700.into()]),
|
||||||
|
// A single Tj that visually spans multiple cells. This mirrors PDFs
|
||||||
|
// where a row's address/role/email columns are emitted as one literal
|
||||||
|
// string with embedded spaces, producing one wide TextItem.
|
||||||
|
Operation::new(
|
||||||
|
"Tj",
|
||||||
|
vec![Object::string_literal("Name JobTitle Email Phone")],
|
||||||
|
),
|
||||||
|
Operation::new("ET", vec![]),
|
||||||
|
];
|
||||||
|
let content = Content { operations }.encode().unwrap();
|
||||||
|
doc.objects
|
||||||
|
.insert(content_id, Stream::new(dictionary! {}, content).into());
|
||||||
|
|
||||||
|
doc.objects.insert(
|
||||||
|
page_id,
|
||||||
|
dictionary! {
|
||||||
|
"Type" => "Page",
|
||||||
|
"Parent" => pages_id,
|
||||||
|
"MediaBox" => vec![0.into(), 0.into(), 200.into(), 800.into()],
|
||||||
|
"Resources" => dictionary! {
|
||||||
|
"Font" => dictionary! {
|
||||||
|
"F1" => font_id,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
"Contents" => content_id,
|
||||||
|
}
|
||||||
|
.into(),
|
||||||
|
);
|
||||||
|
doc.objects.insert(
|
||||||
|
pages_id,
|
||||||
|
dictionary! {
|
||||||
|
"Type" => "Pages",
|
||||||
|
"Kids" => vec![page_id.into()],
|
||||||
|
"Count" => 1,
|
||||||
|
}
|
||||||
|
.into(),
|
||||||
|
);
|
||||||
|
let catalog_id = doc.add_object(dictionary! {
|
||||||
|
"Type" => "Catalog",
|
||||||
|
"Pages" => pages_id,
|
||||||
|
});
|
||||||
|
doc.trailer.set("Root", catalog_id);
|
||||||
|
|
||||||
|
let mut bytes = Vec::new();
|
||||||
|
doc.save_to(&mut bytes).unwrap();
|
||||||
|
bytes
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_extract_tables_with_structure_distributes_wide_item_across_cells() {
|
||||||
|
use pdf_inspector::{extract_tables_with_structure_cells_mem, TsrTableInput};
|
||||||
|
|
||||||
|
// Reproduces failure mode 2: a row of multi-token text rendered as one
|
||||||
|
// Tj produces a single wide TextItem that visually spans multiple cells.
|
||||||
|
// The current first-match-by-center routing parks the entire item in
|
||||||
|
// whichever cell holds the item's center, leaving the other cells empty.
|
||||||
|
// See production samples in scrape_id 019de788-ff41-... where 10-column
|
||||||
|
// grids ended up with row text packed into one cell.
|
||||||
|
let buf = synthetic_wide_row_pdf();
|
||||||
|
|
||||||
|
// Helvetica 10pt with width=0 falls back to char_count*font_size*0.5.
|
||||||
|
// "Name JobTitle Email Phone" is 25 chars → effective_width 125pt,
|
||||||
|
// text starts at PDF (20, 700), top-down y=[90, 100], char_w≈5pt.
|
||||||
|
// Tokens land at:
|
||||||
|
// "Name" chars 0-3 center≈x=30
|
||||||
|
// "JobTitle" chars 5-12 center≈x=65
|
||||||
|
// "Email" chars 14-18 center≈x=100
|
||||||
|
// "Phone" chars 20-24 center≈x=130
|
||||||
|
let cell_bboxes = vec![
|
||||||
|
poly(15.0, 88.0, 50.0, 102.0),
|
||||||
|
poly(50.0, 88.0, 85.0, 102.0),
|
||||||
|
poly(85.0, 88.0, 120.0, 102.0),
|
||||||
|
poly(120.0, 88.0, 155.0, 102.0),
|
||||||
|
];
|
||||||
|
|
||||||
|
let tokens: Vec<String> = [
|
||||||
|
"<table>",
|
||||||
|
"<tbody>",
|
||||||
|
"<tr>",
|
||||||
|
"<td></td>",
|
||||||
|
"<td></td>",
|
||||||
|
"<td></td>",
|
||||||
|
"<td></td>",
|
||||||
|
"</tr>",
|
||||||
|
"</tbody>",
|
||||||
|
"</table>",
|
||||||
|
]
|
||||||
|
.into_iter()
|
||||||
|
.map(String::from)
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
let cells_lists = extract_tables_with_structure_cells_mem(
|
||||||
|
&buf,
|
||||||
|
&[TsrTableInput {
|
||||||
|
page: 0,
|
||||||
|
crop_pdf_pt_bbox: [0.0, 0.0, 200.0, 800.0],
|
||||||
|
render_dpi: 72.0,
|
||||||
|
structure_tokens: tokens,
|
||||||
|
cell_bboxes,
|
||||||
|
}],
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
let cells = &cells_lists[0];
|
||||||
|
assert_eq!(cells.len(), 4);
|
||||||
|
assert_eq!(
|
||||||
|
cells[0].text, "Name",
|
||||||
|
"cell 0 should hold 'Name', got {:?}",
|
||||||
|
cells[0].text
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
cells[1].text, "JobTitle",
|
||||||
|
"cell 1 should hold 'JobTitle', got {:?}",
|
||||||
|
cells[1].text
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
cells[2].text, "Email",
|
||||||
|
"cell 2 should hold 'Email', got {:?}",
|
||||||
|
cells[2].text
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
cells[3].text, "Phone",
|
||||||
|
"cell 3 should hold 'Phone', got {:?}",
|
||||||
|
cells[3].text
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ============================================================================
|
||||||
|
// PROPER TEST: synthetic Type0/Identity-H PDF with malformed ToUnicode CMap
|
||||||
|
// ============================================================================
|
||||||
|
//
|
||||||
|
// Complements the existing real-PDF fixture `shinagawa_identity_h.pdf` by
|
||||||
|
// building a minimal Type0 / Identity-H font in process. We control:
|
||||||
|
// * the byte stream emitted by Tj (a 2-byte CID containing one high byte),
|
||||||
|
// * the malformed ToUnicode contents (junk bytes that won't parse), and
|
||||||
|
// * the DescendantFonts shape (just enough for `parse_type0_widths` to set
|
||||||
|
// `is_cid=true`, which is what the new guard in `extract_text_from_operand`
|
||||||
|
// keys off of).
|
||||||
|
// No fixture file or external license to worry about.
|
||||||
|
|
||||||
|
fn synthetic_type0_broken_tounicode_pdf() -> Vec<u8> {
|
||||||
|
use lopdf::content::{Content, Operation};
|
||||||
|
use lopdf::{dictionary, Document, Object, Stream};
|
||||||
|
|
||||||
|
let mut doc = Document::with_version("1.5");
|
||||||
|
let pages_id = doc.new_object_id();
|
||||||
|
let page_id = doc.new_object_id();
|
||||||
|
let font_id = doc.new_object_id();
|
||||||
|
let cid_font_id = doc.new_object_id();
|
||||||
|
let descriptor_id = doc.new_object_id();
|
||||||
|
let tounicode_id = doc.new_object_id();
|
||||||
|
let cid_system_info_id = doc.new_object_id();
|
||||||
|
let content_id = doc.new_object_id();
|
||||||
|
|
||||||
|
// Type0 font with Identity-H encoding and a broken ToUnicode reference.
|
||||||
|
doc.objects.insert(
|
||||||
|
font_id,
|
||||||
|
dictionary! {
|
||||||
|
"Type" => "Font",
|
||||||
|
"Subtype" => "Type0",
|
||||||
|
"BaseFont" => "AAAAAA+SyntheticCID",
|
||||||
|
"Encoding" => "Identity-H",
|
||||||
|
"DescendantFonts" => vec![cid_font_id.into()],
|
||||||
|
"ToUnicode" => tounicode_id,
|
||||||
|
}
|
||||||
|
.into(),
|
||||||
|
);
|
||||||
|
|
||||||
|
// CIDSystemInfo and a minimal CIDFontType2 descendant. parse_type0_widths
|
||||||
|
// walks DescendantFonts → returns FontWidthInfo with is_cid=true. That's
|
||||||
|
// the only thing the new Latin-1 guard needs to see.
|
||||||
|
doc.objects.insert(
|
||||||
|
cid_system_info_id,
|
||||||
|
dictionary! {
|
||||||
|
"Registry" => Object::string_literal("Adobe"),
|
||||||
|
"Ordering" => Object::string_literal("Identity"),
|
||||||
|
"Supplement" => 0,
|
||||||
|
}
|
||||||
|
.into(),
|
||||||
|
);
|
||||||
|
doc.objects.insert(
|
||||||
|
cid_font_id,
|
||||||
|
dictionary! {
|
||||||
|
"Type" => "Font",
|
||||||
|
"Subtype" => "CIDFontType2",
|
||||||
|
"BaseFont" => "AAAAAA+SyntheticCID",
|
||||||
|
"CIDSystemInfo" => cid_system_info_id,
|
||||||
|
"FontDescriptor" => descriptor_id,
|
||||||
|
"DW" => 1000,
|
||||||
|
}
|
||||||
|
.into(),
|
||||||
|
);
|
||||||
|
doc.objects.insert(
|
||||||
|
descriptor_id,
|
||||||
|
dictionary! {
|
||||||
|
"Type" => "FontDescriptor",
|
||||||
|
"FontName" => "AAAAAA+SyntheticCID",
|
||||||
|
"Flags" => 4,
|
||||||
|
"FontBBox" => vec![Object::Integer(-100), Object::Integer(-100), 1000.into(), 1000.into()],
|
||||||
|
"ItalicAngle" => 0,
|
||||||
|
"Ascent" => 800,
|
||||||
|
"Descent" => Object::Integer(-200),
|
||||||
|
"CapHeight" => 700,
|
||||||
|
"StemV" => 80,
|
||||||
|
}
|
||||||
|
.into(),
|
||||||
|
);
|
||||||
|
|
||||||
|
// Intentionally malformed ToUnicode stream — just junk bytes. ToUnicode
|
||||||
|
// CMap parsing will fail, so `font_cmaps.get_by_obj` returns None and
|
||||||
|
// `has_cmap` stays false. The reference still exists in the font dict,
|
||||||
|
// so `font_tounicode_refs` contains the entry — but the new guard now
|
||||||
|
// routes off `is_cid` from font_widths instead, which is robust to a
|
||||||
|
// failed CMap parse.
|
||||||
|
doc.objects.insert(
|
||||||
|
tounicode_id,
|
||||||
|
Stream::new(dictionary! {}, b"this is not a valid CMap stream".to_vec()).into(),
|
||||||
|
);
|
||||||
|
|
||||||
|
// Tj with a 2-byte CID stream containing a non-ASCII high byte.
|
||||||
|
// Pre-fix this would have decoded as Latin-1 to "\u{00CD}\u{00D9}" ("ÍÙ").
|
||||||
|
// Post-fix it should produce U+FFFD per CID.
|
||||||
|
let cid_bytes = vec![0xCD_u8, 0xD9, 0xCD, 0xD9];
|
||||||
|
let operations = vec![
|
||||||
|
Operation::new("BT", vec![]),
|
||||||
|
Operation::new("Tf", vec!["F0".into(), 12.into()]),
|
||||||
|
Operation::new("Td", vec![50.into(), 100.into()]),
|
||||||
|
Operation::new(
|
||||||
|
"Tj",
|
||||||
|
vec![Object::String(cid_bytes, lopdf::StringFormat::Hexadecimal)],
|
||||||
|
),
|
||||||
|
Operation::new("ET", vec![]),
|
||||||
|
];
|
||||||
|
let content = Content { operations }.encode().unwrap();
|
||||||
|
doc.objects
|
||||||
|
.insert(content_id, Stream::new(dictionary! {}, content).into());
|
||||||
|
|
||||||
|
doc.objects.insert(
|
||||||
|
page_id,
|
||||||
|
dictionary! {
|
||||||
|
"Type" => "Page",
|
||||||
|
"Parent" => pages_id,
|
||||||
|
"MediaBox" => vec![0.into(), 0.into(), 200.into(), 200.into()],
|
||||||
|
"Resources" => dictionary! {
|
||||||
|
"Font" => dictionary! {
|
||||||
|
"F0" => font_id,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
"Contents" => content_id,
|
||||||
|
}
|
||||||
|
.into(),
|
||||||
|
);
|
||||||
|
doc.objects.insert(
|
||||||
|
pages_id,
|
||||||
|
dictionary! {
|
||||||
|
"Type" => "Pages",
|
||||||
|
"Kids" => vec![page_id.into()],
|
||||||
|
"Count" => 1,
|
||||||
|
}
|
||||||
|
.into(),
|
||||||
|
);
|
||||||
|
let catalog_id = doc.add_object(dictionary! {
|
||||||
|
"Type" => "Catalog",
|
||||||
|
"Pages" => pages_id,
|
||||||
|
});
|
||||||
|
doc.trailer.set("Root", catalog_id);
|
||||||
|
|
||||||
|
let mut bytes = Vec::new();
|
||||||
|
doc.save_to(&mut bytes).unwrap();
|
||||||
|
bytes
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_synthetic_type0_broken_tounicode_emits_fffd_not_latin1_mojibake() {
|
||||||
|
let buf = synthetic_type0_broken_tounicode_pdf();
|
||||||
|
|
||||||
|
let items = pdf_inspector::extractor::extract_text_with_positions_mem(&buf).unwrap();
|
||||||
|
let combined: String = items.iter().map(|i| i.text.as_str()).collect();
|
||||||
|
|
||||||
|
// Mojibake leak check: 2-byte CID 0xCDD9 must NOT come out as "ÍÙ"
|
||||||
|
// (U+00CD U+00D9). That was the production scrape symptom.
|
||||||
|
assert!(
|
||||||
|
!combined.contains('\u{00CD}'),
|
||||||
|
"Latin-1 mojibake leaked from Type0 font: {combined:?}"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
!combined.contains('\u{00D9}'),
|
||||||
|
"Latin-1 mojibake leaked from Type0 font: {combined:?}"
|
||||||
|
);
|
||||||
|
|
||||||
|
// Marker presence: Type0/CID + non-ASCII bytes must produce U+FFFD so
|
||||||
|
// `detect_encoding_issues` can flag the page for OCR downstream.
|
||||||
|
assert!(
|
||||||
|
combined.contains('\u{FFFD}'),
|
||||||
|
"Type0 font with malformed ToUnicode CMap should emit U+FFFD per CID; got: {combined:?}"
|
||||||
|
);
|
||||||
|
|
||||||
|
// End-to-end check: the page is correctly routed to OCR.
|
||||||
|
let result = pdf_inspector::process_pdf_mem(&buf).unwrap();
|
||||||
|
assert!(
|
||||||
|
result.pages_needing_ocr.contains(&1),
|
||||||
|
"Type0 page with broken ToUnicode + non-ASCII bytes must be flagged for OCR; \
|
||||||
|
pages_needing_ocr={:?}",
|
||||||
|
result.pages_needing_ocr
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user