diff --git a/src/extractor/content_stream.rs b/src/extractor/content_stream.rs index bba3882..0c202d5 100644 --- a/src/extractor/content_stream.rs +++ b/src/extractor/content_stream.rs @@ -385,6 +385,7 @@ pub(crate) fn extract_page_text_items( &font_encodings, &encoding_cache, &mut cmap_decisions, + &font_widths, ) { let combined = multiply_matrices(&text_matrix, &ctm); let rendered_size = effective_font_size(current_font_size, &combined); @@ -533,6 +534,7 @@ pub(crate) fn extract_page_text_items( &font_encodings, &encoding_cache, &mut cmap_decisions, + &font_widths, ) { current_text.push_str(&text); } @@ -620,6 +622,7 @@ pub(crate) fn extract_page_text_items( &font_encodings, &encoding_cache, &mut cmap_decisions, + &font_widths, ) { if !text.trim().is_empty() { let combined = multiply_matrices(&text_matrix, &ctm); diff --git a/src/extractor/fonts.rs b/src/extractor/fonts.rs index e4f6b84..8637a96 100644 --- a/src/extractor/fonts.rs +++ b/src/extractor/fonts.rs @@ -720,7 +720,11 @@ pub(crate) fn extract_text_from_operand( font_encodings: &PageFontEncodings, encoding_cache: &HashMap>, cmap_decisions: &mut CMapDecisionCache, + font_widths: &PageFontWidths, ) -> Option { + let is_type0_cid_font = font_widths + .get(current_font) + .is_some_and(|info| info.is_cid); let result = (|| -> Option { if let Object::String(bytes, _) = obj { let mut decode_with_entry = |entry: &crate::tounicode::CMapEntry| -> Option { @@ -962,7 +966,31 @@ pub(crate) fn extract_text_from_operand( return Some(symbol_text); } - // Latin-1 fallback + // Latin-1 fallback. Safe ONLY for fonts that use single-byte + // encodings — for these, an unmapped byte is a valid character + // code in Latin-1/WinAnsi space. CID fonts (Type0 / Identity-H) + // emit multi-byte CIDs that aren't characters; per-byte Latin-1 + // produces mojibake (e.g. 2-byte CID 0xCDD9 → "ÍÙ" for the + // production scrape_id 019de78c-... samples). + // + // For a CID font (has_cmap is set OR a /ToUnicode reference + // exists) with any non-ASCII bytes, emit a single U+FFFD per + // CID instead. This both replaces the mojibake with a proper + // "decode failed" marker AND keeps `detect_encoding_issues` + // tripping so the page is flagged for OCR — the existing + // garbage-detection path that the high-Latin-1 mojibake used + // to satisfy by accident. + if is_type0_cid_font && bytes.iter().any(|&b| b > 0x7F) { + // 2-byte CIDs (Identity-H) are by far the common case; for + // an odd byte count we still emit at least one marker so + // detection downstream fires. + let cid_count = (bytes.len() / 2).max(1); + return Some("\u{FFFD}".repeat(cid_count)); + } + // Pure ASCII bytes round-trip safely (Latin-1 == ASCII for + // 0x00..=0x7F), and non-CID (Type1 / TrueType / Type3) fonts + // use single-byte encodings where Latin-1 fallback is the + // canonical interpretation. Some(bytes.iter().map(|&b| b as char).collect()) } else { None @@ -1213,4 +1241,97 @@ mod tests { let bad = "###!!!@@@$$$"; assert!(score_text(good) > score_text(bad)); } + + #[test] + fn cid_font_with_unparseable_cmap_does_not_emit_latin1_mojibake() { + // Type0/CID font (font_widths reports `is_cid=true`) where the + // ToUnicode CMap couldn't be parsed (FontCMaps doesn't have the + // obj_num). Bytes are a 2-byte CID stream containing high bytes + // that aren't valid UTF-8 — exactly the case in the production + // samples (Identity-H text where the ToUnicode CMap was missing + // or malformed, scrape_id 019de78c-..., e.g. "Í Ù Z)¿"). + // + // Without the guard, the function falls through to the byte-by-byte + // Latin-1 fallback and produces "ÍÙ" (U+00CD U+00D9). The correct + // behavior is to emit U+FFFD per CID so downstream + // `detect_encoding_issues` flags the page for OCR. + let bytes = vec![0xCD_u8, 0xD9, 0xCD, 0xD9]; + let obj = Object::String(bytes, lopdf::StringFormat::Hexadecimal); + + let font_cmaps = FontCMaps::default(); + let mut font_tounicode_refs: HashMap = HashMap::new(); + font_tounicode_refs.insert("F0".to_string(), 999); + let inline_cmaps = HashMap::new(); + let font_encodings: PageFontEncodings = HashMap::new(); + let encoding_cache: HashMap> = HashMap::new(); + let mut decisions = CMapDecisionCache::new(); + let mut font_widths: PageFontWidths = HashMap::new(); + font_widths.insert("F0".to_string(), make_font_info(&[], 1000, true)); + + let result = extract_text_from_operand( + &obj, + "F0", + None, + &font_cmaps, + &font_tounicode_refs, + &inline_cmaps, + &font_encodings, + &encoding_cache, + &mut decisions, + &font_widths, + ); + + let text = result.expect("CID font fallback should still emit a marker"); + assert!( + !text.contains('\u{00CD}') && !text.contains('\u{00D9}'), + "CID font with unparseable CMap leaked Latin-1 mojibake: {text:?}" + ); + assert!( + text.contains('\u{FFFD}'), + "CID font with unparseable CMap should emit U+FFFD so detect_encoding_issues fires: {text:?}" + ); + } + + #[test] + fn simple_font_latin1_fallback_passes_high_bytes_through() { + // A Type1/TrueType simple font (is_cid=false) with a `/ToUnicode` + // reference but no usable CMap and no `/Differences` map. + // Per-byte Latin-1 IS the canonical interpretation here — these + // bytes are character codes, not CIDs. The CID guard must NOT + // strip them. Reproduces the false positive that an earlier + // version of the guard introduced for fonts in PDFs like + // pdf-evals/Navigating-Artificial-Intelligence-..., where bytes + // like 0xB6 are legitimate Latin-1 character codes. + let bytes = vec![0x24_u8, 0x47, 0xB6, 0x56]; // "$G¶V" + let obj = Object::String(bytes, lopdf::StringFormat::Hexadecimal); + + let font_cmaps = FontCMaps::default(); + let mut font_tounicode_refs: HashMap = HashMap::new(); + font_tounicode_refs.insert("F1".to_string(), 999); + let inline_cmaps = HashMap::new(); + let font_encodings: PageFontEncodings = HashMap::new(); + let encoding_cache: HashMap> = HashMap::new(); + let mut decisions = CMapDecisionCache::new(); + let mut font_widths: PageFontWidths = HashMap::new(); + font_widths.insert("F1".to_string(), make_font_info(&[], 1000, false)); + + let text = extract_text_from_operand( + &obj, + "F1", + None, + &font_cmaps, + &font_tounicode_refs, + &inline_cmaps, + &font_encodings, + &encoding_cache, + &mut decisions, + &font_widths, + ) + .expect("simple font should round-trip Latin-1 bytes"); + assert_eq!(text, "$G\u{00B6}V"); + assert!( + !text.contains('\u{FFFD}'), + "simple font fallback must not stamp FFFD over legitimate bytes: {text:?}" + ); + } } diff --git a/src/extractor/xobjects.rs b/src/extractor/xobjects.rs index 2cab02f..92c5065 100644 --- a/src/extractor/xobjects.rs +++ b/src/extractor/xobjects.rs @@ -373,6 +373,7 @@ fn extract_form_xobject_text_inner( &font_encodings, &encoding_cache, cmap_decisions, + &font_widths, ) { let combined = multiply_matrices(&text_matrix, &ctm); let rendered_size = effective_font_size(current_font_size, &combined); @@ -517,6 +518,7 @@ fn extract_form_xobject_text_inner( &font_encodings, &encoding_cache, cmap_decisions, + &font_widths, ) { current_text.push_str(&text); } diff --git a/src/lib.rs b/src/lib.rs index 15c53ba..3b19112 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -1556,26 +1556,32 @@ pub fn extract_tables_with_structure_cells_mem( normalize_cell_bands(&mut cells); - // Stage 1: exclusive per-item assignment. For each PDF text item, - // find the cell(s) whose (band-clamped) bbox satisfies the strict - // membership rule (`tsr_region_contains_item`: center inside OR - // >=60% overlap on both axes). If multiple cells qualify, assign - // the item to the cell whose center is geometrically closest. If - // exactly one qualifies, assign to that. If none, the item is an - // orphan and stage 2 below tries to recover it. + // Stage 1: exclusive per-token assignment. Each PDF text item is + // first split into whitespace-separated tokens with estimated x + // positions (see `split_item_into_token_subitems`). For each token + // we find the cell(s) whose (band-clamped) bbox satisfies the + // strict membership rule (`tsr_region_contains_item`: center + // inside OR >=60% overlap on both axes). If multiple cells + // qualify, the closest-center wins. Tokens that don't land in any + // cell are eligible for stage-2 orphan recovery. // - // The exclusivity (one item → one cell) prevents the cell-overlap - // bug where SLANet emits cells whose y-extents overlap between - // rows: under the previous "for each cell, gather items" approach, - // an item whose center fell in two cells' overlap got duplicated - // into both. Closest-center disambiguation routes it to the - // correct row. - let mut claimed: std::collections::HashSet = std::collections::HashSet::new(); - let mut item_to_cell: std::collections::HashMap = - std::collections::HashMap::new(); + // Per-token (rather than per-item) routing is what prevents the + // dense-grid collapse bug: a row rendered as one wide Tj + // ("Marshall Islands 0.9 0.9 0.9") produces one TextItem whose + // CENTER lies in only one cell, so per-item routing parks the + // whole row in that single cell and leaves the rest of the row + // empty. Per-token routing distributes the words to whichever + // cells their (estimated) positions fall into. Single-token items + // collapse to a one-element token list and behave exactly as + // before. + // + // The token-level exclusivity (one token → one cell) still + // prevents the cell-overlap bug where SLANet emits cells whose + // y-extents overlap between rows. Closest-center disambiguation + // routes each token to the correct row. // Pre-compute each cell's bounds + center (in PDF-pt-flipped space) - // so we don't redo the work per item. + // so we don't redo the work per token. let cell_meta: Vec> = cells .iter() .map(|cell| { @@ -1590,44 +1596,53 @@ pub fn extract_tables_with_structure_cells_mem( }) .collect(); - for (item_idx, item) in items.iter().enumerate() { - let item_w = text_utils::effective_width(item); - let item_cx = item.x + item_w * 0.5; - let item_cy = item.y + item.height * 0.5; - let mut best: Option<(usize, f32)> = None; - for (cell_idx, meta) in cell_meta.iter().enumerate() { - let Some((bounds, ccx, ccy)) = meta else { - continue; - }; - if !tsr_region_contains_item(item, *bounds) { - continue; + let mut per_cell_items: Vec> = vec![Vec::new(); cells.len()]; + // Tokens of an item that did NOT land in any cell during stage 1. + // These are the orphan candidates handed to `tsr_assign_orphan_items`. + // Using token-grain orphan candidates (rather than the original wide + // item) lets stage 2 recover individual words that fell just outside + // their cell's clamped band, without re-attributing already-claimed + // tokens. + let mut orphan_token_subitems: Vec = Vec::new(); + + for item in items.iter() { + let token_subitems = split_item_into_token_subitems(item); + for token_item in token_subitems { + let token_w = text_utils::effective_width(&token_item); + let token_cx = token_item.x + token_w * 0.5; + let token_cy = token_item.y + token_item.height * 0.5; + let mut best: Option<(usize, f32)> = None; + for (cell_idx, meta) in cell_meta.iter().enumerate() { + let Some((bounds, ccx, ccy)) = meta else { + continue; + }; + if !tsr_region_contains_item(&token_item, *bounds) { + continue; + } + let dx = token_cx - ccx; + let dy = token_cy - ccy; + let dist_sq = dx * dx + dy * dy; + if best.is_none_or(|(_, d)| dist_sq < d) { + best = Some((cell_idx, dist_sq)); + } } - let dx = item_cx - ccx; - let dy = item_cy - ccy; - let dist_sq = dx * dx + dy * dy; - if best.is_none_or(|(_, d)| dist_sq < d) { - best = Some((cell_idx, dist_sq)); + if let Some((ci, _)) = best { + per_cell_items[ci].push(token_item); + } else { + orphan_token_subitems.push(token_item); } } - if let Some((ci, _)) = best { - claimed.insert(item_idx); - item_to_cell.insert(item_idx, ci); - } } - // Build per-cell text from the assigned items. Markdown cells must + // Build per-cell text from the assigned tokens. Markdown cells must // be one line — collapse line breaks from the line-grouping pass. - let mut per_cell_items: Vec> = vec![Vec::new(); cells.len()]; - for (&item_idx, &cell_idx) in &item_to_cell { - per_cell_items[cell_idx].push(items[item_idx].clone()); - } for (cell_idx, matched) in per_cell_items.into_iter().enumerate() { cells[cell_idx].text = collect_text_from_matched_items(matched, adaptive_threshold) .replace(['\n', '\r'], " "); } - // Stage 2: orphan assignment — text items that didn't land in any - // cell during stage 1 get assigned to their nearest *empty* cell, + // Stage 2: orphan assignment — tokens that didn't land in any cell + // during stage 1 get assigned to their nearest *empty* cell, // clamped by a plausibility cap derived from cell geometry. // // This recovers two failure modes left by `normalize_cell_bands`: @@ -1639,7 +1654,13 @@ pub fn extract_tables_with_structure_cells_mem( // Empty-cell-only is the safety net: a cell already filled by stage 1 // is never overwritten or augmented, so the cell-bleed case PR #62 // closed cannot regress. - tsr_assign_orphan_items(items, &mut cells, &claimed, page_h, coords); + tsr_assign_orphan_items( + &orphan_token_subitems, + &mut cells, + &std::collections::HashSet::new(), + page_h, + coords, + ); results.push(cells); } @@ -1647,6 +1668,70 @@ pub fn extract_tables_with_structure_cells_mem( Ok(results) } +/// Split a `TextItem` into one virtual sub-item per whitespace-separated +/// token, with each token's `x` / `width` estimated from the original item's +/// effective width and the token's character offset. +/// +/// PDFs often render an entire row's content as a single Tj — e.g. +/// "Marshall Islands 0.9 0.9 0.9" — producing one wide TextItem whose +/// center sits in only one of the model-emitted cells. Per-item routing +/// then parks the whole row in that one cell. Splitting on whitespace +/// gives each word its own approximate position so per-cell routing can +/// distribute the words to whichever cells their estimated centers fall +/// into. +/// +/// The character-width estimate is `effective_width / char_count`. +/// `effective_width` returns the explicit `item.width` when known and +/// otherwise falls back to `char_count * font_size * 0.5`. Either way the +/// estimate is uniform across the item — fine for routing, since we only +/// need to know which cell each token's center lands in, not its exact +/// position. Single-token items collapse to a one-element vector +/// equivalent to the input item, making this a no-op for the common case. +fn split_item_into_token_subitems(item: &TextItem) -> Vec { + let total_chars = item.text.chars().count(); + if total_chars == 0 { + return Vec::new(); + } + let item_w = text_utils::effective_width(item); + let char_w = item_w / total_chars as f32; + + let mut tokens: Vec = Vec::new(); + let mut current_token = String::new(); + let mut current_start_idx: Option = None; + + let push_token = + |tokens: &mut Vec, text: String, start_idx: usize, end_idx: usize| { + if text.is_empty() { + return; + } + let mut sub = item.clone(); + sub.text = text; + sub.x = item.x + start_idx as f32 * char_w; + sub.width = (end_idx - start_idx) as f32 * char_w; + tokens.push(sub); + }; + + for (idx, ch) in item.text.chars().enumerate() { + if ch.is_whitespace() { + if let Some(start_idx) = current_start_idx.take() { + let text = std::mem::take(&mut current_token); + push_token(&mut tokens, text, start_idx, idx); + } + } else { + if current_start_idx.is_none() { + current_start_idx = Some(idx); + } + current_token.push(ch); + } + } + if let Some(start_idx) = current_start_idx { + let text = std::mem::take(&mut current_token); + push_token(&mut tokens, text, start_idx, total_chars); + } + + tokens +} + /// Compute plausibility caps for the orphan-assignment pass. Returns /// `(cap_x, cap_y)` — the maximum x/y distance from a text item's center /// to a candidate empty cell's bbox before the candidate is rejected. diff --git a/tests/integration_tests.rs b/tests/integration_tests.rs index ef6d536..4a51aa1 100644 --- a/tests/integration_tests.rs +++ b/tests/integration_tests.rs @@ -1189,9 +1189,38 @@ fn test_firecrawl_tagged_pdf_struct_tree() { #[test] fn test_identity_h_no_tounicode_suppresses_garbage() { // shinagawa_identity_h.pdf uses YuGothic with Identity-H encoding and no - // ToUnicode CMap. The raw CID values look like random Latin characters. - // We should suppress the garbage and flag the page for OCR. + // usable ToUnicode CMap. The raw CID bytes (e.g. 0x08 0x37, 0x0E 0x0F) + // contain non-ASCII high bytes and previously fell through to the + // per-byte Latin-1 fallback, producing high-Latin-1 mojibake that + // `is_cid_garbage` flagged. The Type0/CID guard in + // `extract_text_from_operand` now emits one U+FFFD per CID instead of + // mojibake; `detect_encoding_issues` trips on that and suppresses the + // markdown / flags the page for OCR — so we still pass this test, but + // via the deliberate marker path rather than by accident. let buf = std::fs::read("tests/fixtures/shinagawa_identity_h.pdf").unwrap(); + + // Pre-suppression check: the raw text items must contain the U+FFFD + // markers that prove the Type0/CID fallback fired. This pins the + // mechanism so a future regression that re-enables Latin-1 mojibake + // would fail loudly here, not just silently change the suppression + // chain to one that depends on `is_cid_garbage` + high-Latin-1 chars. + let items = pdf_inspector::extractor::extract_text_with_positions_mem(&buf).unwrap(); + let combined: String = items.iter().map(|i| i.text.as_str()).collect(); + assert!( + combined.contains('\u{FFFD}'), + "Type0/CID font with unparseable ToUnicode CMap should emit U+FFFD per CID; \ + got {} chars: {:?}", + combined.len(), + &combined[..combined.len().min(100)] + ); + assert!( + !combined + .chars() + .any(|c| ('\u{0080}'..='\u{00FF}').contains(&c)), + "Latin-1 mojibake (high bytes) must not leak from Type0/CID fallback; got: {:?}", + &combined[..combined.len().min(100)] + ); + let result = pdf_inspector::process_pdf_mem(&buf).unwrap(); // Page 1 should be flagged for OCR @@ -3000,3 +3029,338 @@ fn test_extract_pages_markdown_path_none_returns_all_pages() { let result = extract_pages_markdown(path, None).unwrap(); assert_eq!(result.pages.len() as u32, page_count); } + +// ============================================================================ +// PROBE: investigate dense-cell text-assignment failure mode (failure mode 2) +// ============================================================================ + +fn synthetic_wide_row_pdf() -> Vec { + use lopdf::content::{Content, Operation}; + use lopdf::{dictionary, Document, Object, Stream}; + + let mut doc = Document::with_version("1.5"); + let pages_id = doc.new_object_id(); + let page_id = doc.new_object_id(); + let font_id = doc.new_object_id(); + let content_id = doc.new_object_id(); + + doc.objects.insert( + font_id, + dictionary! { + "Type" => "Font", + "Subtype" => "Type1", + "BaseFont" => "Helvetica", + } + .into(), + ); + + let operations = vec![ + Operation::new("BT", vec![]), + Operation::new("Tf", vec!["F1".into(), 10.into()]), + Operation::new("Td", vec![20.into(), 700.into()]), + // A single Tj that visually spans multiple cells. This mirrors PDFs + // where a row's address/role/email columns are emitted as one literal + // string with embedded spaces, producing one wide TextItem. + Operation::new( + "Tj", + vec![Object::string_literal("Name JobTitle Email Phone")], + ), + Operation::new("ET", vec![]), + ]; + let content = Content { operations }.encode().unwrap(); + doc.objects + .insert(content_id, Stream::new(dictionary! {}, content).into()); + + doc.objects.insert( + page_id, + dictionary! { + "Type" => "Page", + "Parent" => pages_id, + "MediaBox" => vec![0.into(), 0.into(), 200.into(), 800.into()], + "Resources" => dictionary! { + "Font" => dictionary! { + "F1" => font_id, + }, + }, + "Contents" => content_id, + } + .into(), + ); + doc.objects.insert( + pages_id, + dictionary! { + "Type" => "Pages", + "Kids" => vec![page_id.into()], + "Count" => 1, + } + .into(), + ); + let catalog_id = doc.add_object(dictionary! { + "Type" => "Catalog", + "Pages" => pages_id, + }); + doc.trailer.set("Root", catalog_id); + + let mut bytes = Vec::new(); + doc.save_to(&mut bytes).unwrap(); + bytes +} + +#[test] +fn test_extract_tables_with_structure_distributes_wide_item_across_cells() { + use pdf_inspector::{extract_tables_with_structure_cells_mem, TsrTableInput}; + + // Reproduces failure mode 2: a row of multi-token text rendered as one + // Tj produces a single wide TextItem that visually spans multiple cells. + // The current first-match-by-center routing parks the entire item in + // whichever cell holds the item's center, leaving the other cells empty. + // See production samples in scrape_id 019de788-ff41-... where 10-column + // grids ended up with row text packed into one cell. + let buf = synthetic_wide_row_pdf(); + + // Helvetica 10pt with width=0 falls back to char_count*font_size*0.5. + // "Name JobTitle Email Phone" is 25 chars → effective_width 125pt, + // text starts at PDF (20, 700), top-down y=[90, 100], char_w≈5pt. + // Tokens land at: + // "Name" chars 0-3 center≈x=30 + // "JobTitle" chars 5-12 center≈x=65 + // "Email" chars 14-18 center≈x=100 + // "Phone" chars 20-24 center≈x=130 + let cell_bboxes = vec![ + poly(15.0, 88.0, 50.0, 102.0), + poly(50.0, 88.0, 85.0, 102.0), + poly(85.0, 88.0, 120.0, 102.0), + poly(120.0, 88.0, 155.0, 102.0), + ]; + + let tokens: Vec = [ + "", + "", + "", + "", + "", + "", + "", + "", + "", + "
", + ] + .into_iter() + .map(String::from) + .collect(); + + let cells_lists = extract_tables_with_structure_cells_mem( + &buf, + &[TsrTableInput { + page: 0, + crop_pdf_pt_bbox: [0.0, 0.0, 200.0, 800.0], + render_dpi: 72.0, + structure_tokens: tokens, + cell_bboxes, + }], + ) + .unwrap(); + + let cells = &cells_lists[0]; + assert_eq!(cells.len(), 4); + assert_eq!( + cells[0].text, "Name", + "cell 0 should hold 'Name', got {:?}", + cells[0].text + ); + assert_eq!( + cells[1].text, "JobTitle", + "cell 1 should hold 'JobTitle', got {:?}", + cells[1].text + ); + assert_eq!( + cells[2].text, "Email", + "cell 2 should hold 'Email', got {:?}", + cells[2].text + ); + assert_eq!( + cells[3].text, "Phone", + "cell 3 should hold 'Phone', got {:?}", + cells[3].text + ); +} + +// ============================================================================ +// PROPER TEST: synthetic Type0/Identity-H PDF with malformed ToUnicode CMap +// ============================================================================ +// +// Complements the existing real-PDF fixture `shinagawa_identity_h.pdf` by +// building a minimal Type0 / Identity-H font in process. We control: +// * the byte stream emitted by Tj (a 2-byte CID containing one high byte), +// * the malformed ToUnicode contents (junk bytes that won't parse), and +// * the DescendantFonts shape (just enough for `parse_type0_widths` to set +// `is_cid=true`, which is what the new guard in `extract_text_from_operand` +// keys off of). +// No fixture file or external license to worry about. + +fn synthetic_type0_broken_tounicode_pdf() -> Vec { + use lopdf::content::{Content, Operation}; + use lopdf::{dictionary, Document, Object, Stream}; + + let mut doc = Document::with_version("1.5"); + let pages_id = doc.new_object_id(); + let page_id = doc.new_object_id(); + let font_id = doc.new_object_id(); + let cid_font_id = doc.new_object_id(); + let descriptor_id = doc.new_object_id(); + let tounicode_id = doc.new_object_id(); + let cid_system_info_id = doc.new_object_id(); + let content_id = doc.new_object_id(); + + // Type0 font with Identity-H encoding and a broken ToUnicode reference. + doc.objects.insert( + font_id, + dictionary! { + "Type" => "Font", + "Subtype" => "Type0", + "BaseFont" => "AAAAAA+SyntheticCID", + "Encoding" => "Identity-H", + "DescendantFonts" => vec![cid_font_id.into()], + "ToUnicode" => tounicode_id, + } + .into(), + ); + + // CIDSystemInfo and a minimal CIDFontType2 descendant. parse_type0_widths + // walks DescendantFonts → returns FontWidthInfo with is_cid=true. That's + // the only thing the new Latin-1 guard needs to see. + doc.objects.insert( + cid_system_info_id, + dictionary! { + "Registry" => Object::string_literal("Adobe"), + "Ordering" => Object::string_literal("Identity"), + "Supplement" => 0, + } + .into(), + ); + doc.objects.insert( + cid_font_id, + dictionary! { + "Type" => "Font", + "Subtype" => "CIDFontType2", + "BaseFont" => "AAAAAA+SyntheticCID", + "CIDSystemInfo" => cid_system_info_id, + "FontDescriptor" => descriptor_id, + "DW" => 1000, + } + .into(), + ); + doc.objects.insert( + descriptor_id, + dictionary! { + "Type" => "FontDescriptor", + "FontName" => "AAAAAA+SyntheticCID", + "Flags" => 4, + "FontBBox" => vec![Object::Integer(-100), Object::Integer(-100), 1000.into(), 1000.into()], + "ItalicAngle" => 0, + "Ascent" => 800, + "Descent" => Object::Integer(-200), + "CapHeight" => 700, + "StemV" => 80, + } + .into(), + ); + + // Intentionally malformed ToUnicode stream — just junk bytes. ToUnicode + // CMap parsing will fail, so `font_cmaps.get_by_obj` returns None and + // `has_cmap` stays false. The reference still exists in the font dict, + // so `font_tounicode_refs` contains the entry — but the new guard now + // routes off `is_cid` from font_widths instead, which is robust to a + // failed CMap parse. + doc.objects.insert( + tounicode_id, + Stream::new(dictionary! {}, b"this is not a valid CMap stream".to_vec()).into(), + ); + + // Tj with a 2-byte CID stream containing a non-ASCII high byte. + // Pre-fix this would have decoded as Latin-1 to "\u{00CD}\u{00D9}" ("ÍÙ"). + // Post-fix it should produce U+FFFD per CID. + let cid_bytes = vec![0xCD_u8, 0xD9, 0xCD, 0xD9]; + let operations = vec![ + Operation::new("BT", vec![]), + Operation::new("Tf", vec!["F0".into(), 12.into()]), + Operation::new("Td", vec![50.into(), 100.into()]), + Operation::new( + "Tj", + vec![Object::String(cid_bytes, lopdf::StringFormat::Hexadecimal)], + ), + Operation::new("ET", vec![]), + ]; + let content = Content { operations }.encode().unwrap(); + doc.objects + .insert(content_id, Stream::new(dictionary! {}, content).into()); + + doc.objects.insert( + page_id, + dictionary! { + "Type" => "Page", + "Parent" => pages_id, + "MediaBox" => vec![0.into(), 0.into(), 200.into(), 200.into()], + "Resources" => dictionary! { + "Font" => dictionary! { + "F0" => font_id, + }, + }, + "Contents" => content_id, + } + .into(), + ); + doc.objects.insert( + pages_id, + dictionary! { + "Type" => "Pages", + "Kids" => vec![page_id.into()], + "Count" => 1, + } + .into(), + ); + let catalog_id = doc.add_object(dictionary! { + "Type" => "Catalog", + "Pages" => pages_id, + }); + doc.trailer.set("Root", catalog_id); + + let mut bytes = Vec::new(); + doc.save_to(&mut bytes).unwrap(); + bytes +} + +#[test] +fn test_synthetic_type0_broken_tounicode_emits_fffd_not_latin1_mojibake() { + let buf = synthetic_type0_broken_tounicode_pdf(); + + let items = pdf_inspector::extractor::extract_text_with_positions_mem(&buf).unwrap(); + let combined: String = items.iter().map(|i| i.text.as_str()).collect(); + + // Mojibake leak check: 2-byte CID 0xCDD9 must NOT come out as "ÍÙ" + // (U+00CD U+00D9). That was the production scrape symptom. + assert!( + !combined.contains('\u{00CD}'), + "Latin-1 mojibake leaked from Type0 font: {combined:?}" + ); + assert!( + !combined.contains('\u{00D9}'), + "Latin-1 mojibake leaked from Type0 font: {combined:?}" + ); + + // Marker presence: Type0/CID + non-ASCII bytes must produce U+FFFD so + // `detect_encoding_issues` can flag the page for OCR downstream. + assert!( + combined.contains('\u{FFFD}'), + "Type0 font with malformed ToUnicode CMap should emit U+FFFD per CID; got: {combined:?}" + ); + + // End-to-end check: the page is correctly routed to OCR. + let result = pdf_inspector::process_pdf_mem(&buf).unwrap(); + assert!( + result.pages_needing_ocr.contains(&1), + "Type0 page with broken ToUnicode + non-ASCII bytes must be flagged for OCR; \ + pages_needing_ocr={:?}", + result.pages_needing_ocr + ); +}