Compare commits

..
Author SHA1 Message Date
Abimael MartellandCursor 92b0ca9c68 fix(extractor): bound CID /W range expansion
Type0 /W parsing and the Unicode-CID heuristic expanded every CID in every
range. Repeating a full-width [0 65535 w] entry therefore re-materialized
the same 65,536-key domain on every copy, growing a temporary vector and
HashMap work without bound.

Cap expansion at the 16-bit CID domain, collect unique CIDs for the
median heuristic, and stop width assignment once that many entries have
been written. Legitimate compact /W arrays are unchanged.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-08-12 14:13:45 -07:00
5 changed files with 64 additions and 1020 deletions
-328
View File
@@ -1,328 +0,0 @@
//! Bounded content-stream decoding.
//!
//! `lopdf::content::Content::decode` materializes every operator before any
//! caller can apply a limit. A compact page of `q Q` pairs can therefore
//! allocate hundreds of megabytes and abort. Count operators first (without
//! allocating `Operation` objects) and skip decode when the cap is exceeded.
use crate::PdfError;
use lopdf::content::Content;
/// Maximum content-stream operators decoded for a page or a single Form
/// XObject. Matches the previous post-decode skip threshold.
pub(crate) const MAX_PAGE_OPERATIONS: usize = 1_000_000;
/// Decode `data` unless it contains more than `max_operations` operators.
///
/// Returns `Ok(None)` when the stream exceeds the cap, so callers can skip
/// extraction without first allocating the operation vector.
pub(crate) fn decode_content_bounded(
data: &[u8],
max_operations: usize,
) -> Result<Option<Content>, PdfError> {
if content_exceeds_operation_limit(data, max_operations) {
return Ok(None);
}
Content::decode(data)
.map(Some)
.map_err(|e| PdfError::Parse(e.to_string()))
}
fn content_exceeds_operation_limit(data: &[u8], max_operations: usize) -> bool {
count_content_operators(data, max_operations.saturating_add(1)) > max_operations
}
/// Count operators using the same token rules as lopdf's content parser,
/// stopping at `limit`. Does not allocate `Operation` / `Object` values.
fn count_content_operators(data: &[u8], limit: usize) -> usize {
let mut i = 0;
let mut count = 0;
while i < data.len() && count < limit {
skip_content_space(data, &mut i);
if i >= data.len() {
break;
}
if data[i] == b'%' {
skip_comment(data, &mut i);
continue;
}
match data[i] {
b'(' => i = skip_literal_string(data, i),
b'<' => {
if data.get(i + 1) == Some(&b'<') {
i += 2;
} else {
i = skip_hex_string(data, i);
}
}
b'>' => {
i += 1;
if data.get(i) == Some(&b'>') {
i += 1;
}
}
b'[' | b']' => i += 1,
b'/' => skip_name(data, &mut i),
b'+' | b'-' | b'.' => skip_number(data, &mut i),
b if b.is_ascii_digit() => skip_number(data, &mut i),
b if is_operator_byte(b) => {
let start = i;
i += 1;
while i < data.len() && is_operator_byte(data[i]) {
i += 1;
}
let token = &data[start..i];
if token == b"true" || token == b"false" || token == b"null" {
continue;
}
count += 1;
if token == b"BI" && (i >= data.len() || is_content_space(data[i])) {
i = skip_inline_image_after_bi(data, i);
}
}
_ => i += 1,
}
}
count
}
fn is_content_space(b: u8) -> bool {
// PDF whitespace (ISO 32000): NUL, tab, LF, FF, CR, space. Names must
// stop on these so a following operator is not absorbed into `/Name`.
matches!(b, b'\0' | b'\t' | b'\n' | b'\x0c' | b'\r' | b' ')
}
fn is_operator_byte(b: u8) -> bool {
b.is_ascii_alphabetic() || matches!(b, b'*' | b'\'' | b'"')
}
fn is_delimiter(b: u8) -> bool {
matches!(
b,
b'(' | b')' | b'<' | b'>' | b'[' | b']' | b'{' | b'}' | b'/' | b'%'
)
}
fn skip_content_space(data: &[u8], i: &mut usize) {
while *i < data.len() && is_content_space(data[*i]) {
*i += 1;
}
}
fn skip_comment(data: &[u8], i: &mut usize) {
while *i < data.len() && data[*i] != b'\n' && data[*i] != b'\r' {
*i += 1;
}
}
fn skip_literal_string(data: &[u8], mut i: usize) -> usize {
let mut depth = 1i32;
i += 1;
while i < data.len() && depth > 0 {
match data[i] {
b'\\' => {
i += 1;
if i < data.len() {
i += 1;
}
}
b'(' => {
depth += 1;
i += 1;
}
b')' => {
depth -= 1;
i += 1;
}
_ => i += 1,
}
}
i
}
fn skip_hex_string(data: &[u8], mut i: usize) -> usize {
i += 1;
while i < data.len() && data[i] != b'>' {
i += 1;
}
if i < data.len() {
i += 1;
}
i
}
fn skip_name(data: &[u8], i: &mut usize) {
*i += 1;
while *i < data.len() && !is_content_space(data[*i]) && !is_delimiter(data[*i]) {
*i += 1;
}
}
fn skip_number(data: &[u8], i: &mut usize) {
if *i < data.len() && matches!(data[*i], b'+' | b'-') {
*i += 1;
}
while *i < data.len() && data[*i].is_ascii_digit() {
*i += 1;
}
if *i < data.len() && data[*i] == b'.' {
*i += 1;
while *i < data.len() && data[*i].is_ascii_digit() {
*i += 1;
}
}
}
/// After a `BI` operator, skip inline-image data through `EI`.
/// Uses the same PDF whitespace set as `is_content_space`. If `EI` is not
/// found, leave the cursor in place so later operators are still counted
/// (undercounting would let decode allocate the full vector).
fn skip_inline_image_after_bi(data: &[u8], mut i: usize) -> usize {
skip_content_space(data, &mut i);
let rest = &data[i..];
if let Some(pos) = rest.windows(4).position(|w| {
is_content_space(w[0]) && w[1] == b'E' && w[2] == b'I' && is_content_space(w[3])
}) {
return i + pos + 3;
}
i
}
#[cfg(test)]
mod tests {
use super::*;
fn lopdf_op_count(data: &[u8]) -> usize {
Content::decode(data)
.map(|c| c.operations.len())
.unwrap_or(0)
}
/// DoS safety: never report fewer operators than lopdf would allocate.
/// Overcount is acceptable (skip a page); undercount would re-open decode.
fn assert_count_does_not_undercount(data: &[u8]) {
let ours = count_content_operators(data, usize::MAX);
match Content::decode(data) {
Ok(content) => assert!(
ours >= content.operations.len(),
"undercount: ours={ours} lopdf={} for {:?}",
content.operations.len(),
String::from_utf8_lossy(data)
),
Err(_) => {}
}
}
#[test]
fn operator_count_matches_lopdf_for_typical_streams() {
let samples: &[&[u8]] = &[
b"q 1 0 0 1 0 0 cm BT /F1 12 Tf 72 720 Td (Hello) Tj ET Q",
b"q Q q Q",
b"BT /F1 12 Tf 12 TL 1 0 0 1 100 512 Tm (first) Tj (struck) ' ET",
b"1 0 0 rg 0 0 10 10 re f",
b"true false null q",
b"% comment\nq Q\n",
b"[ (a) 1 (b) ] TJ",
b"1 0 0 1 0 0 cm /Im0 Do",
];
for data in samples {
assert_eq!(
count_content_operators(data, usize::MAX),
lopdf_op_count(data),
"count mismatch for {}",
String::from_utf8_lossy(data)
);
}
}
#[test]
fn strings_and_comments_are_not_operators() {
let data = b"(q Q Tj) Tj % q Q\nET";
assert_eq!(
count_content_operators(data, usize::MAX),
lopdf_op_count(data)
);
assert_eq!(count_content_operators(data, usize::MAX), 2); // Tj, ET
}
#[test]
fn inline_image_counts_as_one_operator() {
let data = b"BI /W 2 /H 2 /CS /RGB /BPC 8 ID \x00\x01\x02\x03 EI q";
assert_eq!(
count_content_operators(data, usize::MAX),
lopdf_op_count(data)
);
assert_eq!(count_content_operators(data, usize::MAX), 2); // BI, q
}
#[test]
fn inline_image_ei_accepts_pdf_whitespace() {
let tab = b"BI /W 1 /H 1 ID \xff\tEI\t q Q";
let nul = b"BI /W 1 /H 1 ID \xff\x00EI\x00 q Q";
let ff = b"BI /W 1 /H 1 ID \xff\x0cEI\x0c q Q";
for data in [tab.as_slice(), nul.as_slice(), ff.as_slice()] {
assert_count_does_not_undercount(data);
assert!(
count_content_operators(data, usize::MAX) >= 3,
"BI plus following q Q must remain visible after EI, got {} for {:?}",
count_content_operators(data, usize::MAX),
String::from_utf8_lossy(data)
);
}
}
#[test]
fn decode_is_skipped_when_operator_cap_is_exceeded() {
let mut data = Vec::new();
for _ in 0..20 {
data.extend_from_slice(b"q Q\n");
}
assert!(decode_content_bounded(&data, 10).unwrap().is_none());
let decoded = decode_content_bounded(&data, 50).unwrap().unwrap();
assert_eq!(decoded.operations.len(), 40);
}
#[test]
fn name_whitespace_does_not_swallow_following_operator() {
// NUL / form-feed end a name (PDF whitespace). Absorbing `q` into
// `/x` would undercount and let decode allocate the operator vector.
let mut nul_sep = Vec::new();
let mut ff_sep = Vec::new();
for _ in 0..8_000 {
nul_sep.extend_from_slice(b"/x\x00q");
ff_sep.extend_from_slice(b"/x\x0cq");
}
assert_count_does_not_undercount(&nul_sep);
assert_count_does_not_undercount(&ff_sep);
assert!(count_content_operators(&ff_sep, usize::MAX) >= 8_000);
}
#[test]
fn edge_streams_do_not_undercount_vs_lopdf() {
let samples: &[&[u8]] = &[
b".5 0 0 .5 0 0 cm",
b"+1 -2 3.0 rg",
b"<0041> Tj",
b"(unbalanced",
b"BI /W 1 /H 1 ID \xff\xff no EI here q Q q Q",
b"q\x00Q\x00q\x00Q",
b"/F1\x0c12 Tf (Hi) Tj",
b"{ 1 2 add } cvx",
];
for data in samples {
assert_count_does_not_undercount(data);
}
}
#[test]
fn million_q_pairs_are_rejected_without_decode() {
let mut data = Vec::with_capacity((MAX_PAGE_OPERATIONS + 1) * 2);
for _ in 0..=MAX_PAGE_OPERATIONS {
data.extend_from_slice(b"q\n");
}
assert!(content_exceeds_operation_limit(&data, MAX_PAGE_OPERATIONS));
assert!(decode_content_bounded(&data, MAX_PAGE_OPERATIONS)
.unwrap()
.is_none());
}
}
+14 -29
View File
@@ -151,6 +151,8 @@ pub(crate) fn extract_page_text_items(
style_cache: &mut FontStyleCache,
form_budget: &mut FormWalkBudget,
) -> Result<(PageExtraction, bool, bool, bool), PdfError> {
use lopdf::content::Content;
let mut items = Vec::new();
let mut rects: Vec<PdfRect> = Vec::new();
let mut clip_rects: Vec<PdfRect> = Vec::new();
@@ -254,20 +256,18 @@ pub(crate) fn extract_page_text_items(
// Content::decode parser, causing it to skip operators like ET and Q.
let content_data = strip_pdf_comments(&content_data);
let content = match super::content_decode::decode_content_bounded(
&content_data,
super::content_decode::MAX_PAGE_OPERATIONS,
)? {
Some(content) => content,
None => {
log::warn!(
"page {}: skipping extraction — content stream exceeds {} operations",
page_num,
super::content_decode::MAX_PAGE_OPERATIONS
);
return Ok(((Vec::new(), Vec::new(), Vec::new()), false, false, false));
}
};
let content = Content::decode(&content_data).map_err(|e| PdfError::Parse(e.to_string()))?;
const MAX_OPERATIONS: usize = 1_000_000;
if content.operations.len() > MAX_OPERATIONS {
log::warn!(
"page {}: skipping extraction — {} operations exceeds limit ({})",
page_num,
content.operations.len(),
MAX_OPERATIONS
);
return Ok(((Vec::new(), Vec::new(), Vec::new()), false, false, false));
}
// Graphics state tracking
let mut ctm = [1.0f32, 0.0, 0.0, 1.0, 0.0, 0.0]; // Current Transformation Matrix
@@ -1937,19 +1937,4 @@ BT 30 700 Tm <41> Tj ET";
let output = strip_pdf_comments(input);
assert_eq!(output, b"(x\\\\) Tj \nET\n");
}
#[test]
fn oversized_content_stream_skips_extraction() {
let mut content =
Vec::with_capacity((super::super::content_decode::MAX_PAGE_OPERATIONS + 1) * 2);
for _ in 0..=super::super::content_decode::MAX_PAGE_OPERATIONS {
content.extend_from_slice(b"q\n");
}
content.extend_from_slice(b"BT /F1 12 Tf 72 720 Td (Hello) Tj ET\n");
let items = extract_simple_items(&content);
assert!(
items.is_empty(),
"pages over the operator cap must not be decoded"
);
}
}
+3 -237
View File
@@ -3,7 +3,6 @@
//! This module extracts text with position information for structure detection.
mod base14;
mod content_decode;
pub(crate) mod content_stream;
mod fonts;
mod layout;
@@ -890,92 +889,6 @@ fn tracked_run_space_floor(group: &[&TextItem], start: usize) -> Option<(usize,
Some((end, floor * fs))
}
/// Fractional font-size band within which `merge_text_items` treats two runs as
/// the same size. Shared with `is_small_caps_continuation`, which exists only to
/// rescue junctions this band would otherwise break.
const MERGE_FONT_SIZE_BAND: f32 = 0.20;
/// Detect a small-caps continuation: typesetters render small caps as a
/// full-size capital immediately followed by shrunken capitals in the same
/// font (`(R) Tj` at 9.98pt, then `(OLANDO) Tj` at 6.74pt). Those runs are one
/// word, but the font-size band in `merge_text_items` would split them,
/// leaving "R" and "OLANDO" as separate items — which then read as separate
/// table columns, since column boundaries cluster on item start positions.
///
/// Gated tightly so it cannot absorb the other reasons a smaller run follows a
/// larger one:
/// - runs the size band already accepts — excluded by requiring the junction
/// to *cross* the band, so within-band pairs keep the normal word-spacing
/// logic instead of having their space suppressed
/// - superscripts / footnote markers — excluded by requiring an uppercase
/// *letter* on both sides, so digits never qualify
/// - drop caps — excluded because the body text that follows is mixed case
/// - adjacent table cells or separate words — excluded by requiring the runs
/// to be visually contiguous (essentially no gap)
fn is_small_caps_continuation(
text_so_far: &str,
first: &TextItem,
next: &TextItem,
gap: f32,
) -> bool {
// Must shrink. Real small caps sit near 0.7-0.8 of the full cap height;
// anything smaller is a superscript or a different run entirely.
if first.font_size <= 0.0 || next.font_size >= first.font_size {
return false;
}
// Only rescue junctions the size band would have broken. Within-band pairs
// merge on their own, and suppressing their space would swallow real word
// gaps between two similarly-sized uppercase words.
if (next.font_size - first.font_size).abs() <= first.font_size * MERGE_FONT_SIZE_BAND {
return false;
}
if next.font_size / first.font_size < 0.55 {
return false;
}
// Visually contiguous: the capital and its small caps touch. A real word
// space or a column gap disqualifies.
if !(-first.font_size * 0.2..=first.font_size * 0.15).contains(&gap) {
return false;
}
// The continuation must be all-uppercase letters (digits and lowercase
// both disqualify), and must contain at least one letter.
let mut saw_letter = false;
for ch in next.text.chars() {
if ch.is_alphabetic() {
saw_letter = true;
if !ch.is_uppercase() {
return false;
}
} else if ch.is_numeric() {
return false;
}
}
if !saw_letter {
return false;
}
// What we are continuing must itself end in a capital. Check the actual
// trailing character rather than skipping back to the nearest letter: after
// "ANGELA M. MAZZARELLI1" the run to continue is the footnote marker, not
// the "I" before it.
let trimmed = text_so_far.trim_end();
if trimmed.chars().last().is_some_and(|c| c.is_numeric()) {
// One legitimate exception: an ordinal suffix set as a smaller run,
// e.g. "JULY 4" + "TH". Only the four English suffixes qualify —
// anything else after a digit is a footnote marker or numeric suffix.
return matches!(trimmed_suffix(next), "TH" | "ST" | "ND" | "RD");
}
trimmed
.chars()
.rev()
.find(|c| c.is_alphabetic())
.is_some_and(|c| c.is_uppercase())
}
/// The continuation run's text, trimmed — used to spot ordinal suffixes.
fn trimmed_suffix(next: &TextItem) -> &str {
next.text.trim()
}
pub(crate) fn merge_text_items(items: Vec<TextItem>) -> Vec<TextItem> {
if items.is_empty() {
return items;
@@ -1034,16 +947,8 @@ pub(crate) fn merge_text_items(items: Vec<TextItem>) -> Vec<TextItem> {
let mut j = i + 1;
while j < group.len() {
let next = group[j];
// A small-caps junction is mid-word: it both survives the
// font-size band below and must never take a space.
let small_caps_join =
is_small_caps_continuation(&text, first, next, next.x - end_x);
// Must be similar font size, except for genuine small-caps
// runs, where the shrunken capitals are the same word as the
// full-size initial (see helper).
if (next.font_size - first.font_size).abs() > first.font_size * MERGE_FONT_SIZE_BAND
&& !small_caps_join
{
// Must be similar font size (within 20%)
if (next.font_size - first.font_size).abs() > first.font_size * 0.20 {
break;
}
// Never merge across style boundaries: the merged item
@@ -1097,7 +1002,7 @@ pub(crate) fn merge_text_items(items: Vec<TextItem>) -> Vec<TextItem> {
Some((run_end, floor)) if j <= run_end => floor,
_ => threshold,
};
if !small_caps_join && (needs_bullet_space || gap > effective_threshold) {
if needs_bullet_space || gap > effective_threshold {
text.push(' ');
}
text.push_str(&next.text);
@@ -3101,145 +3006,6 @@ mod tests {
}
}
/// Small caps as typesetters emit them: a full-size capital at 9.98pt
/// immediately followed by shrunken capitals at 6.74pt, touching.
/// Modelled on `199AD3d.pdf` p.5 ("ROLANDO T. ACOSTA, P.J.").
#[test]
fn small_caps_run_merges_into_one_word() {
let items = vec![
make_item_fs("R", 144.36, 581.84, 7.20, 9.98),
make_item_fs("OLANDO", 151.56, 581.84, 30.56, 6.74),
make_item_fs("T. A", 185.45, 581.84, 17.58, 9.98),
make_item_fs("COSTA", 203.94, 581.84, 23.15, 6.74),
make_item_fs(", P.J.", 227.09, 581.84, 22.56, 9.98),
];
let merged = merge_text_items(items);
assert_eq!(merged.len(), 1, "got {:?}", merged);
assert_eq!(merged[0].text, "ROLANDO T. ACOSTA, P.J.");
}
/// The full two-column row: both names must merge independently and the
/// 72pt column gap between them must survive as an item boundary.
#[test]
fn small_caps_merge_does_not_swallow_a_second_column() {
let items = vec![
// Column 1: "ROLANDO T. ACOSTA, P.J." ending at x=249.65
make_item_fs("R", 144.36, 581.84, 7.20, 9.98),
make_item_fs("OLANDO", 151.56, 581.84, 30.56, 6.74),
make_item_fs("T. A", 185.45, 581.84, 17.58, 9.98),
make_item_fs("COSTA", 203.94, 581.84, 23.15, 6.74),
make_item_fs(", P.J.", 227.09, 581.84, 22.56, 9.98),
// Column 2 starts at x=321.96 — a 72pt gap.
make_item_fs("A", 321.96, 581.84, 7.20, 9.98),
make_item_fs("NIL", 329.17, 581.84, 12.72, 6.74),
make_item_fs("C. S", 345.04, 581.84, 19.59, 9.98),
make_item_fs("INGH", 364.62, 581.84, 19.08, 6.74),
];
let merged = merge_text_items(items);
let texts: Vec<&str> = merged.iter().map(|i| i.text.as_str()).collect();
assert_eq!(
texts,
vec!["ROLANDO T. ACOSTA, P.J.", "ANIL C. SINGH"],
"column gap should keep the two names apart"
);
}
#[test]
fn small_caps_merge_keeps_word_space_between_same_size_capitals() {
// Two uppercase words at sizes the merge band already accepts (9.98 and
// 9.0, a 10% drop) separated by a real word gap. The small-caps path
// must not claim this junction and swallow the space.
let items = vec![
make_item_fs("SEE", 100.0, 500.0, 18.0, 9.98),
make_item_fs("ALSO", 119.2, 500.0, 24.0, 9.0),
];
let merged = merge_text_items(items);
assert_eq!(merged.len(), 1, "got {:?}", merged);
assert_eq!(merged[0].text, "SEE ALSO");
}
#[test]
fn trailing_digit_is_not_a_capital_awaiting_small_caps() {
// "...MAZZARELLI1" ends in a footnote marker; the backward search for an
// uppercase letter must not skip the digit and glue the next run.
assert!(!is_small_caps_continuation(
"ANGELA M. MAZZARELLI1",
&make_item_fs("ANGELA", 100.0, 500.0, 40.0, 9.98),
&make_item_fs("SHULMAN", 140.0, 500.0, 30.0, 6.74),
0.0,
));
}
#[test]
fn ordinal_suffix_after_a_digit_still_merges() {
// "TUESDAY, JULY 4" + "TH" is one word in the source; the digit guard
// must not block the four English ordinal suffixes.
for suffix in ["TH", "ST", "ND", "RD"] {
assert!(
is_small_caps_continuation(
"TUESDAY, JULY 4",
&make_item_fs("JULY", 100.0, 500.0, 30.0, 12.0),
&make_item_fs(suffix, 130.0, 500.0, 8.0, 8.0),
0.0,
),
"{suffix} should merge after a digit"
);
}
}
#[test]
fn superscript_footnote_marker_is_not_a_small_caps_continuation() {
// A digit must never qualify — otherwise footnote markers get glued on
// without the superscript handling.
assert!(!is_small_caps_continuation(
"MAZZARELLI",
&make_item_fs("MAZZARELLI", 100.0, 500.0, 50.0, 9.98),
&make_item_fs("1", 150.0, 503.0, 3.0, 6.74),
0.0,
));
}
#[test]
fn drop_cap_is_not_a_small_caps_continuation() {
// Mixed-case body text after a large initial is a drop cap, not small
// caps.
assert!(!is_small_caps_continuation(
"T",
&make_item_fs("T", 100.0, 500.0, 20.0, 30.0),
&make_item_fs("he court held", 120.0, 500.0, 60.0, 10.0),
0.0,
));
}
#[test]
fn separate_word_is_not_a_small_caps_continuation() {
// A real word space disqualifies even when both runs are uppercase.
let first = make_item_fs("SEE", 100.0, 500.0, 20.0, 9.98);
let next = make_item_fs("ALSO", 128.0, 500.0, 25.0, 6.74);
assert!(!is_small_caps_continuation("SEE", &first, &next, 8.0));
}
#[test]
fn lowercase_continuation_is_not_small_caps() {
assert!(!is_small_caps_continuation(
"SMALL",
&make_item_fs("SMALL", 100.0, 500.0, 30.0, 9.98),
&make_item_fs("caps", 130.0, 500.0, 20.0, 6.74),
0.0,
));
}
#[test]
fn too_small_a_ratio_is_not_small_caps() {
// 0.4 ratio is a superscript/sub-run, outside the small-caps band.
assert!(!is_small_caps_continuation(
"A",
&make_item_fs("A", 100.0, 500.0, 7.0, 10.0),
&make_item_fs("BC", 107.0, 500.0, 8.0, 4.0),
0.0,
));
}
#[test]
fn test_merge_subscript_items_chemical_formula() {
// NH₃: "NH" at fs=8 followed by subscript "3" at fs=4.7
+27 -285
View File
@@ -215,6 +215,8 @@ fn extract_form_xobject_text_inner(
depth: u8,
budget: &mut FormWalkBudget,
) -> Vec<TextItem> {
use lopdf::content::Content;
let mut items = Vec::new();
if !budget.charge_invocation() {
@@ -232,12 +234,8 @@ fn extract_form_xobject_text_inner(
Err(_) => stream.content.clone(),
};
// Decode the content stream. Cap before lopdf materializes the operator
// vector — the walk budget cannot help if decode itself allocates first.
let Ok(Some(content)) = super::content_decode::decode_content_bounded(
&content_data,
super::content_decode::MAX_PAGE_OPERATIONS,
) else {
// Decode the content stream
let Ok(content) = Content::decode(&content_data) else {
return items;
};
@@ -326,29 +324,10 @@ fn extract_form_xobject_text_inner(
let mut current_font = String::new();
let mut current_font_size: f32 = 12.0;
let mut text_matrix = [1.0f32, 0.0, 0.0, 1.0, 0.0, 0.0];
// Text line matrix (TLM) — Td/TD/T* move relative to the start of the
// current line, not to the position left by the last show operator.
let mut line_matrix = [1.0f32, 0.0, 0.0, 1.0, 0.0, 0.0];
let mut text_leading: f32 = 0.0; // TL parameter (text-space units)
let mut char_spacing: f32 = 0.0; // Tc parameter
let mut word_spacing: f32 = 0.0; // Tw parameter
let mut in_text_block = false;
let mut fill_is_white = false;
let mut ctm = base_ctm;
// Text state (Tc/Tw/TL/Tf) and the fill colour are part of the graphics
// state and must be saved/restored by q/Q alongside the CTM.
#[derive(Clone)]
struct GraphicsState {
ctm: [f32; 6],
char_spacing: f32,
word_spacing: f32,
text_leading: f32,
current_font: String,
current_font_size: f32,
fill_is_white: bool,
}
let mut ctm_stack: Vec<GraphicsState> = Vec::new();
let mut ctm_stack: Vec<[f32; 6]> = Vec::new();
for op in &content.operations {
if !budget.charge_operation() {
@@ -356,25 +335,11 @@ fn extract_form_xobject_text_inner(
}
match op.operator.as_str() {
"q" => {
ctm_stack.push(GraphicsState {
ctm,
char_spacing,
word_spacing,
text_leading,
current_font: current_font.clone(),
current_font_size,
fill_is_white,
});
ctm_stack.push(ctm);
}
"Q" => {
if let Some(saved) = ctm_stack.pop() {
ctm = saved.ctm;
char_spacing = saved.char_spacing;
word_spacing = saved.word_spacing;
text_leading = saved.text_leading;
current_font = saved.current_font;
current_font_size = saved.current_font_size;
fill_is_white = saved.fill_is_white;
ctm = saved;
}
}
"cm" => {
@@ -438,7 +403,6 @@ fn extract_form_xobject_text_inner(
"BT" => {
in_text_block = true;
text_matrix = [1.0, 0.0, 0.0, 1.0, 0.0, 0.0];
line_matrix = text_matrix;
}
"ET" => {
in_text_block = false;
@@ -451,33 +415,12 @@ fn extract_form_xobject_text_inner(
current_font_size = get_number(&op.operands[1]).unwrap_or(12.0);
}
}
"TL" => {
// Set text leading (used by T*, ', and ")
if let Some(tl) = op.operands.first().and_then(get_number) {
text_leading = tl;
}
}
"Tc" => {
if let Some(tc) = op.operands.first().and_then(get_number) {
char_spacing = tc;
}
}
"Tw" => {
if let Some(tw) = op.operands.first().and_then(get_number) {
word_spacing = tw;
}
}
"Td" | "TD" => {
// Move text position: TLM = T(tx,ty) x TLM; Tm = TLM
if op.operands.len() >= 2 {
let tx = get_number(&op.operands[0]).unwrap_or(0.0);
let ty = get_number(&op.operands[1]).unwrap_or(0.0);
line_matrix[4] += tx * line_matrix[0] + ty * line_matrix[2];
line_matrix[5] += tx * line_matrix[1] + ty * line_matrix[3];
text_matrix = line_matrix;
if op.operator == "TD" {
text_leading = -ty;
}
text_matrix[4] += tx * text_matrix[0] + ty * text_matrix[2];
text_matrix[5] += tx * text_matrix[1] + ty * text_matrix[3];
}
}
"Tm" => {
@@ -486,20 +429,8 @@ fn extract_form_xobject_text_inner(
text_matrix[i] =
get_number(operand).unwrap_or(if i == 0 || i == 3 { 1.0 } else { 0.0 });
}
line_matrix = text_matrix;
}
}
"T*" => {
// Move to start of next line: equivalent to `0 -TL Td`
let tl = if text_leading != 0.0 {
text_leading
} else {
current_font_size * 1.2
};
line_matrix[4] += (-tl) * line_matrix[2];
line_matrix[5] += (-tl) * line_matrix[3];
text_matrix = line_matrix;
}
"g" => {
if let Some(gray) = op.operands.first().and_then(get_number) {
fill_is_white = gray > 0.95;
@@ -535,33 +466,17 @@ fn extract_form_xobject_text_inner(
_ => fill_is_white = false,
}
}
"Tj" | "'" | "\"" => {
// `'` = move to next line then show; `"` = set word/char spacing,
// move to next line, then show (string is the last operand).
if op.operator != "Tj" {
if op.operator == "\"" && op.operands.len() >= 3 {
word_spacing = get_number(&op.operands[0]).unwrap_or(word_spacing);
char_spacing = get_number(&op.operands[1]).unwrap_or(char_spacing);
}
let tl = if text_leading != 0.0 {
text_leading
} else {
current_font_size * 1.2
};
line_matrix[4] += (-tl) * line_matrix[2];
line_matrix[5] += (-tl) * line_matrix[3];
text_matrix = line_matrix;
}
if let (true, Some(show_operand)) = (in_text_block, op.operands.last()) {
"Tj" => {
if in_text_block && !op.operands.is_empty() {
if fill_is_white {
if let Some(font_info) = font_widths.get(&current_font) {
if let Some(raw_bytes) = get_operand_bytes(show_operand) {
if let Some(raw_bytes) = get_operand_bytes(&op.operands[0]) {
let w_ts = compute_string_width_ts(
raw_bytes,
font_info,
current_font_size,
char_spacing,
word_spacing,
0.0,
0.0,
);
text_matrix[4] += w_ts * text_matrix[0];
text_matrix[5] += w_ts * text_matrix[1];
@@ -570,7 +485,7 @@ fn extract_form_xobject_text_inner(
continue;
}
if let Some(text) = extract_text_from_operand(
show_operand,
&op.operands[0],
&current_font,
font_base_names.get(&current_font).map(|s| s.as_str()),
font_cmaps,
@@ -586,13 +501,13 @@ fn extract_form_xobject_text_inner(
* type3_scales.get(&current_font).copied().unwrap_or(1.0);
let (x, y) = (combined[4], combined[5]);
let width = if let Some(font_info) = font_widths.get(&current_font) {
if let Some(raw_bytes) = get_operand_bytes(show_operand) {
if let Some(raw_bytes) = get_operand_bytes(&op.operands[0]) {
let w_ts = compute_string_width_ts(
raw_bytes,
font_info,
current_font_size,
char_spacing,
word_spacing,
0.0,
0.0,
);
text_matrix[4] += w_ts * text_matrix[0];
text_matrix[5] += w_ts * text_matrix[1];
@@ -715,8 +630,8 @@ fn extract_form_xobject_text_inner(
raw_bytes,
fi,
current_font_size,
char_spacing,
word_spacing,
0.0,
0.0,
);
}
}
@@ -853,9 +768,14 @@ pub(crate) fn get_form_fonts<'a>(
#[cfg(test)]
mod tests {
use super::*;
use super::{
extract_form_xobject_text, CMapDecisionCache, FontStyleCache, FormWalkBudget,
MAX_FORM_XOBJECT_INVOCATIONS, MAX_FORM_XOBJECT_OPERATIONS,
};
use crate::extractor::content_stream::extract_page_text_items;
use lopdf::{dictionary, Dictionary, Stream};
use crate::tounicode::FontCMaps;
use crate::types::TextItem;
use lopdf::{dictionary, Dictionary, Document, Object, ObjectId, Stream};
/// Build an acyclic Form XObject DAG: `levels` form objects, each non-leaf
/// invoking the next form `branches` times. The leaf draws a single `(X)`.
@@ -1083,182 +1003,4 @@ mod tests {
);
assert!(budget.was_truncated());
}
/// Build a document whose page draws *all* of its content through a single
/// Form XObject — the shape emitted by print-to-PDF producers like PDFlib,
/// where the page stream itself is only `q /X1 Do Q`.
fn doc_with_form_content(form_content: &[u8]) -> (Document, ObjectId) {
let mut doc = Document::new();
let widths: Vec<Object> = (0..=255).map(|_| 600.into()).collect();
let font_id = doc.add_object(dictionary! {
"Type" => "Font",
"Subtype" => "Type1",
"BaseFont" => "Helvetica",
"FirstChar" => 0,
"LastChar" => 255,
"Widths" => Object::Array(widths),
});
let form_id = doc.add_object(Object::Stream(Stream::new(
dictionary! {
"Type" => "XObject",
"Subtype" => "Form",
"BBox" => vec![0.into(), 0.into(), 612.into(), 792.into()],
"Resources" => dictionary! {
"Font" => dictionary! { "F1" => Object::Reference(font_id) },
},
},
form_content.to_vec(),
)));
let content_id = doc.add_object(Object::Stream(Stream::new(
dictionary! {},
b"q /X1 Do Q".to_vec(),
)));
let page_id = doc.add_object(dictionary! {
"Type" => "Page",
"Contents" => Object::Reference(content_id),
"Resources" => dictionary! {
"XObject" => dictionary! { "X1" => Object::Reference(form_id) },
},
"MediaBox" => vec![0.into(), 0.into(), 612.into(), 792.into()],
});
let pages_id = doc.add_object(dictionary! {
"Type" => "Pages",
"Count" => Object::Integer(1),
"Kids" => vec![Object::Reference(page_id)],
});
let catalog_id = doc.add_object(dictionary! {
"Type" => "Catalog",
"Pages" => Object::Reference(pages_id),
});
doc.trailer.set("Root", Object::Reference(catalog_id));
(doc, page_id)
}
fn form_items(form_content: &[u8]) -> Vec<TextItem> {
let (doc, page_id) = doc_with_form_content(form_content);
let font_cmaps = FontCMaps::from_doc(&doc);
let ((items, _, _), _, _, _) = extract_page_text_items(
&doc,
page_id,
1,
&font_cmaps,
false,
&mut FontStyleCache::new(),
&mut FormWalkBudget::new(),
)
.unwrap();
items
}
fn find<'a>(items: &'a [TextItem], text: &str) -> &'a TextItem {
items
.iter()
.find(|item| item.text == text)
.unwrap_or_else(|| {
let found: Vec<&String> = items.iter().map(|i| &i.text).collect();
panic!("no item {text:?} in {found:?}")
})
}
#[test]
fn t_star_inside_form_moves_to_next_line() {
// T* was previously unhandled inside Form XObjects, so every line after
// the first piled onto the preceding baseline and drifted right.
let items =
form_items(b"BT /F1 12 Tf 12 TL 1 0 0 1 100 700 Tm (first) Tj T* (second) Tj ET");
let first = find(&items, "first");
let second = find(&items, "second");
assert!((first.y - 700.0).abs() < 0.1, "first y = {}", first.y);
assert!((second.y - 688.0).abs() < 0.1, "second y = {}", second.y);
assert!((second.x - 100.0).abs() < 0.1, "second x = {}", second.x);
}
#[test]
fn td_inside_form_is_relative_to_line_start_not_shown_text() {
// Td moves relative to the text *line* matrix. Applying it to the
// matrix already advanced by Tj marched each line off the right edge.
let items = form_items(b"BT /F1 12 Tf 1 0 0 1 100 700 Tm (AAAAA) Tj 0 -12 Td (B) Tj ET");
let b = find(&items, "B");
assert!((b.x - 100.0).abs() < 0.1, "B x = {} (expected 100)", b.x);
assert!((b.y - 688.0).abs() < 0.1, "B y = {}", b.y);
}
#[test]
fn td_inside_form_sets_leading_for_later_t_star() {
// `TD` sets the leading to -ty as a side effect; a following T* must
// reuse it.
let items = form_items(
b"BT /F1 12 Tf 1 0 0 1 100 700 Tm (one) Tj 0 -15 TD (two) Tj T* (three) Tj ET",
);
assert!((find(&items, "two").y - 685.0).abs() < 0.1);
let three = find(&items, "three");
assert!((three.y - 670.0).abs() < 0.1, "three y = {}", three.y);
assert!((three.x - 100.0).abs() < 0.1, "three x = {}", three.x);
}
#[test]
fn quote_operator_inside_form_moves_to_next_line() {
let items = form_items(b"BT /F1 12 Tf 12 TL 1 0 0 1 100 700 Tm (first) Tj (second) ' ET");
let second = find(&items, "second");
assert!((second.y - 688.0).abs() < 0.1, "second y = {}", second.y);
assert!((second.x - 100.0).abs() < 0.1, "second x = {}", second.x);
}
#[test]
fn double_quote_operator_inside_form_sets_spacing_and_moves() {
// `aw ac (string) "` — set word spacing and char spacing, then T* and show.
let items =
form_items(b"BT /F1 12 Tf 12 TL 1 0 0 1 100 700 Tm (first) Tj 0 0 (second) \" ET");
let second = find(&items, "second");
assert!((second.y - 688.0).abs() < 0.1, "second y = {}", second.y);
assert!((second.x - 100.0).abs() < 0.1, "second x = {}", second.x);
}
#[test]
fn char_spacing_inside_form_widens_advance() {
// Tc was hardcoded to 0 in the form parser, so advance widths drifted.
// 2 glyphs x 600/1000 x 12pt = 14.4, plus 2 x Tc(2.0) = 18.4.
let items = form_items(b"BT /F1 12 Tf 1 0 0 1 100 700 Tm 2 Tc (AB) Tj ET");
let ab = find(&items, "AB");
assert!((ab.width - 18.4).abs() < 0.1, "AB width = {}", ab.width);
}
#[test]
fn q_restores_fill_colour_inside_form() {
// A white fill set inside q/Q must not leak past the Q — otherwise the
// following black text is treated as invisible and dropped entirely.
let items = form_items(
b"BT /F1 12 Tf 12 TL 1 0 0 1 100 700 Tm q 1 g (hidden) Tj Q T* (visible) Tj ET",
);
assert!(
items.iter().any(|item| item.text == "visible"),
"text after Q was dropped: {:?}",
items.iter().map(|i| &i.text).collect::<Vec<_>>()
);
assert!(
!items.iter().any(|item| item.text == "hidden"),
"white-filled text should still be suppressed"
);
}
#[test]
fn q_restores_text_state_inside_form() {
// Tc/TL live in the graphics state; `Q` must roll them back.
let items =
form_items(b"BT /F1 12 Tf 12 TL 1 0 0 1 100 700 Tm q 30 TL (a) Tj Q T* (b) Tj ET");
let b = find(&items, "b");
assert!(
(b.y - 688.0).abs() < 0.1,
"b y = {} (leading should restore to 12)",
b.y
);
}
}
+20 -141
View File
@@ -540,14 +540,18 @@ impl ToUnicodeCMap {
/// Remap a CMap that references pre-subsetting GIDs to sequential post-subsetting GIDs.
/// Collects all source CIDs, sorts them, and reassigns to 1, 2, 3, ...
///
/// Range expansion stops after `MAX_CID_W_EXPANSION` CID visits, counting
/// overwrites, so repeated full-width `bfrange`s cannot re-expand the
/// 16-bit domain. Later overlapping ranges that would have introduced new
/// CIDs after that many visits are truncated.
pub fn remap_to_sequential(&self) -> ToUnicodeCMap {
let mut cid_to_unicode: HashMap<u16, String> = HashMap::new();
expand_bfranges_for_remap(&self.ranges, &mut cid_to_unicode, MAX_CID_W_EXPANSION);
// Expand ranges first
for &(start, end, base) in &self.ranges {
for cid in start..=end {
let unicode_cp = base + (cid - start) as u32;
if let Some(ch) = char::from_u32(unicode_cp) {
cid_to_unicode.insert(cid, ch.to_string());
}
}
}
// char_map entries override range entries
for (&cid, unicode) in &self.char_map {
@@ -572,33 +576,6 @@ impl ToUnicodeCMap {
}
}
/// Expand `bfrange` entries into individual CID→Unicode inserts.
/// Returns how many CIDs were visited. Counts overwrites so a repeated
/// full-width range cannot keep working after `max_assignments`.
fn expand_bfranges_for_remap(
ranges: &[(u16, u16, u32)],
cid_to_unicode: &mut HashMap<u16, String>,
max_assignments: usize,
) -> usize {
let mut assigned = 0usize;
'ranges: for &(start, end, base) in ranges {
if start > end {
continue;
}
for cid in start..=end {
if assigned >= max_assignments {
break 'ranges;
}
assigned += 1;
let unicode_cp = base + (cid - start) as u32;
if let Some(ch) = char::from_u32(unicode_cp) {
cid_to_unicode.insert(cid, ch.to_string());
}
}
}
assigned
}
/// Parse a hex string to u16
fn parse_hex_u16(hex: &str) -> Option<u16> {
u16::from_str_radix(hex.trim(), 16).ok()
@@ -1584,31 +1561,23 @@ fn parse_encoding_cmap_stream(data: &[u8]) -> Option<EncodingCMap> {
}
let mut map = HashMap::new();
let mut assigned = 0usize;
let mut pos = 0;
while let Some(start) = text[pos..].find("begincidchar") {
let section_start = pos + start + "begincidchar".len();
if let Some(end) = text[section_start..].find("endcidchar") {
let section = &text[section_start..section_start + end];
if !parse_cidchar_section(section, &mut map, &mut src_hex_lengths, &mut assigned) {
break;
}
parse_cidchar_section(section, &mut map, &mut src_hex_lengths);
pos = section_start + end;
} else {
break;
}
}
pos = 0;
while assigned < MAX_CID_W_EXPANSION {
let Some(start) = text[pos..].find("begincidrange") else {
break;
};
while let Some(start) = text[pos..].find("begincidrange") {
let section_start = pos + start + "begincidrange".len();
if let Some(end) = text[section_start..].find("endcidrange") {
let section = &text[section_start..section_start + end];
if !parse_cidrange_section(section, &mut map, &mut src_hex_lengths, &mut assigned) {
break;
}
parse_cidrange_section(section, &mut map, &mut src_hex_lengths);
pos = section_start + end;
} else {
break;
@@ -1643,8 +1612,7 @@ fn parse_cidchar_section(
section: &str,
map: &mut HashMap<u16, u16>,
src_hex_lengths: &mut Vec<usize>,
assigned: &mut usize,
) -> bool {
) {
let mut chars = section.chars().peekable();
loop {
while chars.peek().is_some_and(|c| c.is_whitespace()) {
@@ -1675,20 +1643,16 @@ fn parse_cidchar_section(
}
}
if let (Some(code), Ok(cid)) = (parse_hex_u16(&src_hex), cid_str.parse::<u16>()) {
if !assign_encoding_cid(map, code, cid, assigned) {
return false;
}
map.insert(code, cid);
}
}
true
}
fn parse_cidrange_section(
section: &str,
map: &mut HashMap<u16, u16>,
src_hex_lengths: &mut Vec<usize>,
assigned: &mut usize,
) -> bool {
) {
let mut chars = section.chars().peekable();
loop {
while chars.peek().is_some_and(|c| c.is_whitespace()) {
@@ -1739,34 +1703,12 @@ fn parse_cidrange_section(
) else {
continue;
};
if start > end {
continue;
}
let mut cid = start_cid;
for code in start..=end {
if !assign_encoding_cid(map, code, cid, assigned) {
return false;
}
map.insert(code, cid);
cid = cid.saturating_add(1);
}
}
true
}
fn assign_encoding_cid(
map: &mut HashMap<u16, u16>,
code: u16,
cid: u16,
assigned: &mut usize,
) -> bool {
// Count overwrites: unique-key coverage alone would not stop a repeated
// full-width range from re-inserting all 65,536 codes.
if *assigned >= MAX_CID_W_EXPANSION {
return false;
}
map.insert(code, cid);
*assigned += 1;
true
}
fn parse_binary_cmap_encoding(data: &[u8]) -> Result<EncodingCMap, String> {
@@ -1890,11 +1832,9 @@ fn merge_cmaps(mut base: ToUnicodeCMap, overlay: ToUnicodeCMap) -> ToUnicodeCMap
base
}
/// Shared 16-bit CID expansion cap (65,536).
/// Encoding `begincidrange`, `/W` width assignment, and ToUnicode sequential
/// remap count every insert, including overwrites, so a repeated full-width
/// range cannot keep working after the domain is filled. The `/W` unicode
/// heuristic caps unique CIDs with the same number.
/// Upper bound on CID `/W` range expansion. The CID domain is 16-bit, so more
/// than 65,536 unique keys cannot exist; repeating full-width ranges must not
/// re-expand the same domain.
pub(crate) const MAX_CID_W_EXPANSION: usize = 65_536;
/// Check if a CIDFont's /W (widths) array contains CID values that look like
@@ -2947,33 +2887,6 @@ endbfrange
assert!(remapped.ranges.is_empty());
}
#[test]
fn remap_to_sequential_repeated_full_bfranges_stay_bounded() {
// 5,000 copies of `<0003> <ffff>` must stop after 65,536 CID visits,
// not 5,000 × ~65,533 expansions.
let ranges = vec![(3u16, 65535u16, 0x41u32); 5_000];
let mut map = std::collections::HashMap::new();
let assigned = expand_bfranges_for_remap(&ranges, &mut map, MAX_CID_W_EXPANSION);
assert_eq!(assigned, MAX_CID_W_EXPANSION);
assert!(map.len() <= MAX_CID_W_EXPANSION);
let mut body = String::new();
let mut remaining = 5_000usize;
while remaining > 0 {
let n = remaining.min(100);
body.push_str(&format!("{n} beginbfrange\n"));
for _ in 0..n {
body.push_str("<0003> <ffff> <0041>\n");
}
body.push_str("endbfrange\n");
remaining -= n;
}
let data = format!("1 begincodespacerange\n<0000> <ffff>\nendcodespacerange\n{body}");
let cmap = ToUnicodeCMap::parse(data.as_bytes()).unwrap();
let remapped = cmap.remap_to_sequential();
assert_eq!(remapped.lookup(1), Some("A".to_string()));
}
#[test]
fn test_min_source_cid() {
let cmap_content = r#"
@@ -3446,38 +3359,4 @@ endbfrange
dict.set("W", Object::Array(w));
assert!(cid_values_look_like_unicode(&dict));
}
#[test]
fn encoding_cidrange_maps_a_normal_range() {
let data = b"1 begincodespacerange\n<0000> <FFFF>\nendcodespacerange\n\
1 begincidrange\n<0041> <0043> 65\nendcidrange\n";
let enc = parse_encoding_cmap_stream(data).unwrap();
assert_eq!(enc.map.get(&0x41), Some(&65));
assert_eq!(enc.map.get(&0x42), Some(&66));
assert_eq!(enc.map.get(&0x43), Some(&67));
assert_eq!(enc.map.len(), 3);
assert_eq!(enc.code_byte_length, 2);
}
#[test]
fn encoding_cidrange_repeated_full_ranges_stay_bounded() {
// 5,000 copies of `<0000> <ffff> 0` must not re-expand the 16-bit
// domain on every declaration.
let mut body = String::new();
let mut remaining = 5_000usize;
while remaining > 0 {
let n = remaining.min(100);
body.push_str(&format!("{n} begincidrange\n"));
for _ in 0..n {
body.push_str("<0000> <ffff> 0\n");
}
body.push_str("endcidrange\n");
remaining -= n;
}
let data = format!("1 begincodespacerange\n<0000> <FFFF>\nendcodespacerange\n{body}");
let enc = parse_encoding_cmap_stream(data.as_bytes()).unwrap();
assert!(enc.map.len() <= MAX_CID_W_EXPANSION);
assert_eq!(enc.map.get(&0), Some(&0));
assert_eq!(enc.map.get(&65535), Some(&65535));
}
}