Compare commits

...
Author SHA1 Message Date
Abimael MartellandCursor 45be5718b9 fix(extractor): document bfrange remap truncation and assert visit count
The 65,536 cap counts overwrites so repeated ranges cannot keep expanding.
The test now checks the assignment count, not just HashMap size (u16 keys
are always ≤ 65,536).

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-08-13 08:47:44 -07:00
Abimael MartellandCursor 6a5328a437 fix(extractor): bound ToUnicode bfrange expansion during subset remap
Repeated full-width beginbfrange entries were expanded into individual
CID inserts on every copy. Stop after 65,536 assignments, matching the
existing /W and Encoding caps.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-08-13 08:39:38 -07:00
Abimael MartellandCursor c8ba909407 fix(extractor): bound Encoding CMap cidrange expansion (#375)
* fix(extractor): bound Encoding CMap cidrange expansion

Repeating full-width begincidrange declarations re-inserted the entire
16-bit domain on every copy. Stop after 65,536 assignments, matching the
existing /W cap.

Co-authored-by: Cursor <cursoragent@cursor.com>

* docs(extractor): clarify Encoding cidrange cap counts insert operations

The bound includes overwrites so repeated full-width ranges cannot keep
working after the map is full. Unique-key coverage alone would re-open
the CPU blow-up.

Co-authored-by: Cursor <cursoragent@cursor.com>

* docs(extractor): distinguish /W insert vs unique-key CID caps

Encoding cidrange and /W width assignment count every insert; the /W
unicode heuristic caps unique CIDs with the same 65,536 bound.

Co-authored-by: Cursor <cursoragent@cursor.com>

---------

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-08-13 08:36:28 -07:00
Abimael MartellandCursor 076183e2e4 fix(extractor): cap content-stream decode before allocating operators (#373)
* fix(extractor): cap content-stream decode before allocating operators

The 1M operation limit ran after lopdf materialized the full vector, so a
compact page of q/Q pairs could still abort under memory pressure.

Co-authored-by: Cursor <cursoragent@cursor.com>

* fix(extractor): treat NUL and form-feed as PDF whitespace in op counting

Names must stop on the full PDF whitespace set so a following operator is
not absorbed into /Name, which would undercount and skip the decode cap.

Co-authored-by: Cursor <cursoragent@cursor.com>

* fix(extractor): scan inline-image EI with the full PDF whitespace set

A missed EI terminator used to consume the rest of the stream and drop
later operators from the decode cap. If EI is absent, keep scanning.

Co-authored-by: Cursor <cursoragent@cursor.com>

---------

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-08-12 16:13:12 -07:00
5 changed files with 505 additions and 38 deletions
+328
View File
@@ -0,0 +1,328 @@
//! Bounded content-stream decoding.
//!
//! `lopdf::content::Content::decode` materializes every operator before any
//! caller can apply a limit. A compact page of `q Q` pairs can therefore
//! allocate hundreds of megabytes and abort. Count operators first (without
//! allocating `Operation` objects) and skip decode when the cap is exceeded.
use crate::PdfError;
use lopdf::content::Content;
/// Maximum content-stream operators decoded for a page or a single Form
/// XObject. Matches the previous post-decode skip threshold.
pub(crate) const MAX_PAGE_OPERATIONS: usize = 1_000_000;
/// Decode `data` unless it contains more than `max_operations` operators.
///
/// Returns `Ok(None)` when the stream exceeds the cap, so callers can skip
/// extraction without first allocating the operation vector.
pub(crate) fn decode_content_bounded(
data: &[u8],
max_operations: usize,
) -> Result<Option<Content>, PdfError> {
if content_exceeds_operation_limit(data, max_operations) {
return Ok(None);
}
Content::decode(data)
.map(Some)
.map_err(|e| PdfError::Parse(e.to_string()))
}
fn content_exceeds_operation_limit(data: &[u8], max_operations: usize) -> bool {
count_content_operators(data, max_operations.saturating_add(1)) > max_operations
}
/// Count operators using the same token rules as lopdf's content parser,
/// stopping at `limit`. Does not allocate `Operation` / `Object` values.
fn count_content_operators(data: &[u8], limit: usize) -> usize {
let mut i = 0;
let mut count = 0;
while i < data.len() && count < limit {
skip_content_space(data, &mut i);
if i >= data.len() {
break;
}
if data[i] == b'%' {
skip_comment(data, &mut i);
continue;
}
match data[i] {
b'(' => i = skip_literal_string(data, i),
b'<' => {
if data.get(i + 1) == Some(&b'<') {
i += 2;
} else {
i = skip_hex_string(data, i);
}
}
b'>' => {
i += 1;
if data.get(i) == Some(&b'>') {
i += 1;
}
}
b'[' | b']' => i += 1,
b'/' => skip_name(data, &mut i),
b'+' | b'-' | b'.' => skip_number(data, &mut i),
b if b.is_ascii_digit() => skip_number(data, &mut i),
b if is_operator_byte(b) => {
let start = i;
i += 1;
while i < data.len() && is_operator_byte(data[i]) {
i += 1;
}
let token = &data[start..i];
if token == b"true" || token == b"false" || token == b"null" {
continue;
}
count += 1;
if token == b"BI" && (i >= data.len() || is_content_space(data[i])) {
i = skip_inline_image_after_bi(data, i);
}
}
_ => i += 1,
}
}
count
}
fn is_content_space(b: u8) -> bool {
// PDF whitespace (ISO 32000): NUL, tab, LF, FF, CR, space. Names must
// stop on these so a following operator is not absorbed into `/Name`.
matches!(b, b'\0' | b'\t' | b'\n' | b'\x0c' | b'\r' | b' ')
}
fn is_operator_byte(b: u8) -> bool {
b.is_ascii_alphabetic() || matches!(b, b'*' | b'\'' | b'"')
}
fn is_delimiter(b: u8) -> bool {
matches!(
b,
b'(' | b')' | b'<' | b'>' | b'[' | b']' | b'{' | b'}' | b'/' | b'%'
)
}
fn skip_content_space(data: &[u8], i: &mut usize) {
while *i < data.len() && is_content_space(data[*i]) {
*i += 1;
}
}
fn skip_comment(data: &[u8], i: &mut usize) {
while *i < data.len() && data[*i] != b'\n' && data[*i] != b'\r' {
*i += 1;
}
}
fn skip_literal_string(data: &[u8], mut i: usize) -> usize {
let mut depth = 1i32;
i += 1;
while i < data.len() && depth > 0 {
match data[i] {
b'\\' => {
i += 1;
if i < data.len() {
i += 1;
}
}
b'(' => {
depth += 1;
i += 1;
}
b')' => {
depth -= 1;
i += 1;
}
_ => i += 1,
}
}
i
}
fn skip_hex_string(data: &[u8], mut i: usize) -> usize {
i += 1;
while i < data.len() && data[i] != b'>' {
i += 1;
}
if i < data.len() {
i += 1;
}
i
}
fn skip_name(data: &[u8], i: &mut usize) {
*i += 1;
while *i < data.len() && !is_content_space(data[*i]) && !is_delimiter(data[*i]) {
*i += 1;
}
}
fn skip_number(data: &[u8], i: &mut usize) {
if *i < data.len() && matches!(data[*i], b'+' | b'-') {
*i += 1;
}
while *i < data.len() && data[*i].is_ascii_digit() {
*i += 1;
}
if *i < data.len() && data[*i] == b'.' {
*i += 1;
while *i < data.len() && data[*i].is_ascii_digit() {
*i += 1;
}
}
}
/// After a `BI` operator, skip inline-image data through `EI`.
/// Uses the same PDF whitespace set as `is_content_space`. If `EI` is not
/// found, leave the cursor in place so later operators are still counted
/// (undercounting would let decode allocate the full vector).
fn skip_inline_image_after_bi(data: &[u8], mut i: usize) -> usize {
skip_content_space(data, &mut i);
let rest = &data[i..];
if let Some(pos) = rest.windows(4).position(|w| {
is_content_space(w[0]) && w[1] == b'E' && w[2] == b'I' && is_content_space(w[3])
}) {
return i + pos + 3;
}
i
}
#[cfg(test)]
mod tests {
use super::*;
fn lopdf_op_count(data: &[u8]) -> usize {
Content::decode(data)
.map(|c| c.operations.len())
.unwrap_or(0)
}
/// DoS safety: never report fewer operators than lopdf would allocate.
/// Overcount is acceptable (skip a page); undercount would re-open decode.
fn assert_count_does_not_undercount(data: &[u8]) {
let ours = count_content_operators(data, usize::MAX);
match Content::decode(data) {
Ok(content) => assert!(
ours >= content.operations.len(),
"undercount: ours={ours} lopdf={} for {:?}",
content.operations.len(),
String::from_utf8_lossy(data)
),
Err(_) => {}
}
}
#[test]
fn operator_count_matches_lopdf_for_typical_streams() {
let samples: &[&[u8]] = &[
b"q 1 0 0 1 0 0 cm BT /F1 12 Tf 72 720 Td (Hello) Tj ET Q",
b"q Q q Q",
b"BT /F1 12 Tf 12 TL 1 0 0 1 100 512 Tm (first) Tj (struck) ' ET",
b"1 0 0 rg 0 0 10 10 re f",
b"true false null q",
b"% comment\nq Q\n",
b"[ (a) 1 (b) ] TJ",
b"1 0 0 1 0 0 cm /Im0 Do",
];
for data in samples {
assert_eq!(
count_content_operators(data, usize::MAX),
lopdf_op_count(data),
"count mismatch for {}",
String::from_utf8_lossy(data)
);
}
}
#[test]
fn strings_and_comments_are_not_operators() {
let data = b"(q Q Tj) Tj % q Q\nET";
assert_eq!(
count_content_operators(data, usize::MAX),
lopdf_op_count(data)
);
assert_eq!(count_content_operators(data, usize::MAX), 2); // Tj, ET
}
#[test]
fn inline_image_counts_as_one_operator() {
let data = b"BI /W 2 /H 2 /CS /RGB /BPC 8 ID \x00\x01\x02\x03 EI q";
assert_eq!(
count_content_operators(data, usize::MAX),
lopdf_op_count(data)
);
assert_eq!(count_content_operators(data, usize::MAX), 2); // BI, q
}
#[test]
fn inline_image_ei_accepts_pdf_whitespace() {
let tab = b"BI /W 1 /H 1 ID \xff\tEI\t q Q";
let nul = b"BI /W 1 /H 1 ID \xff\x00EI\x00 q Q";
let ff = b"BI /W 1 /H 1 ID \xff\x0cEI\x0c q Q";
for data in [tab.as_slice(), nul.as_slice(), ff.as_slice()] {
assert_count_does_not_undercount(data);
assert!(
count_content_operators(data, usize::MAX) >= 3,
"BI plus following q Q must remain visible after EI, got {} for {:?}",
count_content_operators(data, usize::MAX),
String::from_utf8_lossy(data)
);
}
}
#[test]
fn decode_is_skipped_when_operator_cap_is_exceeded() {
let mut data = Vec::new();
for _ in 0..20 {
data.extend_from_slice(b"q Q\n");
}
assert!(decode_content_bounded(&data, 10).unwrap().is_none());
let decoded = decode_content_bounded(&data, 50).unwrap().unwrap();
assert_eq!(decoded.operations.len(), 40);
}
#[test]
fn name_whitespace_does_not_swallow_following_operator() {
// NUL / form-feed end a name (PDF whitespace). Absorbing `q` into
// `/x` would undercount and let decode allocate the operator vector.
let mut nul_sep = Vec::new();
let mut ff_sep = Vec::new();
for _ in 0..8_000 {
nul_sep.extend_from_slice(b"/x\x00q");
ff_sep.extend_from_slice(b"/x\x0cq");
}
assert_count_does_not_undercount(&nul_sep);
assert_count_does_not_undercount(&ff_sep);
assert!(count_content_operators(&ff_sep, usize::MAX) >= 8_000);
}
#[test]
fn edge_streams_do_not_undercount_vs_lopdf() {
let samples: &[&[u8]] = &[
b".5 0 0 .5 0 0 cm",
b"+1 -2 3.0 rg",
b"<0041> Tj",
b"(unbalanced",
b"BI /W 1 /H 1 ID \xff\xff no EI here q Q q Q",
b"q\x00Q\x00q\x00Q",
b"/F1\x0c12 Tf (Hi) Tj",
b"{ 1 2 add } cvx",
];
for data in samples {
assert_count_does_not_undercount(data);
}
}
#[test]
fn million_q_pairs_are_rejected_without_decode() {
let mut data = Vec::with_capacity((MAX_PAGE_OPERATIONS + 1) * 2);
for _ in 0..=MAX_PAGE_OPERATIONS {
data.extend_from_slice(b"q\n");
}
assert!(content_exceeds_operation_limit(&data, MAX_PAGE_OPERATIONS));
assert!(decode_content_bounded(&data, MAX_PAGE_OPERATIONS)
.unwrap()
.is_none());
}
}
+29 -14
View File
@@ -151,8 +151,6 @@ pub(crate) fn extract_page_text_items(
style_cache: &mut FontStyleCache,
form_budget: &mut FormWalkBudget,
) -> Result<(PageExtraction, bool, bool, bool), PdfError> {
use lopdf::content::Content;
let mut items = Vec::new();
let mut rects: Vec<PdfRect> = Vec::new();
let mut clip_rects: Vec<PdfRect> = Vec::new();
@@ -256,18 +254,20 @@ pub(crate) fn extract_page_text_items(
// Content::decode parser, causing it to skip operators like ET and Q.
let content_data = strip_pdf_comments(&content_data);
let content = Content::decode(&content_data).map_err(|e| PdfError::Parse(e.to_string()))?;
const MAX_OPERATIONS: usize = 1_000_000;
if content.operations.len() > MAX_OPERATIONS {
log::warn!(
"page {}: skipping extraction — {} operations exceeds limit ({})",
page_num,
content.operations.len(),
MAX_OPERATIONS
);
return Ok(((Vec::new(), Vec::new(), Vec::new()), false, false, false));
}
let content = match super::content_decode::decode_content_bounded(
&content_data,
super::content_decode::MAX_PAGE_OPERATIONS,
)? {
Some(content) => content,
None => {
log::warn!(
"page {}: skipping extraction — content stream exceeds {} operations",
page_num,
super::content_decode::MAX_PAGE_OPERATIONS
);
return Ok(((Vec::new(), Vec::new(), Vec::new()), false, false, false));
}
};
// Graphics state tracking
let mut ctm = [1.0f32, 0.0, 0.0, 1.0, 0.0, 0.0]; // Current Transformation Matrix
@@ -1937,4 +1937,19 @@ BT 30 700 Tm <41> Tj ET";
let output = strip_pdf_comments(input);
assert_eq!(output, b"(x\\\\) Tj \nET\n");
}
#[test]
fn oversized_content_stream_skips_extraction() {
let mut content =
Vec::with_capacity((super::super::content_decode::MAX_PAGE_OPERATIONS + 1) * 2);
for _ in 0..=super::super::content_decode::MAX_PAGE_OPERATIONS {
content.extend_from_slice(b"q\n");
}
content.extend_from_slice(b"BT /F1 12 Tf 72 720 Td (Hello) Tj ET\n");
let items = extract_simple_items(&content);
assert!(
items.is_empty(),
"pages over the operator cap must not be decoded"
);
}
}
+1
View File
@@ -3,6 +3,7 @@
//! This module extracts text with position information for structure detection.
mod base14;
mod content_decode;
pub(crate) mod content_stream;
mod fonts;
mod layout;
+6 -4
View File
@@ -215,8 +215,6 @@ fn extract_form_xobject_text_inner(
depth: u8,
budget: &mut FormWalkBudget,
) -> Vec<TextItem> {
use lopdf::content::Content;
let mut items = Vec::new();
if !budget.charge_invocation() {
@@ -234,8 +232,12 @@ fn extract_form_xobject_text_inner(
Err(_) => stream.content.clone(),
};
// Decode the content stream
let Ok(content) = Content::decode(&content_data) else {
// Decode the content stream. Cap before lopdf materializes the operator
// vector — the walk budget cannot help if decode itself allocates first.
let Ok(Some(content)) = super::content_decode::decode_content_bounded(
&content_data,
super::content_decode::MAX_PAGE_OPERATIONS,
) else {
return items;
};
+141 -20
View File
@@ -540,18 +540,14 @@ impl ToUnicodeCMap {
/// Remap a CMap that references pre-subsetting GIDs to sequential post-subsetting GIDs.
/// Collects all source CIDs, sorts them, and reassigns to 1, 2, 3, ...
///
/// Range expansion stops after `MAX_CID_W_EXPANSION` CID visits, counting
/// overwrites, so repeated full-width `bfrange`s cannot re-expand the
/// 16-bit domain. Later overlapping ranges that would have introduced new
/// CIDs after that many visits are truncated.
pub fn remap_to_sequential(&self) -> ToUnicodeCMap {
let mut cid_to_unicode: HashMap<u16, String> = HashMap::new();
// Expand ranges first
for &(start, end, base) in &self.ranges {
for cid in start..=end {
let unicode_cp = base + (cid - start) as u32;
if let Some(ch) = char::from_u32(unicode_cp) {
cid_to_unicode.insert(cid, ch.to_string());
}
}
}
expand_bfranges_for_remap(&self.ranges, &mut cid_to_unicode, MAX_CID_W_EXPANSION);
// char_map entries override range entries
for (&cid, unicode) in &self.char_map {
@@ -576,6 +572,33 @@ impl ToUnicodeCMap {
}
}
/// Expand `bfrange` entries into individual CID→Unicode inserts.
/// Returns how many CIDs were visited. Counts overwrites so a repeated
/// full-width range cannot keep working after `max_assignments`.
fn expand_bfranges_for_remap(
ranges: &[(u16, u16, u32)],
cid_to_unicode: &mut HashMap<u16, String>,
max_assignments: usize,
) -> usize {
let mut assigned = 0usize;
'ranges: for &(start, end, base) in ranges {
if start > end {
continue;
}
for cid in start..=end {
if assigned >= max_assignments {
break 'ranges;
}
assigned += 1;
let unicode_cp = base + (cid - start) as u32;
if let Some(ch) = char::from_u32(unicode_cp) {
cid_to_unicode.insert(cid, ch.to_string());
}
}
}
assigned
}
/// Parse a hex string to u16
fn parse_hex_u16(hex: &str) -> Option<u16> {
u16::from_str_radix(hex.trim(), 16).ok()
@@ -1561,23 +1584,31 @@ fn parse_encoding_cmap_stream(data: &[u8]) -> Option<EncodingCMap> {
}
let mut map = HashMap::new();
let mut assigned = 0usize;
let mut pos = 0;
while let Some(start) = text[pos..].find("begincidchar") {
let section_start = pos + start + "begincidchar".len();
if let Some(end) = text[section_start..].find("endcidchar") {
let section = &text[section_start..section_start + end];
parse_cidchar_section(section, &mut map, &mut src_hex_lengths);
if !parse_cidchar_section(section, &mut map, &mut src_hex_lengths, &mut assigned) {
break;
}
pos = section_start + end;
} else {
break;
}
}
pos = 0;
while let Some(start) = text[pos..].find("begincidrange") {
while assigned < MAX_CID_W_EXPANSION {
let Some(start) = text[pos..].find("begincidrange") else {
break;
};
let section_start = pos + start + "begincidrange".len();
if let Some(end) = text[section_start..].find("endcidrange") {
let section = &text[section_start..section_start + end];
parse_cidrange_section(section, &mut map, &mut src_hex_lengths);
if !parse_cidrange_section(section, &mut map, &mut src_hex_lengths, &mut assigned) {
break;
}
pos = section_start + end;
} else {
break;
@@ -1612,7 +1643,8 @@ fn parse_cidchar_section(
section: &str,
map: &mut HashMap<u16, u16>,
src_hex_lengths: &mut Vec<usize>,
) {
assigned: &mut usize,
) -> bool {
let mut chars = section.chars().peekable();
loop {
while chars.peek().is_some_and(|c| c.is_whitespace()) {
@@ -1643,16 +1675,20 @@ fn parse_cidchar_section(
}
}
if let (Some(code), Ok(cid)) = (parse_hex_u16(&src_hex), cid_str.parse::<u16>()) {
map.insert(code, cid);
if !assign_encoding_cid(map, code, cid, assigned) {
return false;
}
}
}
true
}
fn parse_cidrange_section(
section: &str,
map: &mut HashMap<u16, u16>,
src_hex_lengths: &mut Vec<usize>,
) {
assigned: &mut usize,
) -> bool {
let mut chars = section.chars().peekable();
loop {
while chars.peek().is_some_and(|c| c.is_whitespace()) {
@@ -1703,12 +1739,34 @@ fn parse_cidrange_section(
) else {
continue;
};
if start > end {
continue;
}
let mut cid = start_cid;
for code in start..=end {
map.insert(code, cid);
if !assign_encoding_cid(map, code, cid, assigned) {
return false;
}
cid = cid.saturating_add(1);
}
}
true
}
fn assign_encoding_cid(
map: &mut HashMap<u16, u16>,
code: u16,
cid: u16,
assigned: &mut usize,
) -> bool {
// Count overwrites: unique-key coverage alone would not stop a repeated
// full-width range from re-inserting all 65,536 codes.
if *assigned >= MAX_CID_W_EXPANSION {
return false;
}
map.insert(code, cid);
*assigned += 1;
true
}
fn parse_binary_cmap_encoding(data: &[u8]) -> Result<EncodingCMap, String> {
@@ -1832,9 +1890,11 @@ fn merge_cmaps(mut base: ToUnicodeCMap, overlay: ToUnicodeCMap) -> ToUnicodeCMap
base
}
/// Upper bound on CID `/W` range expansion. The CID domain is 16-bit, so more
/// than 65,536 unique keys cannot exist; repeating full-width ranges must not
/// re-expand the same domain.
/// Shared 16-bit CID expansion cap (65,536).
/// Encoding `begincidrange`, `/W` width assignment, and ToUnicode sequential
/// remap count every insert, including overwrites, so a repeated full-width
/// range cannot keep working after the domain is filled. The `/W` unicode
/// heuristic caps unique CIDs with the same number.
pub(crate) const MAX_CID_W_EXPANSION: usize = 65_536;
/// Check if a CIDFont's /W (widths) array contains CID values that look like
@@ -2887,6 +2947,33 @@ endbfrange
assert!(remapped.ranges.is_empty());
}
#[test]
fn remap_to_sequential_repeated_full_bfranges_stay_bounded() {
// 5,000 copies of `<0003> <ffff>` must stop after 65,536 CID visits,
// not 5,000 × ~65,533 expansions.
let ranges = vec![(3u16, 65535u16, 0x41u32); 5_000];
let mut map = std::collections::HashMap::new();
let assigned = expand_bfranges_for_remap(&ranges, &mut map, MAX_CID_W_EXPANSION);
assert_eq!(assigned, MAX_CID_W_EXPANSION);
assert!(map.len() <= MAX_CID_W_EXPANSION);
let mut body = String::new();
let mut remaining = 5_000usize;
while remaining > 0 {
let n = remaining.min(100);
body.push_str(&format!("{n} beginbfrange\n"));
for _ in 0..n {
body.push_str("<0003> <ffff> <0041>\n");
}
body.push_str("endbfrange\n");
remaining -= n;
}
let data = format!("1 begincodespacerange\n<0000> <ffff>\nendcodespacerange\n{body}");
let cmap = ToUnicodeCMap::parse(data.as_bytes()).unwrap();
let remapped = cmap.remap_to_sequential();
assert_eq!(remapped.lookup(1), Some("A".to_string()));
}
#[test]
fn test_min_source_cid() {
let cmap_content = r#"
@@ -3359,4 +3446,38 @@ endbfrange
dict.set("W", Object::Array(w));
assert!(cid_values_look_like_unicode(&dict));
}
#[test]
fn encoding_cidrange_maps_a_normal_range() {
let data = b"1 begincodespacerange\n<0000> <FFFF>\nendcodespacerange\n\
1 begincidrange\n<0041> <0043> 65\nendcidrange\n";
let enc = parse_encoding_cmap_stream(data).unwrap();
assert_eq!(enc.map.get(&0x41), Some(&65));
assert_eq!(enc.map.get(&0x42), Some(&66));
assert_eq!(enc.map.get(&0x43), Some(&67));
assert_eq!(enc.map.len(), 3);
assert_eq!(enc.code_byte_length, 2);
}
#[test]
fn encoding_cidrange_repeated_full_ranges_stay_bounded() {
// 5,000 copies of `<0000> <ffff> 0` must not re-expand the 16-bit
// domain on every declaration.
let mut body = String::new();
let mut remaining = 5_000usize;
while remaining > 0 {
let n = remaining.min(100);
body.push_str(&format!("{n} begincidrange\n"));
for _ in 0..n {
body.push_str("<0000> <ffff> 0\n");
}
body.push_str("endcidrange\n");
remaining -= n;
}
let data = format!("1 begincodespacerange\n<0000> <FFFF>\nendcodespacerange\n{body}");
let enc = parse_encoding_cmap_stream(data.as_bytes()).unwrap();
assert!(enc.map.len() <= MAX_CID_W_EXPANSION);
assert_eq!(enc.map.get(&0), Some(&0));
assert_eq!(enc.map.get(&65535), Some(&65535));
}
}