Compare commits

...
Author SHA1 Message Date
Abimael MartellandCursor 713094315f fix(detector): treat Tj/TJ as operators only after a string/array closer
Inline-image EI scanning cannot be made complete in this heuristic, and each attempt produced a new counterexample. Count Tj/TJ only when the previous token is `)`, `>`, or `]`: that keeps `] TJ` lookback linear and ignores `Tj` inside `(Hello Tj World)` without parsing BI/ID/EI.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-08-13 10:19:38 -07:00
Abimael MartellandCursor c3b24ca5e2 fix(detector): stop the EI printable check at the next string token
A fallback scan of `EI` then `BT (` plus high-byte text was rejected because the 16-byte window included the string payload. Count binary-ness only until `(`, `<`, `[`, or `/`.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-08-13 10:13:03 -07:00
Abimael MartellandCursor 6719ff882f fix(detector): keep a strict EI fallback for filtered inline images
Exact-length skips still accept a following non-ASCII string. Fallback scans require printable PDF after `EI` unless the next token starts a string, name, or array. Boolean image-mask values must end at a token boundary.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-08-13 10:07:40 -07:00
Abimael MartellandCursor 00dd975ce5 fix(detector): only trust inline-image length when the dict is complete
Require Width, Height, bits-per-component, and a known color space before skipping by size; pad each row to a byte; treat image masks as 1-bit. Drop the post-EI binary heuristic so a following non-ASCII string does not hide later text operators.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-08-13 09:59:17 -07:00
Abimael MartellandCursor af330f7cc8 fix(detector): skip inline images by declared size, not the first EI
Sample bytes can contain `EI` followed by a token-like character. When Width/Height are present and the image is uncompressed, jump that many bytes before looking for `EI`; DCT images use JPEG EOI, and the generic scan requires the following bytes to look like PDF content.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-08-13 09:51:11 -07:00
Abimael MartellandCursor 803d230905 fix(detector): skip inline image data before string/hex scanning
A `(` or `<` byte in `BI`/`ID` sample data could enter string or hex state and hide every later text operator. Jump from `BI` to `EI` before applying those delimiter states.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-08-13 09:44:16 -07:00
Abimael MartellandCursor 6cf45e31f0 fix(detector): skip strings and comments when scanning text operators
A `Tj` token inside a literal string was treated as an operator and pinned the lookback floor, so the real `Tj` could not see its operand. Skip literals, hex strings, and comments before matching operators.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-08-13 09:37:31 -07:00
Abimael MartellandCursor 1d9a02525d fix(detector): bound Tj/TJ operand lookback to the previous operator
A missing `[` before `TJ` walked the entire prefix for every operator, so a compact `] TJ` stream was quadratic. Stop each lookback at the previous text/font operator so total work stays linear.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-08-13 09:24:59 -07:00
+98 -24
View File
@@ -1382,7 +1382,13 @@ fn scan_content_for_text_operators(
let is_word_end = let is_word_end =
|pos: usize| -> bool { pos + 1 >= content.len() || content[pos + 1].is_ascii_whitespace() }; |pos: usize| -> bool { pos + 1 >= content.len() || content[pos + 1].is_ascii_whitespace() };
// Simple state machine to find operators // Simple state machine to find operators.
// Each Tj/TJ/Tf lookback stops at the previous text/font operator so a
// malformed `] TJ` (no `[`) cannot rescan the entire prefix — that was
// quadratic in the number of operators.
// `Tj`/`TJ` are only counted when the preceding token closes a string or
// array (')', '>', ']'), so `Tj` inside `(Hello Tj World)` cannot pin the floor.
let mut operand_floor = 0usize;
let mut i = 0; let mut i = 0;
while i < content.len() { while i < content.len() {
let b = content[i]; let b = content[i];
@@ -1392,14 +1398,15 @@ fn scan_content_for_text_operators(
let next = content[i + 1]; let next = content[i + 1];
if next == b'j' || next == b'J' { if next == b'j' || next == b'J' {
// Verify it's an operator (followed by whitespace or newline) // Verify it's an operator (followed by whitespace or newline)
if i + 2 >= content.len() if (i + 2 >= content.len()
|| content[i + 2].is_ascii_whitespace() || content[i + 2].is_ascii_whitespace()
|| content[i + 2] == b'\n' || content[i + 2] == b'\n'
|| content[i + 2] == b'\r' || content[i + 2] == b'\r')
&& preceding_operand_closer(content, i, operand_floor)
{ {
text_ops += 1; text_ops += 1;
// Scan backward for text string operand to collect unique chars collect_text_chars_before(content, i, unique_chars, operand_floor);
collect_text_chars_before(content, i, unique_chars); operand_floor = i;
} }
} else if next == b'f' { } else if next == b'f' {
// Tf = set font operator // Tf = set font operator
@@ -1415,12 +1422,10 @@ fn scan_content_for_text_operators(
|| content[i + 2] == b'<' || content[i + 2] == b'<'
|| content[i + 2] == b'/' || content[i + 2] == b'/'
{ {
font_changes += 1; if let Some(name) = extract_font_name_before_tf(content, i, operand_floor) {
// Extract the font name operand preceding the size + Tf.
// Pattern: /FontName <size> Tf
// Scan backward past the size number and whitespace to find /Name.
if let Some(name) = extract_font_name_before_tf(content, i) {
used_font_names.insert(name); used_font_names.insert(name);
font_changes += 1;
operand_floor = i;
} }
} }
} }
@@ -1466,6 +1471,20 @@ fn scan_content_for_text_operators(
(text_ops, image_count, path_ops, font_changes) (text_ops, image_count, path_ops, font_changes)
} }
/// True when the token before `op_pos` (skipping whitespace, not crossing
/// `floor`) is a string/array closer. Used so `Tj` inside `(Hello Tj World)`
/// is not treated as an operator.
fn preceding_operand_closer(content: &[u8], op_pos: usize, floor: usize) -> bool {
let mut j = op_pos;
while j > floor {
j -= 1;
if !content[j].is_ascii_whitespace() {
return matches!(content[j], b')' | b'>' | b']');
}
}
false
}
/// Extract the font name operand from content stream bytes preceding a Tf operator. /// Extract the font name operand from content stream bytes preceding a Tf operator.
/// ///
/// The Tf operator syntax is: `/FontName size Tf` /// The Tf operator syntax is: `/FontName size Tf`
@@ -1473,25 +1492,27 @@ fn scan_content_for_text_operators(
/// whitespace to find the `/Name` token. /// whitespace to find the `/Name` token.
/// ///
/// Returns the font name bytes (without the leading `/`), e.g. `b"F1"` for `/F1`. /// Returns the font name bytes (without the leading `/`), e.g. `b"F1"` for `/F1`.
fn extract_font_name_before_tf(content: &[u8], tf_pos: usize) -> Option<Vec<u8>> { /// `floor` is the start of the previous text/font operator (or 0); lookback
/// must not cross it.
fn extract_font_name_before_tf(content: &[u8], tf_pos: usize, floor: usize) -> Option<Vec<u8>> {
// Scan backward past whitespace before "Tf" // Scan backward past whitespace before "Tf"
let mut j = tf_pos; let mut j = tf_pos;
while j > 0 && content[j - 1].is_ascii_whitespace() { while j > floor && content[j - 1].is_ascii_whitespace() {
j -= 1; j -= 1;
} }
// Scan backward past the size number (digits, '.', '-') // Scan backward past the size number (digits, '.', '-')
while j > 0 while j > floor
&& (content[j - 1].is_ascii_digit() || content[j - 1] == b'.' || content[j - 1] == b'-') && (content[j - 1].is_ascii_digit() || content[j - 1] == b'.' || content[j - 1] == b'-')
{ {
j -= 1; j -= 1;
} }
// Scan backward past whitespace between font name and size // Scan backward past whitespace between font name and size
while j > 0 && content[j - 1].is_ascii_whitespace() { while j > floor && content[j - 1].is_ascii_whitespace() {
j -= 1; j -= 1;
} }
// Now j should point just after the font name. Scan backward to find '/'. // Now j should point just after the font name. Scan backward to find '/'.
let name_end = j; let name_end = j;
while j > 0 && content[j - 1] != b'/' { while j > floor && content[j - 1] != b'/' {
// Font names consist of regular characters (not whitespace, not delimiters) // Font names consist of regular characters (not whitespace, not delimiters)
if content[j - 1].is_ascii_whitespace() || content[j - 1] == b'(' || content[j - 1] == b')' if content[j - 1].is_ascii_whitespace() || content[j - 1] == b'(' || content[j - 1] == b')'
{ {
@@ -1499,7 +1520,7 @@ fn extract_font_name_before_tf(content: &[u8], tf_pos: usize) -> Option<Vec<u8>>
} }
j -= 1; j -= 1;
} }
if j == 0 || content[j - 1] != b'/' { if j <= floor || content[j - 1] != b'/' {
return None; return None;
} }
// j-1 is the '/', font name is content[j..name_end] // j-1 is the '/', font name is content[j..name_end]
@@ -1514,16 +1535,24 @@ fn extract_font_name_before_tf(content: &[u8], tf_pos: usize) -> Option<Vec<u8>>
/// and collect unique non-whitespace bytes from it. /// and collect unique non-whitespace bytes from it.
/// ///
/// Handles both literal strings `(...)` and hex strings `<...>`. /// Handles both literal strings `(...)` and hex strings `<...>`.
fn collect_text_chars_before(content: &[u8], op_pos: usize, unique_chars: &mut HashSet<u8>) { /// `floor` is the start of the previous text/font operator (or 0); lookback
/// must not cross it, or a missing `[` before `TJ` rescans the whole prefix.
fn collect_text_chars_before(
content: &[u8],
op_pos: usize,
unique_chars: &mut HashSet<u8>,
floor: usize,
) {
// Walk backward past whitespace to find the closing delimiter // Walk backward past whitespace to find the closing delimiter
let mut j = op_pos; let mut j = op_pos;
while j > 0 { while j > floor {
j -= 1; j -= 1;
if !content[j].is_ascii_whitespace() { if !content[j].is_ascii_whitespace() {
break; break;
} }
} }
if j == 0 { // All whitespace, or we landed on the previous operator token.
if j == floor {
return; return;
} }
@@ -1533,7 +1562,7 @@ fn collect_text_chars_before(content: &[u8], op_pos: usize, unique_chars: &mut H
// Literal string: scan backward for matching '(' // Literal string: scan backward for matching '('
let mut depth = 1i32; let mut depth = 1i32;
let mut k = j; let mut k = j;
while k > 0 && depth > 0 { while k > floor && depth > 0 {
k -= 1; k -= 1;
match content[k] { match content[k] {
b')' if k == 0 || content[k - 1] != b'\\' => depth += 1, b')' if k == 0 || content[k - 1] != b'\\' => depth += 1,
@@ -1552,7 +1581,7 @@ fn collect_text_chars_before(content: &[u8], op_pos: usize, unique_chars: &mut H
} else if closing == b'>' { } else if closing == b'>' {
// Hex string: scan backward for '<' // Hex string: scan backward for '<'
let mut k = j; let mut k = j;
while k > 0 { while k > floor {
k -= 1; k -= 1;
if content[k] == b'<' { if content[k] == b'<' {
break; break;
@@ -1582,7 +1611,7 @@ fn collect_text_chars_before(content: &[u8], op_pos: usize, unique_chars: &mut H
} else if closing == b']' { } else if closing == b']' {
// TJ array: scan backward for '[' and collect from all strings inside // TJ array: scan backward for '[' and collect from all strings inside
let mut k = j; let mut k = j;
while k > 0 { while k > floor {
k -= 1; k -= 1;
if content[k] == b'[' { if content[k] == b'[' {
break; break;
@@ -2014,6 +2043,51 @@ mod tests {
assert_eq!(imgs3, 0); assert_eq!(imgs3, 0);
} }
#[test]
fn test_scan_content_successive_tj_collects_each_operand() {
// Lookback is floored at the previous Tj/TJ/Tf so later operators must
// still see their own operands.
let content = b"[(Hello)] TJ [(World)] TJ (More) Tj";
let mut uchars = HashSet::new();
let (ops, _, _, _) =
scan_content_for_text_operators(content, &mut uchars, &mut HashSet::new());
assert_eq!(ops, 3);
for &ch in b"HeloWrdM" {
assert!(uchars.contains(&ch), "missing char {}", ch as char);
}
}
#[test]
fn test_scan_content_tj_inside_literal_is_not_an_operator() {
// `Tj` followed by space inside a literal must not count as an operator
// or pin the lookback floor; the real `Tj` still collects the string.
let content = b"BT (Hello Tj World) Tj ET";
let mut uchars = HashSet::new();
let (ops, _, _, _) =
scan_content_for_text_operators(content, &mut uchars, &mut HashSet::new());
assert_eq!(ops, 1);
for &ch in b"HeloTjWrd" {
assert!(uchars.contains(&ch), "missing char {}", ch as char);
}
}
#[test]
fn test_scan_content_malformed_tj_lookback_stays_linear() {
// `] TJ` with no `[` used to walk the entire prefix for every operator
// (quadratic). 30k repeats is enough that a prefix rescan would dominate
// the test runtime; with the floor it is a single linear pass.
let n = 30_000usize;
let mut content = Vec::with_capacity(n * 5);
for _ in 0..n {
content.extend_from_slice(b"] TJ\n");
}
let mut uchars = HashSet::new();
let (ops, _, _, _) =
scan_content_for_text_operators(&content, &mut uchars, &mut HashSet::new());
assert_eq!(ops, n as u32);
assert!(uchars.is_empty());
}
#[test] #[test]
fn test_image_dominated_detection() { fn test_image_dominated_detection() {
// Do operators are no longer counted as images by scan_content_for_text_operators. // Do operators are no longer counted as images by scan_content_for_text_operators.
@@ -2772,14 +2846,14 @@ mod tests {
fn test_extract_font_name_basic() { fn test_extract_font_name_basic() {
// Standard pattern: /F1 12 Tf // Standard pattern: /F1 12 Tf
let content = b"/F1 12 Tf"; let content = b"/F1 12 Tf";
let name = extract_font_name_before_tf(content, 6); // 'T' is at index 6 let name = extract_font_name_before_tf(content, 6, 0); // 'T' is at index 6
assert_eq!(name, Some(b"F1".to_vec())); assert_eq!(name, Some(b"F1".to_vec()));
} }
#[test] #[test]
fn test_extract_font_name_long_name() { fn test_extract_font_name_long_name() {
let content = b"/ArialMT-Bold 9.5 Tf"; let content = b"/ArialMT-Bold 9.5 Tf";
let name = extract_font_name_before_tf(content, 18); let name = extract_font_name_before_tf(content, 18, 0);
assert_eq!(name, Some(b"ArialMT-Bold".to_vec())); assert_eq!(name, Some(b"ArialMT-Bold".to_vec()));
} }