perf(extractor): Cache font encodings to avoid re-parsing CMap per text op
lopdf's get_font_encoding() re-parses ToUnicode CMap streams via nom for every Tj/TJ operator. Cache the encoding once per font during page setup. Also add binary search for CMap range lookup and de-dup CMap extraction. Reduces Arabic PDF processing from 45s to <1s (53x speedup). Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.6
parent
fb6d98fde4
commit
b929bbd92e
+29
-18
@@ -5,7 +5,7 @@
|
|||||||
use crate::glyph_names::glyph_to_char;
|
use crate::glyph_names::glyph_to_char;
|
||||||
use crate::tounicode::FontCMaps;
|
use crate::tounicode::FontCMaps;
|
||||||
use crate::PdfError;
|
use crate::PdfError;
|
||||||
use lopdf::{Document, Object, ObjectId};
|
use lopdf::{Document, Encoding, Object, ObjectId};
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
use std::path::Path;
|
use std::path::Path;
|
||||||
|
|
||||||
@@ -961,6 +961,16 @@ fn extract_page_text_items(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Cache font encodings from lopdf (once per font, not per text operand).
|
||||||
|
// This avoids re-parsing ToUnicode CMap streams for every Tj/TJ operator.
|
||||||
|
let mut encoding_cache: HashMap<String, Encoding<'_>> = HashMap::new();
|
||||||
|
for (font_name, font_dict) in &fonts {
|
||||||
|
let name = String::from_utf8_lossy(font_name).to_string();
|
||||||
|
if let Ok(enc) = font_dict.get_font_encoding(doc) {
|
||||||
|
encoding_cache.insert(name, enc);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// Get XObjects (images) from page resources
|
// Get XObjects (images) from page resources
|
||||||
let xobjects = get_page_xobjects(doc, page_id);
|
let xobjects = get_page_xobjects(doc, page_id);
|
||||||
|
|
||||||
@@ -1120,13 +1130,12 @@ fn extract_page_text_items(
|
|||||||
}
|
}
|
||||||
if let Some(text) = extract_text_from_operand(
|
if let Some(text) = extract_text_from_operand(
|
||||||
&op.operands[0],
|
&op.operands[0],
|
||||||
doc,
|
|
||||||
&fonts,
|
|
||||||
¤t_font,
|
¤t_font,
|
||||||
font_cmaps,
|
font_cmaps,
|
||||||
&font_base_names,
|
&font_base_names,
|
||||||
&font_tounicode_refs,
|
&font_tounicode_refs,
|
||||||
&font_encodings,
|
&font_encodings,
|
||||||
|
&encoding_cache,
|
||||||
) {
|
) {
|
||||||
if !text.trim().is_empty() {
|
if !text.trim().is_empty() {
|
||||||
let rendered_size =
|
let rendered_size =
|
||||||
@@ -1224,13 +1233,12 @@ fn extract_page_text_items(
|
|||||||
if !fill_is_white {
|
if !fill_is_white {
|
||||||
if let Some(text) = extract_text_from_operand(
|
if let Some(text) = extract_text_from_operand(
|
||||||
element,
|
element,
|
||||||
doc,
|
|
||||||
&fonts,
|
|
||||||
¤t_font,
|
¤t_font,
|
||||||
font_cmaps,
|
font_cmaps,
|
||||||
&font_base_names,
|
&font_base_names,
|
||||||
&font_tounicode_refs,
|
&font_tounicode_refs,
|
||||||
&font_encodings,
|
&font_encodings,
|
||||||
|
&encoding_cache,
|
||||||
) {
|
) {
|
||||||
combined_text.push_str(&text);
|
combined_text.push_str(&text);
|
||||||
}
|
}
|
||||||
@@ -1287,13 +1295,12 @@ fn extract_page_text_items(
|
|||||||
if !fill_is_white && !op.operands.is_empty() {
|
if !fill_is_white && !op.operands.is_empty() {
|
||||||
if let Some(text) = extract_text_from_operand(
|
if let Some(text) = extract_text_from_operand(
|
||||||
&op.operands[0],
|
&op.operands[0],
|
||||||
doc,
|
|
||||||
&fonts,
|
|
||||||
¤t_font,
|
¤t_font,
|
||||||
font_cmaps,
|
font_cmaps,
|
||||||
&font_base_names,
|
&font_base_names,
|
||||||
&font_tounicode_refs,
|
&font_tounicode_refs,
|
||||||
&font_encodings,
|
&font_encodings,
|
||||||
|
&encoding_cache,
|
||||||
) {
|
) {
|
||||||
if !text.trim().is_empty() {
|
if !text.trim().is_empty() {
|
||||||
let rendered_size =
|
let rendered_size =
|
||||||
@@ -1498,6 +1505,15 @@ fn extract_form_xobject_text(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Cache font encodings for form fonts
|
||||||
|
let mut encoding_cache: HashMap<String, Encoding<'_>> = HashMap::new();
|
||||||
|
for (font_name, font_dict) in &form_fonts {
|
||||||
|
let name = String::from_utf8_lossy(font_name).to_string();
|
||||||
|
if let Ok(enc) = font_dict.get_font_encoding(doc) {
|
||||||
|
encoding_cache.insert(name, enc);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// Process the content stream
|
// Process the content stream
|
||||||
let mut current_font = String::new();
|
let mut current_font = String::new();
|
||||||
let mut current_font_size: f32 = 12.0;
|
let mut current_font_size: f32 = 12.0;
|
||||||
@@ -1578,13 +1594,12 @@ fn extract_form_xobject_text(
|
|||||||
}
|
}
|
||||||
if let Some(text) = extract_text_from_operand(
|
if let Some(text) = extract_text_from_operand(
|
||||||
&op.operands[0],
|
&op.operands[0],
|
||||||
doc,
|
|
||||||
&form_fonts,
|
|
||||||
¤t_font,
|
¤t_font,
|
||||||
font_cmaps,
|
font_cmaps,
|
||||||
&font_base_names,
|
&font_base_names,
|
||||||
&font_tounicode_refs,
|
&font_tounicode_refs,
|
||||||
&font_encodings,
|
&font_encodings,
|
||||||
|
&encoding_cache,
|
||||||
) {
|
) {
|
||||||
if !text.trim().is_empty() {
|
if !text.trim().is_empty() {
|
||||||
let rendered_size =
|
let rendered_size =
|
||||||
@@ -1682,13 +1697,12 @@ fn extract_form_xobject_text(
|
|||||||
if !fill_is_white {
|
if !fill_is_white {
|
||||||
if let Some(text) = extract_text_from_operand(
|
if let Some(text) = extract_text_from_operand(
|
||||||
element,
|
element,
|
||||||
doc,
|
|
||||||
&form_fonts,
|
|
||||||
¤t_font,
|
¤t_font,
|
||||||
font_cmaps,
|
font_cmaps,
|
||||||
&font_base_names,
|
&font_base_names,
|
||||||
&font_tounicode_refs,
|
&font_tounicode_refs,
|
||||||
&font_encodings,
|
&font_encodings,
|
||||||
|
&encoding_cache,
|
||||||
) {
|
) {
|
||||||
combined_text.push_str(&text);
|
combined_text.push_str(&text);
|
||||||
}
|
}
|
||||||
@@ -1965,13 +1979,12 @@ pub fn is_italic_font(font_name: &str) -> bool {
|
|||||||
#[allow(clippy::too_many_arguments)]
|
#[allow(clippy::too_many_arguments)]
|
||||||
fn extract_text_from_operand(
|
fn extract_text_from_operand(
|
||||||
obj: &Object,
|
obj: &Object,
|
||||||
doc: &Document,
|
|
||||||
fonts: &std::collections::BTreeMap<Vec<u8>, &lopdf::Dictionary>,
|
|
||||||
current_font: &str,
|
current_font: &str,
|
||||||
font_cmaps: &FontCMaps,
|
font_cmaps: &FontCMaps,
|
||||||
font_base_names: &std::collections::HashMap<String, String>,
|
font_base_names: &std::collections::HashMap<String, String>,
|
||||||
font_tounicode_refs: &std::collections::HashMap<String, u32>,
|
font_tounicode_refs: &std::collections::HashMap<String, u32>,
|
||||||
font_encodings: &PageFontEncodings,
|
font_encodings: &PageFontEncodings,
|
||||||
|
encoding_cache: &HashMap<String, Encoding<'_>>,
|
||||||
) -> Option<String> {
|
) -> Option<String> {
|
||||||
if let Object::String(bytes, _) = obj {
|
if let Object::String(bytes, _) = obj {
|
||||||
// First, try to look up CMap by ToUnicode object reference (most reliable)
|
// First, try to look up CMap by ToUnicode object reference (most reliable)
|
||||||
@@ -2043,14 +2056,12 @@ fn extract_text_from_operand(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Try to decode using font encoding from lopdf
|
// Try to decode using cached font encoding from lopdf
|
||||||
if let Some(font_dict) = fonts.get(current_font.as_bytes()) {
|
if let Some(encoding) = encoding_cache.get(current_font) {
|
||||||
if let Ok(encoding) = font_dict.get_font_encoding(doc) {
|
if let Ok(text) = Document::decode_text(encoding, bytes) {
|
||||||
if let Ok(text) = Document::decode_text(&encoding, bytes) {
|
|
||||||
return Some(text);
|
return Some(text);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
|
||||||
|
|
||||||
// Fallback: try UTF-16BE then Latin-1
|
// Fallback: try UTF-16BE then Latin-1
|
||||||
if bytes.len() >= 2 && bytes[0] == 0xFE && bytes[1] == 0xFF {
|
if bytes.len() >= 2 && bytes[0] == 0xFE && bytes[1] == 0xFF {
|
||||||
|
|||||||
+27
-6
@@ -108,6 +108,9 @@ impl ToUnicodeCMap {
|
|||||||
2 // Default to 2-byte
|
2 // Default to 2-byte
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// Sort ranges by start CID for binary search in lookup()
|
||||||
|
cmap.ranges.sort_unstable_by_key(|&(start, _, _)| start);
|
||||||
|
|
||||||
Some(cmap)
|
Some(cmap)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -308,11 +311,28 @@ impl ToUnicodeCMap {
|
|||||||
return Some(s.clone());
|
return Some(s.clone());
|
||||||
}
|
}
|
||||||
|
|
||||||
// Then check ranges
|
// Binary search through sorted ranges
|
||||||
for &(start, end, base) in &self.ranges {
|
let idx = self
|
||||||
|
.ranges
|
||||||
|
.binary_search_by(|&(start, _, _)| start.cmp(&cid))
|
||||||
|
.unwrap_or_else(|i| i);
|
||||||
|
|
||||||
|
// Check the range at idx (where start == cid)
|
||||||
|
if idx < self.ranges.len() {
|
||||||
|
let (start, end, base) = self.ranges[idx];
|
||||||
if cid >= start && cid <= end {
|
if cid >= start && cid <= end {
|
||||||
let offset = (cid - start) as u32;
|
let unicode = base + (cid - start) as u32;
|
||||||
let unicode = base + offset;
|
if let Some(c) = char::from_u32(unicode) {
|
||||||
|
return Some(c.to_string());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Check the range before idx (cid may fall within a range that starts before it)
|
||||||
|
if idx > 0 {
|
||||||
|
let (start, end, base) = self.ranges[idx - 1];
|
||||||
|
if cid >= start && cid <= end {
|
||||||
|
let unicode = base + (cid - start) as u32;
|
||||||
if let Some(c) = char::from_u32(unicode) {
|
if let Some(c) = char::from_u32(unicode) {
|
||||||
return Some(c.to_string());
|
return Some(c.to_string());
|
||||||
}
|
}
|
||||||
@@ -499,10 +519,11 @@ pub fn extract_tounicode_cmaps(pdf_bytes: &[u8]) -> HashMap<u32, ToUnicodeCMap>
|
|||||||
}
|
}
|
||||||
|
|
||||||
if let Ok(obj_num) = num_str.parse::<u32>() {
|
if let Ok(obj_num) = num_str.parse::<u32>() {
|
||||||
// Try to extract the stream for this object
|
if let std::collections::hash_map::Entry::Vacant(e) = cmaps.entry(obj_num) {
|
||||||
if let Some(stream_data) = extract_stream_from_raw_pdf(pdf_bytes, obj_num) {
|
if let Some(stream_data) = extract_stream_from_raw_pdf(pdf_bytes, obj_num) {
|
||||||
if let Some(cmap) = ToUnicodeCMap::parse(&stream_data) {
|
if let Some(cmap) = ToUnicodeCMap::parse(&stream_data) {
|
||||||
cmaps.insert(obj_num, cmap);
|
e.insert(cmap);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user