diff --git a/src/bin/debug_spacing.rs b/src/bin/debug_spacing.rs deleted file mode 100644 index 4d95d1e..0000000 --- a/src/bin/debug_spacing.rs +++ /dev/null @@ -1,32 +0,0 @@ -use pdf_inspector::extract_text_with_positions; -use std::env; - -fn main() { - let path = env::args().nth(1).expect("Need PDF path"); - let items = extract_text_with_positions(&path).expect("Failed"); - - // Look at consecutive items on same Y line - let mut prev_item: Option<&pdf_inspector::TextItem> = None; - for item in items.iter() { - if let Some(prev) = prev_item { - // Same line (similar Y) - if (item.y - prev.y).abs() < 5.0 && item.x > prev.x { - let gap = item.x - prev.x - prev.width; - let char_width = if prev.width > 0.0 && !prev.text.is_empty() { - prev.width / prev.text.len() as f32 - } else { - prev.font_size * 0.5 // Approximate - }; - - println!( - "Gap: {:.1} (charW: {:.1}) | '{}' -> '{}'", - gap, - char_width, - prev.text.chars().take(20).collect::(), - item.text.chars().take(20).collect::() - ); - } - } - prev_item = Some(item); - } -} diff --git a/src/extractor.rs b/src/extractor.rs index 6a39da2..ef79412 100644 --- a/src/extractor.rs +++ b/src/extractor.rs @@ -2,11 +2,109 @@ //! //! This module extracts text with position information for structure detection. +use crate::glyph_names::glyph_to_char; use crate::tounicode::FontCMaps; use crate::PdfError; use lopdf::{Document, Object, ObjectId}; +use std::collections::HashMap; use std::path::Path; +/// Font encoding map: maps byte codes to Unicode characters +type FontEncodingMap = HashMap; + +/// All font encodings for a page +type PageFontEncodings = HashMap; + +/// Build encoding maps for all fonts on a page +fn build_font_encodings( + doc: &Document, + fonts: &std::collections::BTreeMap, &lopdf::Dictionary>, +) -> PageFontEncodings { + let mut encodings = PageFontEncodings::new(); + + for (font_name, font_dict) in fonts { + let resource_name = String::from_utf8_lossy(font_name).to_string(); + + if let Some(encoding_map) = parse_font_encoding(doc, font_dict) { + encodings.insert(resource_name, encoding_map); + } + } + + encodings +} + +/// Parse font encoding from a font dictionary +fn parse_font_encoding(doc: &Document, font_dict: &lopdf::Dictionary) -> Option { + let encoding_obj = font_dict.get(b"Encoding").ok()?; + + // Encoding can be a name or a dictionary + match encoding_obj { + Object::Name(name) => { + // Standard encoding name (e.g., MacRomanEncoding, WinAnsiEncoding) + // For standard encodings, we can use the standard tables + // But we still need to check for Differences + None // Let lopdf handle standard encodings + } + Object::Reference(obj_ref) => { + // Reference to encoding dictionary + if let Ok(enc_dict) = doc.get_dictionary(*obj_ref) { + parse_encoding_dictionary(doc, enc_dict) + } else { + None + } + } + Object::Dictionary(enc_dict) => parse_encoding_dictionary(doc, enc_dict), + _ => None, + } +} + +/// Parse an encoding dictionary with Differences array +fn parse_encoding_dictionary( + doc: &Document, + enc_dict: &lopdf::Dictionary, +) -> Option { + let differences = enc_dict.get(b"Differences").ok()?; + + let diff_array = match differences { + Object::Array(arr) => arr.clone(), + Object::Reference(obj_ref) => { + if let Ok(Object::Array(arr)) = doc.get_object(*obj_ref) { + arr.clone() + } else { + return None; + } + } + _ => return None, + }; + + let mut encoding_map = FontEncodingMap::new(); + let mut current_code: u8 = 0; + + for item in diff_array { + match item { + Object::Integer(n) => { + // This sets the starting code for subsequent glyph names + current_code = n as u8; + } + Object::Name(name) => { + // Map current code to glyph name -> Unicode + let glyph_name = String::from_utf8_lossy(&name).to_string(); + if let Some(ch) = glyph_to_char(&glyph_name) { + encoding_map.insert(current_code, ch); + } + current_code = current_code.wrapping_add(1); + } + _ => {} + } + } + + if encoding_map.is_empty() { + None + } else { + Some(encoding_map) + } +} + /// Type of content item #[derive(Debug, Clone, PartialEq, Default)] pub enum ItemType { @@ -350,6 +448,9 @@ fn extract_page_text_items( // Get fonts for encoding let fonts = doc.get_page_fonts(page_id).unwrap_or_default(); + // Build font encoding maps from Differences arrays + let font_encodings = build_font_encodings(doc, &fonts); + // Build maps of font resource names to their base font names and ToUnicode object refs let mut font_base_names: std::collections::HashMap = std::collections::HashMap::new(); @@ -477,6 +578,7 @@ fn extract_page_text_items( font_cmaps, &font_base_names, &font_tounicode_refs, + &font_encodings, ) { if !text.trim().is_empty() { let rendered_size = @@ -512,6 +614,27 @@ fn extract_page_text_items( if let Ok(array) = op.operands[0].as_array() { let mut combined_text = String::new(); for item in array { + // Check for spacing values - large negative values indicate word spaces + // In PDF, TJ arrays contain: strings and positioning adjustments + // Negative values move right (create space), positive move left (tighten) + // Values are in thousandths of an em unit + match item { + Object::Integer(n) => { + // Threshold: -200 or more negative typically indicates a word space + // (roughly 1/5 of an em or more) + if *n < -200 && !combined_text.ends_with(' ') { + combined_text.push(' '); + } + continue; + } + Object::Real(n) => { + if *n < -200.0 && !combined_text.ends_with(' ') { + combined_text.push(' '); + } + continue; + } + _ => {} + } if let Some(text) = extract_text_from_operand( item, doc, @@ -520,6 +643,7 @@ fn extract_page_text_items( font_cmaps, &font_base_names, &font_tounicode_refs, + &font_encodings, ) { combined_text.push_str(&text); } @@ -565,6 +689,7 @@ fn extract_page_text_items( font_cmaps, &font_base_names, &font_tounicode_refs, + &font_encodings, ) { if !text.trim().is_empty() { let rendered_size = @@ -595,31 +720,42 @@ fn extract_page_text_items( } } "Do" => { - // XObject invocation - could be an image + // XObject invocation - could be an image or form if !op.operands.is_empty() { if let Ok(name) = op.operands[0].as_name() { let xobj_name = String::from_utf8_lossy(name).to_string(); - // Check if this XObject is an image - if xobjects.contains(&xobj_name) { - // Get position from CTM - let (x, y) = (ctm[4], ctm[5]); - // Get dimensions from CTM scale factors - let width = ctm[0].abs(); - let height = ctm[3].abs(); - items.push(TextItem { - text: format!("[Image: {}]", xobj_name), - x, - y, - width, - height, - font: String::new(), - font_size: 0.0, - page: page_num, - is_bold: false, - is_italic: false, - item_type: ItemType::Image, - }); + if let Some(xobj_type) = xobjects.get(&xobj_name) { + match xobj_type { + XObjectType::Image => { + // Get position from CTM + let (x, y) = (ctm[4], ctm[5]); + // Get dimensions from CTM scale factors + let width = ctm[0].abs(); + let height = ctm[3].abs(); + + items.push(TextItem { + text: format!("[Image: {}]", xobj_name), + x, + y, + width, + height, + font: String::new(), + font_size: 0.0, + page: page_num, + is_bold: false, + is_italic: false, + item_type: ItemType::Image, + }); + } + XObjectType::Form(form_id) => { + // Extract text from Form XObject + let form_items = extract_form_xobject_text( + doc, *form_id, page_num, font_cmaps, &ctm, + ); + items.extend(form_items); + } + } } } } @@ -641,8 +777,19 @@ fn get_number(obj: &Object) -> Option { } /// Get XObject names that are images from page resources -fn get_page_xobjects(doc: &Document, page_id: ObjectId) -> std::collections::HashSet { - let mut image_names = std::collections::HashSet::new(); +/// XObject info - either Image or Form +#[derive(Debug)] +enum XObjectType { + Image, + Form(ObjectId), +} + +/// Get XObjects from page resources, categorized by type +fn get_page_xobjects( + doc: &Document, + page_id: ObjectId, +) -> std::collections::HashMap { + let mut xobject_types = std::collections::HashMap::new(); // Try to get the page dictionary if let Ok(page_dict) = doc.get_dictionary(page_id) { @@ -670,14 +817,16 @@ fn get_page_xobjects(doc: &Document, page_id: ObjectId) -> std::collections::Has for (name, value) in xobjects.iter() { let name_str = String::from_utf8_lossy(name).to_string(); - // Check if this XObject is an Image - // XObjects are typically Stream objects, not Dictionary + // Check XObject subtype if let Ok(obj_ref) = value.as_reference() { if let Ok(Object::Stream(stream)) = doc.get_object(obj_ref) { if let Ok(subtype) = stream.dict.get(b"Subtype") { if let Ok(subtype_name) = subtype.as_name() { if subtype_name == b"Image" { - image_names.insert(name_str); + xobject_types.insert(name_str, XObjectType::Image); + } else if subtype_name == b"Form" { + xobject_types + .insert(name_str, XObjectType::Form(obj_ref)); } } } @@ -689,7 +838,269 @@ fn get_page_xobjects(doc: &Document, page_id: ObjectId) -> std::collections::Has } } - image_names + xobject_types +} + +/// Extract text items from a Form XObject +fn extract_form_xobject_text( + doc: &Document, + form_id: ObjectId, + page_num: u32, + font_cmaps: &FontCMaps, + parent_ctm: &[f32; 6], +) -> Vec { + use lopdf::content::Content; + + let mut items = Vec::new(); + + // Get the Form XObject stream + let Ok(Object::Stream(stream)) = doc.get_object(form_id) else { + return items; + }; + + // Decompress the content stream + let Ok(content_data) = stream.decompressed_content() else { + return items; + }; + + // Decode the content stream + let Ok(content) = Content::decode(&content_data) else { + return items; + }; + + // Get fonts from the Form's Resources + let form_fonts = get_form_fonts(doc, &stream.dict); + let font_encodings = build_font_encodings(doc, &form_fonts); + + // Build font base names and ToUnicode refs for the form + let mut font_base_names: std::collections::HashMap = + std::collections::HashMap::new(); + let mut font_tounicode_refs: std::collections::HashMap = + std::collections::HashMap::new(); + + for (font_name, font_dict) in &form_fonts { + let resource_name = String::from_utf8_lossy(font_name).to_string(); + if let Ok(base_font) = font_dict.get(b"BaseFont") { + if let Ok(name) = base_font.as_name() { + let base_name = String::from_utf8_lossy(name).to_string(); + font_base_names.insert(resource_name.clone(), base_name); + } + } + if let Ok(tounicode) = font_dict.get(b"ToUnicode") { + if let Ok(obj_ref) = tounicode.as_reference() { + font_tounicode_refs.insert(resource_name, obj_ref.0); + } + } + } + + // Process the content stream (simplified version) + let mut current_font = String::new(); + let mut current_font_size: f32 = 12.0; + let mut text_matrix = [1.0f32, 0.0, 0.0, 1.0, 0.0, 0.0]; + let mut in_text_block = false; + + for op in &content.operations { + match op.operator.as_str() { + "BT" => { + in_text_block = true; + text_matrix = [1.0, 0.0, 0.0, 1.0, 0.0, 0.0]; + } + "ET" => { + in_text_block = false; + } + "Tf" => { + if op.operands.len() >= 2 { + if let Ok(name) = op.operands[0].as_name() { + current_font = String::from_utf8_lossy(name).to_string(); + } + current_font_size = get_number(&op.operands[1]).unwrap_or(12.0); + } + } + "Td" | "TD" => { + if op.operands.len() >= 2 { + let tx = get_number(&op.operands[0]).unwrap_or(0.0); + let ty = get_number(&op.operands[1]).unwrap_or(0.0); + text_matrix[4] += tx; + text_matrix[5] += ty; + } + } + "Tm" => { + if op.operands.len() >= 6 { + for (i, operand) in op.operands.iter().take(6).enumerate() { + text_matrix[i] = + get_number(operand).unwrap_or(if i == 0 || i == 3 { 1.0 } else { 0.0 }); + } + } + } + "Tj" => { + if in_text_block && !op.operands.is_empty() { + if let Some(text) = extract_text_from_operand( + &op.operands[0], + doc, + &form_fonts, + ¤t_font, + font_cmaps, + &font_base_names, + &font_tounicode_refs, + &font_encodings, + ) { + if !text.trim().is_empty() { + let rendered_size = + effective_font_size(current_font_size, &text_matrix); + // Transform position through parent CTM + let combined = multiply_matrices(&text_matrix, parent_ctm); + let (x, y) = (combined[4], combined[5]); + let base_font = font_base_names + .get(¤t_font) + .map(|s| s.as_str()) + .unwrap_or(¤t_font); + items.push(TextItem { + text, + x, + y, + width: 0.0, + height: rendered_size, + font: current_font.clone(), + font_size: rendered_size, + page: page_num, + is_bold: is_bold_font(base_font), + is_italic: is_italic_font(base_font), + item_type: ItemType::Text, + }); + } + } + } + } + "TJ" => { + if in_text_block && !op.operands.is_empty() { + if let Ok(array) = op.operands[0].as_array() { + let mut combined_text = String::new(); + // Track if the last item was a large spacing (potential word boundary) + let mut last_was_large_space = false; + for item in array { + // Check for spacing values + match item { + Object::Integer(n) => { + if *n < -200 && !combined_text.ends_with(' ') { + combined_text.push(' '); + } + last_was_large_space = *n < -200; + continue; + } + Object::Real(n) => { + if *n < -200.0 && !combined_text.ends_with(' ') { + combined_text.push(' '); + } + last_was_large_space = *n < -200.0; + continue; + } + _ => { + last_was_large_space = false; + } + } + if let Some(text) = extract_text_from_operand( + item, + doc, + &form_fonts, + ¤t_font, + font_cmaps, + &font_base_names, + &font_tounicode_refs, + &font_encodings, + ) { + combined_text.push_str(&text); + } + } + // Add trailing space if the text ends with a letter + // This helps with Form XObjects where text_matrix doesn't advance between TJs + if !combined_text.is_empty() { + let last_char = combined_text.chars().last(); + if let Some(c) = last_char { + if c.is_alphabetic() && !combined_text.ends_with(' ') { + combined_text.push(' '); + } + } + } + if !combined_text.trim().is_empty() { + let rendered_size = + effective_font_size(current_font_size, &text_matrix); + let combined = multiply_matrices(&text_matrix, parent_ctm); + let (x, y) = (combined[4], combined[5]); + let base_font = font_base_names + .get(¤t_font) + .map(|s| s.as_str()) + .unwrap_or(¤t_font); + items.push(TextItem { + text: combined_text, + x, + y, + width: 0.0, + height: rendered_size, + font: current_font.clone(), + font_size: rendered_size, + page: page_num, + is_bold: is_bold_font(base_font), + is_italic: is_italic_font(base_font), + item_type: ItemType::Text, + }); + } + } + } + } + _ => {} + } + } + + items +} + +/// Get fonts from a Form XObject's Resources +fn get_form_fonts<'a>( + doc: &'a Document, + form_dict: &lopdf::Dictionary, +) -> std::collections::BTreeMap, &'a lopdf::Dictionary> { + let mut fonts = std::collections::BTreeMap::new(); + + // Get Resources from Form dictionary + let resources = if let Ok(res_ref) = form_dict.get(b"Resources") { + if let Ok(obj_ref) = res_ref.as_reference() { + doc.get_dictionary(obj_ref).ok() + } else { + res_ref.as_dict().ok() + } + } else { + return fonts; + }; + + let Some(resources) = resources else { + return fonts; + }; + + // Get Font dictionary + let font_dict = if let Ok(font_ref) = resources.get(b"Font") { + if let Ok(obj_ref) = font_ref.as_reference() { + doc.get_dictionary(obj_ref).ok() + } else { + font_ref.as_dict().ok() + } + } else { + return fonts; + }; + + let Some(font_dict) = font_dict else { + return fonts; + }; + + // Collect fonts + for (name, value) in font_dict.iter() { + if let Ok(obj_ref) = value.as_reference() { + if let Ok(dict) = doc.get_dictionary(obj_ref) { + fonts.insert(name.clone(), dict); + } + } + } + + fonts } /// Extract hyperlinks from page annotations @@ -859,6 +1270,7 @@ fn extract_text_from_operand( font_cmaps: &FontCMaps, font_base_names: &std::collections::HashMap, font_tounicode_refs: &std::collections::HashMap, + font_encodings: &PageFontEncodings, ) -> Option { if let Object::String(bytes, _) = obj { // First, try to look up CMap by ToUnicode object reference (most reliable) @@ -903,6 +1315,17 @@ fn extract_text_from_operand( } } + // Try our custom encoding map from Differences arrays + if let Some(encoding_map) = font_encodings.get(current_font) { + let decoded: String = bytes + .iter() + .filter_map(|&b| encoding_map.get(&b).copied()) + .collect(); + if !decoded.is_empty() { + return Some(decoded); + } + } + // Try to decode using font encoding from lopdf if let Some(font_dict) = fonts.get(current_font.as_bytes()) { if let Ok(encoding) = font_dict.get_font_encoding(doc) { diff --git a/src/glyph_names.rs b/src/glyph_names.rs new file mode 100644 index 0000000..195ffd8 --- /dev/null +++ b/src/glyph_names.rs @@ -0,0 +1,329 @@ +//! Adobe Glyph List mapping from glyph names to Unicode +//! This is a subset of the most common glyph names + +use std::collections::HashMap; +use std::sync::LazyLock; + +/// Maps Adobe glyph names to Unicode code points +pub static GLYPH_TO_UNICODE: LazyLock> = LazyLock::new(|| { + let mut m = HashMap::new(); + + // Basic Latin + m.insert("space", ' '); + m.insert("exclam", '!'); + m.insert("quotedbl", '"'); + m.insert("numbersign", '#'); + m.insert("dollar", '$'); + m.insert("percent", '%'); + m.insert("ampersand", '&'); + m.insert("quotesingle", '\''); + m.insert("quoteright", '\u{2019}'); + m.insert("parenleft", '('); + m.insert("parenright", ')'); + m.insert("asterisk", '*'); + m.insert("plus", '+'); + m.insert("comma", ','); + m.insert("hyphen", '-'); + m.insert("period", '.'); + m.insert("slash", '/'); + m.insert("zero", '0'); + m.insert("one", '1'); + m.insert("two", '2'); + m.insert("three", '3'); + m.insert("four", '4'); + m.insert("five", '5'); + m.insert("six", '6'); + m.insert("seven", '7'); + m.insert("eight", '8'); + m.insert("nine", '9'); + m.insert("colon", ':'); + m.insert("semicolon", ';'); + m.insert("less", '<'); + m.insert("equal", '='); + m.insert("greater", '>'); + m.insert("question", '?'); + m.insert("at", '@'); + + // Uppercase letters + m.insert("A", 'A'); + m.insert("B", 'B'); + m.insert("C", 'C'); + m.insert("D", 'D'); + m.insert("E", 'E'); + m.insert("F", 'F'); + m.insert("G", 'G'); + m.insert("H", 'H'); + m.insert("I", 'I'); + m.insert("J", 'J'); + m.insert("K", 'K'); + m.insert("L", 'L'); + m.insert("M", 'M'); + m.insert("N", 'N'); + m.insert("O", 'O'); + m.insert("P", 'P'); + m.insert("Q", 'Q'); + m.insert("R", 'R'); + m.insert("S", 'S'); + m.insert("T", 'T'); + m.insert("U", 'U'); + m.insert("V", 'V'); + m.insert("W", 'W'); + m.insert("X", 'X'); + m.insert("Y", 'Y'); + m.insert("Z", 'Z'); + + m.insert("bracketleft", '['); + m.insert("backslash", '\\'); + m.insert("bracketright", ']'); + m.insert("asciicircum", '^'); + m.insert("underscore", '_'); + m.insert("grave", '`'); + m.insert("quoteleft", '\u{2018}'); + + // Lowercase letters + m.insert("a", 'a'); + m.insert("b", 'b'); + m.insert("c", 'c'); + m.insert("d", 'd'); + m.insert("e", 'e'); + m.insert("f", 'f'); + m.insert("g", 'g'); + m.insert("h", 'h'); + m.insert("i", 'i'); + m.insert("j", 'j'); + m.insert("k", 'k'); + m.insert("l", 'l'); + m.insert("m", 'm'); + m.insert("n", 'n'); + m.insert("o", 'o'); + m.insert("p", 'p'); + m.insert("q", 'q'); + m.insert("r", 'r'); + m.insert("s", 's'); + m.insert("t", 't'); + m.insert("u", 'u'); + m.insert("v", 'v'); + m.insert("w", 'w'); + m.insert("x", 'x'); + m.insert("y", 'y'); + m.insert("z", 'z'); + + m.insert("braceleft", '{'); + m.insert("bar", '|'); + m.insert("braceright", '}'); + m.insert("asciitilde", '~'); + + // Extended Latin and punctuation + m.insert("exclamdown", '¡'); + m.insert("cent", '¢'); + m.insert("sterling", '£'); + m.insert("currency", '¤'); + m.insert("yen", '¥'); + m.insert("brokenbar", '¦'); + m.insert("section", '§'); + m.insert("dieresis", '¨'); + m.insert("copyright", '©'); + m.insert("ordfeminine", 'ª'); + m.insert("guillemotleft", '«'); + m.insert("logicalnot", '¬'); + m.insert("registered", '®'); + m.insert("macron", '¯'); + m.insert("degree", '°'); + m.insert("plusminus", '±'); + m.insert("twosuperior", '²'); + m.insert("threesuperior", '³'); + m.insert("acute", '´'); + m.insert("mu", 'µ'); + m.insert("paragraph", '¶'); + m.insert("periodcentered", '·'); + m.insert("cedilla", '¸'); + m.insert("onesuperior", '¹'); + m.insert("ordmasculine", 'º'); + m.insert("guillemotright", '»'); + m.insert("onequarter", '¼'); + m.insert("onehalf", '½'); + m.insert("threequarters", '¾'); + m.insert("questiondown", '¿'); + + // Accented capitals + m.insert("Agrave", 'À'); + m.insert("Aacute", 'Á'); + m.insert("Acircumflex", 'Â'); + m.insert("Atilde", 'Ã'); + m.insert("Adieresis", 'Ä'); + m.insert("Aring", 'Å'); + m.insert("AE", 'Æ'); + m.insert("Ccedilla", 'Ç'); + m.insert("Egrave", 'È'); + m.insert("Eacute", 'É'); + m.insert("Ecircumflex", 'Ê'); + m.insert("Edieresis", 'Ë'); + m.insert("Igrave", 'Ì'); + m.insert("Iacute", 'Í'); + m.insert("Icircumflex", 'Î'); + m.insert("Idieresis", 'Ï'); + m.insert("Eth", 'Ð'); + m.insert("Ntilde", 'Ñ'); + m.insert("Ograve", 'Ò'); + m.insert("Oacute", 'Ó'); + m.insert("Ocircumflex", 'Ô'); + m.insert("Otilde", 'Õ'); + m.insert("Odieresis", 'Ö'); + m.insert("multiply", '×'); + m.insert("Oslash", 'Ø'); + m.insert("Ugrave", 'Ù'); + m.insert("Uacute", 'Ú'); + m.insert("Ucircumflex", 'Û'); + m.insert("Udieresis", 'Ü'); + m.insert("Yacute", 'Ý'); + m.insert("Thorn", 'Þ'); + m.insert("germandbls", 'ß'); + + // Accented lowercase + m.insert("agrave", 'à'); + m.insert("aacute", 'á'); + m.insert("acircumflex", 'â'); + m.insert("atilde", 'ã'); + m.insert("adieresis", 'ä'); + m.insert("aring", 'å'); + m.insert("ae", 'æ'); + m.insert("ccedilla", 'ç'); + m.insert("egrave", 'è'); + m.insert("eacute", 'é'); + m.insert("ecircumflex", 'ê'); + m.insert("edieresis", 'ë'); + m.insert("igrave", 'ì'); + m.insert("iacute", 'í'); + m.insert("icircumflex", 'î'); + m.insert("idieresis", 'ï'); + m.insert("eth", 'ð'); + m.insert("ntilde", 'ñ'); + m.insert("ograve", 'ò'); + m.insert("oacute", 'ó'); + m.insert("ocircumflex", 'ô'); + m.insert("otilde", 'õ'); + m.insert("odieresis", 'ö'); + m.insert("divide", '÷'); + m.insert("oslash", 'ø'); + m.insert("ugrave", 'ù'); + m.insert("uacute", 'ú'); + m.insert("ucircumflex", 'û'); + m.insert("udieresis", 'ü'); + m.insert("yacute", 'ý'); + m.insert("thorn", 'þ'); + m.insert("ydieresis", 'ÿ'); + + // Ligatures and special (Unicode ligature characters) + m.insert("fi", '\u{FB01}'); // fi + m.insert("fl", '\u{FB02}'); // fl + m.insert("ff", '\u{FB00}'); // ff + m.insert("ffi", '\u{FB03}'); // ffi + m.insert("ffl", '\u{FB04}'); // ffl + + // Quotes and dashes + m.insert("endash", '–'); + m.insert("emdash", '—'); + m.insert("quotedblleft", '"'); + m.insert("quotedblright", '"'); + m.insert("quoteleft", '\u{2018}'); + m.insert("quoteright", '\u{2019}'); + m.insert("quotesinglbase", '‚'); + m.insert("quotedblbase", '„'); + m.insert("dagger", '†'); + m.insert("daggerdbl", '‡'); + m.insert("bullet", '•'); + m.insert("ellipsis", '…'); + m.insert("perthousand", '‰'); + m.insert("guilsinglleft", '‹'); + m.insert("guilsinglright", '›'); + m.insert("fraction", '⁄'); + m.insert("trademark", '™'); + m.insert("minus", '−'); + + // Math symbols + m.insert("infinity", '∞'); + m.insert("notequal", '≠'); + m.insert("lessequal", '≤'); + m.insert("greaterequal", '≥'); + m.insert("partialdiff", '∂'); + m.insert("summation", '∑'); + m.insert("product", '∏'); + m.insert("radical", '√'); + m.insert("approxequal", '≈'); + m.insert("Delta", 'Δ'); + m.insert("lozenge", '◊'); + + // Greek letters (common ones) + m.insert("Alpha", 'Α'); + m.insert("Beta", 'Β'); + m.insert("Gamma", 'Γ'); + m.insert("Epsilon", 'Ε'); + m.insert("Zeta", 'Ζ'); + m.insert("Eta", 'Η'); + m.insert("Theta", 'Θ'); + m.insert("Iota", 'Ι'); + m.insert("Kappa", 'Κ'); + m.insert("Lambda", 'Λ'); + m.insert("Mu", 'Μ'); + m.insert("Nu", 'Ν'); + m.insert("Xi", 'Ξ'); + m.insert("Omicron", 'Ο'); + m.insert("Pi", 'Π'); + m.insert("Rho", 'Ρ'); + m.insert("Sigma", 'Σ'); + m.insert("Tau", 'Τ'); + m.insert("Upsilon", 'Υ'); + m.insert("Phi", 'Φ'); + m.insert("Chi", 'Χ'); + m.insert("Psi", 'Ψ'); + m.insert("Omega", 'Ω'); + m.insert("alpha", 'α'); + m.insert("beta", 'β'); + m.insert("gamma", 'γ'); + m.insert("delta", 'δ'); + m.insert("epsilon", 'ε'); + m.insert("zeta", 'ζ'); + m.insert("eta", 'η'); + m.insert("theta", 'θ'); + m.insert("iota", 'ι'); + m.insert("kappa", 'κ'); + m.insert("lambda", 'λ'); + m.insert("nu", 'ν'); + m.insert("xi", 'ξ'); + m.insert("omicron", 'ο'); + m.insert("pi", 'π'); + m.insert("rho", 'ρ'); + m.insert("sigma", 'σ'); + m.insert("tau", 'τ'); + m.insert("upsilon", 'υ'); + m.insert("phi", 'φ'); + m.insert("chi", 'χ'); + m.insert("psi", 'ψ'); + m.insert("omega", 'ω'); + + m +}); + +/// Convert a glyph name to its Unicode character +pub fn glyph_to_char(name: &str) -> Option { + // First check our mapping + if let Some(&c) = GLYPH_TO_UNICODE.get(name) { + return Some(c); + } + + // Try to parse uniXXXX format + if name.starts_with("uni") && name.len() >= 7 { + if let Ok(code) = u32::from_str_radix(&name[3..7], 16) { + return char::from_u32(code); + } + } + + // Try to parse uXXXX or uXXXXX format + if name.starts_with('u') && name.len() >= 5 { + if let Ok(code) = u32::from_str_radix(&name[1..], 16) { + return char::from_u32(code); + } + } + + None +} diff --git a/src/lib.rs b/src/lib.rs index a5c3cd9..d86c95d 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -7,6 +7,7 @@ pub mod detector; pub mod extractor; +pub mod glyph_names; pub mod markdown; pub mod tables; pub mod tounicode; diff --git a/src/markdown.rs b/src/markdown.rs index c2922e9..8425d6a 100644 --- a/src/markdown.rs +++ b/src/markdown.rs @@ -1146,33 +1146,95 @@ fn format_urls(text: &str) -> String { let url = mat.as_str(); // Check if this URL is already in a markdown link by looking at preceding chars - let before = if start >= 2 { - &text[start - 2..start] - } else { - "" + // Use safe character boundary checking for multi-byte UTF-8 + let before = { + let mut check_start = start.saturating_sub(2); + // Find a valid character boundary + while check_start > 0 && !text.is_char_boundary(check_start) { + check_start -= 1; + } + if check_start < start && text.is_char_boundary(start) { + &text[check_start..start] + } else { + "" + } }; let already_linked = before.ends_with("](") || before.ends_with("]("); // Also check if it's inside square brackets (link text) - let prefix = &text[..start]; + // Ensure we're slicing at a valid char boundary + let prefix = if text.is_char_boundary(start) { + &text[..start] + } else { + // Find the nearest valid boundary before start + let mut safe_start = start; + while safe_start > 0 && !text.is_char_boundary(safe_start) { + safe_start -= 1; + } + &text[..safe_start] + }; let open_brackets = prefix.matches('[').count(); let close_brackets = prefix.matches(']').count(); let inside_link_text = open_brackets > close_brackets; + // Ensure mat boundaries are valid char boundaries + let safe_last_end = if text.is_char_boundary(last_end) { + last_end + } else { + let mut pos = last_end; + while pos < text.len() && !text.is_char_boundary(pos) { + pos += 1; + } + pos + }; + let safe_start = if text.is_char_boundary(start) { + start + } else { + let mut pos = start; + while pos < text.len() && !text.is_char_boundary(pos) { + pos += 1; + } + pos + }; + let safe_end = if text.is_char_boundary(mat.end()) { + mat.end() + } else { + let mut pos = mat.end(); + while pos < text.len() && !text.is_char_boundary(pos) { + pos += 1; + } + pos + }; + if already_linked || inside_link_text { // Already formatted, keep as-is - result.push_str(&text[last_end..mat.end()]); + if safe_last_end <= safe_end { + result.push_str(&text[safe_last_end..safe_end]); + } } else { // Add text before this URL - result.push_str(&text[last_end..start]); + if safe_last_end <= safe_start { + result.push_str(&text[safe_last_end..safe_start]); + } // Format as markdown link result.push_str(&format!("[{}]({})", url, url)); } - last_end = mat.end(); + last_end = safe_end; } - // Add remaining text - result.push_str(&text[last_end..]); + // Add remaining text (ensure valid char boundary) + let safe_last_end = if text.is_char_boundary(last_end) { + last_end + } else { + let mut pos = last_end; + while pos < text.len() && !text.is_char_boundary(pos) { + pos += 1; + } + pos + }; + if safe_last_end < text.len() { + result.push_str(&text[safe_last_end..]); + } result }