Formatting Improvements
1. Word Fragment Joining (extractor.rs:63-94) Added is_word_continuation() to detect when text items are fragments of the same word and should be joined without spaces: 2. Caption Detection (markdown.rs:621-658) Added is_caption_line() to detect figures, tables, and source citations: - Ensures captions are on their own line with paragraph breaks 3. Paragraph Threshold Adjustment Changed from base_size * 2.0 to base_size * 1.8 for better paragraph detection.
This commit is contained in:
+70
-2
@@ -264,7 +264,7 @@ fn to_markdown_from_lines_with_tables(
|
||||
|
||||
// Paragraph break (large Y gap)
|
||||
let y_gap = prev_y - line.y;
|
||||
let is_para_break = y_gap > base_size * 2.0;
|
||||
let is_para_break = y_gap > base_size * 1.8; // Slightly lower threshold
|
||||
if is_para_break {
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
@@ -283,6 +283,18 @@ fn to_markdown_from_lines_with_tables(
|
||||
continue;
|
||||
}
|
||||
|
||||
// Detect figure/table captions and source citations
|
||||
// These should be on their own line followed by a paragraph break
|
||||
if is_caption_line(trimmed) {
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
in_paragraph = false;
|
||||
}
|
||||
output.push_str(trimmed);
|
||||
output.push_str("\n\n");
|
||||
continue;
|
||||
}
|
||||
|
||||
// Detect headers by font size
|
||||
if options.detect_headers && trimmed.len() > 3 {
|
||||
let line_font_size = line.items.first().map(|i| i.font_size).unwrap_or(base_size);
|
||||
@@ -395,7 +407,7 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
|
||||
// Paragraph break (large Y gap)
|
||||
let y_gap = prev_y - line.y;
|
||||
let is_para_break = y_gap > base_size * 2.0;
|
||||
let is_para_break = y_gap > base_size * 1.8; // Slightly lower threshold
|
||||
if is_para_break {
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
@@ -414,6 +426,18 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
continue;
|
||||
}
|
||||
|
||||
// Detect figure/table captions and source citations
|
||||
// These should be on their own line followed by a paragraph break
|
||||
if is_caption_line(trimmed) {
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
in_paragraph = false;
|
||||
}
|
||||
output.push_str(trimmed);
|
||||
output.push_str("\n\n");
|
||||
continue;
|
||||
}
|
||||
|
||||
// Detect headers by font size
|
||||
// Skip very short text (likely drop caps or labels)
|
||||
if options.detect_headers && trimmed.len() > 3 {
|
||||
@@ -605,6 +629,50 @@ fn detect_header_level(font_size: f32, base_size: f32) -> Option<usize> {
|
||||
}
|
||||
}
|
||||
|
||||
/// Check if text is a figure/table caption or source citation
|
||||
fn is_caption_line(text: &str) -> bool {
|
||||
let trimmed = text.trim();
|
||||
|
||||
// Common caption prefixes in multiple languages
|
||||
let caption_prefixes = [
|
||||
"Figure ",
|
||||
"Figura ",
|
||||
"Fig. ",
|
||||
"Fig ",
|
||||
"Table ",
|
||||
"Tabela ",
|
||||
"Source:",
|
||||
"Fonte:",
|
||||
"Source ",
|
||||
"Fonte ",
|
||||
"Note:",
|
||||
"Nota:",
|
||||
"Chart ",
|
||||
"Gráfico ",
|
||||
"Graph ",
|
||||
"Diagram ",
|
||||
"Image ",
|
||||
"Imagem ",
|
||||
"Photo ",
|
||||
"Foto ",
|
||||
];
|
||||
|
||||
// Check if line starts with a caption prefix
|
||||
for prefix in &caption_prefixes {
|
||||
if trimmed.starts_with(prefix) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// Check case-insensitive patterns
|
||||
let lower = trimmed.to_lowercase();
|
||||
if lower.starts_with("figure ") || lower.starts_with("table ") || lower.starts_with("source:") {
|
||||
return true;
|
||||
}
|
||||
|
||||
false
|
||||
}
|
||||
|
||||
/// Check if text looks like a list item
|
||||
fn is_list_item(text: &str) -> bool {
|
||||
let trimmed = text.trim_start();
|
||||
|
||||
Reference in New Issue
Block a user