Compare commits
5
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4b1c7b21b6 | ||
|
|
da1323a044 | ||
|
|
83754ddad0 | ||
|
|
f2f49bdac2 | ||
|
|
d28594ef94 |
@@ -130,6 +130,108 @@ pub(crate) fn has_dot_leaders(text: &str) -> bool {
|
|||||||
dot_groups >= 2
|
dot_groups >= 2
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Detect a table-of-contents entry: a line ending in a page number preceded by
|
||||||
|
/// a dot-leader group (e.g. "Measurement Lab worksheet ... 3"). `has_dot_leaders`
|
||||||
|
/// misses single-group leaders ("..."), but a trailing "<dots> <number>" is a
|
||||||
|
/// strong TOC signal on its own. Such lines must never be promoted to headings.
|
||||||
|
pub(crate) fn is_toc_entry_line(text: &str) -> bool {
|
||||||
|
let trimmed = text.trim_end();
|
||||||
|
let digits = trimmed
|
||||||
|
.chars()
|
||||||
|
.rev()
|
||||||
|
.take_while(|c| c.is_ascii_digit())
|
||||||
|
.count();
|
||||||
|
if digits == 0 || digits > 4 {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
let before_number = trimmed[..trimmed.len() - digits].trim_end();
|
||||||
|
let dots = before_number
|
||||||
|
.chars()
|
||||||
|
.rev()
|
||||||
|
.take_while(|c| *c == '.')
|
||||||
|
.count();
|
||||||
|
dots >= 3
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A heading that announces a table of contents ("Contents", "Table of
|
||||||
|
/// Contents"). Lines after it on the same page are ToC entries — section
|
||||||
|
/// titles that look exactly like headings but must not be promoted.
|
||||||
|
pub(crate) fn is_toc_marker_heading(text: &str) -> bool {
|
||||||
|
let t = text.trim().trim_end_matches(':').trim().to_lowercase();
|
||||||
|
matches!(t.as_str(), "contents" | "table of contents")
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Lines that resemble headings structurally but are display-math fragments:
|
||||||
|
/// equations ending in an equation number ("S = kB ln W, (2)") or equation
|
||||||
|
/// lead-ins ("Rearranging Equation (8) gives:"). Both carry an "(N)" equation
|
||||||
|
/// reference — but a trailing "(N)" alone is not enough: real headings end
|
||||||
|
/// with parenthesized numbers too ("Nicaea (325)", appendix numbering), so
|
||||||
|
/// the suffix form additionally requires math evidence — an "=" in the line
|
||||||
|
/// or a comma immediately before the number, both present in every display
|
||||||
|
/// equation and absent from name-plus-number headings. A bare trailing colon
|
||||||
|
/// is NOT a fragment signal either: real headings frequently end with colons
|
||||||
|
/// ("Procedure:", "Steps for Using the Microscope:").
|
||||||
|
pub(crate) fn is_heading_fragment(text: &str) -> bool {
|
||||||
|
let t = text.trim_end();
|
||||||
|
|
||||||
|
fn is_equation_number(s: &str) -> bool {
|
||||||
|
s.strip_prefix('(')
|
||||||
|
.and_then(|r| r.strip_suffix(')'))
|
||||||
|
.is_some_and(|inner| {
|
||||||
|
!inner.is_empty() && inner.len() <= 3 && inner.chars().all(|c| c.is_ascii_digit())
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// Equation-number suffix with math evidence: "S = kB ln W, (2)"
|
||||||
|
let mut rev = t.rsplit(' ');
|
||||||
|
let last = rev.next().unwrap_or("");
|
||||||
|
if is_equation_number(last) {
|
||||||
|
// Page-of-total running headers: "LIVSMEDELSVERKET PM 2 (10)"
|
||||||
|
if let Some(prev_word) = t.rsplit(' ').nth(1) {
|
||||||
|
if let (Ok(page), Some(total)) = (
|
||||||
|
prev_word.parse::<u32>(),
|
||||||
|
last.trim_start_matches('(')
|
||||||
|
.trim_end_matches(')')
|
||||||
|
.parse::<u32>()
|
||||||
|
.ok(),
|
||||||
|
) {
|
||||||
|
if page <= total {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let punct_before = rev
|
||||||
|
.next()
|
||||||
|
.is_some_and(|w| w.ends_with(',') || w.ends_with(':'));
|
||||||
|
let has_math_op = t.chars().any(|c| {
|
||||||
|
matches!(
|
||||||
|
c,
|
||||||
|
'=' | '<'
|
||||||
|
| '>'
|
||||||
|
| '≤'
|
||||||
|
| '≥'
|
||||||
|
| '≪'
|
||||||
|
| '≫'
|
||||||
|
| '≈'
|
||||||
|
| '≠'
|
||||||
|
| '±'
|
||||||
|
| '∑'
|
||||||
|
| '∫'
|
||||||
|
| '√'
|
||||||
|
| '∝'
|
||||||
|
)
|
||||||
|
});
|
||||||
|
if punct_before || has_math_op {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Lead-in: ends with a colon AND references an equation number inline
|
||||||
|
if t.ends_with(':') && t.split_whitespace().any(is_equation_number) {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
false
|
||||||
|
}
|
||||||
|
|
||||||
/// Compute the Y-gap threshold for paragraph break detection.
|
/// Compute the Y-gap threshold for paragraph break detection.
|
||||||
///
|
///
|
||||||
/// Instead of using a fixed multiple of base_size (which fails for double-spaced
|
/// Instead of using a fixed multiple of base_size (which fails for double-spaced
|
||||||
@@ -320,3 +422,65 @@ pub(crate) fn detect_header_level(
|
|||||||
Some(4)
|
Some(4)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn toc_entry_with_single_dot_group() {
|
||||||
|
assert!(is_toc_entry_line("Measurement Lab worksheet ... 3"));
|
||||||
|
assert!(is_toc_entry_line("Results ........ 12"));
|
||||||
|
assert!(is_toc_entry_line("Appendix B...42"));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn non_toc_lines_pass() {
|
||||||
|
assert!(!is_toc_entry_line(
|
||||||
|
"6.2. Expectations for Re-Hiring Employees"
|
||||||
|
));
|
||||||
|
assert!(!is_toc_entry_line("What happened in 2020"));
|
||||||
|
assert!(!is_toc_entry_line("IMPLEMENTATION"));
|
||||||
|
// Ellipsis without a trailing page number
|
||||||
|
assert!(!is_toc_entry_line("and so it goes ..."));
|
||||||
|
// Long numbers are data, not page refs
|
||||||
|
assert!(!is_toc_entry_line("ISBN ... 97814"));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn toc_marker_headings() {
|
||||||
|
assert!(is_toc_marker_heading("Contents"));
|
||||||
|
assert!(is_toc_marker_heading("CONTENTS"));
|
||||||
|
assert!(is_toc_marker_heading("Table of Contents"));
|
||||||
|
assert!(is_toc_marker_heading("Table of contents:"));
|
||||||
|
assert!(!is_toc_marker_heading("Contents of the Shipment"));
|
||||||
|
assert!(!is_toc_marker_heading("Introduction"));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn heading_fragments() {
|
||||||
|
// Equation lead-ins: colon ending + inline equation reference
|
||||||
|
assert!(is_heading_fragment("Rearranging Equation (8) gives:"));
|
||||||
|
// Display-equation neighbours ending in an equation number
|
||||||
|
assert!(is_heading_fragment("S = kB ln W, (2)"));
|
||||||
|
assert!(is_heading_fragment("E = mc2 (12)"));
|
||||||
|
assert!(is_heading_fragment("x + y = z, (3)"));
|
||||||
|
// Page-of-total running headers
|
||||||
|
assert!(is_heading_fragment("LIVSMEDELSVERKET PM 2 (10)"));
|
||||||
|
// Comparison-operator evidence and colon-before-number
|
||||||
|
assert!(is_heading_fragment(
|
||||||
|
"PLL\u{fe} PHH\u{226a} PLH\u{fe} PHL: (12)"
|
||||||
|
));
|
||||||
|
// Real headings pass — including name-plus-number and colon-ended ones
|
||||||
|
assert!(!is_heading_fragment("Nicaea (325)"));
|
||||||
|
assert!(!is_heading_fragment(
|
||||||
|
"\u{627}\u{644}\u{645}\u{644}\u{62d}\u{642} \u{631}\u{642}\u{645} (1)"
|
||||||
|
));
|
||||||
|
assert!(!is_heading_fragment("4. Entropy"));
|
||||||
|
assert!(!is_heading_fragment("Procedure:"));
|
||||||
|
assert!(!is_heading_fragment("Steps for Using the Microscope:"));
|
||||||
|
assert!(!is_heading_fragment("Changing objectives:"));
|
||||||
|
assert!(!is_heading_fragment("Sales by Region (2024)"));
|
||||||
|
assert!(!is_heading_fragment("Results (preliminary)"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+24
-3
@@ -7,7 +7,8 @@ use crate::types::TextLine;
|
|||||||
|
|
||||||
use super::analysis::{
|
use super::analysis::{
|
||||||
bold_heading_level, calculate_font_stats, compute_heading_tiers, compute_paragraph_threshold,
|
bold_heading_level, calculate_font_stats, compute_heading_tiers, compute_paragraph_threshold,
|
||||||
detect_header_level, font_size_rarity, has_dot_leaders,
|
detect_header_level, font_size_rarity, has_dot_leaders, is_heading_fragment, is_toc_entry_line,
|
||||||
|
is_toc_marker_heading,
|
||||||
};
|
};
|
||||||
use super::classify::{
|
use super::classify::{
|
||||||
format_list_item, is_caption_line, is_list_item, is_monospace_font, starts_with_bullet_marker,
|
format_list_item, is_caption_line, is_list_item, is_monospace_font, starts_with_bullet_marker,
|
||||||
@@ -486,6 +487,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
|||||||
let mut in_code_block = false;
|
let mut in_code_block = false;
|
||||||
let mut prev_had_dot_leaders = false;
|
let mut prev_had_dot_leaders = false;
|
||||||
let mut paragraph_in_wrapped_bold_run = false;
|
let mut paragraph_in_wrapped_bold_run = false;
|
||||||
|
let mut toc_suppress_page: Option<u32> = None;
|
||||||
let mut inserted_tables: HashSet<(u32, usize)> = HashSet::new();
|
let mut inserted_tables: HashSet<(u32, usize)> = HashSet::new();
|
||||||
let mut inserted_images: HashSet<(u32, usize)> = HashSet::new();
|
let mut inserted_images: HashSet<(u32, usize)> = HashSet::new();
|
||||||
|
|
||||||
@@ -703,6 +705,9 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
|||||||
&& plain_trimmed.len() > 3
|
&& plain_trimmed.len() > 3
|
||||||
&& plain_trimmed.split_whitespace().count() <= 15
|
&& plain_trimmed.split_whitespace().count() <= 15
|
||||||
&& !starts_with_bullet_marker(plain_trimmed)
|
&& !starts_with_bullet_marker(plain_trimmed)
|
||||||
|
&& !is_toc_entry_line(plain_trimmed)
|
||||||
|
&& !is_heading_fragment(plain_trimmed)
|
||||||
|
&& toc_suppress_page != Some(line.page)
|
||||||
{
|
{
|
||||||
let line_font_size = line.items.first().map(|i| i.font_size).unwrap_or(base_size);
|
let line_font_size = line.items.first().map(|i| i.font_size).unwrap_or(base_size);
|
||||||
detect_header_level(line_font_size, base_size, &heading_tiers).or_else(|| {
|
detect_header_level(line_font_size, base_size, &heading_tiers).or_else(|| {
|
||||||
@@ -738,7 +743,11 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
|||||||
// paragraph continuity and minor font-size variation
|
// paragraph continuity and minor font-size variation
|
||||||
// inflates rarity scores.
|
// inflates rarity scores.
|
||||||
let has_strong_signal = all_bold || isolated || (rarity >= 0.97 && word_count <= 8);
|
let has_strong_signal = all_bold || isolated || (rarity >= 0.97 && word_count <= 8);
|
||||||
if score >= 0.5 && standalone && word_count >= 2 && has_strong_signal {
|
// Single-word headings ("IMPLEMENTATION", "CONTENTS") are common;
|
||||||
|
// accept them only with the strongest signal combination.
|
||||||
|
let enough_words =
|
||||||
|
word_count >= 2 || (all_bold && isolated && plain_trimmed.len() >= 4);
|
||||||
|
if score >= 0.5 && standalone && enough_words && has_strong_signal {
|
||||||
Some(bold_heading_level(&heading_tiers))
|
Some(bold_heading_level(&heading_tiers))
|
||||||
} else {
|
} else {
|
||||||
None
|
None
|
||||||
@@ -763,6 +772,9 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
|||||||
plain_text.clone()
|
plain_text.clone()
|
||||||
};
|
};
|
||||||
output.push_str(&format!("{} {}\n\n", prefix, heading_text.trim()));
|
output.push_str(&format!("{} {}\n\n", prefix, heading_text.trim()));
|
||||||
|
if is_toc_marker_heading(plain_trimmed) {
|
||||||
|
toc_suppress_page = Some(line.page);
|
||||||
|
}
|
||||||
in_list = false;
|
in_list = false;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
@@ -959,6 +971,7 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
|||||||
let mut last_list_x: Option<f32> = None;
|
let mut last_list_x: Option<f32> = None;
|
||||||
let mut prev_had_dot_leaders = false;
|
let mut prev_had_dot_leaders = false;
|
||||||
let mut paragraph_in_wrapped_bold_run = false;
|
let mut paragraph_in_wrapped_bold_run = false;
|
||||||
|
let mut toc_suppress_page: Option<u32> = None;
|
||||||
|
|
||||||
for (line_idx, line) in lines.iter().enumerate() {
|
for (line_idx, line) in lines.iter().enumerate() {
|
||||||
// Page break
|
// Page break
|
||||||
@@ -1037,6 +1050,9 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
|||||||
if options.detect_headers
|
if options.detect_headers
|
||||||
&& plain_trimmed.len() > 3
|
&& plain_trimmed.len() > 3
|
||||||
&& plain_trimmed.split_whitespace().count() <= 15
|
&& plain_trimmed.split_whitespace().count() <= 15
|
||||||
|
&& !is_toc_entry_line(plain_trimmed)
|
||||||
|
&& !is_heading_fragment(plain_trimmed)
|
||||||
|
&& toc_suppress_page != Some(line.page)
|
||||||
{
|
{
|
||||||
let line_font_size = line.items.first().map(|i| i.font_size).unwrap_or(base_size);
|
let line_font_size = line.items.first().map(|i| i.font_size).unwrap_or(base_size);
|
||||||
if let Some(header_level) =
|
if let Some(header_level) =
|
||||||
@@ -1059,7 +1075,9 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
|||||||
+ if all_bold { 0.3 } else { 0.0 }
|
+ if all_bold { 0.3 } else { 0.0 }
|
||||||
+ if standalone { 0.2 } else { 0.0 }
|
+ if standalone { 0.2 } else { 0.0 }
|
||||||
+ if isolated { 0.3 } else { 0.0 };
|
+ if isolated { 0.3 } else { 0.0 };
|
||||||
if score >= 0.5 && standalone && word_count >= 2 {
|
let enough_words =
|
||||||
|
word_count >= 2 || (all_bold && isolated && plain_trimmed.len() >= 4);
|
||||||
|
if score >= 0.5 && standalone && enough_words {
|
||||||
return Some(bold_heading_level(&heading_tiers));
|
return Some(bold_heading_level(&heading_tiers));
|
||||||
}
|
}
|
||||||
None
|
None
|
||||||
@@ -1078,6 +1096,9 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
|||||||
plain_text.clone()
|
plain_text.clone()
|
||||||
};
|
};
|
||||||
output.push_str(&format!("{} {}\n\n", prefix, heading_text.trim()));
|
output.push_str(&format!("{} {}\n\n", prefix, heading_text.trim()));
|
||||||
|
if is_toc_marker_heading(plain_trimmed) {
|
||||||
|
toc_suppress_page = Some(line.page);
|
||||||
|
}
|
||||||
in_list = false;
|
in_list = false;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|||||||
+100
-1
@@ -3,7 +3,7 @@
|
|||||||
use std::collections::{HashMap, HashSet};
|
use std::collections::{HashMap, HashSet};
|
||||||
|
|
||||||
use crate::structure_tree::StructRole;
|
use crate::structure_tree::StructRole;
|
||||||
use crate::types::TextLine;
|
use crate::types::{TextItem, TextLine};
|
||||||
|
|
||||||
use super::analysis::detect_header_level;
|
use super::analysis::detect_header_level;
|
||||||
|
|
||||||
@@ -87,6 +87,41 @@ pub(crate) fn merge_heading_lines(
|
|||||||
false
|
false
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// Bold headings at body font size never reach a tier, so wrapped ones
|
||||||
|
// split into two output headings ("…of wood pellets and cost" /
|
||||||
|
// "structure in Japan"). Merge a fully-bold line into the previous
|
||||||
|
// fully-bold line when it reads as a wrap continuation: starts
|
||||||
|
// lowercase, tiny Y gap, and the previous line has no terminal
|
||||||
|
// punctuation. Kept deliberately narrow — bold list labels and bold
|
||||||
|
// sentences start with markers or capitals and are unaffected.
|
||||||
|
let should_merge = should_merge
|
||||||
|
|| if let Some(prev) = result.last() {
|
||||||
|
let all_bold = |l: &TextLine| {
|
||||||
|
!l.items.is_empty() && l.items.iter().all(|i: &TextItem| i.is_bold)
|
||||||
|
};
|
||||||
|
let prev_text = prev.text();
|
||||||
|
let prev_trim = prev_text.trim_end();
|
||||||
|
let curr_text = line.text();
|
||||||
|
let curr_trim = curr_text.trim();
|
||||||
|
let y_gap = prev.y - line.y;
|
||||||
|
// Both lines must be tier-less: a tiered/tagged bold heading
|
||||||
|
// followed by bold body text must not absorb it.
|
||||||
|
line_level.is_none()
|
||||||
|
&& effective_heading_level(prev, base_size, heading_tiers, struct_roles)
|
||||||
|
.is_none()
|
||||||
|
&& prev.page == line.page
|
||||||
|
&& all_bold(prev)
|
||||||
|
&& all_bold(&line)
|
||||||
|
&& y_gap > 0.0
|
||||||
|
&& y_gap < line_font * 1.6
|
||||||
|
&& curr_trim.chars().next().is_some_and(|c| c.is_lowercase())
|
||||||
|
&& !prev_trim.ends_with(['.', ':', ';', '!', '?'])
|
||||||
|
&& prev_trim.split_whitespace().count() + curr_trim.split_whitespace().count()
|
||||||
|
<= 20
|
||||||
|
} else {
|
||||||
|
false
|
||||||
|
};
|
||||||
|
|
||||||
if should_merge {
|
if should_merge {
|
||||||
// Append this line's items to the previous line
|
// Append this line's items to the previous line
|
||||||
let prev = result.last_mut().unwrap();
|
let prev = result.last_mut().unwrap();
|
||||||
@@ -685,4 +720,68 @@ mod tests {
|
|||||||
.unwrap();
|
.unwrap();
|
||||||
assert_eq!(first_header.page, 1, "first occurrence should be on page 1");
|
assert_eq!(first_header.page, 1, "first occurrence should be on page 1");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn make_bold_line(text: &str, page: u32, y: f32) -> TextLine {
|
||||||
|
let mut item = make_item(text, 12.0, None);
|
||||||
|
item.is_bold = true;
|
||||||
|
TextLine {
|
||||||
|
items: vec![item],
|
||||||
|
y,
|
||||||
|
page,
|
||||||
|
adaptive_threshold: 0.10,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn merge_wrapped_bold_heading_lowercase_continuation() {
|
||||||
|
// Bold-at-body-size heading wrapped across two lines: the second line
|
||||||
|
// starts lowercase and must merge into the first.
|
||||||
|
let lines = vec![
|
||||||
|
make_bold_line(
|
||||||
|
"3. Perspective of supply and demand balance and cost",
|
||||||
|
1,
|
||||||
|
700.0,
|
||||||
|
),
|
||||||
|
make_bold_line("structure in Japan", 1, 686.0),
|
||||||
|
make_line("Body text paragraph follows here.", 12.0, 1, 660.0, None),
|
||||||
|
];
|
||||||
|
let result = merge_heading_lines(lines, 12.0, &[], None);
|
||||||
|
assert_eq!(result.len(), 2, "wrapped bold heading should merge");
|
||||||
|
assert!(result[0].text().contains("cost structure in Japan"));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn no_merge_for_bold_sentences_or_new_headings() {
|
||||||
|
// Second bold line starts with a capital — a new heading or label,
|
||||||
|
// not a wrap continuation.
|
||||||
|
let lines = vec![
|
||||||
|
make_bold_line("Replace", 1, 700.0),
|
||||||
|
make_bold_line("Trash", 1, 686.0),
|
||||||
|
];
|
||||||
|
let result = merge_heading_lines(lines, 12.0, &[], None);
|
||||||
|
assert_eq!(result.len(), 2, "distinct bold lines must not merge");
|
||||||
|
|
||||||
|
// Previous line ends a sentence — continuation must not merge.
|
||||||
|
let lines = vec![
|
||||||
|
make_bold_line("This is a bold sentence.", 1, 700.0),
|
||||||
|
make_bold_line("another bold line", 1, 686.0),
|
||||||
|
];
|
||||||
|
let result = merge_heading_lines(lines, 12.0, &[], None);
|
||||||
|
assert_eq!(result.len(), 2, "sentence-final bold line must not merge");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn tiered_bold_heading_does_not_absorb_bold_body() {
|
||||||
|
// Previous line is a tier-level bold heading (16pt vs 12pt body);
|
||||||
|
// a following lowercase bold body line must NOT merge into it.
|
||||||
|
let mut heading = make_bold_line("Section Title", 1, 700.0);
|
||||||
|
heading.items[0].font_size = 16.0;
|
||||||
|
heading.items[0].height = 16.0;
|
||||||
|
let lines = vec![
|
||||||
|
heading,
|
||||||
|
make_bold_line("emphasized body text continues here", 1, 686.0),
|
||||||
|
];
|
||||||
|
let result = merge_heading_lines(lines, 12.0, &[16.0], None);
|
||||||
|
assert_eq!(result.len(), 2, "tiered heading must not absorb bold body");
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1477,6 +1477,10 @@ fn detect_row_stripe_table(
|
|||||||
debug!(" row-stripe rejected: sparse outline/prose continuation shape");
|
debug!(" row-stripe rejected: sparse outline/prose continuation shape");
|
||||||
return None;
|
return None;
|
||||||
}
|
}
|
||||||
|
if has_dominant_prose_cell(&cells) {
|
||||||
|
debug!(" row-stripe rejected: dominant prose cell (chart/figure region over body text)");
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
|
||||||
let column_centers: Vec<f32> = (0..num_cols)
|
let column_centers: Vec<f32> = (0..num_cols)
|
||||||
.map(|c| (col_edges[c] + col_edges[c + 1]) / 2.0)
|
.map(|c| (col_edges[c] + col_edges[c + 1]) / 2.0)
|
||||||
@@ -1495,6 +1499,34 @@ fn detect_row_stripe_table(
|
|||||||
Some(Table::new(column_centers, row_centers, cells, item_indices))
|
Some(Table::new(column_centers, row_centers, cells, item_indices))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Detect a grid that swallowed body text instead of tabular data.
|
||||||
|
///
|
||||||
|
/// Charts (bar graphs, axis gridlines) emit fields of drawing rects that can
|
||||||
|
/// pass the row-stripe shape test; the resulting "table" then captures the
|
||||||
|
/// page's prose. The signature: one cell holds an entire paragraph — ≥60 words
|
||||||
|
/// AND at least a third of all words in the table.
|
||||||
|
///
|
||||||
|
/// There is deliberately no row-count exemption. A small table whose single
|
||||||
|
/// long cell dominates its word count is indistinguishable by content from a
|
||||||
|
/// phantom grid over body text, and across the regression corpora every such
|
||||||
|
/// grid observed has been swallowed prose, never a real note table. The costs
|
||||||
|
/// are also asymmetric: rejecting a real table degrades it to readable prose,
|
||||||
|
/// while accepting a phantom scrambles the page into Y-interleaved cells.
|
||||||
|
/// Larger legitimate tables are safe because the one-third-of-total threshold
|
||||||
|
/// scales with table size.
|
||||||
|
fn has_dominant_prose_cell(cells: &[Vec<String>]) -> bool {
|
||||||
|
let mut total_words = 0usize;
|
||||||
|
let mut max_cell_words = 0usize;
|
||||||
|
for row in cells {
|
||||||
|
for cell in row {
|
||||||
|
let words = cell.split_whitespace().count();
|
||||||
|
total_words += words;
|
||||||
|
max_cell_words = max_cell_words.max(words);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
max_cell_words >= 60 && max_cell_words * 3 >= total_words
|
||||||
|
}
|
||||||
|
|
||||||
fn row_stripe_is_sparse_prose_outline(cells: &[Vec<String>]) -> bool {
|
fn row_stripe_is_sparse_prose_outline(cells: &[Vec<String>]) -> bool {
|
||||||
let Some(num_cols) = cells.first().map(|row| row.len()) else {
|
let Some(num_cols) = cells.first().map(|row| row.len()) else {
|
||||||
return false;
|
return false;
|
||||||
@@ -2294,6 +2326,12 @@ fn detect_merged_cluster_table(
|
|||||||
);
|
);
|
||||||
return None;
|
return None;
|
||||||
}
|
}
|
||||||
|
if has_dominant_prose_cell(&cells) {
|
||||||
|
debug!(
|
||||||
|
" merged-cluster rejected: dominant prose cell (chart/figure region over body text)"
|
||||||
|
);
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
|
||||||
// No empty columns
|
// No empty columns
|
||||||
for col in 0..num_cols {
|
for col in 0..num_cols {
|
||||||
@@ -2398,6 +2436,89 @@ mod tests {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- has_dominant_prose_cell ---
|
||||||
|
|
||||||
|
fn cells_of(rows: &[&[&str]]) -> Vec<Vec<String>> {
|
||||||
|
rows.iter()
|
||||||
|
.map(|r| r.iter().map(|c| c.to_string()).collect())
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn dominant_prose_cell_rejects_swallowed_paragraph() {
|
||||||
|
// Two cells hold paragraphs (the shape every observed phantom grid
|
||||||
|
// has: swallowed body text spans multiple cells), rest are chart labels
|
||||||
|
let para = ["word"; 70].join(" ");
|
||||||
|
let para2 = ["word"; 35].join(" ");
|
||||||
|
let cells = cells_of(&[
|
||||||
|
&[para.as_str(), "81", "76"],
|
||||||
|
&[para2.as_str(), "56", "9"],
|
||||||
|
&["2019", "2020", ""],
|
||||||
|
]);
|
||||||
|
assert!(has_dominant_prose_cell(&cells));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn dominant_prose_cell_rejects_small_table_dominated_by_one_cell() {
|
||||||
|
// Boundary case, documented as INTENDED: a small grid whose single
|
||||||
|
// long cell dominates the word count is rejected even at 4+ rows.
|
||||||
|
// By content alone this shape is indistinguishable from a phantom
|
||||||
|
// grid over body text, and every observed instance in the regression
|
||||||
|
// corpora was swallowed prose (chart/figure regions), not a real
|
||||||
|
// note table. Rejection degrades gracefully — the text is still
|
||||||
|
// extracted as prose — while accepting a phantom scrambles reading
|
||||||
|
// order.
|
||||||
|
let note = ["word"; 70].join(" ");
|
||||||
|
let cells = cells_of(&[
|
||||||
|
&["Purpose", note.as_str()],
|
||||||
|
&["Owner", "Facilities team"],
|
||||||
|
&["Date", "2024-06-01"],
|
||||||
|
&["Status", "Active"],
|
||||||
|
]);
|
||||||
|
assert!(has_dominant_prose_cell(&cells));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn dominant_prose_cell_allows_description_column() {
|
||||||
|
// Long-ish description cells, but text is spread across the table
|
||||||
|
let desc = ["word"; 25].join(" ");
|
||||||
|
let cells = cells_of(&[
|
||||||
|
&["Item A", desc.as_str(), "100"],
|
||||||
|
&["Item B", desc.as_str(), "200"],
|
||||||
|
&["Item C", desc.as_str(), "300"],
|
||||||
|
&["Item D", desc.as_str(), "400"],
|
||||||
|
]);
|
||||||
|
assert!(!has_dominant_prose_cell(&cells));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn dominant_prose_cell_allows_short_tables() {
|
||||||
|
let cells = cells_of(&[&["Name", "Value"], &["Total", "42"]]);
|
||||||
|
assert!(!has_dominant_prose_cell(&cells));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn dominant_prose_cell_allows_data_table_with_long_note() {
|
||||||
|
// A real 4+ row table with one verbose remark cell: the note is ≥60
|
||||||
|
// words but the table's other content carries more than 2× its word
|
||||||
|
// count, so concentration stays below the 1/3 threshold. The
|
||||||
|
// denominator scales with table size — this is what keeps large
|
||||||
|
// legitimate tables safe where a bare length cap would not.
|
||||||
|
let note = ["word"; 60].join(" ");
|
||||||
|
let row_text = ["data"; 12].join(" ");
|
||||||
|
let mut rows: Vec<Vec<String>> = (0..11)
|
||||||
|
.map(|i| {
|
||||||
|
vec![
|
||||||
|
format!("Item {i}"),
|
||||||
|
row_text.clone(),
|
||||||
|
format!("{}", i * 100),
|
||||||
|
]
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
rows.push(vec!["Note".into(), note, String::new()]);
|
||||||
|
assert!(!has_dominant_prose_cell(&rows));
|
||||||
|
}
|
||||||
|
|
||||||
// --- rects_overlap ---
|
// --- rects_overlap ---
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
|
|||||||
@@ -48,7 +48,7 @@ tips of directly from customers received other employees paid tips rec’d. entr
|
|||||||
|
|
||||||
**Page 3**
|
**Page 3**
|
||||||
|
|
||||||
27 28 29 30 31 **Subtotals** **from pages** **1, 2, and 3** **Totals**
|
27 28 29 30 31 **Subtotals from pages** **1, 2, and 3** **Totals**
|
||||||
|
|
||||||
**1.** Report total cash tips (col. **a**) on Form 4070, line **1.**
|
**1.** Report total cash tips (col. **a**) on Form 4070, line **1.**
|
||||||
**2.** Report total credit card tips (col. **b**) on Form 4070, line **2.**
|
**2.** Report total credit card tips (col. **b**) on Form 4070, line **2.**
|
||||||
@@ -79,4 +79,3 @@ forms simpler, we would be happy to hear from you. You can write to the Tax Form
|
|||||||
**Instructions** *(continued)*
|
**Instructions** *(continued)*
|
||||||
|
|
||||||
Use this space to total your tips for the year
|
Use this space to total your tips for the year
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user