fix: don't treat bullet-marker columns as real text columns (#55)

On PDFs where every list item starts with ● at the left margin and content
at a fixed offset, histogram column detection sees the gap between marker
and content as a gutter and splits each line across two phantom "columns,"
scrambling the reading order (Anthropic's Mythos system card p.73–74).

- layout: reject gutter candidates where the smaller side is ≥80%
  standalone bullet-marker glyphs (•, ●, ○, ◦, ▪, ▫, ◆, ◇, ■, □)
- markdown/classify: add starts_with_bullet_marker helper (narrower than
  is_list_item — excludes numbered/lettered patterns like 1. and a) so
  numbered section headings stay as headings)
- markdown/convert: skip heuristic heading detection on lines that start
  with a bullet marker
- markdown/classify: strip a leading bullet wrapped in a bold/italic run
  (e.g. "**● Label:**" → "- **Label:**") — some PDFs put the marker inside
  the same bold run as the label

Co-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
Abimael Martell
2026-04-20 22:47:34 -07:00
committed by GitHub
co-authored by Claude Opus 4.7
parent 2e933ba8c1
commit f44509ac3a
3 changed files with 166 additions and 1 deletions
+65
View File
@@ -64,6 +64,22 @@ pub(crate) fn is_caption_line(text: &str) -> bool {
false
}
/// Check if text starts with an unambiguous bullet marker (●, •, ○, ◦).
///
/// Narrower than [`is_list_item`]: it excludes numbered/lettered patterns
/// like `1.` or `a)`, which legitimately appear as section headings in many
/// documents. Used by the heading classifier to reject bullet lines without
/// also demoting numbered headings.
pub(crate) fn starts_with_bullet_marker(text: &str) -> bool {
let trimmed = text.trim_start();
trimmed.starts_with("")
|| trimmed.starts_with("")
|| trimmed.starts_with("")
|| trimmed.starts_with("")
|| trimmed.starts_with("- ")
|| trimmed.starts_with("* ")
}
/// Check if text looks like a list item
pub(crate) fn is_list_item(text: &str) -> bool {
let trimmed = text.trim_start();
@@ -115,6 +131,16 @@ pub(crate) fn format_list_item(text: &str) -> String {
if let Some(rest) = trimmed.strip_prefix(*bullet) {
return format!("- {}", rest.trim_start());
}
// Bullet inside a leading bold/italic run (e.g. "**● Label:** rest").
// The run wraps both the marker and the following label because both
// use a bold font in the PDF.
for wrapper in ["**", "*"] {
if let Some(after_open) = trimmed.strip_prefix(wrapper) {
if let Some(rest) = after_open.strip_prefix(*bullet) {
return format!("- {}{}", wrapper, rest.trim_start());
}
}
}
}
if trimmed.starts_with("- ") || trimmed.starts_with("* ") {
@@ -198,3 +224,42 @@ pub(crate) fn is_monospace_font(font_name: &str) -> bool {
patterns.iter().any(|p| lower.contains(p))
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn format_list_item_plain_bullet() {
assert_eq!(format_list_item("● Item"), "- Item");
assert_eq!(format_list_item("• Item"), "- Item");
}
#[test]
fn format_list_item_bullet_inside_bold() {
// PDF that uses bold font for both the marker and the label produces
// a single bold run like "**● Label:** rest"; the bullet must still
// be stripped and the bold wrapper preserved on the label.
assert_eq!(
format_list_item("**● Fraud: Willing cooperation;**"),
"- **Fraud: Willing cooperation;**"
);
assert_eq!(
format_list_item("**● Label:** rest of line"),
"- **Label:** rest of line"
);
assert_eq!(format_list_item("*● Italic:* rest"), "- *Italic:* rest");
}
#[test]
fn format_list_item_already_dash() {
assert_eq!(format_list_item("- existing"), "- existing");
}
#[test]
fn is_list_item_with_bullet_space() {
assert!(is_list_item("● Item"));
assert!(is_list_item("• Item"));
assert!(is_list_item("- Item"));
}
}