Compare commits
19
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ef96c68e90 | ||
|
|
c72fc8a221 | ||
|
|
c6f291aafa | ||
|
|
faed30c1a6 | ||
|
|
b630f8c308 | ||
|
|
4c604b79dd | ||
|
|
db03d4f13b | ||
|
|
bae8d34e97 | ||
|
|
d2306fed74 | ||
|
|
6bd28fb37f | ||
|
|
2be669bf7b | ||
|
|
adc54151b6 | ||
|
|
c60a0f3825 | ||
|
|
b2e202866c | ||
|
|
4252959e63 | ||
|
|
1a783417b7 | ||
|
|
59c7eda117 | ||
|
|
489d0b9966 | ||
|
|
173cc76f44 |
+769
-15
@@ -960,22 +960,713 @@ fn spans_multiple_columns(item: &TextItem, columns: &[ColumnRegion]) -> bool {
|
||||
overlap_count >= 2
|
||||
}
|
||||
|
||||
/// Check if a text item is likely a page number
|
||||
fn is_page_number(item: &TextItem) -> bool {
|
||||
const PAGE_NUMBER_Y_TOLERANCE: f32 = 3.0;
|
||||
const PAGE_NUMBER_CONTEXT_GAP_EM: f32 = 1.5;
|
||||
const PAGE_NUMBER_BOTTOM_Y: f32 = 100.0;
|
||||
const PAGE_NUMBER_TOP_Y: f32 = 720.0;
|
||||
const SPREAD_MIN_CONTENT_WIDTH_EM: f32 = 40.0;
|
||||
const SPREAD_EDGE_FRACTION: f32 = 0.25;
|
||||
const ADJACENT_PAGE_MIN_CONTENT_WIDTH_EM: f32 = 26.0;
|
||||
|
||||
type ContextualCandidateOccurrence = (u32, f32, Vec<(usize, u32)>);
|
||||
|
||||
fn page_number_value(item: &TextItem) -> Option<u32> {
|
||||
if !matches!(
|
||||
item.item_type,
|
||||
crate::types::ItemType::Text | crate::types::ItemType::FormField
|
||||
) {
|
||||
return None;
|
||||
}
|
||||
|
||||
let text = item.text.trim();
|
||||
|
||||
// Must be 1-4 digits only
|
||||
if text.is_empty() || text.len() > 4 {
|
||||
return false;
|
||||
}
|
||||
if !text.chars().all(|c| c.is_ascii_digit()) {
|
||||
return false;
|
||||
if text.is_empty() || text.len() > 4 || !text.chars().all(|c| c.is_ascii_digit()) {
|
||||
return None;
|
||||
}
|
||||
|
||||
// Must be at top or bottom of page.
|
||||
// US Letter = 792pt, A4 = 841pt. Page numbers are typically in the
|
||||
// top ~5% or bottom ~12% of the page.
|
||||
item.y > 720.0 || item.y < 100.0
|
||||
if item.y <= PAGE_NUMBER_TOP_Y && item.y >= PAGE_NUMBER_BOTTOM_Y {
|
||||
return None;
|
||||
}
|
||||
|
||||
text.parse().ok()
|
||||
}
|
||||
|
||||
/// Mark numeric slots that advance inside a repeated deep-margin line.
|
||||
///
|
||||
/// A folio can be emitted as part of a footer text run (for example,
|
||||
/// `42 Company report`) and therefore look contextual on a single page. Across
|
||||
/// the document, however, the surrounding text and Y position repeat while the
|
||||
/// numeric slot advances. Require that full signal before treating the slot as
|
||||
/// a folio so constant metadata and substantive rows near the page edge remain
|
||||
/// untouched.
|
||||
fn mark_repeated_folio_candidates(
|
||||
occurrences_by_signature: HashMap<String, Vec<ContextualCandidateOccurrence>>,
|
||||
document_page_count: usize,
|
||||
explicit_folio: &mut [bool],
|
||||
) {
|
||||
// Both thresholds are evidence floors: short documents still need four
|
||||
// occurrences, while long documents also need meaningful coverage. Using
|
||||
// `min` here would make one occurrence sufficient in a one-page document.
|
||||
let min_pages = 4usize.max((document_page_count * 30).div_ceil(100));
|
||||
|
||||
for occurrences in occurrences_by_signature.into_values() {
|
||||
if occurrences.len() < min_pages {
|
||||
continue;
|
||||
}
|
||||
|
||||
let distinct_pages: HashSet<u32> = occurrences.iter().map(|(page, _, _)| *page).collect();
|
||||
// Repeated table rows or duplicated drawing labels can share a
|
||||
// signature multiple times on one page. They are not running folios.
|
||||
if distinct_pages.len() != occurrences.len() || distinct_pages.len() < min_pages {
|
||||
continue;
|
||||
}
|
||||
|
||||
let min_y = occurrences
|
||||
.iter()
|
||||
.map(|(_, y, _)| *y)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let max_y = occurrences
|
||||
.iter()
|
||||
.map(|(_, y, _)| *y)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
if max_y - min_y >= PAGE_NUMBER_Y_TOLERANCE {
|
||||
continue;
|
||||
}
|
||||
|
||||
let slot_count = occurrences[0].2.len();
|
||||
if slot_count == 0
|
||||
|| occurrences
|
||||
.iter()
|
||||
.any(|(_, _, candidates)| candidates.len() != slot_count)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
for slot in 0..slot_count {
|
||||
let mut values: Vec<(u32, u32, usize)> = occurrences
|
||||
.iter()
|
||||
.map(|(page, _, candidates)| {
|
||||
let (index, value) = candidates[slot];
|
||||
(*page, value, index)
|
||||
})
|
||||
.collect();
|
||||
values.sort_by_key(|(page, _, _)| *page);
|
||||
|
||||
let unique_values: HashSet<u32> = values.iter().map(|(_, value, _)| *value).collect();
|
||||
let mostly_unique = unique_values.len() * 5 >= values.len() * 4;
|
||||
let page_tracking_pairs = values
|
||||
.windows(2)
|
||||
.filter(|pair| {
|
||||
let page_delta = pair[1].0 - pair[0].0;
|
||||
let value_delta = pair[1].1.saturating_sub(pair[0].1);
|
||||
value_delta == page_delta || value_delta == page_delta.saturating_mul(2)
|
||||
})
|
||||
.count();
|
||||
let mostly_tracks_page_order = page_tracking_pairs * 5 >= (values.len() - 1) * 4;
|
||||
// A running folio can be offset by front matter or advance twice per
|
||||
// PDF page in a two-page spread, but its magnitude should still be
|
||||
// plausible for the document. This keeps changing metadata such as
|
||||
// a sequence of years from becoming a deletion signal.
|
||||
let max_plausible_folio = (document_page_count as u32).saturating_mul(4).max(100);
|
||||
let plausible_magnitude = values
|
||||
.iter()
|
||||
.all(|(_, value, _)| *value <= max_plausible_folio);
|
||||
|
||||
if mostly_unique && mostly_tracks_page_order && plausible_magnitude {
|
||||
for (_, _, index) in values {
|
||||
explicit_folio[index] = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Mark the contextual half of a facing-page folio pair.
|
||||
///
|
||||
/// A landscape PDF can contain two printed pages per PDF page. One folio may be
|
||||
/// isolated while the other touches footer text; they remain a pair because
|
||||
/// they are consecutive, share a deep-margin baseline, and sit on opposite
|
||||
/// sides of the spread. The isolated half is strong evidence that the touching
|
||||
/// half is also a folio.
|
||||
fn mark_spread_folio_pairs(
|
||||
items: &[TextItem],
|
||||
candidate_values: &[Option<u32>],
|
||||
contextual: &[bool],
|
||||
explicit_folio: &mut [bool],
|
||||
) {
|
||||
let mut candidates_by_page: HashMap<u32, Vec<usize>> = HashMap::new();
|
||||
let mut page_bounds: HashMap<u32, (f32, f32)> = HashMap::new();
|
||||
for (index, value) in candidate_values.iter().enumerate() {
|
||||
if value.is_some() {
|
||||
candidates_by_page
|
||||
.entry(items[index].page)
|
||||
.or_default()
|
||||
.push(index);
|
||||
}
|
||||
}
|
||||
for item in items.iter().filter(|item| !item.text.trim().is_empty()) {
|
||||
let bounds = page_bounds
|
||||
.entry(item.page)
|
||||
.or_insert((f32::INFINITY, f32::NEG_INFINITY));
|
||||
bounds.0 = bounds.0.min(item.x);
|
||||
bounds.1 = bounds.1.max(item.x + effective_width(item));
|
||||
}
|
||||
|
||||
for (page, page_candidates) in candidates_by_page {
|
||||
let Some(&(page_left, page_right)) = page_bounds.get(&page) else {
|
||||
continue;
|
||||
};
|
||||
let page_width = page_right - page_left;
|
||||
if page_width <= 0.0 {
|
||||
continue;
|
||||
}
|
||||
let left_edge = page_left + page_width * SPREAD_EDGE_FRACTION;
|
||||
let right_edge = page_right - page_width * SPREAD_EDGE_FRACTION;
|
||||
let max_pair_font_size = page_width / SPREAD_MIN_CONTENT_WIDTH_EM;
|
||||
let edge_side = |index: usize| {
|
||||
let center = items[index].x + effective_width(&items[index]) / 2.0;
|
||||
if center <= left_edge {
|
||||
Some(false)
|
||||
} else if center >= right_edge {
|
||||
Some(true)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
};
|
||||
|
||||
// Index strong folio evidence by value and spread edge. Sorted
|
||||
// baselines let each contextual candidate query only the two adjacent
|
||||
// values on the opposite edge in O(log n), rather than comparing every
|
||||
// candidate pair on numeric-heavy pages.
|
||||
let mut known_baselines: HashMap<(u32, bool), Vec<f32>> = HashMap::new();
|
||||
for &index in &page_candidates {
|
||||
if (contextual[index] && !explicit_folio[index])
|
||||
|| items[index].font_size >= max_pair_font_size
|
||||
{
|
||||
continue;
|
||||
}
|
||||
let Some(side) = edge_side(index) else {
|
||||
continue;
|
||||
};
|
||||
let value = candidate_values[index].unwrap();
|
||||
known_baselines
|
||||
.entry((value, side))
|
||||
.or_default()
|
||||
.push(items[index].y);
|
||||
}
|
||||
for baselines in known_baselines.values_mut() {
|
||||
baselines.sort_by(f32::total_cmp);
|
||||
}
|
||||
|
||||
for index in page_candidates {
|
||||
if !contextual[index]
|
||||
|| explicit_folio[index]
|
||||
|| items[index].font_size >= max_pair_font_size
|
||||
{
|
||||
continue;
|
||||
}
|
||||
let Some(value) = candidate_values[index] else {
|
||||
continue;
|
||||
};
|
||||
let Some(side) = edge_side(index) else {
|
||||
continue;
|
||||
};
|
||||
let y = items[index].y;
|
||||
let paired = [value.checked_sub(1), value.checked_add(1)]
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|other_value| known_baselines.get(&(other_value, !side)))
|
||||
.any(|baselines| {
|
||||
let first = baselines
|
||||
.partition_point(|baseline| *baseline <= y - PAGE_NUMBER_Y_TOLERANCE);
|
||||
baselines
|
||||
.get(first)
|
||||
.is_some_and(|baseline| *baseline < y + PAGE_NUMBER_Y_TOLERANCE)
|
||||
});
|
||||
if paired {
|
||||
explicit_folio[index] = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Mark a contextual folio that alternates with an isolated folio on the
|
||||
/// neighboring PDF page.
|
||||
///
|
||||
/// Facing pages commonly put folios on opposite outer edges. A running header
|
||||
/// can touch the right-hand folio while the next left-hand folio is isolated.
|
||||
/// A single isolated candidate is not enough to remove nearby contextual text.
|
||||
/// Require a second pre-existing anchor in the same advancing sequence, along
|
||||
/// with a genuinely wide content span, stable baselines/font sizes, and
|
||||
/// alternating outer edges.
|
||||
fn mark_adjacent_page_folio_pairs(
|
||||
items: &[TextItem],
|
||||
candidate_values: &[Option<u32>],
|
||||
contextual: &[bool],
|
||||
document_page_count: usize,
|
||||
explicit_folio: &mut [bool],
|
||||
) {
|
||||
let max_plausible_folio = (document_page_count as u32).saturating_mul(4).max(100);
|
||||
// Do not let newly inferred candidates recursively become evidence for
|
||||
// later candidates; every match must be anchored by evidence established
|
||||
// before this cross-page pass.
|
||||
let strong_folio_evidence = explicit_folio.to_vec();
|
||||
let mut page_bounds: HashMap<u32, (f32, f32)> = HashMap::new();
|
||||
for item in items.iter().filter(|item| !item.text.trim().is_empty()) {
|
||||
let bounds = page_bounds
|
||||
.entry(item.page)
|
||||
.or_insert((f32::INFINITY, f32::NEG_INFINITY));
|
||||
bounds.0 = bounds.0.min(item.x);
|
||||
bounds.1 = bounds.1.max(item.x + effective_width(item));
|
||||
}
|
||||
|
||||
let edge_side = |index: usize| {
|
||||
let &(page_left, page_right) = page_bounds.get(&items[index].page)?;
|
||||
let page_width = page_right - page_left;
|
||||
if page_width < items[index].font_size * ADJACENT_PAGE_MIN_CONTENT_WIDTH_EM {
|
||||
return None;
|
||||
}
|
||||
let center = items[index].x + effective_width(&items[index]) / 2.0;
|
||||
let left_edge = page_left + page_width * SPREAD_EDGE_FRACTION;
|
||||
let right_edge = page_right - page_width * SPREAD_EDGE_FRACTION;
|
||||
if center <= left_edge {
|
||||
Some(false)
|
||||
} else if center >= right_edge {
|
||||
Some(true)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
};
|
||||
|
||||
// Anchor sequences by outer edge and the value/page offset. This enforces
|
||||
// forward page tracking and lets candidates query adjacent pages directly,
|
||||
// while a baseline-sorted index finds a second independent anchor without
|
||||
// a document-wide quadratic scan.
|
||||
let mut anchors_by_page: HashMap<(u32, bool, i64), Vec<usize>> = HashMap::new();
|
||||
let mut anchors_by_sequence: HashMap<(bool, i64), Vec<usize>> = HashMap::new();
|
||||
for (index, value) in candidate_values.iter().enumerate() {
|
||||
let Some(value) = value else {
|
||||
continue;
|
||||
};
|
||||
if *value > max_plausible_folio || (contextual[index] && !strong_folio_evidence[index]) {
|
||||
continue;
|
||||
}
|
||||
let Some(side) = edge_side(index) else {
|
||||
continue;
|
||||
};
|
||||
let page = items[index].page;
|
||||
let offset = i64::from(*value) - i64::from(page);
|
||||
anchors_by_page
|
||||
.entry((page, side, offset))
|
||||
.or_default()
|
||||
.push(index);
|
||||
anchors_by_sequence
|
||||
.entry((side, offset))
|
||||
.or_default()
|
||||
.push(index);
|
||||
}
|
||||
for anchors in anchors_by_sequence.values_mut() {
|
||||
anchors.sort_by(|&left, &right| items[left].y.total_cmp(&items[right].y));
|
||||
}
|
||||
|
||||
for (index, value) in candidate_values.iter().enumerate() {
|
||||
let Some(value) = value else {
|
||||
continue;
|
||||
};
|
||||
if !contextual[index] || explicit_folio[index] || *value > max_plausible_folio {
|
||||
continue;
|
||||
}
|
||||
let Some(side) = edge_side(index) else {
|
||||
continue;
|
||||
};
|
||||
|
||||
let page = items[index].page;
|
||||
let offset = i64::from(*value) - i64::from(page);
|
||||
let Some(sequence_anchors) = anchors_by_sequence.get(&(!side, offset)) else {
|
||||
continue;
|
||||
};
|
||||
let neighbor = [page.checked_sub(1), page.checked_add(1)]
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|neighbor_page| anchors_by_page.get(&(neighbor_page, !side, offset)))
|
||||
.flatten()
|
||||
.copied()
|
||||
.find(|&neighbor_index| {
|
||||
(items[index].y - items[neighbor_index].y).abs() < PAGE_NUMBER_Y_TOLERANCE
|
||||
&& (items[index].font_size - items[neighbor_index].font_size).abs() < 1.0
|
||||
});
|
||||
let Some(neighbor_index) = neighbor else {
|
||||
continue;
|
||||
};
|
||||
|
||||
let first = sequence_anchors.partition_point(|&anchor_index| {
|
||||
items[anchor_index].y <= items[index].y - PAGE_NUMBER_Y_TOLERANCE
|
||||
});
|
||||
let has_second_anchor = sequence_anchors[first..]
|
||||
.iter()
|
||||
.take_while(|&&anchor_index| {
|
||||
items[anchor_index].y < items[index].y + PAGE_NUMBER_Y_TOLERANCE
|
||||
})
|
||||
.any(|&anchor_index| {
|
||||
items[anchor_index].page != page
|
||||
&& items[anchor_index].page != items[neighbor_index].page
|
||||
&& (items[index].font_size - items[anchor_index].font_size).abs() < 1.0
|
||||
});
|
||||
if has_second_anchor {
|
||||
explicit_folio[index] = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Identify page-edge numeric items that belong to a nearby content run.
|
||||
///
|
||||
/// Numeric candidates on their own do not establish context for one another.
|
||||
/// A connected same-baseline run is contextual only when it also contains a
|
||||
/// non-candidate item, preserving lines such as `Chapter 1 2026` while still
|
||||
/// removing isolated numeric footer clusters.
|
||||
fn page_number_context_masks(
|
||||
items: &[TextItem],
|
||||
candidate_values: &[Option<u32>],
|
||||
document_page_count: usize,
|
||||
) -> (Vec<bool>, Vec<bool>) {
|
||||
let mut contextual = vec![false; items.len()];
|
||||
let mut explicit_folio = vec![false; items.len()];
|
||||
let mut occurrences_by_signature: HashMap<String, Vec<ContextualCandidateOccurrence>> =
|
||||
HashMap::new();
|
||||
let mut indices_by_page: HashMap<u32, Vec<usize>> = HashMap::new();
|
||||
for (index, item) in items.iter().enumerate() {
|
||||
if matches!(
|
||||
item.item_type,
|
||||
crate::types::ItemType::Text | crate::types::ItemType::FormField
|
||||
) && !item.text.trim().is_empty()
|
||||
{
|
||||
indices_by_page.entry(item.page).or_default().push(index);
|
||||
}
|
||||
}
|
||||
for mut page_indices in indices_by_page.into_values() {
|
||||
page_indices.sort_by(|&left, &right| {
|
||||
items[right]
|
||||
.y
|
||||
.total_cmp(&items[left].y)
|
||||
.then(items[left].x.total_cmp(&items[right].x))
|
||||
});
|
||||
|
||||
let mut rows: Vec<Vec<usize>> = Vec::new();
|
||||
for index in page_indices {
|
||||
if rows.last().is_some_and(|row| {
|
||||
(items[row[0]].y - items[index].y).abs() < PAGE_NUMBER_Y_TOLERANCE
|
||||
}) {
|
||||
rows.last_mut().unwrap().push(index);
|
||||
} else {
|
||||
rows.push(vec![index]);
|
||||
}
|
||||
}
|
||||
|
||||
for mut row in rows {
|
||||
row.sort_by(|&left, &right| items[left].x.total_cmp(&items[right].x));
|
||||
let mut start = 0;
|
||||
while start < row.len() {
|
||||
let mut end = start + 1;
|
||||
let first = &items[row[start]];
|
||||
let mut group_right = first.x + effective_width(first);
|
||||
let mut group_font_size = first.font_size;
|
||||
|
||||
while end < row.len() {
|
||||
let item = &items[row[end]];
|
||||
let gap = item.x - group_right;
|
||||
if gap > group_font_size.max(item.font_size) * PAGE_NUMBER_CONTEXT_GAP_EM {
|
||||
break;
|
||||
}
|
||||
group_right = group_right.max(item.x + effective_width(item));
|
||||
group_font_size = group_font_size.max(item.font_size);
|
||||
end += 1;
|
||||
}
|
||||
|
||||
let group = &row[start..end];
|
||||
let has_lexical_context = group.iter().any(|&index| {
|
||||
candidate_values[index].is_none()
|
||||
&& items[index]
|
||||
.text
|
||||
.chars()
|
||||
.any(|character| character.is_alphabetic())
|
||||
});
|
||||
// Numeric data near a page edge also needs protection, but a
|
||||
// lone long integer beside a short candidate is not enough to
|
||||
// establish context. Preserve explicit numeric structures
|
||||
// (list markers, ranges, comma-formatted values, dotted index
|
||||
// entries) and dense runs with at least one long integer.
|
||||
let numeric_like = |text: &str| {
|
||||
text.chars().any(|character| character.is_numeric())
|
||||
&& !text.chars().any(|character| character.is_alphabetic())
|
||||
};
|
||||
let is_structured_numeric_context = |index: usize| {
|
||||
if candidate_values[index].is_some() {
|
||||
return false;
|
||||
}
|
||||
let text = items[index].text.trim();
|
||||
numeric_like(text)
|
||||
&& text
|
||||
.chars()
|
||||
.any(|character| !character.is_numeric() && !character.is_whitespace())
|
||||
};
|
||||
let has_structured_numeric_context = group
|
||||
.iter()
|
||||
.any(|&index| is_structured_numeric_context(index));
|
||||
let numeric_item_count = group
|
||||
.iter()
|
||||
.filter(|&&index| numeric_like(items[index].text.trim()))
|
||||
.count();
|
||||
let has_long_integer = group.iter().any(|&index| {
|
||||
candidate_values[index].is_none()
|
||||
&& items[index]
|
||||
.text
|
||||
.trim()
|
||||
.chars()
|
||||
.all(|character| character.is_numeric())
|
||||
});
|
||||
let has_dense_numeric_context = numeric_item_count >= 3 && has_long_integer;
|
||||
let has_context = has_lexical_context
|
||||
|| has_structured_numeric_context
|
||||
|| has_dense_numeric_context;
|
||||
let has_candidate = row[start..end]
|
||||
.iter()
|
||||
.any(|&index| candidate_values[index].is_some());
|
||||
// Decorative centered folios have no lexical context, so
|
||||
// recognize the complete delimiter-number-delimiter triplet
|
||||
// before the contextual-content gate. This prevents `- 42 -`
|
||||
// from leaving a malformed `- -` line.
|
||||
if group.len() == 3
|
||||
&& items[group[0]].text.trim() == "-"
|
||||
&& candidate_values[group[1]].is_some()
|
||||
&& items[group[2]].text.trim() == "-"
|
||||
{
|
||||
for &index in group {
|
||||
explicit_folio[index] = true;
|
||||
}
|
||||
}
|
||||
if has_context && has_candidate {
|
||||
let group = &row[start..end];
|
||||
let group_text = group
|
||||
.iter()
|
||||
.map(|&index| items[index].text.trim())
|
||||
.filter(|text| !text.is_empty())
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
let group_is_folio =
|
||||
crate::text_utils::is_explicit_page_number_expression(&group_text);
|
||||
let context_text = group
|
||||
.iter()
|
||||
.filter(|&&index| candidate_values[index].is_none())
|
||||
.map(|&index| items[index].text.trim())
|
||||
.filter(|text| !text.is_empty())
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
let candidates: Vec<(usize, u32)> = group
|
||||
.iter()
|
||||
.filter_map(|&index| candidate_values[index].map(|value| (index, value)))
|
||||
.collect();
|
||||
// Recurrence is only evidence for numeric slots at the
|
||||
// outer boundary of a contextual run. An embedded number
|
||||
// in repeated prose such as `Page 42 explains the result`
|
||||
// is substantive content, not a running folio.
|
||||
let recurrence_candidates: Vec<(usize, u32)> = candidates
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|(index, _)| {
|
||||
group.first() == Some(index) || group.last() == Some(index)
|
||||
})
|
||||
.collect();
|
||||
let in_deep_margin = candidates.iter().all(|(index, _)| {
|
||||
items[*index].y < PAGE_NUMBER_BOTTOM_Y
|
||||
|| items[*index].y > PAGE_NUMBER_TOP_Y
|
||||
});
|
||||
if in_deep_margin
|
||||
&& !recurrence_candidates.is_empty()
|
||||
&& context_text
|
||||
.chars()
|
||||
.filter(|character| character.is_alphanumeric())
|
||||
.count()
|
||||
>= 8
|
||||
{
|
||||
let signature = group
|
||||
.iter()
|
||||
.map(|&index| {
|
||||
if candidate_values[index].is_some() {
|
||||
"{number}".to_string()
|
||||
} else {
|
||||
items[index]
|
||||
.text
|
||||
.split_whitespace()
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ")
|
||||
.to_lowercase()
|
||||
}
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
occurrences_by_signature
|
||||
.entry(signature)
|
||||
.or_default()
|
||||
.push((
|
||||
items[group[0]].page,
|
||||
items[group[0]].y,
|
||||
recurrence_candidates,
|
||||
));
|
||||
}
|
||||
for (position, &index) in group.iter().enumerate() {
|
||||
if let Some(value) = candidate_values[index] {
|
||||
let adjacent_context = [position.checked_sub(1), Some(position + 1)]
|
||||
.into_iter()
|
||||
.flatten()
|
||||
.filter_map(|position| group.get(position).copied())
|
||||
.any(|adjacent| {
|
||||
candidate_values[adjacent].is_none()
|
||||
&& items[adjacent].text.chars().any(|character| {
|
||||
!character.is_numeric() && !character.is_whitespace()
|
||||
})
|
||||
});
|
||||
let max_plausible_folio =
|
||||
(document_page_count as u32).saturating_mul(4).max(100);
|
||||
// Large year/identifier-like values stay attached
|
||||
// to their lexical run even when a smaller numeric
|
||||
// candidate sits between them and the text.
|
||||
let implausible_folio_with_lexical_context =
|
||||
value > max_plausible_folio && has_lexical_context;
|
||||
contextual[index] = adjacent_context
|
||||
|| has_dense_numeric_context
|
||||
|| implausible_folio_with_lexical_context;
|
||||
let previous = position
|
||||
.checked_sub(1)
|
||||
.map(|position| items[group[position]].text.trim());
|
||||
let next = group
|
||||
.get(position + 1)
|
||||
.map(|&index| items[index].text.trim());
|
||||
let follows_page_label =
|
||||
previous.is_some_and(|text| text.eq_ignore_ascii_case("page"));
|
||||
let starts_of_expression =
|
||||
next.is_some_and(|text| text.eq_ignore_ascii_case("of"));
|
||||
let is_centered_folio = previous == Some("-") && next == Some("-");
|
||||
explicit_folio[index] |= group_is_folio
|
||||
&& (follows_page_label
|
||||
|| starts_of_expression
|
||||
|| is_centered_folio);
|
||||
}
|
||||
}
|
||||
// Remove the complete labeled expression rather than
|
||||
// leaving fragments such as `Page of 15`. A trailing
|
||||
// running-header suffix remains untouched.
|
||||
if group_is_folio
|
||||
&& group.len() >= 4
|
||||
&& items[group[0]].text.trim().eq_ignore_ascii_case("page")
|
||||
&& candidate_values[group[1]].is_some()
|
||||
&& items[group[2]].text.trim().eq_ignore_ascii_case("of")
|
||||
&& items[group[3]]
|
||||
.text
|
||||
.trim()
|
||||
.chars()
|
||||
.all(|character| character.is_ascii_digit())
|
||||
{
|
||||
for &index in &group[..4] {
|
||||
explicit_folio[index] = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
start = end;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
mark_repeated_folio_candidates(
|
||||
occurrences_by_signature,
|
||||
document_page_count,
|
||||
&mut explicit_folio,
|
||||
);
|
||||
mark_spread_folio_pairs(items, candidate_values, &contextual, &mut explicit_folio);
|
||||
mark_adjacent_page_folio_pairs(
|
||||
items,
|
||||
candidate_values,
|
||||
&contextual,
|
||||
document_page_count,
|
||||
&mut explicit_folio,
|
||||
);
|
||||
|
||||
(contextual, explicit_folio)
|
||||
}
|
||||
|
||||
/// Decide which digit-only page-edge items can be removed before layout.
|
||||
///
|
||||
/// PDF producers commonly emit one text-showing operation per word. A numeric
|
||||
/// item attached to neighboring content on the same baseline is therefore kept.
|
||||
/// Complete page-number expressions such as `Page 42` remain removable even
|
||||
/// though their numeric item has lexical context.
|
||||
fn page_number_removal_mask(items: &[TextItem], document_page_count: usize) -> Vec<bool> {
|
||||
let candidate_values: Vec<Option<u32>> = items.iter().map(page_number_value).collect();
|
||||
let (contextual, explicit_folio) =
|
||||
page_number_context_masks(items, &candidate_values, document_page_count);
|
||||
|
||||
candidate_values
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(index, value)| explicit_folio[index] || (value.is_some() && !contextual[index]))
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Return whether selected-page extraction contains a page-edge number whose
|
||||
/// folio status depends on evidence from other pages. Isolated and explicitly
|
||||
/// labeled folios can be decided locally; only contextual candidates require
|
||||
/// a document-wide extraction pass.
|
||||
pub(super) fn needs_document_page_number_context(
|
||||
items: &[TextItem],
|
||||
document_page_count: usize,
|
||||
) -> bool {
|
||||
let candidate_values: Vec<Option<u32>> = items.iter().map(page_number_value).collect();
|
||||
let (contextual, explicit_folio) =
|
||||
page_number_context_masks(items, &candidate_values, document_page_count);
|
||||
|
||||
candidate_values
|
||||
.iter()
|
||||
.enumerate()
|
||||
.any(|(index, value)| value.is_some() && contextual[index] && !explicit_folio[index])
|
||||
}
|
||||
|
||||
/// Remove numeric folios using complete document context before downstream
|
||||
/// non-table layout partitions could separate the evidence needed to recognize
|
||||
/// them.
|
||||
#[cfg(test)]
|
||||
pub(crate) fn filter_markdown_page_numbers(
|
||||
items: Vec<TextItem>,
|
||||
document_page_count: u32,
|
||||
) -> Vec<TextItem> {
|
||||
filter_markdown_page_numbers_with_removed_pages(items, document_page_count).0
|
||||
}
|
||||
|
||||
/// Filter Markdown folios while retaining the pages where items were removed.
|
||||
///
|
||||
/// The page set lets downstream table-continuation classification preserve its
|
||||
/// pre-filter semantics even though structural layout consumes the cleaned
|
||||
/// item collection.
|
||||
pub(crate) fn filter_markdown_page_numbers_with_removed_pages(
|
||||
items: Vec<TextItem>,
|
||||
document_page_count: u32,
|
||||
) -> (Vec<TextItem>, HashSet<u32>, Vec<bool>) {
|
||||
let remove = page_number_removal_mask(&items, document_page_count as usize);
|
||||
let mut removed_pages = HashSet::new();
|
||||
let items = items
|
||||
.into_iter()
|
||||
.zip(remove.iter().copied())
|
||||
.filter_map(|(item, remove)| {
|
||||
if remove {
|
||||
removed_pages.insert(item.page);
|
||||
None
|
||||
} else {
|
||||
Some(item)
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
(items, removed_pages, remove)
|
||||
}
|
||||
|
||||
/// Group text items into lines, with multi-column support
|
||||
@@ -1205,6 +1896,27 @@ pub(crate) fn group_into_lines_with_thresholds_and_charts(
|
||||
)
|
||||
}
|
||||
|
||||
/// Group items after document-level page-number filtering has already run.
|
||||
///
|
||||
/// Partitioned Markdown layout uses this path so a contextual candidate that
|
||||
/// was preserved with its complete baseline context is not reconsidered after
|
||||
/// its neighboring text lands in another band or chart/prose zone.
|
||||
pub(crate) fn group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
table_pages: &HashSet<u32>,
|
||||
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
|
||||
) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds_and_regions_impl(
|
||||
items,
|
||||
page_thresholds,
|
||||
table_pages,
|
||||
chart_regions,
|
||||
&HashMap::new(),
|
||||
false,
|
||||
)
|
||||
}
|
||||
|
||||
pub(crate) fn group_into_lines_with_thresholds_and_regions(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
@@ -1222,6 +1934,23 @@ pub(crate) fn group_into_lines_with_thresholds_and_regions(
|
||||
)
|
||||
}
|
||||
|
||||
pub(crate) fn group_prefiltered_items_into_lines_with_thresholds_and_regions(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
table_pages: &HashSet<u32>,
|
||||
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
|
||||
image_regions: &HashMap<u32, Vec<super::reading_order::ImageRegion>>,
|
||||
) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds_and_regions_impl(
|
||||
items,
|
||||
page_thresholds,
|
||||
table_pages,
|
||||
chart_regions,
|
||||
image_regions,
|
||||
false,
|
||||
)
|
||||
}
|
||||
|
||||
fn group_into_lines_with_thresholds_and_regions_impl(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
@@ -1234,12 +1963,25 @@ fn group_into_lines_with_thresholds_and_regions_impl(
|
||||
return Vec::new();
|
||||
}
|
||||
|
||||
// Markdown output omits standalone numeric headers/footers. Plain-text
|
||||
// callers opt out because dropping extracted text violates that API.
|
||||
// Markdown output omits standalone numeric headers/footers. Determine
|
||||
// standalone status from rough baseline context before layout analysis so
|
||||
// removed page numbers cannot affect column detection. Plain-text callers
|
||||
// opt out because dropping extracted text violates that API.
|
||||
let items = if filter_page_numbers {
|
||||
// Item-only grouping has no document metadata, so use the highest
|
||||
// observed 1-based page as its best available coverage denominator.
|
||||
// The Markdown document path passes the authoritative PDF page count
|
||||
// through `filter_markdown_page_numbers` before reaching this helper.
|
||||
let observed_page_count = items
|
||||
.iter()
|
||||
.map(|item| item.page as usize)
|
||||
.max()
|
||||
.unwrap_or(0);
|
||||
let remove = page_number_removal_mask(&items, observed_page_count);
|
||||
items
|
||||
.into_iter()
|
||||
.filter(|item| !is_page_number(item))
|
||||
.zip(remove)
|
||||
.filter_map(|(item, remove)| (!remove).then_some(item))
|
||||
.collect()
|
||||
} else {
|
||||
items
|
||||
@@ -1254,6 +1996,15 @@ fn group_into_lines_with_thresholds_and_regions_impl(
|
||||
|
||||
for page in pages {
|
||||
let page_items: Vec<TextItem> = items.iter().filter(|i| i.page == page).cloned().collect();
|
||||
// Page-edge numeric runs are weak evidence for column geometry. Keep
|
||||
// contextual values for line assembly, but prevent their preservation
|
||||
// from changing the page's inferred layout.
|
||||
let column_detection_items: Vec<TextItem> = page_items
|
||||
.iter()
|
||||
.filter(|item| page_number_value(item).is_none())
|
||||
.cloned()
|
||||
.collect();
|
||||
let column_detection_items = column_detection_items.as_slice();
|
||||
|
||||
// Use pre-computed threshold from fix_letterspaced_items if available
|
||||
// (computed before embedded-space removal, with full signal).
|
||||
@@ -1265,9 +2016,9 @@ fn group_into_lines_with_thresholds_and_regions_impl(
|
||||
// their own positioned-region ordering and therefore stay on that path.
|
||||
if !chart_regions.contains_key(&page) {
|
||||
let preliminary_columns =
|
||||
detect_columns(&page_items, page, table_pages.contains(&page));
|
||||
detect_columns(column_detection_items, page, table_pages.contains(&page));
|
||||
let detected_split =
|
||||
(preliminary_columns.len() == 2).then_some(preliminary_columns[0].x_max);
|
||||
(preliminary_columns.len() == 2).then(|| preliminary_columns[0].x_max);
|
||||
if let Some(band) = image_regions.get(&page).and_then(|regions| {
|
||||
super::reading_order::infer_image_anchored_flow(
|
||||
&page_items,
|
||||
@@ -1307,6 +2058,9 @@ fn group_into_lines_with_thresholds_and_regions_impl(
|
||||
let col_input: Vec<TextItem> = page_items
|
||||
.iter()
|
||||
.filter(|it| {
|
||||
if page_number_value(it).is_some() {
|
||||
return false;
|
||||
}
|
||||
let cx = it.x + it.width / 2.0;
|
||||
// Tight bounds: this only blinds the histogram to
|
||||
// chart-internal text; rows adjacent to the chart
|
||||
@@ -1319,7 +2073,7 @@ fn group_into_lines_with_thresholds_and_regions_impl(
|
||||
.collect();
|
||||
detect_columns(&col_input, page, table_pages.contains(&page))
|
||||
}
|
||||
None => detect_columns(&page_items, page, table_pages.contains(&page)),
|
||||
None => detect_columns(column_detection_items, page, table_pages.contains(&page)),
|
||||
};
|
||||
|
||||
if columns.len() <= 1 {
|
||||
|
||||
+90
-2
@@ -153,16 +153,54 @@ pub(crate) fn extract_form_fields(
|
||||
},
|
||||
Err(_) => return items,
|
||||
};
|
||||
if fields.is_empty() {
|
||||
return items;
|
||||
}
|
||||
let annotation_pages = annotation_page_map(doc, page_map);
|
||||
|
||||
for field_obj in &fields {
|
||||
if let Ok(field_ref) = field_obj.as_reference() {
|
||||
walk_form_fields(doc, field_ref, None, "", page_map, &mut items);
|
||||
walk_form_fields(
|
||||
doc,
|
||||
field_ref,
|
||||
None,
|
||||
"",
|
||||
page_map,
|
||||
&annotation_pages,
|
||||
&mut items,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
items
|
||||
}
|
||||
|
||||
/// Map widget annotation objects back to the page whose `/Annots` array owns
|
||||
/// them. Some valid widgets omit `/P`, so the page tree is the only reliable
|
||||
/// ownership signal available for page-filtered extraction.
|
||||
fn annotation_page_map(
|
||||
doc: &Document,
|
||||
page_map: &HashMap<ObjectId, u32>,
|
||||
) -> HashMap<ObjectId, u32> {
|
||||
let mut annotation_pages = HashMap::new();
|
||||
for (&page_id, &page_num) in page_map {
|
||||
let Some(annotations) = doc
|
||||
.get_dictionary(page_id)
|
||||
.ok()
|
||||
.and_then(|page| page.get(b"Annots").ok())
|
||||
.and_then(|annotations| resolve_array(doc, annotations))
|
||||
else {
|
||||
continue;
|
||||
};
|
||||
for annotation in annotations {
|
||||
if let Ok(annotation_id) = annotation.as_reference() {
|
||||
annotation_pages.insert(annotation_id, page_num);
|
||||
}
|
||||
}
|
||||
}
|
||||
annotation_pages
|
||||
}
|
||||
|
||||
/// Recursively walk the form field tree, extracting leaf field values.
|
||||
pub(crate) fn walk_form_fields(
|
||||
doc: &Document,
|
||||
@@ -170,6 +208,7 @@ pub(crate) fn walk_form_fields(
|
||||
parent_ft: Option<&[u8]>,
|
||||
parent_name: &str,
|
||||
page_map: &HashMap<ObjectId, u32>,
|
||||
annotation_pages: &HashMap<ObjectId, u32>,
|
||||
items: &mut Vec<TextItem>,
|
||||
) {
|
||||
let field_dict = match doc.get_dictionary(field_id) {
|
||||
@@ -206,7 +245,15 @@ pub(crate) fn walk_form_fields(
|
||||
let kids = kids.clone();
|
||||
for kid in &kids {
|
||||
if let Ok(kid_ref) = kid.as_reference() {
|
||||
walk_form_fields(doc, kid_ref, ft, &full_name, page_map, items);
|
||||
walk_form_fields(
|
||||
doc,
|
||||
kid_ref,
|
||||
ft,
|
||||
&full_name,
|
||||
page_map,
|
||||
annotation_pages,
|
||||
items,
|
||||
);
|
||||
}
|
||||
}
|
||||
return;
|
||||
@@ -299,6 +346,7 @@ pub(crate) fn walk_form_fields(
|
||||
.ok()
|
||||
.and_then(|o| o.as_reference().ok())
|
||||
.and_then(|p| page_map.get(&p).copied())
|
||||
.or_else(|| annotation_pages.get(&field_id).copied())
|
||||
.unwrap_or(1);
|
||||
|
||||
let text = if full_name.is_empty() {
|
||||
@@ -324,3 +372,43 @@ pub(crate) fn walk_form_fields(
|
||||
mcid: None,
|
||||
});
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use lopdf::{dictionary, Object};
|
||||
|
||||
#[test]
|
||||
fn widget_without_page_reference_uses_owning_page_annotation() {
|
||||
let mut doc = Document::new();
|
||||
let widget_id = doc.add_object(dictionary! {
|
||||
"Type" => "Annot",
|
||||
"Subtype" => "Widget",
|
||||
"FT" => "Tx",
|
||||
"T" => Object::string_literal("customer"),
|
||||
"V" => Object::string_literal("Alice"),
|
||||
"Rect" => vec![10.into(), 20.into(), 110.into(), 40.into()],
|
||||
});
|
||||
let page_one_id = doc.add_object(dictionary! {
|
||||
"Type" => "Page",
|
||||
});
|
||||
let page_two_id = doc.add_object(dictionary! {
|
||||
"Type" => "Page",
|
||||
"Annots" => vec![Object::Reference(widget_id)],
|
||||
});
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"AcroForm" => dictionary! {
|
||||
"Fields" => vec![Object::Reference(widget_id)],
|
||||
},
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
|
||||
let page_map = HashMap::from([(page_one_id, 1), (page_two_id, 2)]);
|
||||
let items = extract_form_fields(&doc, &page_map);
|
||||
|
||||
assert_eq!(items.len(), 1);
|
||||
assert_eq!(items[0].page, 2);
|
||||
assert_eq!(items[0].text, "customer: Alice");
|
||||
}
|
||||
}
|
||||
|
||||
+826
-17
@@ -27,9 +27,12 @@ pub use crate::text_utils::{is_bold_font, is_italic_font};
|
||||
pub use crate::types::{ItemType, TextLine};
|
||||
pub(crate) use fonts::FontStyleCache;
|
||||
pub(crate) use layout::detect_columns;
|
||||
#[cfg(test)]
|
||||
use layout::filter_markdown_page_numbers;
|
||||
pub(crate) use layout::filter_markdown_page_numbers_with_removed_pages;
|
||||
pub(crate) use layout::group_into_lines_with_thresholds;
|
||||
pub(crate) use layout::group_into_lines_with_thresholds_and_charts;
|
||||
pub(crate) use layout::group_into_lines_with_thresholds_and_regions;
|
||||
pub(crate) use layout::group_prefiltered_items_into_lines_with_thresholds_and_charts;
|
||||
pub(crate) use layout::group_prefiltered_items_into_lines_with_thresholds_and_regions;
|
||||
pub(crate) use layout::is_newspaper_layout;
|
||||
pub(crate) use layout::ColumnRegion;
|
||||
pub use layout::{group_into_lines, group_into_lines_preserving_all_text};
|
||||
@@ -140,17 +143,92 @@ pub(crate) fn extract_positioned_text_from_doc(
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
extract_positioned_text_impl(doc, font_cmaps, page_filter, false)
|
||||
extract_positioned_text_impl(doc, font_cmaps, page_filter, false, None)
|
||||
}
|
||||
|
||||
/// Extract with option to include invisible (Tr=3) text.
|
||||
/// Used for Mixed/template PDFs where the OCR text layer is invisible.
|
||||
pub(crate) fn extract_positioned_text_include_invisible(
|
||||
/// Extract selected pages and gather document-wide folio evidence only when a
|
||||
/// selected page contains an ambiguous contextual page-edge number. Errors on
|
||||
/// selected pages remain fatal; errors on context-only pages are skipped.
|
||||
pub(crate) fn extract_positioned_text_with_folio_context(
|
||||
doc: &Document,
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
extract_positioned_text_impl(doc, font_cmaps, page_filter, true)
|
||||
extract_positioned_text_with_folio_context_impl(doc, font_cmaps, page_filter, false)
|
||||
}
|
||||
|
||||
/// Invisible-text variant of [`extract_positioned_text_with_folio_context`].
|
||||
pub(crate) fn extract_positioned_text_include_invisible_with_folio_context(
|
||||
doc: &Document,
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
extract_positioned_text_with_folio_context_impl(doc, font_cmaps, page_filter, true)
|
||||
}
|
||||
|
||||
fn extract_positioned_text_with_folio_context_impl(
|
||||
doc: &Document,
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
include_invisible: bool,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
let Some(required_pages) = page_filter else {
|
||||
return extract_positioned_text_impl(doc, font_cmaps, None, include_invisible, None);
|
||||
};
|
||||
|
||||
let (
|
||||
(mut selected_items, mut selected_rects, mut selected_lines),
|
||||
mut page_thresholds,
|
||||
mut gid_encoded_pages,
|
||||
) = extract_positioned_text_impl(
|
||||
doc,
|
||||
font_cmaps,
|
||||
Some(required_pages),
|
||||
include_invisible,
|
||||
None,
|
||||
)?;
|
||||
if !layout::needs_document_page_number_context(&selected_items, doc.get_pages().len()) {
|
||||
return Ok((
|
||||
(selected_items, selected_rects, selected_lines),
|
||||
page_thresholds,
|
||||
gid_encoded_pages,
|
||||
));
|
||||
}
|
||||
|
||||
let context_pages: HashSet<u32> = doc
|
||||
.get_pages()
|
||||
.keys()
|
||||
.copied()
|
||||
.filter(|page| !required_pages.contains(page))
|
||||
.collect();
|
||||
let ((context_items, context_rects, context_lines), context_thresholds, context_gid_pages) =
|
||||
extract_positioned_text_impl(
|
||||
doc,
|
||||
font_cmaps,
|
||||
Some(&context_pages),
|
||||
include_invisible,
|
||||
Some(required_pages),
|
||||
)?;
|
||||
selected_items.extend(context_items);
|
||||
selected_rects.extend(context_rects);
|
||||
selected_lines.extend(context_lines);
|
||||
page_thresholds.extend(context_thresholds);
|
||||
gid_encoded_pages.extend(context_gid_pages);
|
||||
Ok((
|
||||
(selected_items, selected_rects, selected_lines),
|
||||
page_thresholds,
|
||||
gid_encoded_pages,
|
||||
))
|
||||
}
|
||||
|
||||
/// Extract all pages for document-wide analysis while allowing malformed
|
||||
/// unselected pages to be skipped. Any requested page still fails normally.
|
||||
pub(crate) fn extract_positioned_text_for_document_analysis(
|
||||
doc: &Document,
|
||||
font_cmaps: &FontCMaps,
|
||||
required_pages: &HashSet<u32>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
extract_positioned_text_impl(doc, font_cmaps, None, false, Some(required_pages))
|
||||
}
|
||||
|
||||
fn extract_positioned_text_impl(
|
||||
@@ -158,6 +236,7 @@ fn extract_positioned_text_impl(
|
||||
font_cmaps: &FontCMaps,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
include_invisible: bool,
|
||||
required_pages: Option<&HashSet<u32>>,
|
||||
) -> Result<(PageExtraction, PageThresholds, HashSet<u32>), PdfError> {
|
||||
let pages = doc.get_pages();
|
||||
let mut all_items = Vec::new();
|
||||
@@ -179,15 +258,25 @@ fn extract_positioned_text_impl(
|
||||
continue;
|
||||
}
|
||||
}
|
||||
let ((mut items, mut rects, mut lines), has_gid_fonts, coords_rotated) =
|
||||
extract_page_text_items(
|
||||
doc,
|
||||
page_id,
|
||||
*page_num,
|
||||
font_cmaps,
|
||||
include_invisible,
|
||||
&mut style_cache,
|
||||
)?;
|
||||
let page_result = extract_page_text_items(
|
||||
doc,
|
||||
page_id,
|
||||
*page_num,
|
||||
font_cmaps,
|
||||
include_invisible,
|
||||
&mut style_cache,
|
||||
);
|
||||
let ((mut items, mut rects, mut lines), has_gid_fonts, coords_rotated) = match page_result {
|
||||
Ok(extraction) => extraction,
|
||||
Err(error) if required_pages.is_some_and(|required| !required.contains(page_num)) => {
|
||||
debug!(
|
||||
"page {}: skipping context-only extraction error: {}",
|
||||
page_num, error
|
||||
);
|
||||
continue;
|
||||
}
|
||||
Err(error) => return Err(error),
|
||||
};
|
||||
// Clip to the visible page box: single-page extracts and imposed
|
||||
// spreads keep neighboring pages' content in the stream, positioned
|
||||
// outside the CropBox. Extracting it interleaves invisible text into
|
||||
@@ -317,7 +406,9 @@ fn extract_positioned_text_impl(
|
||||
}
|
||||
|
||||
// Extract AcroForm field values
|
||||
let form_items = extract_form_fields(doc, &page_id_to_num);
|
||||
let form_items = extract_form_fields(doc, &page_id_to_num)
|
||||
.into_iter()
|
||||
.filter(|item| page_filter.is_none_or(|filter| filter.contains(&item.page)));
|
||||
all_items.extend(form_items);
|
||||
|
||||
Ok((
|
||||
@@ -1531,6 +1622,724 @@ mod tests {
|
||||
assert_eq!(lines[0].text(), "42");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn inline_numeric_run_near_page_edge_is_not_removed() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Total", 100.0, 30.0),
|
||||
make_merge_item("730", 136.0, 18.0),
|
||||
make_merge_item("seats", 160.0, 30.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 780.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Total 730 seats");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn numeric_page_footer_separated_from_label_is_removed() {
|
||||
let mut page_number = make_merge_item("42", 25.0, 12.0);
|
||||
page_number.y = 50.0;
|
||||
let mut footer_label = make_merge_item("DOCUMENT FOOTER", 60.0, 100.0);
|
||||
footer_label.y = 50.0;
|
||||
|
||||
let lines = group_into_lines(vec![page_number, footer_label]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "DOCUMENT FOOTER");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decorative_marker_does_not_contextualize_numeric_page_footer() {
|
||||
let mut marker = make_merge_item("•", 19.0, 6.0);
|
||||
marker.y = 30.0;
|
||||
let mut page_number = make_merge_item("42", 37.0, 10.0);
|
||||
page_number.y = 30.0;
|
||||
let mut footer_label = make_merge_item("Company report footer", 68.0, 120.0);
|
||||
footer_label.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![marker, page_number, footer_label]);
|
||||
|
||||
assert!(lines.iter().all(|line| !line.text().contains("42")));
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text().contains("Company report footer")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn labeled_page_number_is_removed_in_a_short_document() {
|
||||
let mut label = make_merge_item("Page", 25.0, 28.0);
|
||||
label.y = 50.0;
|
||||
let mut page_number = make_merge_item("42", 57.0, 12.0);
|
||||
page_number.y = 50.0;
|
||||
|
||||
let lines = group_into_lines(vec![label, page_number]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Page");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn labeled_page_number_with_running_header_suffix_is_removed() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Page", 25.0, 28.0),
|
||||
make_merge_item("42", 57.0, 12.0),
|
||||
make_merge_item("of", 73.0, 12.0),
|
||||
make_merge_item("100", 89.0, 18.0),
|
||||
make_merge_item("Report header", 111.0, 78.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 50.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Report header");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_of_total_expression_is_removed_without_leaving_fragments() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Page", 482.0, 27.0),
|
||||
make_merge_item("1", 513.0, 6.0),
|
||||
make_merge_item("of", 523.0, 10.0),
|
||||
make_merge_item("15", 537.0, 12.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 46.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert!(lines.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn document_folio_filter_survives_per_page_layout_splitting() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "42"), (2, "43"), (3, "44")] {
|
||||
let mut label = make_merge_item("Page", 25.0, 28.0);
|
||||
label.page = page;
|
||||
label.y = 50.0;
|
||||
let mut page_number = make_merge_item(value, 57.0, 12.0);
|
||||
page_number.page = page;
|
||||
page_number.y = 50.0;
|
||||
items.extend([label, page_number]);
|
||||
}
|
||||
|
||||
let filtered = filter_markdown_page_numbers(items, 3);
|
||||
assert!(filtered
|
||||
.iter()
|
||||
.all(|item| !matches!(item.text.as_str(), "42" | "43" | "44")));
|
||||
let mut lines = Vec::new();
|
||||
for page in 1..=3 {
|
||||
let page_items = filtered
|
||||
.iter()
|
||||
.filter(|item| item.page == page)
|
||||
.cloned()
|
||||
.collect();
|
||||
lines.extend(
|
||||
group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
page_items,
|
||||
&HashMap::new(),
|
||||
&HashSet::new(),
|
||||
&HashMap::new(),
|
||||
),
|
||||
);
|
||||
}
|
||||
|
||||
assert_eq!(lines.len(), 3);
|
||||
assert!(lines.iter().all(|line| line.text() == "Page"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn numeric_only_page_edge_runs_do_not_contextualize_folios() {
|
||||
let mut page_number = make_merge_item("42", 25.0, 12.0);
|
||||
page_number.y = 50.0;
|
||||
let mut long_number = make_merge_item("12345", 43.0, 30.0);
|
||||
long_number.y = 50.0;
|
||||
|
||||
let lines = group_into_lines(vec![page_number, long_number]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "12345");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn structured_and_dense_numeric_page_edge_runs_are_preserved() {
|
||||
let mut list_marker = make_merge_item("11)", 25.0, 18.0);
|
||||
list_marker.y = 50.0;
|
||||
let mut chapter = make_merge_item("13", 47.0, 12.0);
|
||||
chapter.y = 50.0;
|
||||
|
||||
let mut isbn_prefix = make_merge_item("9", 25.0, 6.0);
|
||||
isbn_prefix.page = 2;
|
||||
isbn_prefix.y = 50.0;
|
||||
let mut isbn_mid = make_merge_item("780113", 35.0, 36.0);
|
||||
isbn_mid.page = 2;
|
||||
isbn_mid.y = 50.0;
|
||||
let mut isbn_end = make_merge_item("227426", 75.0, 36.0);
|
||||
isbn_end.page = 2;
|
||||
isbn_end.y = 50.0;
|
||||
|
||||
let lines = group_into_lines(vec![list_marker, chapter, isbn_prefix, isbn_mid, isbn_end]);
|
||||
|
||||
assert_eq!(lines.len(), 2);
|
||||
assert_eq!(lines[0].text(), "11) 13");
|
||||
assert_eq!(lines[1].text(), "9 780113 227426");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn incrementing_numeric_body_column_is_not_treated_as_a_folio() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "13"), (2, "14"), (3, "15")] {
|
||||
let mut row_number = make_merge_item(value, 72.0, 12.0);
|
||||
row_number.page = page;
|
||||
row_number.y = 730.0;
|
||||
let mut name = make_merge_item("Person", 90.0, 42.0);
|
||||
name.page = page;
|
||||
name.y = 730.0;
|
||||
items.extend([row_number, name]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 3);
|
||||
assert_eq!(lines[0].text(), "13 Person");
|
||||
assert_eq!(lines[1].text(), "14 Person");
|
||||
assert_eq!(lines[2].text(), "15 Person");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn advancing_number_in_repeated_deep_margin_footer_is_removed() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "2"), (2, "4"), (3, "6"), (4, "8")] {
|
||||
let mut page_number = make_merge_item(value, 25.0, 12.0);
|
||||
page_number.page = page;
|
||||
page_number.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 30.0;
|
||||
items.extend([page_number, footer]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
assert!(lines
|
||||
.iter()
|
||||
.all(|line| line.text() == "Company report footer"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repeated_substantive_page_number_prose_is_preserved() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "42"), (2, "43"), (3, "44"), (4, "45")] {
|
||||
let mut page_label = make_merge_item("Page", 25.0, 28.0);
|
||||
page_label.page = page;
|
||||
page_label.y = 30.0;
|
||||
let mut number = make_merge_item(value, 57.0, 12.0);
|
||||
number.page = page;
|
||||
number.y = 30.0;
|
||||
let mut explanation = make_merge_item("explains the result", 73.0, 108.0);
|
||||
explanation.page = page;
|
||||
explanation.y = 30.0;
|
||||
items.extend([page_label, number, explanation]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
for (line, value) in lines.iter().zip(["42", "43", "44", "45"]) {
|
||||
assert_eq!(line.text(), format!("Page {value} explains the result"));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn numeric_candidates_do_not_bridge_lexical_context() {
|
||||
let mut report = make_merge_item("Report", 25.0, 40.0);
|
||||
report.y = 30.0;
|
||||
let mut year = make_merge_item("2026", 69.0, 24.0);
|
||||
year.y = 30.0;
|
||||
let mut folio = make_merge_item("42", 97.0, 12.0);
|
||||
folio.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![report, year, folio]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Report 2026");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn centered_folio_delimiters_are_removed_with_the_number() {
|
||||
let mut left = make_merge_item("-", 270.0, 6.0);
|
||||
left.y = 30.0;
|
||||
let mut number = make_merge_item("42", 280.0, 12.0);
|
||||
number.y = 30.0;
|
||||
let mut right = make_merge_item("-", 296.0, 6.0);
|
||||
right.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![left, number, right]);
|
||||
|
||||
assert!(lines.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn centered_delimiters_inside_substantive_text_are_preserved() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Result", 240.0, 36.0),
|
||||
make_merge_item("-", 280.0, 6.0),
|
||||
make_merge_item("42", 290.0, 12.0),
|
||||
make_merge_item("-", 306.0, 6.0),
|
||||
make_merge_item("approved", 316.0, 48.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 30.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Result-42-approved");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn changing_year_in_repeated_deep_margin_header_is_preserved() {
|
||||
let mut items = Vec::new();
|
||||
for (page, year) in [(1, "2020"), (2, "2021"), (3, "2022"), (4, "2023")] {
|
||||
let mut year = make_merge_item(year, 25.0, 24.0);
|
||||
year.page = page;
|
||||
year.y = 780.0;
|
||||
let mut header = make_merge_item("Annual report", 53.0, 78.0);
|
||||
header.page = page;
|
||||
header.y = 780.0;
|
||||
items.extend([year, header]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
assert_eq!(lines[0].text(), "2020 Annual report");
|
||||
assert_eq!(lines[3].text(), "2023 Annual report");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sparse_repeated_margin_numbers_do_not_meet_the_folio_evidence_floor() {
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=4 {
|
||||
if page <= 2 {
|
||||
let value = if page == 1 { "2" } else { "4" };
|
||||
let mut number = make_merge_item(value, 25.0, 12.0);
|
||||
number.page = page;
|
||||
number.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 30.0;
|
||||
items.extend([number, footer]);
|
||||
} else {
|
||||
let mut body = make_merge_item("Body text", 72.0, 54.0);
|
||||
body.page = page;
|
||||
body.y = 400.0;
|
||||
items.push(body);
|
||||
}
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text() == "2 Company report footer"));
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text() == "4 Company report footer"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sparse_document_pages_count_toward_repeated_folio_coverage() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "1"), (10, "10"), (19, "19"), (28, "28")] {
|
||||
let mut number = make_merge_item(value, 25.0, 12.0);
|
||||
number.page = page;
|
||||
number.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 30.0;
|
||||
items.extend([number, footer]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text() == "1 Company report footer"));
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text() == "28 Company report footer"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn trailing_blank_pages_count_toward_repeated_folio_coverage() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "1"), (2, "2"), (3, "3"), (4, "4")] {
|
||||
let mut number = make_merge_item(value, 25.0, 12.0);
|
||||
number.page = page;
|
||||
number.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 30.0;
|
||||
items.extend([number, footer]);
|
||||
}
|
||||
|
||||
let filtered = filter_markdown_page_numbers(items, 20);
|
||||
|
||||
assert!(filtered.iter().any(|item| item.text == "1"));
|
||||
assert!(filtered.iter().any(|item| item.text == "4"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn prefiltered_contextual_number_survives_layout_partitioning() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Total", 100.0, 30.0),
|
||||
make_merge_item("730", 136.0, 18.0),
|
||||
make_merge_item("seats", 160.0, 30.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 780.0;
|
||||
}
|
||||
|
||||
let filtered = filter_markdown_page_numbers(items, 1);
|
||||
let partitioned_number: Vec<TextItem> = filtered
|
||||
.into_iter()
|
||||
.filter(|item| item.text == "730")
|
||||
.collect();
|
||||
let lines = group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
partitioned_number,
|
||||
&HashMap::new(),
|
||||
&HashSet::new(),
|
||||
&HashMap::new(),
|
||||
);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "730");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn numeric_only_partition_does_not_define_columns() {
|
||||
let mut items = Vec::new();
|
||||
for row in 0..20 {
|
||||
let y = 90.0 - row as f32 * 4.0;
|
||||
let mut left = make_merge_item(&(row + 1).to_string(), 50.0, 20.0);
|
||||
left.y = y;
|
||||
let mut right = make_merge_item(&(row + 101).to_string(), 350.0, 20.0);
|
||||
right.y = y;
|
||||
items.extend([left, right]);
|
||||
}
|
||||
assert_eq!(detect_columns(&items, 1, false).len(), 2);
|
||||
|
||||
let lines = group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
items,
|
||||
&HashMap::new(),
|
||||
&HashSet::new(),
|
||||
&HashMap::new(),
|
||||
);
|
||||
|
||||
assert_eq!(lines.len(), 20);
|
||||
assert!(lines.iter().all(|line| line.items.len() == 2));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn separated_content_is_not_treated_as_a_spread_folio_pair() {
|
||||
let mut value = make_merge_item("12", 100.0, 12.0);
|
||||
value.y = 30.0;
|
||||
let mut label = make_merge_item("Total", 116.0, 30.0);
|
||||
label.y = 30.0;
|
||||
let mut unrelated_number = make_merge_item("13", 300.0, 12.0);
|
||||
unrelated_number.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![value, label, unrelated_number]);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "12 Total");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repeated_folio_uses_the_full_page_edge_band() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "2"), (2, "4"), (3, "6"), (4, "8")] {
|
||||
let mut page_number = make_merge_item(value, 25.0, 12.0);
|
||||
page_number.page = page;
|
||||
page_number.y = 80.0;
|
||||
let mut footer = make_merge_item("Company report footer", 41.0, 120.0);
|
||||
footer.page = page;
|
||||
footer.y = 80.0;
|
||||
items.extend([page_number, footer]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
assert!(lines
|
||||
.iter()
|
||||
.all(|line| line.text() == "Company report footer"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn contextual_folio_on_facing_page_spread_is_removed() {
|
||||
let mut marker = make_merge_item("•", 19.0, 6.0);
|
||||
marker.y = 30.0;
|
||||
let mut left_folio = make_merge_item("326", 35.0, 17.0);
|
||||
left_folio.y = 30.0;
|
||||
let mut footer = make_merge_item("Company report footer", 61.0, 120.0);
|
||||
footer.y = 30.0;
|
||||
let mut right_folio = make_merge_item("327", 1148.0, 17.0);
|
||||
right_folio.y = 30.0;
|
||||
|
||||
let lines = group_into_lines(vec![marker, left_folio, footer, right_folio]);
|
||||
|
||||
assert!(lines
|
||||
.iter()
|
||||
.all(|line| !line.text().contains("326") && !line.text().contains("327")));
|
||||
assert!(lines
|
||||
.iter()
|
||||
.any(|line| line.text().contains("Company report footer")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn contextual_folios_alternating_across_pages_are_removed() {
|
||||
let headers = [
|
||||
"Letter to shareholders",
|
||||
"Corporate governance report",
|
||||
"Business environment overview",
|
||||
"Consolidated financial statements",
|
||||
];
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=8 {
|
||||
let mut body = make_merge_item("Body text", 50.0, 500.0);
|
||||
body.page = page;
|
||||
body.y = 400.0;
|
||||
items.push(body);
|
||||
|
||||
let mut folio = make_merge_item(&(page + 22).to_string(), 0.0, 14.0);
|
||||
folio.page = page;
|
||||
folio.y = 780.0;
|
||||
if page % 2 == 0 {
|
||||
folio.x = 50.0;
|
||||
items.push(folio);
|
||||
} else {
|
||||
let mut header = make_merge_item(headers[(page / 2) as usize], 350.0, 180.0);
|
||||
header.page = page;
|
||||
header.y = 780.0;
|
||||
folio.x = 536.0;
|
||||
items.extend([header, folio]);
|
||||
}
|
||||
}
|
||||
|
||||
let filtered = filter_markdown_page_numbers(items, 8);
|
||||
|
||||
assert!(filtered.iter().all(|item| {
|
||||
!matches!(
|
||||
item.text.as_str(),
|
||||
"23" | "24" | "25" | "26" | "27" | "28" | "29" | "30"
|
||||
)
|
||||
}));
|
||||
assert!(headers
|
||||
.iter()
|
||||
.all(|header| filtered.iter().any(|item| item.text == *header)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn one_isolated_neighbor_does_not_remove_contextual_number() {
|
||||
let mut body_one = make_merge_item("Body text", 50.0, 500.0);
|
||||
body_one.y = 400.0;
|
||||
let mut label = make_merge_item("Report", 450.0, 70.0);
|
||||
label.y = 780.0;
|
||||
let mut contextual = make_merge_item("1", 526.0, 7.0);
|
||||
contextual.y = 780.0;
|
||||
|
||||
let mut body_two = body_one.clone();
|
||||
body_two.page = 2;
|
||||
let mut isolated = make_merge_item("2", 50.0, 7.0);
|
||||
isolated.page = 2;
|
||||
isolated.y = 780.0;
|
||||
|
||||
let filtered =
|
||||
filter_markdown_page_numbers(vec![body_one, label, contextual, body_two, isolated], 2);
|
||||
|
||||
assert!(filtered.iter().any(|item| item.text == "Report"));
|
||||
assert!(filtered.iter().any(|item| item.text == "1"));
|
||||
assert!(filtered.iter().all(|item| item.text != "2"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn narrow_content_span_does_not_establish_adjacent_page_edges() {
|
||||
let mut body_one = make_merge_item("Body text", 100.0, 120.0);
|
||||
body_one.y = 400.0;
|
||||
let mut label = make_merge_item("Report", 170.0, 60.0);
|
||||
label.y = 780.0;
|
||||
let mut contextual = make_merge_item("1", 235.0, 7.0);
|
||||
contextual.y = 780.0;
|
||||
|
||||
let mut body_two = body_one.clone();
|
||||
body_two.page = 2;
|
||||
let mut isolated_two = make_merge_item("2", 100.0, 7.0);
|
||||
isolated_two.page = 2;
|
||||
isolated_two.y = 780.0;
|
||||
|
||||
let mut body_four = body_one.clone();
|
||||
body_four.page = 4;
|
||||
let mut isolated_four = make_merge_item("4", 100.0, 7.0);
|
||||
isolated_four.page = 4;
|
||||
isolated_four.y = 780.0;
|
||||
|
||||
let filtered = filter_markdown_page_numbers(
|
||||
vec![
|
||||
body_one,
|
||||
label,
|
||||
contextual,
|
||||
body_two,
|
||||
isolated_two,
|
||||
body_four,
|
||||
isolated_four,
|
||||
],
|
||||
4,
|
||||
);
|
||||
|
||||
assert!(filtered.iter().any(|item| item.text == "Report"));
|
||||
assert!(filtered.iter().any(|item| item.text == "1"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn same_edge_number_on_an_adjacent_page_is_not_folio_evidence() {
|
||||
let mut body_one = make_merge_item("Body text", 50.0, 500.0);
|
||||
body_one.y = 400.0;
|
||||
let mut isolated = make_merge_item("42", 50.0, 14.0);
|
||||
isolated.y = 780.0;
|
||||
|
||||
let mut body_two = body_one.clone();
|
||||
body_two.page = 2;
|
||||
let mut contextual = make_merge_item("43", 50.0, 14.0);
|
||||
contextual.page = 2;
|
||||
contextual.y = 780.0;
|
||||
let mut label = make_merge_item("cases reviewed", 70.0, 90.0);
|
||||
label.page = 2;
|
||||
label.y = 780.0;
|
||||
|
||||
let filtered =
|
||||
filter_markdown_page_numbers(vec![body_one, isolated, body_two, contextual, label], 2);
|
||||
|
||||
assert!(filtered.iter().any(|item| item.text == "43"));
|
||||
assert!(filtered.iter().any(|item| item.text == "cases reviewed"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn constant_number_in_repeated_deep_margin_header_is_preserved() {
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=4 {
|
||||
let mut year = make_merge_item("2026", 25.0, 24.0);
|
||||
year.page = page;
|
||||
year.y = 780.0;
|
||||
let mut header = make_merge_item("Annual report", 53.0, 78.0);
|
||||
header.page = page;
|
||||
header.y = 780.0;
|
||||
items.extend([year, header]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 4);
|
||||
assert!(lines.iter().all(|line| line.text() == "2026 Annual report"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_prefix_does_not_remove_substantive_text_during_layout() {
|
||||
let mut items = vec![
|
||||
make_merge_item("Page", 25.0, 28.0),
|
||||
make_merge_item("42", 57.0, 12.0),
|
||||
make_merge_item("explains", 73.0, 44.0),
|
||||
make_merge_item("the result", 121.0, 55.0),
|
||||
];
|
||||
for item in &mut items {
|
||||
item.y = 50.0;
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Page 42 explains the result");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repeated_page_number_prefix_with_substantive_text_is_preserved_during_layout() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value, chapter) in [(1, "42", "Chapter 1"), (2, "43", "Chapter 2")] {
|
||||
let mut label = make_merge_item("Page", 25.0, 28.0);
|
||||
label.page = page;
|
||||
label.y = 50.0;
|
||||
let mut page_number = make_merge_item(value, 57.0, 12.0);
|
||||
page_number.page = page;
|
||||
page_number.y = 50.0;
|
||||
let mut suffix = make_merge_item(chapter, 73.0, 58.0);
|
||||
suffix.page = page;
|
||||
suffix.y = 50.0;
|
||||
items.extend([label, page_number, suffix]);
|
||||
}
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 2);
|
||||
assert_eq!(lines[0].text(), "Page 42 Chapter 1");
|
||||
assert_eq!(lines[1].text(), "Page 43 Chapter 2");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_phrase_in_the_page_body_is_preserved() {
|
||||
let items = vec![
|
||||
make_merge_item("Page", 25.0, 28.0),
|
||||
make_merge_item("42", 57.0, 12.0),
|
||||
];
|
||||
|
||||
let lines = group_into_lines(items);
|
||||
|
||||
assert_eq!(lines.len(), 1);
|
||||
assert_eq!(lines[0].text(), "Page 42");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn short_numeric_context_near_page_edge_is_preserved() {
|
||||
let mut chapter = make_merge_item("Chapter", 100.0, 45.0);
|
||||
chapter.y = 760.0;
|
||||
let mut chapter_number = make_merge_item("1", 151.0, 6.0);
|
||||
chapter_number.y = 760.0;
|
||||
|
||||
let chapter_lines = group_into_lines(vec![chapter, chapter_number]);
|
||||
assert_eq!(chapter_lines.len(), 1);
|
||||
assert_eq!(chapter_lines[0].text(), "Chapter 1");
|
||||
|
||||
let mut year = make_merge_item("2026", 100.0, 24.0);
|
||||
year.y = 760.0;
|
||||
let mut report = make_merge_item("Report", 130.0, 36.0);
|
||||
report.y = 760.0;
|
||||
|
||||
let report_lines = group_into_lines(vec![year, report]);
|
||||
assert_eq!(report_lines.len(), 1);
|
||||
assert_eq!(report_lines[0].text(), "2026 Report");
|
||||
|
||||
let mut chapter = make_merge_item("Chapter", 100.0, 45.0);
|
||||
chapter.y = 760.0;
|
||||
let mut chapter_number = make_merge_item("1", 151.0, 6.0);
|
||||
chapter_number.y = 760.0;
|
||||
let mut edition_year = make_merge_item("2026", 163.0, 24.0);
|
||||
edition_year.y = 760.0;
|
||||
|
||||
let chained_lines = group_into_lines(vec![chapter, chapter_number, edition_year]);
|
||||
assert_eq!(chained_lines.len(), 1);
|
||||
assert_eq!(chained_lines[0].text(), "Chapter 1 2026");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_bold_italic_detection() {
|
||||
// Test bold detection
|
||||
|
||||
+210
-24
@@ -53,8 +53,8 @@ pub use extractor::{
|
||||
extract_text_with_positions_pages,
|
||||
};
|
||||
pub use markdown::{
|
||||
to_markdown, to_markdown_from_items, to_markdown_from_items_with_rects, MarkdownOptions,
|
||||
MarkdownProfile,
|
||||
to_markdown, to_markdown_from_items, to_markdown_from_items_with_rects,
|
||||
to_markdown_from_items_with_rects_and_page_count, MarkdownOptions, MarkdownProfile,
|
||||
};
|
||||
pub use process_mode::ProcessMode;
|
||||
pub use types::{LayoutComplexity, PdfLine, PdfRect, TextItem};
|
||||
@@ -462,16 +462,39 @@ pub fn extract_pages_markdown_mem(
|
||||
let (doc, page_count) = load_document_from_mem(buffer)?;
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
|
||||
// Extract ALL pages to get accurate, document-wide font stats.
|
||||
// Extract ALL pages to get accurate, document-wide font stats. A malformed
|
||||
// unselected page cannot make a valid requested page fail, but errors on a
|
||||
// requested page retain the normal extraction semantics.
|
||||
let required_pages: Option<HashSet<u32>> = pages.map(|pages| {
|
||||
pages
|
||||
.iter()
|
||||
.filter_map(|page| page.checked_add(1))
|
||||
.collect()
|
||||
});
|
||||
let ((all_items, all_rects, all_lines), page_thresholds, gid_pages) =
|
||||
extractor::extract_positioned_text_from_doc(&doc, &font_cmaps, None)?;
|
||||
if let Some(required_pages) = required_pages.as_ref() {
|
||||
extractor::extract_positioned_text_for_document_analysis(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
required_pages,
|
||||
)?
|
||||
} else {
|
||||
extractor::extract_positioned_text_from_doc(&doc, &font_cmaps, None)?
|
||||
};
|
||||
let text_quality = analyze_text_quality(&all_items);
|
||||
|
||||
// Compute layout complexity from full document (near-zero cost).
|
||||
let complexity = compute_layout_complexity(&all_items, &all_rects, &all_lines);
|
||||
// Resolve page numbers with full-document context before partitioning.
|
||||
// Per-page Markdown receives the original items plus these decisions so
|
||||
// table detection can retain legitimate numeric cells.
|
||||
let (filtered_items, removed_page_number_pages, page_number_removal_mask) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(all_items.clone(), page_count);
|
||||
|
||||
// Tables need the original numeric cells; columns use folio-cleaned
|
||||
// evidence so removed page numbers cannot create false layout metadata.
|
||||
let complexity = compute_layout_complexity(&all_items, &filtered_items, &all_rects, &all_lines);
|
||||
|
||||
// Compute font stats from full document (cross-page consistency).
|
||||
let font_stats = markdown::analysis::calculate_font_stats_from_items(&all_items);
|
||||
let font_stats = markdown::analysis::calculate_font_stats_from_items(&filtered_items);
|
||||
|
||||
// When caller doesn't specify pages, return every page in document order.
|
||||
let all_pages: Vec<u32>;
|
||||
@@ -502,12 +525,13 @@ pub fn extract_pages_markdown_mem(
|
||||
|
||||
let page_1idx = page_0idx + 1;
|
||||
|
||||
// Filter items/rects for this page only
|
||||
let page_items: Vec<TextItem> = all_items
|
||||
// Partition items, removal decisions, and rects for this page only.
|
||||
let (page_items, page_number_removal_mask): (Vec<TextItem>, Vec<bool>) = all_items
|
||||
.iter()
|
||||
.filter(|i| i.page == page_1idx)
|
||||
.cloned()
|
||||
.collect();
|
||||
.zip(&page_number_removal_mask)
|
||||
.filter(|(item, _)| item.page == page_1idx)
|
||||
.map(|(item, remove)| (item.clone(), *remove))
|
||||
.unzip();
|
||||
|
||||
let page_rects: Vec<PdfRect> = all_rects
|
||||
.iter()
|
||||
@@ -534,9 +558,14 @@ pub fn extract_pages_markdown_mem(
|
||||
options,
|
||||
&page_rects,
|
||||
&[],
|
||||
&page_thresholds,
|
||||
None,
|
||||
&[],
|
||||
markdown::MarkdownDocumentContext {
|
||||
page_thresholds: &page_thresholds,
|
||||
struct_roles: None,
|
||||
struct_tables: &[],
|
||||
page_count,
|
||||
prefiltered_page_number_pages: Some(&removed_page_number_pages),
|
||||
prefiltered_page_number_mask: Some(&page_number_removal_mask),
|
||||
},
|
||||
)
|
||||
};
|
||||
|
||||
@@ -3605,7 +3634,10 @@ fn process_document(
|
||||
// Step 2 — Extraction (reuses the already-loaded document)
|
||||
let extracted = {
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let result = extractor::extract_positioned_text_from_doc(
|
||||
// Most page-filtered requests extract only the selected pages. Gather
|
||||
// other pages only when a selected contextual folio needs cross-page
|
||||
// evidence; failures on those context-only pages are non-fatal.
|
||||
let result = extractor::extract_positioned_text_with_folio_context(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
options.page_filter.as_ref(),
|
||||
@@ -3616,9 +3648,19 @@ fn process_document(
|
||||
// This unlocks OCR text layers behind scanned images.
|
||||
if pdf_type == PdfType::Mixed {
|
||||
if let Ok((ref items, _, _)) = result.as_ref().map(|(e, _, _)| e) {
|
||||
let sample: String = items.iter().take(200).map(|i| i.text.as_str()).collect();
|
||||
let sample: String = items
|
||||
.iter()
|
||||
.filter(|item| {
|
||||
options
|
||||
.page_filter
|
||||
.as_ref()
|
||||
.is_none_or(|filter| filter.contains(&item.page))
|
||||
})
|
||||
.take(200)
|
||||
.map(|item| item.text.as_str())
|
||||
.collect();
|
||||
if is_garbage_text(&sample) || sample.trim().is_empty() {
|
||||
extractor::extract_positioned_text_include_invisible(
|
||||
extractor::extract_positioned_text_include_invisible_with_folio_context(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
options.page_filter.as_ref(),
|
||||
@@ -3628,7 +3670,7 @@ fn process_document(
|
||||
}
|
||||
} else {
|
||||
// Normal extraction failed — try invisible as fallback
|
||||
extractor::extract_positioned_text_include_invisible(
|
||||
extractor::extract_positioned_text_include_invisible_with_folio_context(
|
||||
&doc,
|
||||
&font_cmaps,
|
||||
options.page_filter.as_ref(),
|
||||
@@ -3692,6 +3734,13 @@ fn process_document(
|
||||
let mut garbage_pages: std::collections::HashSet<u32> =
|
||||
std::collections::HashSet::new();
|
||||
for &pg in &ocr_set {
|
||||
if options
|
||||
.page_filter
|
||||
.as_ref()
|
||||
.is_some_and(|filter| !filter.contains(&pg))
|
||||
{
|
||||
continue;
|
||||
}
|
||||
let page_text: String = items
|
||||
.iter()
|
||||
.filter(|i| i.page == pg)
|
||||
@@ -3731,9 +3780,38 @@ fn process_document(
|
||||
}
|
||||
};
|
||||
|
||||
let selected_page = |page: u32| {
|
||||
options
|
||||
.page_filter
|
||||
.as_ref()
|
||||
.is_none_or(|filter| filter.contains(&page))
|
||||
};
|
||||
let rects: Vec<_> = rects
|
||||
.into_iter()
|
||||
.filter(|rect| selected_page(rect.page))
|
||||
.collect();
|
||||
let lines: Vec<_> = lines
|
||||
.into_iter()
|
||||
.filter(|line| selected_page(line.page))
|
||||
.collect();
|
||||
let gid_encoded_pages: HashSet<_> = gid_encoded_pages
|
||||
.into_iter()
|
||||
.filter(|page| selected_page(*page))
|
||||
.collect();
|
||||
let FolioFilteredItems {
|
||||
items,
|
||||
layout_items,
|
||||
removal_mask,
|
||||
removed_pages,
|
||||
} = select_items_with_document_folio_context(
|
||||
items,
|
||||
page_count,
|
||||
options.page_filter.as_ref(),
|
||||
);
|
||||
|
||||
let text_quality = analyze_text_quality(&items);
|
||||
merge_ocr_reasons(&mut ocr_reasons_by_page, text_quality.reasons_by_page);
|
||||
let layout = compute_layout_complexity(&items, &rects, &lines);
|
||||
let layout = compute_layout_complexity(&items, &layout_items, &rects, &lines);
|
||||
|
||||
let md = if options.mode == ProcessMode::Analyze {
|
||||
None
|
||||
@@ -3743,9 +3821,14 @@ fn process_document(
|
||||
options.markdown,
|
||||
&rects,
|
||||
&lines,
|
||||
&page_thresholds,
|
||||
struct_roles.as_ref(),
|
||||
&struct_tables,
|
||||
markdown::MarkdownDocumentContext {
|
||||
page_thresholds: &page_thresholds,
|
||||
struct_roles: struct_roles.as_ref(),
|
||||
struct_tables: &struct_tables,
|
||||
page_count,
|
||||
prefiltered_page_number_pages: Some(&removed_pages),
|
||||
prefiltered_page_number_mask: Some(removal_mask.as_slice()),
|
||||
},
|
||||
))
|
||||
};
|
||||
|
||||
@@ -5504,9 +5587,50 @@ mod looks_like_partial_table_tests {
|
||||
}
|
||||
}
|
||||
|
||||
struct FolioFilteredItems {
|
||||
items: Vec<types::TextItem>,
|
||||
layout_items: Vec<types::TextItem>,
|
||||
removal_mask: Vec<bool>,
|
||||
removed_pages: HashSet<u32>,
|
||||
}
|
||||
|
||||
/// Resolve folios with complete document context, then select the caller's
|
||||
/// requested pages without losing those decisions.
|
||||
fn select_items_with_document_folio_context(
|
||||
all_items: Vec<types::TextItem>,
|
||||
page_count: u32,
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> FolioFilteredItems {
|
||||
let (all_layout_items, all_removed_pages, all_removal_mask) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(all_items.clone(), page_count);
|
||||
let selected_page = |page: u32| page_filter.is_none_or(|filter| filter.contains(&page));
|
||||
|
||||
let (items, removal_mask) = all_items
|
||||
.into_iter()
|
||||
.zip(all_removal_mask)
|
||||
.filter(|(item, _)| selected_page(item.page))
|
||||
.unzip();
|
||||
let layout_items = all_layout_items
|
||||
.into_iter()
|
||||
.filter(|item| selected_page(item.page))
|
||||
.collect();
|
||||
let removed_pages = all_removed_pages
|
||||
.into_iter()
|
||||
.filter(|page| selected_page(*page))
|
||||
.collect();
|
||||
|
||||
FolioFilteredItems {
|
||||
items,
|
||||
layout_items,
|
||||
removal_mask,
|
||||
removed_pages,
|
||||
}
|
||||
}
|
||||
|
||||
/// Analyse extracted items and rects for layout complexity.
|
||||
fn compute_layout_complexity(
|
||||
items: &[types::TextItem],
|
||||
column_items: &[types::TextItem],
|
||||
rects: &[types::PdfRect],
|
||||
lines: &[types::PdfLine],
|
||||
) -> LayoutComplexity {
|
||||
@@ -5591,7 +5715,7 @@ fn compute_layout_complexity(
|
||||
|
||||
let mut pages_with_columns: Vec<u32> = Vec::new();
|
||||
for page in seen_pages {
|
||||
let cols = extractor::detect_columns(items, page, pages_with_tables.contains(&page));
|
||||
let cols = extractor::detect_columns(column_items, page, pages_with_tables.contains(&page));
|
||||
if cols.len() >= 2 {
|
||||
pages_with_columns.push(page);
|
||||
}
|
||||
@@ -5803,6 +5927,68 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn removed_sparse_folios_leave_no_layout_evidence() {
|
||||
let items = vec![
|
||||
test_item("1", 25.0, 20.0, 12.0, 10.0),
|
||||
test_item("2", 520.0, 60.0, 12.0, 10.0),
|
||||
];
|
||||
let (filtered, _, _) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(items.clone(), 1);
|
||||
assert!(filtered.is_empty());
|
||||
|
||||
let filtered = compute_layout_complexity(&items, &filtered, &[], &[]);
|
||||
|
||||
assert!(!filtered.is_complex);
|
||||
assert!(filtered.pages_with_tables.is_empty());
|
||||
assert!(filtered.pages_with_columns.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_selection_keeps_document_wide_folio_layout_decisions() {
|
||||
let mut items = Vec::new();
|
||||
for page in 1..=4 {
|
||||
for row in 0..8 {
|
||||
let y = 20.0 + row as f32 * 8.0;
|
||||
let mut folio = test_item(&(row * 10 + page).to_string(), 25.0, y, 12.0, 10.0);
|
||||
folio.page = page;
|
||||
let mut footer =
|
||||
test_item(&format!("Footer row {row} summary"), 43.0, y, 470.0, 10.0);
|
||||
footer.page = page;
|
||||
let mut body = test_item(&format!("Body{row}"), 530.0, y, 55.0, 10.0);
|
||||
body.page = page;
|
||||
items.extend([folio, footer, body]);
|
||||
}
|
||||
}
|
||||
|
||||
let page_one_items: Vec<_> = items
|
||||
.iter()
|
||||
.filter(|item| item.page == 1)
|
||||
.cloned()
|
||||
.collect();
|
||||
let (page_local_layout, _, _) =
|
||||
extractor::filter_markdown_page_numbers_with_removed_pages(page_one_items.clone(), 4);
|
||||
let page_local = compute_layout_complexity(&page_one_items, &page_local_layout, &[], &[]);
|
||||
assert!(
|
||||
page_local.pages_with_columns.contains(&1),
|
||||
"fixture must reproduce page-local folio column evidence"
|
||||
);
|
||||
|
||||
let selected =
|
||||
select_items_with_document_folio_context(items, 4, Some(&HashSet::from([1])));
|
||||
assert_eq!(
|
||||
selected
|
||||
.removal_mask
|
||||
.iter()
|
||||
.filter(|remove| **remove)
|
||||
.count(),
|
||||
8
|
||||
);
|
||||
let document_wide =
|
||||
compute_layout_complexity(&selected.items, &selected.layout_items, &[], &[]);
|
||||
assert!(!document_wide.pages_with_columns.contains(&1));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_detect_encoding_issues_fffd() {
|
||||
assert!(detect_encoding_issues(
|
||||
|
||||
+150
-24
@@ -975,18 +975,54 @@ pub fn to_markdown_from_items_with_rects(
|
||||
items: Vec<TextItem>,
|
||||
options: MarkdownOptions,
|
||||
rects: &[crate::types::PdfRect],
|
||||
) -> String {
|
||||
let document_page_count = items.iter().map(|item| item.page).max().unwrap_or(0);
|
||||
to_markdown_from_items_with_rects_and_page_count(items, options, rects, document_page_count)
|
||||
}
|
||||
|
||||
/// Convert positioned text items to Markdown with an authoritative PDF page count.
|
||||
///
|
||||
/// Use this overload when the owning PDF is available so trailing blank or
|
||||
/// unextracted pages are included in document-level header and folio coverage.
|
||||
/// Item-only callers can continue using [`to_markdown_from_items_with_rects`],
|
||||
/// which falls back to the highest observed item page.
|
||||
pub fn to_markdown_from_items_with_rects_and_page_count(
|
||||
items: Vec<TextItem>,
|
||||
options: MarkdownOptions,
|
||||
rects: &[crate::types::PdfRect],
|
||||
document_page_count: u32,
|
||||
) -> String {
|
||||
to_markdown_from_items_with_rects_and_lines(
|
||||
items,
|
||||
options,
|
||||
rects,
|
||||
&[],
|
||||
&HashMap::new(),
|
||||
None,
|
||||
&[],
|
||||
MarkdownDocumentContext {
|
||||
page_thresholds: &HashMap::new(),
|
||||
struct_roles: None,
|
||||
struct_tables: &[],
|
||||
page_count: document_page_count,
|
||||
prefiltered_page_number_pages: None,
|
||||
prefiltered_page_number_mask: None,
|
||||
},
|
||||
)
|
||||
}
|
||||
|
||||
pub(crate) struct MarkdownDocumentContext<'a> {
|
||||
pub(crate) page_thresholds: &'a HashMap<u32, f32>,
|
||||
pub(crate) struct_roles:
|
||||
Option<&'a HashMap<u32, HashMap<i64, crate::structure_tree::StructRole>>>,
|
||||
pub(crate) struct_tables: &'a [crate::structure_tree::StructTable],
|
||||
pub(crate) page_count: u32,
|
||||
/// Pages where an upstream document-level pass removed folios. This keeps
|
||||
/// table-continuation classification consistent after masked items drop.
|
||||
pub(crate) prefiltered_page_number_pages: Option<&'a HashSet<u32>>,
|
||||
/// Document-level removal decisions aligned with this call's input items.
|
||||
/// Table detection consumes the original items; the mask is applied only
|
||||
/// after table claims have been established.
|
||||
pub(crate) prefiltered_page_number_mask: Option<&'a [bool]>,
|
||||
}
|
||||
|
||||
/// Convert positioned text items to markdown, using rectangles and line segments for table detection.
|
||||
///
|
||||
/// Line-based detection runs first (strongest structural evidence), then rect-based,
|
||||
@@ -996,9 +1032,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
options: MarkdownOptions,
|
||||
rects: &[crate::types::PdfRect],
|
||||
pdf_lines: &[crate::types::PdfLine],
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
struct_roles: Option<&HashMap<u32, HashMap<i64, crate::structure_tree::StructRole>>>,
|
||||
struct_tables: &[crate::structure_tree::StructTable],
|
||||
context: MarkdownDocumentContext<'_>,
|
||||
) -> String {
|
||||
use crate::tables::{
|
||||
detect_tables, detect_tables_from_lines, detect_tables_from_rects,
|
||||
@@ -1006,17 +1040,35 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
};
|
||||
use crate::types::ItemType;
|
||||
|
||||
let MarkdownDocumentContext {
|
||||
page_thresholds,
|
||||
struct_roles,
|
||||
struct_tables,
|
||||
page_count: document_page_count,
|
||||
prefiltered_page_number_pages,
|
||||
prefiltered_page_number_mask,
|
||||
} = context;
|
||||
|
||||
if items.is_empty() {
|
||||
return String::new();
|
||||
}
|
||||
|
||||
// Table detection must retain the original collection because short
|
||||
// numeric table cells can be indistinguishable from folios until
|
||||
// structural context is available. A precomputed mask carries the
|
||||
// document-wide decision without removing items before table claims.
|
||||
debug_assert!(prefiltered_page_number_mask.is_none_or(|mask| mask.len() == items.len()));
|
||||
let has_precomputed_page_number_mask = prefiltered_page_number_mask.is_some();
|
||||
let removed_page_number_pages = prefiltered_page_number_pages.cloned().unwrap_or_default();
|
||||
|
||||
// Separate images and links from text items
|
||||
let mut images: Vec<TextItem> = Vec::new();
|
||||
let mut page_image_regions: HashMap<u32, Vec<(f32, f32, f32, f32)>> = HashMap::new();
|
||||
let mut links: Vec<TextItem> = Vec::new();
|
||||
let mut text_items: Vec<TextItem> = Vec::new();
|
||||
let mut text_item_page_number_mask: Vec<bool> = Vec::new();
|
||||
|
||||
for item in items {
|
||||
for (input_index, item) in items.into_iter().enumerate() {
|
||||
match &item.item_type {
|
||||
ItemType::Image => {
|
||||
page_image_regions.entry(item.page).or_default().push((
|
||||
@@ -1035,6 +1087,12 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
}
|
||||
}
|
||||
ItemType::Text | ItemType::FormField => {
|
||||
text_item_page_number_mask.push(
|
||||
prefiltered_page_number_mask
|
||||
.and_then(|mask| mask.get(input_index))
|
||||
.copied()
|
||||
.unwrap_or(false),
|
||||
);
|
||||
text_items.push(item);
|
||||
}
|
||||
}
|
||||
@@ -1075,7 +1133,6 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
|
||||
let mut pages: Vec<u32> = page_groups.keys().copied().collect();
|
||||
pages.sort();
|
||||
let page_count = pages.last().copied().unwrap_or(0) + 1;
|
||||
|
||||
// Track band splits per page so we can split non-table items later
|
||||
let mut page_band_splits: HashMap<u32, Vec<(f32, f32)>> = HashMap::new();
|
||||
@@ -1128,7 +1185,10 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
});
|
||||
let chart_prose_columns = chart_prose_split.is_some();
|
||||
|
||||
// Check for side-by-side layout (e.g. two tables placed left and right)
|
||||
// Check for side-by-side table layout using the original items. Sparse
|
||||
// numeric cells need table context before they can be distinguished
|
||||
// safely from folios; cleaned evidence is reserved for column and
|
||||
// final non-table layout decisions.
|
||||
let mut bands = split_side_by_side(&page_items);
|
||||
// A rect table crossing a proposed split boundary means the "gutter"
|
||||
// is really the gap between ruled and borderless table columns —
|
||||
@@ -1634,16 +1694,20 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
};
|
||||
|
||||
// Filter out table items and process the rest
|
||||
let non_table_items: Vec<TextItem> = text_items
|
||||
let non_table_items: Vec<(usize, TextItem)> = text_items
|
||||
.into_iter()
|
||||
.enumerate()
|
||||
.filter(|(idx, _)| !table_items.contains(idx))
|
||||
.map(|(_, item)| item)
|
||||
.collect();
|
||||
|
||||
// Find pages that are table-only (no remaining non-table text)
|
||||
let table_only_pages: HashSet<u32> = {
|
||||
let pages_with_text: HashSet<u32> = non_table_items.iter().map(|i| i.page).collect();
|
||||
let mut pages_with_text: HashSet<u32> =
|
||||
non_table_items.iter().map(|(_, item)| item.page).collect();
|
||||
// Preserve the pre-filter continuation classification: a page that
|
||||
// originally also contained a folio does not become table-only merely
|
||||
// because an upstream document-level pass removed it.
|
||||
pages_with_text.extend(removed_page_number_pages);
|
||||
page_tables
|
||||
.keys()
|
||||
.filter(|p| !pages_with_text.contains(p))
|
||||
@@ -1658,11 +1722,25 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
// column detection on pages where table column gaps would be misidentified.
|
||||
let table_page_set: HashSet<u32> = page_tables.keys().copied().collect();
|
||||
|
||||
let non_table_items = if has_precomputed_page_number_mask {
|
||||
non_table_items
|
||||
.into_iter()
|
||||
.filter(|(index, _)| !text_item_page_number_mask[*index])
|
||||
.map(|(_, item)| item)
|
||||
.collect()
|
||||
} else {
|
||||
crate::extractor::filter_markdown_page_numbers_with_removed_pages(
|
||||
non_table_items.into_iter().map(|(_, item)| item).collect(),
|
||||
document_page_count,
|
||||
)
|
||||
.0
|
||||
};
|
||||
|
||||
// Split non-table items by band boundaries before line grouping so that
|
||||
// items from different side-by-side zones (e.g. left/right month columns
|
||||
// in a calendar) don't merge into the same line.
|
||||
let lines = if page_band_splits.is_empty() && page_chart_prose_splits.is_empty() {
|
||||
crate::extractor::group_into_lines_with_thresholds_and_regions(
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_regions(
|
||||
non_table_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
@@ -1690,13 +1768,14 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
}
|
||||
}
|
||||
// Process unsplit pages normally
|
||||
let mut all_lines = crate::extractor::group_into_lines_with_thresholds_and_regions(
|
||||
unsplit_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
&page_chart_map,
|
||||
&page_image_regions,
|
||||
);
|
||||
let mut all_lines =
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_regions(
|
||||
unsplit_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
&page_chart_map,
|
||||
&page_image_regions,
|
||||
);
|
||||
// Process each split page's bands independently, then interleave
|
||||
// by Y position so paired zones (e.g. left/right months) appear together.
|
||||
let mut split_pages: Vec<u32> = split_page_items.keys().copied().collect();
|
||||
@@ -1714,7 +1793,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
.collect();
|
||||
if !band_items.is_empty() {
|
||||
page_lines.extend(
|
||||
crate::extractor::group_into_lines_with_thresholds_and_charts(
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
band_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
@@ -1748,7 +1827,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
.collect();
|
||||
if !column_items.is_empty() {
|
||||
zone_lines.extend(
|
||||
crate::extractor::group_into_lines_with_thresholds_and_charts(
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
column_items,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
@@ -1789,7 +1868,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
item.y >= low || item_is_in_chart_region(item, chart_regions)
|
||||
});
|
||||
all_lines.extend(
|
||||
crate::extractor::group_into_lines_with_thresholds_and_charts(
|
||||
crate::extractor::group_prefiltered_items_into_lines_with_thresholds_and_charts(
|
||||
chart_zone,
|
||||
page_thresholds,
|
||||
&table_page_set,
|
||||
@@ -1808,7 +1887,7 @@ pub(crate) fn to_markdown_from_items_with_rects_and_lines(
|
||||
|
||||
// Strip repeated headers/footers before conversion
|
||||
let lines = if options.strip_headers_footers {
|
||||
preprocess::strip_repeated_lines(lines, page_count)
|
||||
preprocess::strip_repeated_lines(lines, document_page_count)
|
||||
} else {
|
||||
lines
|
||||
};
|
||||
@@ -1921,6 +2000,53 @@ mod tests {
|
||||
it
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn precomputed_folio_mask_preserves_numeric_table_cells() {
|
||||
let mut items = Vec::new();
|
||||
let mut rects = Vec::new();
|
||||
for row in 0..4 {
|
||||
for column in 0..2 {
|
||||
let mut item = make_item_w(
|
||||
110.0 + column as f32 * 100.0,
|
||||
30.0 + row as f32 * 20.0,
|
||||
20.0,
|
||||
1,
|
||||
);
|
||||
item.text = (row * 2 + column + 1).to_string();
|
||||
items.push(item);
|
||||
rects.push(PdfRect {
|
||||
x: 100.0 + column as f32 * 100.0,
|
||||
y: 20.0 + row as f32 * 20.0,
|
||||
width: 100.0,
|
||||
height: 20.0,
|
||||
page: 1,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Simulate document-level folio decisions that would remove every
|
||||
// short numeric item if applied before structural table detection.
|
||||
let removal_mask = vec![true; items.len()];
|
||||
let removed_pages = HashSet::from([1]);
|
||||
let markdown = to_markdown_from_items_with_rects_and_lines(
|
||||
items,
|
||||
MarkdownOptions::default(),
|
||||
&rects,
|
||||
&[],
|
||||
MarkdownDocumentContext {
|
||||
page_thresholds: &HashMap::new(),
|
||||
struct_roles: None,
|
||||
struct_tables: &[],
|
||||
page_count: 1,
|
||||
prefiltered_page_number_pages: Some(&removed_pages),
|
||||
prefiltered_page_number_mask: Some(&removal_mask),
|
||||
},
|
||||
);
|
||||
|
||||
assert!(markdown.contains("|1|2|"), "{markdown}");
|
||||
assert!(markdown.contains("|7|8|"), "{markdown}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn early_layout_excludes_chart_items_before_column_detection() {
|
||||
let mut items = Vec::new();
|
||||
|
||||
+20
-64
@@ -3,6 +3,7 @@
|
||||
use regex::Regex;
|
||||
|
||||
use super::{MarkdownOptions, MarkdownProfile};
|
||||
use crate::text_utils::is_page_number_line;
|
||||
|
||||
/// Clean up markdown output with post-processing
|
||||
pub(crate) fn clean_markdown(mut text: String, options: &MarkdownOptions) -> String {
|
||||
@@ -145,7 +146,7 @@ fn fix_hyphenation(text: &str) -> String {
|
||||
result
|
||||
}
|
||||
|
||||
/// Remove standalone page numbers (lines that are just 1-4 digit numbers)
|
||||
/// Remove isolated page-number expressions from Markdown.
|
||||
fn remove_page_numbers(text: &str) -> String {
|
||||
let mut result = Vec::new();
|
||||
let lines: Vec<&str> = text.lines().collect();
|
||||
@@ -183,69 +184,6 @@ fn remove_page_numbers(text: &str) -> String {
|
||||
result.join("\n")
|
||||
}
|
||||
|
||||
/// Check if a line looks like a page number
|
||||
fn is_page_number_line(trimmed: &str) -> bool {
|
||||
// Empty lines are not page numbers
|
||||
if trimmed.is_empty() {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Pattern 1: Just a number (1-4 digits)
|
||||
if trimmed.len() <= 4 && trimmed.chars().all(|c| c.is_ascii_digit()) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Pattern 2: "Page X of Y" or "Page X" or "Page of" (placeholder)
|
||||
let lower = trimmed.to_lowercase();
|
||||
if let Some(rest) = lower.strip_prefix("page") {
|
||||
let rest = rest.trim();
|
||||
// "Page of" (empty page numbers)
|
||||
if rest == "of" || rest.starts_with("of ") {
|
||||
return true;
|
||||
}
|
||||
// "Page X" or "Page X of Y"
|
||||
if rest
|
||||
.chars()
|
||||
.next()
|
||||
.map(|c| c.is_ascii_digit())
|
||||
.unwrap_or(false)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
// Just "Page" followed by whitespace and maybe "of"
|
||||
if rest.is_empty()
|
||||
|| rest
|
||||
.split_whitespace()
|
||||
.all(|w| w == "of" || w.chars().all(|c| c.is_ascii_digit()))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// Pattern 3: "X of Y" where X and Y are numbers
|
||||
if let Some(of_idx) = trimmed.find(" of ") {
|
||||
let before = trimmed[..of_idx].trim();
|
||||
let after = trimmed[of_idx + 4..].trim();
|
||||
if before.chars().all(|c| c.is_ascii_digit())
|
||||
&& after.chars().all(|c| c.is_ascii_digit())
|
||||
&& !before.is_empty()
|
||||
&& !after.is_empty()
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// Pattern 4: "- X -" centered page number
|
||||
if trimmed.len() >= 3 && trimmed.starts_with('-') && trimmed.ends_with('-') {
|
||||
let inner = trimmed[1..trimmed.len() - 1].trim();
|
||||
if inner.chars().all(|c| c.is_ascii_digit()) && !inner.is_empty() {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
false
|
||||
}
|
||||
|
||||
/// Convert URLs to markdown links
|
||||
fn format_urls(text: &str) -> String {
|
||||
use once_cell::sync::Lazy;
|
||||
@@ -489,12 +427,14 @@ mod tests {
|
||||
fn test_is_page_number_page_x() {
|
||||
assert!(is_page_number_line("Page 5"));
|
||||
assert!(is_page_number_line("page 12"));
|
||||
assert!(is_page_number_line("Page123"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_is_page_number_page_x_of_y() {
|
||||
assert!(is_page_number_line("Page 3 of 10"));
|
||||
assert!(is_page_number_line("page 1 of 5"));
|
||||
assert!(is_page_number_line("Page 3 of 10 Report header"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -526,6 +466,12 @@ mod tests {
|
||||
assert!(!is_page_number_line("Total: 500"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_is_page_number_labeled_running_header() {
|
||||
assert!(is_page_number_line("Page 42 Chapter 5"));
|
||||
assert!(is_page_number_line("Page 42 explains the result"));
|
||||
}
|
||||
|
||||
// --- remove_page_numbers ---
|
||||
|
||||
#[test]
|
||||
@@ -551,6 +497,16 @@ mod tests {
|
||||
assert!(result.contains("42"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_remove_page_numbers_labeled_header_with_content() {
|
||||
let input = "Content\n\nPage 42 explains the result\n---\nEnd";
|
||||
let result = remove_page_numbers(input);
|
||||
|
||||
assert!(!result.contains("Page 42 explains the result"));
|
||||
assert!(result.contains("Content"));
|
||||
assert!(result.contains("End"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_remove_page_numbers_multiple_patterns() {
|
||||
let input = "\n1\n\nContent\n\n2\n\n---\nMore\n\n3\n";
|
||||
|
||||
@@ -7,6 +7,76 @@
|
||||
use crate::types::TextItem;
|
||||
use unicode_normalization::UnicodeNormalization;
|
||||
|
||||
/// Return whether text is an explicit page-number expression.
|
||||
///
|
||||
/// This strict form is suitable before layout, where removing one numeric item
|
||||
/// from substantive text such as `Page 42 explains the result` would lose data.
|
||||
pub(crate) fn is_explicit_page_number_expression(text: &str) -> bool {
|
||||
let trimmed = text.trim();
|
||||
if trimmed.is_empty() {
|
||||
return false;
|
||||
}
|
||||
|
||||
let is_number = |value: &str| {
|
||||
!value.is_empty() && value.chars().all(|character| character.is_ascii_digit())
|
||||
};
|
||||
|
||||
if trimmed.len() <= 4 && is_number(trimmed) {
|
||||
return true;
|
||||
}
|
||||
|
||||
if trimmed.len() >= 3 && trimmed.starts_with('-') && trimmed.ends_with('-') {
|
||||
let inner = trimmed[1..trimmed.len() - 1].trim();
|
||||
if is_number(inner) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
let lowercase = trimmed.to_ascii_lowercase();
|
||||
if let Some(rest) = lowercase.strip_prefix("page") {
|
||||
let words: Vec<&str> = rest.split_whitespace().collect();
|
||||
if words.len() >= 3 && is_number(words[0]) && words[1] == "of" && is_number(words[2]) {
|
||||
return true;
|
||||
}
|
||||
if words.len() >= 2 && words[0] == "of" && is_number(words[1]) {
|
||||
return true;
|
||||
}
|
||||
return match words.as_slice() {
|
||||
[] | ["of"] => true,
|
||||
[number] => is_number(number),
|
||||
["of", total] => is_number(total),
|
||||
[number, "of", total] => is_number(number) && is_number(total),
|
||||
_ => false,
|
||||
};
|
||||
}
|
||||
|
||||
let words: Vec<&str> = lowercase.split_whitespace().collect();
|
||||
match words.as_slice() {
|
||||
[number, "of", total] => is_number(number) && is_number(total),
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
|
||||
/// Return whether a completed Markdown line looks like a page number or a
|
||||
/// labeled running header.
|
||||
///
|
||||
/// At this stage the complete line and surrounding breaks are available, so a
|
||||
/// leading `Page N` remains compatible with the existing header cleanup even
|
||||
/// when the PDF appends a chapter or document title.
|
||||
pub(crate) fn is_page_number_line(text: &str) -> bool {
|
||||
if is_explicit_page_number_expression(text) {
|
||||
return true;
|
||||
}
|
||||
|
||||
let lowercase = text.trim().to_ascii_lowercase();
|
||||
lowercase.strip_prefix("page").is_some_and(|rest| {
|
||||
rest.trim_start()
|
||||
.chars()
|
||||
.next()
|
||||
.is_some_and(|character| character.is_ascii_digit())
|
||||
})
|
||||
}
|
||||
|
||||
/// Check if a character is CJK (Chinese, Japanese, Korean).
|
||||
/// CJK languages don't use spaces between words, so word-boundary
|
||||
/// heuristics should not apply when CJK characters are involved.
|
||||
|
||||
+276
-4
@@ -8,12 +8,13 @@ use pdf_inspector::{
|
||||
detect_pdf_type, detect_vector_grid_in_region_mem, extract_pages_markdown,
|
||||
extract_pages_markdown_mem, extract_tables_in_regions_mem, extract_text,
|
||||
extract_text_in_regions_mem, extract_text_with_positions, extract_text_with_positions_mem,
|
||||
process_pdf_mem, process_pdf_with_options, to_markdown, MarkdownOptions, PdfError, PdfOptions,
|
||||
process_pdf_mem, process_pdf_mem_with_options, process_pdf_with_options, to_markdown,
|
||||
to_markdown_from_items_with_rects_and_page_count, MarkdownOptions, PdfError, PdfOptions,
|
||||
PdfType, TextItem,
|
||||
};
|
||||
use std::collections::HashSet;
|
||||
|
||||
fn make_minimal_text_pdf() -> Vec<u8> {
|
||||
fn make_text_pdf(content: &str, media_box: &str) -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
@@ -40,10 +41,11 @@ fn make_minimal_text_pdf() -> Vec<u8> {
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
3,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>",
|
||||
&format!(
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [{media_box}] /Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>"
|
||||
),
|
||||
);
|
||||
|
||||
let content = "BT /F1 12 Tf 100 700 Td (Hello World) Tj 0 -14 Td (Second Line) Tj 0 -14 Td (Third Line) Tj ET";
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
@@ -79,6 +81,186 @@ fn make_minimal_text_pdf() -> Vec<u8> {
|
||||
pdf
|
||||
}
|
||||
|
||||
fn make_recurring_contextual_folio_pdf() -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
fn add_object(pdf: &mut Vec<u8>, offsets: &mut Vec<usize>, id: usize, body: &str) {
|
||||
offsets.push(pdf.len());
|
||||
pdf.extend_from_slice(format!("{id} 0 obj\n").as_bytes());
|
||||
pdf.extend_from_slice(body.as_bytes());
|
||||
pdf.extend_from_slice(b"\nendobj\n");
|
||||
}
|
||||
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
1,
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
2,
|
||||
"<< /Type /Pages /Kids [3 0 R 5 0 R 7 0 R 9 0 R] /Count 4 >>",
|
||||
);
|
||||
for page_index in 0..4 {
|
||||
let page_id = 3 + page_index * 2;
|
||||
let content_id = page_id + 1;
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
page_id,
|
||||
&format!(
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 11 0 R >> >> /Contents {content_id} 0 R >>"
|
||||
),
|
||||
);
|
||||
let page_number = page_index + 1;
|
||||
let content = format!(
|
||||
"BT /F1 12 Tf 1 0 0 1 25 30 Tm ({page_number}) Tj 1 0 0 1 41 30 Tm (Company report footer) Tj 1 0 0 1 72 700 Tm (Body page {page_number}) Tj ET"
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
content_id,
|
||||
&format!(
|
||||
"<< /Length {} >>\nstream\n{}\nendstream",
|
||||
content.len(),
|
||||
content
|
||||
),
|
||||
);
|
||||
}
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
11,
|
||||
"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
|
||||
);
|
||||
|
||||
let xref_start = pdf.len();
|
||||
pdf.extend_from_slice(format!("xref\n0 {}\n", offsets.len()).as_bytes());
|
||||
pdf.extend_from_slice(b"0000000000 65535 f \n");
|
||||
for offset in offsets.iter().skip(1) {
|
||||
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
|
||||
}
|
||||
pdf.extend_from_slice(
|
||||
format!(
|
||||
"trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{}\n%%EOF",
|
||||
offsets.len(),
|
||||
xref_start
|
||||
)
|
||||
.as_bytes(),
|
||||
);
|
||||
|
||||
pdf
|
||||
}
|
||||
|
||||
fn make_pdf_with_malformed_unselected_page() -> Vec<u8> {
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = vec![0usize];
|
||||
|
||||
fn add_object(pdf: &mut Vec<u8>, offsets: &mut Vec<usize>, id: usize, body: &str) {
|
||||
offsets.push(pdf.len());
|
||||
pdf.extend_from_slice(format!("{id} 0 obj\n").as_bytes());
|
||||
pdf.extend_from_slice(body.as_bytes());
|
||||
pdf.extend_from_slice(b"\nendobj\n");
|
||||
}
|
||||
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
1,
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
2,
|
||||
"<< /Type /Pages /Kids [3 0 R 5 0 R] /Count 2 >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
3,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 7 0 R >> >> /Contents 4 0 R >>",
|
||||
);
|
||||
let content = "BT /F1 12 Tf 1 0 0 1 25 30 Tm (1) Tj 1 0 0 1 41 30 Tm (Company report footer) Tj 1 0 0 1 72 700 Tm (Selected page text) Tj 0 -16 Td (More selected text) Tj 0 -16 Td (Still selected text) Tj ET";
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
4,
|
||||
&format!(
|
||||
"<< /Length {} >>\nstream\n{}\nendstream",
|
||||
content.len(),
|
||||
content
|
||||
),
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
5,
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 7 0 R >> >> /Contents 6 0 R >>",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
6,
|
||||
"<< /Length 3 >>\nstream\nBI \nendstream",
|
||||
);
|
||||
add_object(
|
||||
&mut pdf,
|
||||
&mut offsets,
|
||||
7,
|
||||
"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
|
||||
);
|
||||
|
||||
let xref_start = pdf.len();
|
||||
pdf.extend_from_slice(format!("xref\n0 {}\n", offsets.len()).as_bytes());
|
||||
pdf.extend_from_slice(b"0000000000 65535 f \n");
|
||||
for offset in offsets.iter().skip(1) {
|
||||
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
|
||||
}
|
||||
pdf.extend_from_slice(
|
||||
format!(
|
||||
"trailer\n<< /Size {} /Root 1 0 R >>\nstartxref\n{}\n%%EOF",
|
||||
offsets.len(),
|
||||
xref_start
|
||||
)
|
||||
.as_bytes(),
|
||||
);
|
||||
|
||||
pdf
|
||||
}
|
||||
|
||||
fn make_minimal_text_pdf() -> Vec<u8> {
|
||||
make_text_pdf(
|
||||
"BT /F1 12 Tf 100 700 Td (Hello World) Tj 0 -14 Td (Second Line) Tj 0 -14 Td (Third Line) Tj ET",
|
||||
"0 0 612 792",
|
||||
)
|
||||
}
|
||||
|
||||
fn make_digit_run_repro_pdf() -> Vec<u8> {
|
||||
let content = r#"BT
|
||||
/F1 12 Tf
|
||||
1 0 0 1 72 780 Tm (A\)) Tj
|
||||
1 0 0 1 96 780 Tm (The) Tj
|
||||
1 0 0 1 126 780 Tm (total) Tj
|
||||
1 0 0 1 166 780 Tm (of) Tj
|
||||
1 0 0 1 186 780 Tm (730) Tj
|
||||
1 0 0 1 220 780 Tm (seats) Tj
|
||||
1 0 0 1 262 780 Tm (was) Tj
|
||||
1 0 0 1 296 780 Tm (approved.) Tj
|
||||
1 0 0 1 72 755 Tm (B\)) Tj
|
||||
1 0 0 1 96 755 Tm (let) Tj
|
||||
1 0 0 1 120 755 Tm (log) Tj
|
||||
1 0 0 1 150 755 Tm (2) Tj
|
||||
1 0 0 1 164 755 Tm (=) Tj
|
||||
1 0 0 1 180 755 Tm (a) Tj
|
||||
1 0 0 1 72 720 Tm (C\) Control: The total of 730 seats was approved. let log 2 = a) Tj
|
||||
ET"#;
|
||||
make_text_pdf(content, "0 0 595 842")
|
||||
}
|
||||
|
||||
fn truncate_eof_marker(mut pdf: Vec<u8>) -> Vec<u8> {
|
||||
assert!(pdf.ends_with(b"%%EOF"));
|
||||
pdf.pop();
|
||||
@@ -330,6 +512,21 @@ fn test_group_into_lines_sorting_by_x() {
|
||||
assert_eq!(lines[0].text(), "First Second Third");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_digit_only_text_runs_are_preserved_in_markdown() {
|
||||
let pdf = make_digit_run_repro_pdf();
|
||||
|
||||
let items = extract_text_with_positions_mem(&pdf).expect("extract positioned text");
|
||||
assert!(items.iter().any(|item| item.text == "730"));
|
||||
assert!(items.iter().any(|item| item.text == "2"));
|
||||
|
||||
let result = process_pdf_mem(&pdf).expect("convert PDF to markdown");
|
||||
assert_eq!(
|
||||
result.markdown.expect("markdown output").trim(),
|
||||
"A) The total of 730 seats was approved.\nB) let log 2 = a\nC) Control: The total of 730 seats was approved. let log 2 = a"
|
||||
);
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// MarkdownOptions Tests
|
||||
// ============================================================================
|
||||
@@ -547,6 +744,30 @@ fn test_markdown_from_items_page_breaks() {
|
||||
assert!(md.contains("Content on second page"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_markdown_page_count_overload_includes_trailing_blank_pages_in_folio_coverage() {
|
||||
let mut items = Vec::new();
|
||||
for (page, value) in [(1, "1"), (2, "2"), (3, "3"), (4, "4")] {
|
||||
items.push(make_text_item(value, 25.0, 30.0, 12.0, page));
|
||||
items.push(make_text_item(
|
||||
"Company report footer",
|
||||
41.0,
|
||||
30.0,
|
||||
12.0,
|
||||
page,
|
||||
));
|
||||
}
|
||||
let options = MarkdownOptions {
|
||||
strip_headers_footers: false,
|
||||
..MarkdownOptions::default()
|
||||
};
|
||||
|
||||
let md = to_markdown_from_items_with_rects_and_page_count(items, options, &[], 20);
|
||||
|
||||
assert!(md.contains("1 Company report footer"));
|
||||
assert!(md.contains("4 Company report footer"));
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Markdown From Lines Tests
|
||||
// ============================================================================
|
||||
@@ -2915,6 +3136,57 @@ fn test_extract_pages_markdown_basic() {
|
||||
assert!(!result.pages[0].needs_ocr);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_uses_document_wide_folio_context() {
|
||||
let pdf = make_recurring_contextual_folio_pdf();
|
||||
let result = extract_pages_markdown_mem(&pdf, None).unwrap();
|
||||
|
||||
assert_eq!(result.pages.len(), 4);
|
||||
for (index, page) in result.pages.iter().enumerate() {
|
||||
assert!(page.markdown.contains("Company report footer"));
|
||||
assert!(
|
||||
!page
|
||||
.markdown
|
||||
.contains(&format!("{} Company report footer", index + 1)),
|
||||
"recurring contextual folio survived on page {}: {}",
|
||||
index + 1,
|
||||
page.markdown
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_process_pdf_page_filter_uses_document_wide_folio_context() {
|
||||
let pdf = make_recurring_contextual_folio_pdf();
|
||||
let result = process_pdf_mem_with_options(&pdf, PdfOptions::new().pages([1])).unwrap();
|
||||
let markdown = result.markdown.unwrap();
|
||||
|
||||
assert!(markdown.contains("Company report footer"));
|
||||
assert!(!markdown.contains("1 Company report footer"), "{markdown}");
|
||||
assert!(markdown.contains("Body page 1"));
|
||||
assert!(!markdown.contains("Body page 2"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_selected_page_ignores_context_only_extraction_failure() {
|
||||
let pdf = make_pdf_with_malformed_unselected_page();
|
||||
|
||||
let pages = extract_pages_markdown_mem(&pdf, Some(&[0])).unwrap();
|
||||
assert_eq!(pages.pages.len(), 1);
|
||||
assert!(pages.pages[0].markdown.contains("Selected page text"));
|
||||
|
||||
let result = process_pdf_mem_with_options(&pdf, PdfOptions::new().pages([1])).unwrap();
|
||||
let markdown = result.markdown.unwrap();
|
||||
assert!(markdown.contains("Selected page text"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_requested_page_extraction_failure_remains_fatal() {
|
||||
let pdf = make_pdf_with_malformed_unselected_page();
|
||||
|
||||
assert!(extract_pages_markdown_mem(&pdf, Some(&[1])).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_pages_markdown_page_ordering() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
|
||||
@@ -56,7 +56,7 @@ tips of directly from customers received other employees paid tips rec’d. entr
|
||||
**2.** Report total credit card tips (col. **b**) on Form 4070, line **2.**
|
||||
**3.** Report total tips paid out (col. **c**) on Form 4070, line **3.** **Page 4**
|
||||
|
||||
Form Employee’s Report (Rev. July 1996)
|
||||
Form **4070** Employee’s Report (Rev. July 1996)
|
||||
|
||||
## of Tips to EmployerOMB No. 1545-0065
|
||||
|
||||
@@ -81,4 +81,3 @@ forms simpler, we would be happy to hear from you. You can write to the Tax Form
|
||||
### Instructions (continued)
|
||||
|
||||
Use this space to total your tips for the year
|
||||
|
||||
|
||||
Reference in New Issue
Block a user