Compare commits
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@firecrawl/pdf-inspector",
|
||||
"version": "1.6.3",
|
||||
"version": "1.8.12",
|
||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||
"main": "index.js",
|
||||
"types": "index.d.ts",
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
import { readFileSync } from "node:fs";
|
||||
import { createRequire } from "node:module";
|
||||
|
||||
const require = createRequire(import.meta.url);
|
||||
const { detectVectorGridInRegion } = require("./index.js");
|
||||
|
||||
const pdfPath =
|
||||
process.argv[2] ?? "/tmp/pdf_inspector_indent_fixtures/cis_edge_benchmark.pdf";
|
||||
const pdf = readFileSync(pdfPath);
|
||||
const dpi = Number(process.argv[3] ?? 200);
|
||||
|
||||
const crops = [
|
||||
{ pageIdx: 29, box: [0, 0, 612, 792], label: "page30-full" },
|
||||
{ pageIdx: 16, box: [0, 0, 612, 792], label: "page17-full" },
|
||||
{ pageIdx: 23, box: [0, 0, 612, 792], label: "page24-full" },
|
||||
];
|
||||
|
||||
for (const { pageIdx, box, label } of crops) {
|
||||
const result = detectVectorGridInRegion(pdf, pageIdx, box, dpi);
|
||||
if (!result) {
|
||||
console.log(`${label}: null`);
|
||||
continue;
|
||||
}
|
||||
const rows = result.structureTokens.filter((token) => token === "<tr>").length;
|
||||
const cols = rows > 0 ? result.cellBboxes.length / rows : 0;
|
||||
console.log(
|
||||
`${label}: cells=${result.cellBboxes.length} rows=${rows} cols=${cols}`,
|
||||
);
|
||||
}
|
||||
+102
@@ -99,6 +99,13 @@ pub struct PageRegionTexts {
|
||||
pub regions: Vec<RegionText>,
|
||||
}
|
||||
|
||||
/// Vector-grid detection result compatible with `extractTablesWithStructure*`.
|
||||
#[napi(object)]
|
||||
pub struct VectorGridDetectionJs {
|
||||
pub structure_tokens: Vec<String>,
|
||||
pub cell_bboxes: Vec<Vec<f64>>,
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -317,6 +324,53 @@ pub fn extract_tables_in_regions(
|
||||
})
|
||||
}
|
||||
|
||||
/// Detect a vector ruled-line / rectangle grid inside one page region.
|
||||
///
|
||||
/// Returns TSR-compatible structure tokens plus crop-pixel cell bboxes, or
|
||||
/// `null` when the region does not contain a valid vector grid.
|
||||
///
|
||||
/// `pageIdx` is 0-indexed. `regionPdfPtBbox` is `[x1,y1,x2,y2]` in PDF
|
||||
/// points with top-left origin. `renderDpi` is the DPI of the crop image that
|
||||
/// will consume the returned cell bboxes.
|
||||
#[napi]
|
||||
pub fn detect_vector_grid_in_region(
|
||||
buffer: Buffer,
|
||||
page_idx: u32,
|
||||
region_pdf_pt_bbox: Vec<f64>,
|
||||
render_dpi: f64,
|
||||
) -> Result<Option<VectorGridDetectionJs>> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
let region = if region_pdf_pt_bbox.len() == 4 {
|
||||
[
|
||||
region_pdf_pt_bbox[0] as f32,
|
||||
region_pdf_pt_bbox[1] as f32,
|
||||
region_pdf_pt_bbox[2] as f32,
|
||||
region_pdf_pt_bbox[3] as f32,
|
||||
]
|
||||
} else {
|
||||
[0.0, 0.0, 0.0, 0.0]
|
||||
};
|
||||
|
||||
catch_panic("detect_vector_grid_in_region", move || {
|
||||
let result = pdf_inspector::detect_vector_grid_in_region_mem(
|
||||
&bytes,
|
||||
page_idx,
|
||||
region,
|
||||
render_dpi as f32,
|
||||
)
|
||||
.map_err(|e| to_napi_err(e, "detect_vector_grid_in_region"))?;
|
||||
|
||||
Ok(result.map(|r| VectorGridDetectionJs {
|
||||
structure_tokens: r.structure_tokens,
|
||||
cell_bboxes: r
|
||||
.cell_bboxes
|
||||
.into_iter()
|
||||
.map(|bbox| bbox.into_iter().map(|v| v as f64).collect())
|
||||
.collect(),
|
||||
}))
|
||||
})
|
||||
}
|
||||
|
||||
/// One cropped table region plus its raw structure-recovery output, for
|
||||
/// `extractTablesWithStructure`.
|
||||
///
|
||||
@@ -422,6 +476,54 @@ pub fn extract_tables_with_structure_cells(
|
||||
})
|
||||
}
|
||||
|
||||
/// One result from `extractTablesWithStructureAuto` — markdown plus a
|
||||
/// diagnostic flag identifying which path produced it.
|
||||
///
|
||||
/// `fallbackReason` is `null` when the TSR-hybrid path produced the
|
||||
/// markdown directly. When stage 1's quality check fires (the cells
|
||||
/// look like a SLANet detection pathology — phantom rows or multi-row
|
||||
/// content in a single cell), the auto path may expand the TSR cells
|
||||
/// in-place or run the heuristic table extractor on the same region.
|
||||
/// `fallbackReason` carries the diagnostic label (for example
|
||||
/// `"multi_row_in_cell_expanded"` or `"phantom_empty_row"`).
|
||||
#[napi(object)]
|
||||
pub struct TableExtractionResultJs {
|
||||
pub markdown: String,
|
||||
pub fallback_reason: Option<String>,
|
||||
}
|
||||
|
||||
/// Auto-fallback variant of [`extractTablesWithStructure`].
|
||||
///
|
||||
/// Runs the TSR-hybrid path, checks the resulting cells for known
|
||||
/// SLANet detection pathologies, expands multi-row cells in-place when
|
||||
/// possible, and otherwise falls back to the heuristic
|
||||
/// `extractTablesInRegions` for inputs where the TSR path looks
|
||||
/// compromised.
|
||||
///
|
||||
/// On clean inputs this returns identical markdown to
|
||||
/// `extractTablesWithStructure`; on flagged inputs `fallbackReason` is
|
||||
/// set to the recovery path that produced the result.
|
||||
#[napi]
|
||||
pub fn extract_tables_with_structure_auto(
|
||||
buffer: Buffer,
|
||||
inputs: Vec<TsrTableInputJs>,
|
||||
) -> Result<Vec<TableExtractionResultJs>> {
|
||||
let bytes: Vec<u8> = buffer.to_vec();
|
||||
let parsed = parse_tsr_inputs(&inputs);
|
||||
|
||||
catch_panic("extract_tables_with_structure_auto", move || {
|
||||
let result = pdf_inspector::extract_tables_with_structure_auto_mem(&bytes, &parsed)
|
||||
.map_err(|e| to_napi_err(e, "extract_tables_with_structure_auto"))?;
|
||||
Ok(result
|
||||
.into_iter()
|
||||
.map(|r| TableExtractionResultJs {
|
||||
markdown: r.markdown,
|
||||
fallback_reason: r.fallback_reason,
|
||||
})
|
||||
.collect())
|
||||
})
|
||||
}
|
||||
|
||||
fn parse_tsr_inputs(inputs: &[TsrTableInputJs]) -> Vec<pdf_inspector::TsrTableInput> {
|
||||
inputs
|
||||
.iter()
|
||||
|
||||
@@ -7,6 +7,7 @@ import {
|
||||
extractText,
|
||||
extractTextWithPositions,
|
||||
extractTextInRegions,
|
||||
detectVectorGridInRegion,
|
||||
extractPagesMarkdown,
|
||||
} from './index.js';
|
||||
|
||||
@@ -90,6 +91,17 @@ assert.equal(typeof regionResults[0].regions[0].text, 'string');
|
||||
assert.equal(typeof regionResults[0].regions[0].needsOcr, 'boolean');
|
||||
console.log(' extractTextInRegions: OK');
|
||||
|
||||
// --- detectVectorGridInRegion ---
|
||||
console.log('Testing detectVectorGridInRegion...');
|
||||
const vectorGrid = detectVectorGridInRegion(fixture, 0, [0, 0, 600, 800], 72);
|
||||
assert.ok(vectorGrid === null || typeof vectorGrid === 'object');
|
||||
if (vectorGrid) {
|
||||
assert.ok(Array.isArray(vectorGrid.structureTokens));
|
||||
assert.ok(Array.isArray(vectorGrid.cellBboxes));
|
||||
assert.ok(vectorGrid.cellBboxes.every(bbox => Array.isArray(bbox) && bbox.length === 4));
|
||||
}
|
||||
console.log(' detectVectorGridInRegion: OK');
|
||||
|
||||
// --- extractPagesMarkdown ---
|
||||
console.log('Testing extractPagesMarkdown...');
|
||||
|
||||
|
||||
+33
-11
@@ -1,8 +1,12 @@
|
||||
//! CLI tool for detecting PDF type (text-based vs scanned)
|
||||
|
||||
use pdf_inspector::{detect_pdf_type, process_pdf_with_options, PdfOptions, PdfType, ProcessMode};
|
||||
use pdf_inspector::{
|
||||
detect_pdf_type, detector::estimate_page_count_from_bytes, process_pdf_with_options,
|
||||
PdfOptions, PdfType, ProcessMode,
|
||||
};
|
||||
use std::env;
|
||||
use std::fmt::Write;
|
||||
use std::fs;
|
||||
use std::process;
|
||||
use std::time::Instant;
|
||||
|
||||
@@ -64,6 +68,32 @@ fn pdf_type_str(pdf_type: &PdfType) -> &'static str {
|
||||
}
|
||||
}
|
||||
|
||||
fn page_count_hint(pdf_path: &str) -> Option<u32> {
|
||||
fs::read(pdf_path)
|
||||
.ok()
|
||||
.map(|bytes| estimate_page_count_from_bytes(&bytes))
|
||||
.filter(|&count| count > 0)
|
||||
}
|
||||
|
||||
fn print_error(e: &pdf_inspector::PdfError, pdf_path: &str, json_output: bool) {
|
||||
if json_output {
|
||||
if let Some(count) = page_count_hint(pdf_path) {
|
||||
println!(
|
||||
r#"{{"error":"{}","page_count_hint":{}}}"#,
|
||||
json_escape(&e.to_string()),
|
||||
count
|
||||
);
|
||||
} else {
|
||||
println!(r#"{{"error":"{}"}}"#, json_escape(&e.to_string()));
|
||||
}
|
||||
} else {
|
||||
eprintln!("Error: {}", e);
|
||||
if let Some(count) = page_count_hint(pdf_path) {
|
||||
eprintln!("Page count hint: {}", count);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn run_analyze(pdf_path: &str, json_output: bool, start: Instant) {
|
||||
match process_pdf_with_options(pdf_path, PdfOptions::new().mode(ProcessMode::Analyze)) {
|
||||
Ok(result) => {
|
||||
@@ -135,11 +165,7 @@ fn run_analyze(pdf_path: &str, json_output: bool, start: Instant) {
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
if json_output {
|
||||
println!(r#"{{"error":"{}"}}"#, e);
|
||||
} else {
|
||||
eprintln!("Error: {}", e);
|
||||
}
|
||||
print_error(&e, pdf_path, json_output);
|
||||
process::exit(1);
|
||||
}
|
||||
}
|
||||
@@ -236,11 +262,7 @@ fn run_detect_only(pdf_path: &str, json_output: bool, start: Instant) {
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
if json_output {
|
||||
println!(r#"{{"error":"{}"}}"#, e);
|
||||
} else {
|
||||
eprintln!("Error: {}", e);
|
||||
}
|
||||
print_error(&e, pdf_path, json_output);
|
||||
process::exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
+58
-36
@@ -97,26 +97,9 @@ pub fn detect_pdf_type_with_config<P: AsRef<Path>>(
|
||||
) -> Result<PdfTypeResult, PdfError> {
|
||||
crate::validate_pdf_file(&path)?;
|
||||
|
||||
// First, load metadata only (fast operation)
|
||||
let metadata = match Document::load_metadata(&path) {
|
||||
Ok(m) => m,
|
||||
Err(ref e) if crate::is_encrypted_lopdf_error(e) => {
|
||||
Document::load_metadata_with_password(&path, "")?
|
||||
}
|
||||
Err(e) => return Err(e.into()),
|
||||
};
|
||||
let (doc, page_count) = crate::load_document_from_path(&path)?;
|
||||
|
||||
// Then load the full document for content inspection
|
||||
// We use filtered loading to skip heavy objects we don't need
|
||||
let doc = match Document::load(&path) {
|
||||
Ok(d) => d,
|
||||
Err(ref e) if crate::is_encrypted_lopdf_error(e) => {
|
||||
Document::load_with_password(&path, "")?
|
||||
}
|
||||
Err(e) => return Err(e.into()),
|
||||
};
|
||||
|
||||
detect_from_document(&doc, metadata.page_count, &config)
|
||||
detect_from_document(&doc, page_count, &config)
|
||||
}
|
||||
|
||||
/// Detect PDF type from memory buffer
|
||||
@@ -131,25 +114,64 @@ pub fn detect_pdf_type_mem_with_config(
|
||||
) -> Result<PdfTypeResult, PdfError> {
|
||||
crate::validate_pdf_bytes(buffer)?;
|
||||
|
||||
// Load metadata first (fast)
|
||||
let metadata = match Document::load_metadata_mem(buffer) {
|
||||
Ok(m) => m,
|
||||
Err(ref e) if crate::is_encrypted_lopdf_error(e) => {
|
||||
Document::load_metadata_mem_with_password(buffer, "")?
|
||||
}
|
||||
Err(e) => return Err(e.into()),
|
||||
};
|
||||
let (doc, page_count) = crate::load_document_from_mem(buffer)?;
|
||||
|
||||
// Load document for inspection
|
||||
let doc = match Document::load_mem(buffer) {
|
||||
Ok(d) => d,
|
||||
Err(ref e) if crate::is_encrypted_lopdf_error(e) => {
|
||||
Document::load_mem_with_options(buffer, lopdf::LoadOptions::with_password(""))?
|
||||
}
|
||||
Err(e) => return Err(e.into()),
|
||||
};
|
||||
detect_from_document(&doc, page_count, &config)
|
||||
}
|
||||
|
||||
detect_from_document(&doc, metadata.page_count, &config)
|
||||
/// Heuristic page-count fallback for malformed PDFs that cannot be parsed.
|
||||
///
|
||||
/// This scans raw bytes for page dictionaries (`/Type /Page`) while excluding
|
||||
/// the page tree node (`/Type /Pages`). It is intended as a low-confidence hint
|
||||
/// for diagnostics; parsed page-tree counts remain authoritative.
|
||||
pub fn estimate_page_count_from_bytes(buffer: &[u8]) -> u32 {
|
||||
let mut count = 0u32;
|
||||
let mut pos = 0usize;
|
||||
|
||||
while let Some(rel_idx) = find_bytes(&buffer[pos..], b"/Type") {
|
||||
let mut value_pos = pos + rel_idx + b"/Type".len();
|
||||
value_pos = skip_pdf_whitespace(buffer, value_pos);
|
||||
|
||||
if buffer.get(value_pos) == Some(&b'/') {
|
||||
let name_start = value_pos + 1;
|
||||
let name_end = name_start + b"Page".len();
|
||||
if name_end <= buffer.len()
|
||||
&& &buffer[name_start..name_end] == b"Page"
|
||||
&& buffer
|
||||
.get(name_end)
|
||||
.is_none_or(|b| is_pdf_name_delimiter(*b))
|
||||
{
|
||||
count += 1;
|
||||
}
|
||||
}
|
||||
|
||||
pos += rel_idx + b"/Type".len();
|
||||
}
|
||||
|
||||
count
|
||||
}
|
||||
|
||||
fn find_bytes(haystack: &[u8], needle: &[u8]) -> Option<usize> {
|
||||
haystack.windows(needle.len()).position(|w| w == needle)
|
||||
}
|
||||
|
||||
fn skip_pdf_whitespace(buffer: &[u8], mut pos: usize) -> usize {
|
||||
while pos < buffer.len() && is_pdf_whitespace(buffer[pos]) {
|
||||
pos += 1;
|
||||
}
|
||||
pos
|
||||
}
|
||||
|
||||
fn is_pdf_whitespace(byte: u8) -> bool {
|
||||
matches!(byte, b'\0' | b'\t' | b'\n' | 0x0C | b'\r' | b' ')
|
||||
}
|
||||
|
||||
fn is_pdf_name_delimiter(byte: u8) -> bool {
|
||||
is_pdf_whitespace(byte)
|
||||
|| matches!(
|
||||
byte,
|
||||
b'(' | b')' | b'<' | b'>' | b'[' | b']' | b'{' | b'}' | b'/' | b'%'
|
||||
)
|
||||
}
|
||||
|
||||
/// Detection logic on a pre-loaded document.
|
||||
|
||||
@@ -385,6 +385,7 @@ pub(crate) fn extract_page_text_items(
|
||||
&font_encodings,
|
||||
&encoding_cache,
|
||||
&mut cmap_decisions,
|
||||
&font_widths,
|
||||
) {
|
||||
let combined = multiply_matrices(&text_matrix, &ctm);
|
||||
let rendered_size = effective_font_size(current_font_size, &combined);
|
||||
@@ -533,6 +534,7 @@ pub(crate) fn extract_page_text_items(
|
||||
&font_encodings,
|
||||
&encoding_cache,
|
||||
&mut cmap_decisions,
|
||||
&font_widths,
|
||||
) {
|
||||
current_text.push_str(&text);
|
||||
}
|
||||
@@ -620,6 +622,7 @@ pub(crate) fn extract_page_text_items(
|
||||
&font_encodings,
|
||||
&encoding_cache,
|
||||
&mut cmap_decisions,
|
||||
&font_widths,
|
||||
) {
|
||||
if !text.trim().is_empty() {
|
||||
let combined = multiply_matrices(&text_matrix, &ctm);
|
||||
@@ -1012,9 +1015,17 @@ pub(crate) fn extract_page_text_items(
|
||||
// producing thousands of identical rects that yield a degenerate grid.
|
||||
// After dedup, if too few unique clip rects remain we fall through to
|
||||
// fill rects (explicitly drawn visible rectangles).
|
||||
//
|
||||
// When fill rects substantially outnumber clip rects, the clips are
|
||||
// typically section-level wrappers and the fills are the actual table
|
||||
// cell backgrounds (e.g. shaded-header tables drawn with `m`/`l`/`h`/`f*`
|
||||
// sequences). In that case, prefer fills.
|
||||
if rects.is_empty() {
|
||||
dedup_rects(&mut clip_rects);
|
||||
if clip_rects.len() >= 4 {
|
||||
let prefer_fills = !fill_rects.is_empty() && fill_rects.len() >= clip_rects.len() * 3;
|
||||
if prefer_fills {
|
||||
rects = fill_rects;
|
||||
} else if clip_rects.len() >= 4 {
|
||||
rects = clip_rects;
|
||||
} else if !fill_rects.is_empty() {
|
||||
rects = fill_rects;
|
||||
|
||||
+122
-1
@@ -720,7 +720,11 @@ pub(crate) fn extract_text_from_operand(
|
||||
font_encodings: &PageFontEncodings,
|
||||
encoding_cache: &HashMap<String, Encoding<'_>>,
|
||||
cmap_decisions: &mut CMapDecisionCache,
|
||||
font_widths: &PageFontWidths,
|
||||
) -> Option<String> {
|
||||
let is_type0_cid_font = font_widths
|
||||
.get(current_font)
|
||||
.is_some_and(|info| info.is_cid);
|
||||
let result = (|| -> Option<String> {
|
||||
if let Object::String(bytes, _) = obj {
|
||||
let mut decode_with_entry = |entry: &crate::tounicode::CMapEntry| -> Option<String> {
|
||||
@@ -962,7 +966,31 @@ pub(crate) fn extract_text_from_operand(
|
||||
return Some(symbol_text);
|
||||
}
|
||||
|
||||
// Latin-1 fallback
|
||||
// Latin-1 fallback. Safe ONLY for fonts that use single-byte
|
||||
// encodings — for these, an unmapped byte is a valid character
|
||||
// code in Latin-1/WinAnsi space. CID fonts (Type0 / Identity-H)
|
||||
// emit multi-byte CIDs that aren't characters; per-byte Latin-1
|
||||
// produces mojibake (e.g. 2-byte CID 0xCDD9 → "ÍÙ" for the
|
||||
// production scrape_id 019de78c-... samples).
|
||||
//
|
||||
// For a CID font (has_cmap is set OR a /ToUnicode reference
|
||||
// exists) with any non-ASCII bytes, emit a single U+FFFD per
|
||||
// CID instead. This both replaces the mojibake with a proper
|
||||
// "decode failed" marker AND keeps `detect_encoding_issues`
|
||||
// tripping so the page is flagged for OCR — the existing
|
||||
// garbage-detection path that the high-Latin-1 mojibake used
|
||||
// to satisfy by accident.
|
||||
if is_type0_cid_font && bytes.iter().any(|&b| b > 0x7F) {
|
||||
// 2-byte CIDs (Identity-H) are by far the common case; for
|
||||
// an odd byte count we still emit at least one marker so
|
||||
// detection downstream fires.
|
||||
let cid_count = (bytes.len() / 2).max(1);
|
||||
return Some("\u{FFFD}".repeat(cid_count));
|
||||
}
|
||||
// Pure ASCII bytes round-trip safely (Latin-1 == ASCII for
|
||||
// 0x00..=0x7F), and non-CID (Type1 / TrueType / Type3) fonts
|
||||
// use single-byte encodings where Latin-1 fallback is the
|
||||
// canonical interpretation.
|
||||
Some(bytes.iter().map(|&b| b as char).collect())
|
||||
} else {
|
||||
None
|
||||
@@ -1213,4 +1241,97 @@ mod tests {
|
||||
let bad = "###!!!@@@$$$";
|
||||
assert!(score_text(good) > score_text(bad));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cid_font_with_unparseable_cmap_does_not_emit_latin1_mojibake() {
|
||||
// Type0/CID font (font_widths reports `is_cid=true`) where the
|
||||
// ToUnicode CMap couldn't be parsed (FontCMaps doesn't have the
|
||||
// obj_num). Bytes are a 2-byte CID stream containing high bytes
|
||||
// that aren't valid UTF-8 — exactly the case in the production
|
||||
// samples (Identity-H text where the ToUnicode CMap was missing
|
||||
// or malformed, scrape_id 019de78c-..., e.g. "Í Ù Z)¿").
|
||||
//
|
||||
// Without the guard, the function falls through to the byte-by-byte
|
||||
// Latin-1 fallback and produces "ÍÙ" (U+00CD U+00D9). The correct
|
||||
// behavior is to emit U+FFFD per CID so downstream
|
||||
// `detect_encoding_issues` flags the page for OCR.
|
||||
let bytes = vec![0xCD_u8, 0xD9, 0xCD, 0xD9];
|
||||
let obj = Object::String(bytes, lopdf::StringFormat::Hexadecimal);
|
||||
|
||||
let font_cmaps = FontCMaps::default();
|
||||
let mut font_tounicode_refs: HashMap<String, u32> = HashMap::new();
|
||||
font_tounicode_refs.insert("F0".to_string(), 999);
|
||||
let inline_cmaps = HashMap::new();
|
||||
let font_encodings: PageFontEncodings = HashMap::new();
|
||||
let encoding_cache: HashMap<String, Encoding<'_>> = HashMap::new();
|
||||
let mut decisions = CMapDecisionCache::new();
|
||||
let mut font_widths: PageFontWidths = HashMap::new();
|
||||
font_widths.insert("F0".to_string(), make_font_info(&[], 1000, true));
|
||||
|
||||
let result = extract_text_from_operand(
|
||||
&obj,
|
||||
"F0",
|
||||
None,
|
||||
&font_cmaps,
|
||||
&font_tounicode_refs,
|
||||
&inline_cmaps,
|
||||
&font_encodings,
|
||||
&encoding_cache,
|
||||
&mut decisions,
|
||||
&font_widths,
|
||||
);
|
||||
|
||||
let text = result.expect("CID font fallback should still emit a marker");
|
||||
assert!(
|
||||
!text.contains('\u{00CD}') && !text.contains('\u{00D9}'),
|
||||
"CID font with unparseable CMap leaked Latin-1 mojibake: {text:?}"
|
||||
);
|
||||
assert!(
|
||||
text.contains('\u{FFFD}'),
|
||||
"CID font with unparseable CMap should emit U+FFFD so detect_encoding_issues fires: {text:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn simple_font_latin1_fallback_passes_high_bytes_through() {
|
||||
// A Type1/TrueType simple font (is_cid=false) with a `/ToUnicode`
|
||||
// reference but no usable CMap and no `/Differences` map.
|
||||
// Per-byte Latin-1 IS the canonical interpretation here — these
|
||||
// bytes are character codes, not CIDs. The CID guard must NOT
|
||||
// strip them. Reproduces the false positive that an earlier
|
||||
// version of the guard introduced for fonts in PDFs like
|
||||
// pdf-evals/Navigating-Artificial-Intelligence-..., where bytes
|
||||
// like 0xB6 are legitimate Latin-1 character codes.
|
||||
let bytes = vec![0x24_u8, 0x47, 0xB6, 0x56]; // "$G¶V"
|
||||
let obj = Object::String(bytes, lopdf::StringFormat::Hexadecimal);
|
||||
|
||||
let font_cmaps = FontCMaps::default();
|
||||
let mut font_tounicode_refs: HashMap<String, u32> = HashMap::new();
|
||||
font_tounicode_refs.insert("F1".to_string(), 999);
|
||||
let inline_cmaps = HashMap::new();
|
||||
let font_encodings: PageFontEncodings = HashMap::new();
|
||||
let encoding_cache: HashMap<String, Encoding<'_>> = HashMap::new();
|
||||
let mut decisions = CMapDecisionCache::new();
|
||||
let mut font_widths: PageFontWidths = HashMap::new();
|
||||
font_widths.insert("F1".to_string(), make_font_info(&[], 1000, false));
|
||||
|
||||
let text = extract_text_from_operand(
|
||||
&obj,
|
||||
"F1",
|
||||
None,
|
||||
&font_cmaps,
|
||||
&font_tounicode_refs,
|
||||
&inline_cmaps,
|
||||
&font_encodings,
|
||||
&encoding_cache,
|
||||
&mut decisions,
|
||||
&font_widths,
|
||||
)
|
||||
.expect("simple font should round-trip Latin-1 bytes");
|
||||
assert_eq!(text, "$G\u{00B6}V");
|
||||
assert!(
|
||||
!text.contains('\u{FFFD}'),
|
||||
"simple font fallback must not stamp FFFD over legitimate bytes: {text:?}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
+4
-28
@@ -36,26 +36,14 @@ pub(crate) use layout::ColumnRegion;
|
||||
/// Extract text from PDF file as plain string
|
||||
pub fn extract_text<P: AsRef<Path>>(path: P) -> Result<String, PdfError> {
|
||||
crate::validate_pdf_file(&path)?;
|
||||
let doc = match Document::load(&path) {
|
||||
Ok(d) => d,
|
||||
Err(ref e) if crate::is_encrypted_lopdf_error(e) => {
|
||||
Document::load_with_password(&path, "")?
|
||||
}
|
||||
Err(e) => return Err(e.into()),
|
||||
};
|
||||
let (doc, _) = crate::load_document_from_path(&path)?;
|
||||
extract_text_from_doc(&doc)
|
||||
}
|
||||
|
||||
/// Extract text from PDF memory buffer
|
||||
pub fn extract_text_mem(buffer: &[u8]) -> Result<String, PdfError> {
|
||||
crate::validate_pdf_bytes(buffer)?;
|
||||
let doc = match Document::load_mem(buffer) {
|
||||
Ok(d) => d,
|
||||
Err(ref e) if crate::is_encrypted_lopdf_error(e) => {
|
||||
Document::load_mem_with_options(buffer, lopdf::LoadOptions::with_password(""))?
|
||||
}
|
||||
Err(e) => return Err(e.into()),
|
||||
};
|
||||
let (doc, _) = crate::load_document_from_mem(buffer)?;
|
||||
extract_text_from_doc(&doc)
|
||||
}
|
||||
|
||||
@@ -91,13 +79,7 @@ pub(crate) fn extract_text_with_positions_and_rects<P: AsRef<Path>>(
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> Result<PageExtraction, PdfError> {
|
||||
crate::validate_pdf_file(&path)?;
|
||||
let doc = match Document::load(&path) {
|
||||
Ok(d) => d,
|
||||
Err(ref e) if crate::is_encrypted_lopdf_error(e) => {
|
||||
Document::load_with_password(&path, "")?
|
||||
}
|
||||
Err(e) => return Err(e.into()),
|
||||
};
|
||||
let (doc, _) = crate::load_document_from_path(&path)?;
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let (extraction, _thresholds, _gid_pages) =
|
||||
extract_positioned_text_from_doc(&doc, &font_cmaps, page_filter)?;
|
||||
@@ -124,13 +106,7 @@ pub(crate) fn extract_text_with_positions_mem_and_rects(
|
||||
page_filter: Option<&HashSet<u32>>,
|
||||
) -> Result<PageExtraction, PdfError> {
|
||||
crate::validate_pdf_bytes(buffer)?;
|
||||
let doc = match Document::load_mem(buffer) {
|
||||
Ok(d) => d,
|
||||
Err(ref e) if crate::is_encrypted_lopdf_error(e) => {
|
||||
Document::load_mem_with_options(buffer, lopdf::LoadOptions::with_password(""))?
|
||||
}
|
||||
Err(e) => return Err(e.into()),
|
||||
};
|
||||
let (doc, _) = crate::load_document_from_mem(buffer)?;
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let (extraction, _thresholds, _gid_pages) =
|
||||
extract_positioned_text_from_doc(&doc, &font_cmaps, page_filter)?;
|
||||
|
||||
@@ -373,6 +373,7 @@ fn extract_form_xobject_text_inner(
|
||||
&font_encodings,
|
||||
&encoding_cache,
|
||||
cmap_decisions,
|
||||
&font_widths,
|
||||
) {
|
||||
let combined = multiply_matrices(&text_matrix, &ctm);
|
||||
let rendered_size = effective_font_size(current_font_size, &combined);
|
||||
@@ -517,6 +518,7 @@ fn extract_form_xobject_text_inner(
|
||||
&font_encodings,
|
||||
&encoding_cache,
|
||||
cmap_decisions,
|
||||
&font_widths,
|
||||
) {
|
||||
current_text.push_str(&text);
|
||||
}
|
||||
|
||||
+2029
-68
File diff suppressed because it is too large
Load Diff
+261
-29
@@ -4,11 +4,74 @@
|
||||
//! gridlines. Many IRS forms and government PDFs use these instead of
|
||||
//! `re` (rectangle) operators.
|
||||
|
||||
use std::collections::HashSet;
|
||||
|
||||
use crate::tables::Table;
|
||||
use crate::types::{PdfLine, TextItem};
|
||||
|
||||
use super::detect_rects::{assign_items_to_grid, snap_edges};
|
||||
|
||||
/// Derive column edges from the x-endpoints of horizontal-rule
|
||||
/// segments when no vertical lines were drawn.
|
||||
///
|
||||
/// Catalog and archival-finding-aid tables are commonly drawn with
|
||||
/// per-row horizontal rules broken into N segments (one segment per
|
||||
/// cell), with no vertical dividers at all. The segment break points
|
||||
/// (e.g. `[50, 127], [127, 485], [485, 562]` per row) implicitly
|
||||
/// encode the column boundaries.
|
||||
///
|
||||
/// Returns column edges if ≥3 distinct x-positions each show up as a
|
||||
/// segment endpoint on ≥50% of the unique horizontal-line rows.
|
||||
/// Returns `None` otherwise — decorative rules with varying widths
|
||||
/// shouldn't be mistaken for a table.
|
||||
fn derive_columns_from_horizontal_segments(horizontals: &[(f32, f32, f32)]) -> Option<Vec<f32>> {
|
||||
if horizontals.len() < 3 {
|
||||
return None;
|
||||
}
|
||||
|
||||
let mut endpoints: Vec<f32> = Vec::with_capacity(horizontals.len() * 2);
|
||||
for &(_, x_min, x_max) in horizontals {
|
||||
endpoints.push(x_min);
|
||||
endpoints.push(x_max);
|
||||
}
|
||||
let clusters = snap_edges(&endpoints, 5.0);
|
||||
if clusters.len() < 3 {
|
||||
return None;
|
||||
}
|
||||
|
||||
// Bucket y-values to count unique rows. Tolerance ~0.1pt (×10
|
||||
// rounding) tolerates the snap_edges 3pt clustering used later
|
||||
// for row edges.
|
||||
let unique_rows: HashSet<i32> = horizontals
|
||||
.iter()
|
||||
.map(|&(y, _, _)| (y * 10.0).round() as i32)
|
||||
.collect();
|
||||
if unique_rows.len() < 2 {
|
||||
return None;
|
||||
}
|
||||
let min_rows = (unique_rows.len() as f32 * 0.5).ceil() as usize;
|
||||
|
||||
let qualifying: Vec<f32> = clusters
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|&cluster_x| {
|
||||
let rows_touched: HashSet<i32> = horizontals
|
||||
.iter()
|
||||
.filter(|&&(_, x_min, x_max)| {
|
||||
(x_min - cluster_x).abs() < 5.0 || (x_max - cluster_x).abs() < 5.0
|
||||
})
|
||||
.map(|&(y, _, _)| (y * 10.0).round() as i32)
|
||||
.collect();
|
||||
rows_touched.len() >= min_rows
|
||||
})
|
||||
.collect();
|
||||
|
||||
if qualifying.len() < 3 {
|
||||
return None;
|
||||
}
|
||||
Some(qualifying)
|
||||
}
|
||||
|
||||
/// Detect tables from line segments on a given page.
|
||||
///
|
||||
/// Lines are classified as horizontal or vertical, snapped into grid edges,
|
||||
@@ -52,25 +115,50 @@ pub fn detect_tables_from_lines(items: &[TextItem], lines: &[PdfLine], page: u32
|
||||
// Diagonal lines are ignored
|
||||
}
|
||||
|
||||
if horizontals.len() < 3 || verticals.len() < 2 {
|
||||
if horizontals.len() < 3 {
|
||||
return Vec::new();
|
||||
}
|
||||
|
||||
// If no/very-few vertical lines are drawn, try to derive column edges
|
||||
// from the x-endpoints of the horizontal-rule segments. Catalog and
|
||||
// archival-finding-aid layouts commonly draw each row's horizontal
|
||||
// rule as N segments (one per cell), with no vertical dividers at
|
||||
// all — the segment break points encode the column boundaries.
|
||||
let implicit_col_edges: Option<Vec<f32>> = if verticals.len() < 2 {
|
||||
derive_columns_from_horizontal_segments(&horizontals)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
if verticals.len() < 2 && implicit_col_edges.is_none() {
|
||||
return Vec::new();
|
||||
}
|
||||
let cols_from_segments = implicit_col_edges.is_some();
|
||||
|
||||
log::debug!(
|
||||
"detect_lines p{}: {} horiz, {} vert lines (of {} total on page)",
|
||||
"detect_lines p{}: {} horiz, {} vert lines (of {} total on page){}",
|
||||
page,
|
||||
horizontals.len(),
|
||||
verticals.len(),
|
||||
page_lines.len()
|
||||
page_lines.len(),
|
||||
if cols_from_segments {
|
||||
" — columns from horizontal segments"
|
||||
} else {
|
||||
""
|
||||
}
|
||||
);
|
||||
|
||||
// Snap Y-values of horizontal lines → row edges
|
||||
let h_ys: Vec<f32> = horizontals.iter().map(|(y, _, _)| *y).collect();
|
||||
let row_edges = snap_edges(&h_ys, 3.0);
|
||||
|
||||
// Snap X-values of vertical lines → column edges
|
||||
let v_xs: Vec<f32> = verticals.iter().map(|(x, _, _)| *x).collect();
|
||||
let col_edges = snap_edges(&v_xs, 3.0);
|
||||
// Column edges from drawn verticals when present, else from the
|
||||
// horizontal-segment endpoints derived above.
|
||||
let col_edges = if let Some(c) = implicit_col_edges {
|
||||
c
|
||||
} else {
|
||||
let v_xs: Vec<f32> = verticals.iter().map(|(x, _, _)| *x).collect();
|
||||
snap_edges(&v_xs, 3.0)
|
||||
};
|
||||
|
||||
log::debug!(
|
||||
"detect_lines p{}: {} row edges, {} col edges after snap",
|
||||
@@ -110,15 +198,21 @@ pub fn detect_tables_from_lines(items: &[TextItem], lines: &[PdfLine], page: u32
|
||||
return Vec::new();
|
||||
}
|
||||
|
||||
// Reject page-spanning frames: if the grid covers >90% of a standard page
|
||||
// dimension in both axes, it's a border frame, not a table.
|
||||
// Reject page-spanning frames: a decorative outer border has just 4
|
||||
// edges (top/bottom/left/right). Real full-page tables — common in
|
||||
// governmental ledgers, financial reports, etc. — span the same A4 /
|
||||
// Letter dimensions but have many internal row/column rules. Only
|
||||
// reject when the line set looks like a bare frame, not a grid.
|
||||
// Standard pages are ~595×842 (A4) or ~612×792 (Letter).
|
||||
if table_width > 500.0 && table_height > 700.0 {
|
||||
if table_width > 500.0 && table_height > 700.0 && horizontals.len() <= 4 && verticals.len() <= 4
|
||||
{
|
||||
log::debug!(
|
||||
"detect_lines p{}: rejected — page-spanning frame ({:.0}×{:.0})",
|
||||
"detect_lines p{}: rejected — page-spanning frame ({:.0}×{:.0}, {} h + {} v)",
|
||||
page,
|
||||
table_width,
|
||||
table_height
|
||||
table_height,
|
||||
horizontals.len(),
|
||||
verticals.len()
|
||||
);
|
||||
return Vec::new();
|
||||
}
|
||||
@@ -146,24 +240,33 @@ pub fn detect_tables_from_lines(items: &[TextItem], lines: &[PdfLine], page: u32
|
||||
|
||||
// Validate vertical lines: at least 2 should span a meaningful height.
|
||||
// Full spanning (>30%) is ideal, but accept many shorter lines (>10%)
|
||||
// for tables with partial column separators.
|
||||
let spanning_v = verticals
|
||||
.iter()
|
||||
.filter(|(_, y_min, y_max)| (y_max - y_min) > table_height * 0.3)
|
||||
.count();
|
||||
let partial_v = verticals
|
||||
.iter()
|
||||
.filter(|(_, y_min, y_max)| (y_max - y_min) > table_height * 0.10)
|
||||
.count();
|
||||
if spanning_v < 2 && partial_v < 4 {
|
||||
log::debug!(
|
||||
"detect_lines p{}: rejected — {} spanning + {} partial V lines",
|
||||
page,
|
||||
spanning_v,
|
||||
partial_v
|
||||
);
|
||||
return Vec::new();
|
||||
}
|
||||
// for tables with partial column separators. Skipped entirely when
|
||||
// columns came from horizontal-segment endpoints — there are no
|
||||
// vertical lines to validate against, and the segment-endpoint
|
||||
// consistency check in `derive_columns_from_horizontal_segments`
|
||||
// is the equivalent guard.
|
||||
let spanning_v = if cols_from_segments {
|
||||
0
|
||||
} else {
|
||||
let s = verticals
|
||||
.iter()
|
||||
.filter(|(_, y_min, y_max)| (y_max - y_min) > table_height * 0.3)
|
||||
.count();
|
||||
let p = verticals
|
||||
.iter()
|
||||
.filter(|(_, y_min, y_max)| (y_max - y_min) > table_height * 0.10)
|
||||
.count();
|
||||
if s < 2 && p < 4 {
|
||||
log::debug!(
|
||||
"detect_lines p{}: rejected — {} spanning + {} partial V lines",
|
||||
page,
|
||||
s,
|
||||
p
|
||||
);
|
||||
return Vec::new();
|
||||
}
|
||||
s
|
||||
};
|
||||
|
||||
// Row edges need to be in descending order (top of page = higher Y first)
|
||||
let mut row_edges_desc = row_edges;
|
||||
@@ -410,6 +513,135 @@ mod tests {
|
||||
assert!(tables.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_horizontal_segments_only_implicit_columns_accepted() {
|
||||
// Catalog/finding-aid pattern: each row's horizontal rule is
|
||||
// drawn as 3 segments at consistent x-endpoints (50, 127, 485,
|
||||
// 562), with no vertical lines anywhere. The segment break
|
||||
// points must be inferred as column edges.
|
||||
let mut lines = Vec::new();
|
||||
// Slightly uneven row spacing so the chart-gridline rejector
|
||||
// (CV < 0.02) doesn't fire.
|
||||
let row_ys = [80.0_f32, 145.0, 215.0, 280.0, 350.0, 415.0, 485.0];
|
||||
for &y in &row_ys {
|
||||
lines.push(make_hline(y, 50.0, 127.0, 1));
|
||||
lines.push(make_hline(y, 127.0, 485.0, 1));
|
||||
lines.push(make_hline(y, 485.0, 562.0, 1));
|
||||
}
|
||||
// Populate every cell so capture / density checks pass.
|
||||
let mut items = Vec::new();
|
||||
for w in row_ys.windows(2) {
|
||||
let row_y = (w[0] + w[1]) / 2.0;
|
||||
items.push(make_item("id", 80.0, row_y, 1));
|
||||
items.push(make_item("description here", 200.0, row_y, 1));
|
||||
items.push(make_item("date", 510.0, row_y, 1));
|
||||
}
|
||||
let tables = detect_tables_from_lines(&items, &lines, 1);
|
||||
assert_eq!(
|
||||
tables.len(),
|
||||
1,
|
||||
"horizontal-segment-only grid should be accepted"
|
||||
);
|
||||
let t = &tables[0];
|
||||
assert!(
|
||||
t.cells.len() >= 4,
|
||||
"expected ≥4 rows, got {}",
|
||||
t.cells.len()
|
||||
);
|
||||
assert_eq!(t.cells[0].len(), 3, "expected 3 columns");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_horizontal_segments_with_inconsistent_endpoints_rejected() {
|
||||
// Decorative rules of varying widths shouldn't be detected as a
|
||||
// table — each line has its own x-endpoints, no consistent
|
||||
// column boundary survives the 50%-of-rows threshold.
|
||||
let lines = vec![
|
||||
make_hline(100.0, 50.0, 150.0, 1),
|
||||
make_hline(200.0, 50.0, 220.0, 1),
|
||||
make_hline(300.0, 50.0, 310.0, 1),
|
||||
make_hline(400.0, 50.0, 470.0, 1),
|
||||
];
|
||||
let items = vec![
|
||||
make_item("decorative", 100.0, 150.0, 1),
|
||||
make_item("text", 100.0, 250.0, 1),
|
||||
];
|
||||
let tables = detect_tables_from_lines(&items, &lines, 1);
|
||||
assert!(
|
||||
tables.is_empty(),
|
||||
"varying-width decorative rules should not be detected"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_page_spanning_bare_frame_rejected() {
|
||||
// Just an outer A4-sized rectangle: 2 horizontals + 2 verticals.
|
||||
// No internal structure → decorative border, not a table.
|
||||
let lines = vec![
|
||||
make_hline(20.0, 20.0, 575.0, 1), // top
|
||||
make_hline(820.0, 20.0, 575.0, 1), // bottom
|
||||
make_vline(20.0, 20.0, 820.0, 1), // left
|
||||
make_vline(575.0, 20.0, 820.0, 1), // right
|
||||
];
|
||||
let items = vec![
|
||||
make_item("title", 100.0, 100.0, 1),
|
||||
make_item("body", 100.0, 200.0, 1),
|
||||
];
|
||||
let tables = detect_tables_from_lines(&items, &lines, 1);
|
||||
assert!(
|
||||
tables.is_empty(),
|
||||
"Page-sized 4-edge frame should be rejected as decoration"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_page_spanning_grid_with_internal_lines_accepted() {
|
||||
// Full-page table (governmental-ledger pattern): A4-sized grid
|
||||
// that previously hit the "page-spanning frame" early reject
|
||||
// before downstream validation could even look at it.
|
||||
// Verticals span the full table height so we isolate the
|
||||
// frame-vs-grid decision under test.
|
||||
let mut lines = Vec::new();
|
||||
// 13 horizontal rules: header + 12 row separators
|
||||
let h_ys = [
|
||||
22.5, 37.9, 95.5, 144.5, 184.9, 233.9, 291.7, 340.7, 415.8, 499.6, 574.7, 623.7, 698.8,
|
||||
];
|
||||
for &y in &h_ys {
|
||||
lines.push(make_hline(y, 22.6, 566.6, 1));
|
||||
}
|
||||
// 7 column dividers spanning full table height.
|
||||
let v_xs = [22.6, 66.3, 116.3, 186.6, 263.1, 493.5, 566.5];
|
||||
for &x in &v_xs {
|
||||
lines.push(make_vline(x, 22.5, 698.8, 1));
|
||||
}
|
||||
// Populate every cell so the capture-ratio + density checks pass.
|
||||
let mut items = Vec::new();
|
||||
for r in 0..(h_ys.len() - 1) {
|
||||
let row_y = (h_ys[r] + h_ys[r + 1]) / 2.0;
|
||||
for c in 0..(v_xs.len() - 1) {
|
||||
let col_x = (v_xs[c] + v_xs[c + 1]) / 2.0;
|
||||
items.push(make_item("x", col_x, row_y, 1));
|
||||
}
|
||||
}
|
||||
let tables = detect_tables_from_lines(&items, &lines, 1);
|
||||
assert_eq!(
|
||||
tables.len(),
|
||||
1,
|
||||
"Full-page table with internal grid should be accepted"
|
||||
);
|
||||
let t = &tables[0];
|
||||
assert!(
|
||||
t.cells.len() >= 6,
|
||||
"expected ≥6 rows, got {}",
|
||||
t.cells.len()
|
||||
);
|
||||
assert!(
|
||||
t.cells[0].len() >= 3,
|
||||
"expected ≥3 columns, got {}",
|
||||
t.cells[0].len()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_single_column_rejected() {
|
||||
// Only 2 col edges (1 column) — not a table even with verticals
|
||||
|
||||
+626
-41
@@ -285,7 +285,11 @@ pub fn detect_tables_from_rects(
|
||||
//
|
||||
// Only remove when the container is a similarly-sized cell (height
|
||||
// ratio < 4×), NOT when the container is a table-wide background
|
||||
// that dwarfs the sub-rect.
|
||||
// that dwarfs the sub-rect. Origin-anchored page-background rects
|
||||
// also disqualify as containers — they normally exceed the 4× ratio,
|
||||
// but when the sub-rect is itself a tall table-frame the ratio can
|
||||
// fall under the gate, and dropping the frame collapses cluster
|
||||
// adjacency between adjacent column-cell groups.
|
||||
//
|
||||
// Skip this O(n²) dedup when there are too many rects — pages with
|
||||
// thousands of vector-drawing rects won't benefit from cell dedup.
|
||||
@@ -295,9 +299,11 @@ pub fn detect_tables_from_rects(
|
||||
page_rects.retain(|&(ax, ay, aw, ah)| {
|
||||
let tol = 2.0;
|
||||
!snapshot.iter().any(|&(bx, by, bw, bh)| {
|
||||
let container_is_page_bg = bx < 5.0 && by < 5.0;
|
||||
// b must strictly contain a (b is larger in area)
|
||||
bw * bh > aw * ah * 1.2
|
||||
&& bh < ah * 4.0 // container must be similarly sized, not a table background
|
||||
&& !container_is_page_bg
|
||||
&& bx <= ax + tol
|
||||
&& (bx + bw) >= (ax + aw) - tol
|
||||
&& by <= ay + tol
|
||||
@@ -1388,13 +1394,15 @@ fn detect_row_stripe_table(
|
||||
.max()
|
||||
.unwrap_or(0);
|
||||
// Allow longer cells for multi-column tables (descriptions in one column
|
||||
// are common). Single-column or 2-column "tables" with giant cells are
|
||||
// almost always layout backgrounds.
|
||||
// are common). Narrow grids with giant cells are usually layout
|
||||
// backgrounds — but only when the row count is also small. A 4+-row
|
||||
// key/value table with one descriptive column reads as a real table
|
||||
// on every other gate, so don't reject it on cell length alone.
|
||||
let max_allowed = if num_cols >= 3 { 2000 } else { 500 };
|
||||
if max_cell_len > max_allowed {
|
||||
if max_cell_len > max_allowed && non_empty_rows < 4 {
|
||||
debug!(
|
||||
" row-stripe rejected: max cell length {} > {} (layout background)",
|
||||
max_cell_len, max_allowed
|
||||
" row-stripe rejected: max cell length {} > {} (layout background, {} rows)",
|
||||
max_cell_len, max_allowed, non_empty_rows
|
||||
);
|
||||
return None;
|
||||
}
|
||||
@@ -1587,25 +1595,111 @@ fn detect_row_stripe_table_from_cell_rects(
|
||||
return None;
|
||||
}
|
||||
|
||||
// Derive columns from text X-position clustering
|
||||
// Derive columns from text X-position clustering, but prefer rect
|
||||
// X-edges when they already provide a tighter scaffold. Some PDFs draw
|
||||
// only the row-index cells in the body plus a full header row; that is
|
||||
// not dense enough for `try_build_grid`, but the header rects still define
|
||||
// the real columns. Text starts inside wide cells can otherwise split the
|
||||
// table into spurious sub-columns.
|
||||
let columns = cluster_x_positions(&page_items, 15.0);
|
||||
if columns.len() < 2 {
|
||||
let text_col_edges = if columns.len() >= 2 {
|
||||
let mut edges: Vec<f32> = Vec::with_capacity(columns.len() + 1);
|
||||
let min_x = page_items.iter().map(|(_, i)| i.x).reduce(f32::min)?;
|
||||
edges.push(min_x - 5.0);
|
||||
for pair in columns.windows(2) {
|
||||
edges.push((pair[0] + pair[1]) / 2.0);
|
||||
}
|
||||
let max_x_right = page_items
|
||||
.iter()
|
||||
.map(|(_, i)| i.x + i.width)
|
||||
.reduce(f32::max)?;
|
||||
edges.push(max_x_right + 5.0);
|
||||
Some(edges)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
|
||||
let rect_col_edges = {
|
||||
let mut x_vals = Vec::with_capacity(content_rects.len() * 2);
|
||||
for &&(x, _, w, _) in &content_rects {
|
||||
x_vals.push(x);
|
||||
x_vals.push(x + w);
|
||||
}
|
||||
let mut edges = snap_edges(&x_vals, 6.0);
|
||||
edges.sort_by(|a, b| a.total_cmp(b));
|
||||
if (3..=26).contains(&edges.len()) {
|
||||
Some(edges)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
};
|
||||
|
||||
// For wired-grid tables whose header text is centered/right-aligned but
|
||||
// whose data is left-aligned, cluster_x_positions can drop the header-only
|
||||
// x-cluster in its singleton-filter pass and merge adjacent data clusters
|
||||
// when the gap is below threshold, losing a column. Rect borders are
|
||||
// ground truth in that case — but only when each rect column actually
|
||||
// holds text. Decorative or background rects (prose laid out in a frame,
|
||||
// cell-fill rects with extra borders) can produce more rect-derived
|
||||
// columns than the text supports; preferring rects there would split a
|
||||
// logical column into spurious sub-columns.
|
||||
let rect_cols_match_text = match (&rect_col_edges, &text_col_edges) {
|
||||
(Some(rect_edges), _) if rect_edges.len() >= 4 => {
|
||||
let num_rect_cols = rect_edges.len() - 1;
|
||||
let mut col_item_counts = vec![0usize; num_rect_cols];
|
||||
for (_, item) in &page_items {
|
||||
let cx = item.x + item.width / 2.0;
|
||||
for c in 0..num_rect_cols {
|
||||
if cx >= rect_edges[c] - 2.0 && cx <= rect_edges[c + 1] + 2.0 {
|
||||
col_item_counts[c] += 1;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
// Require every rect column to hold multiple text items. A rect
|
||||
// column with no (or only one) item is decorative or the rect grid
|
||||
// is detecting a spurious column the data does not need; in those
|
||||
// cases the old text-cluster preference is the safer fallback.
|
||||
col_item_counts.iter().all(|&n| n >= 2)
|
||||
}
|
||||
_ => false,
|
||||
};
|
||||
|
||||
let (col_edges, columns_from_text) = match (rect_col_edges, text_col_edges) {
|
||||
(Some(rect_edges), text_edges_opt) if rect_cols_match_text => {
|
||||
debug!(
|
||||
" cell-rect using {} rect-derived columns (text clusters: {}; rect cols well-distributed)",
|
||||
rect_edges.len() - 1,
|
||||
text_edges_opt
|
||||
.as_ref()
|
||||
.map(|e| (e.len() - 1) as i32)
|
||||
.unwrap_or(-1)
|
||||
);
|
||||
(rect_edges, false)
|
||||
}
|
||||
(Some(rect_edges), Some(text_edges)) if rect_edges.len() <= text_edges.len() => {
|
||||
debug!(
|
||||
" cell-rect using {} rect-derived columns over {} text clusters",
|
||||
rect_edges.len() - 1,
|
||||
text_edges.len() - 1
|
||||
);
|
||||
(rect_edges, false)
|
||||
}
|
||||
(_, Some(text_edges)) => (text_edges, true),
|
||||
(Some(rect_edges), None) => (rect_edges, false),
|
||||
(None, None) => {
|
||||
debug!(
|
||||
" cell-rect rejected: only {} columns from text clustering",
|
||||
columns.len()
|
||||
);
|
||||
return None;
|
||||
}
|
||||
};
|
||||
|
||||
if col_edges.len() < 3 {
|
||||
return None;
|
||||
}
|
||||
|
||||
// Build column edges
|
||||
let mut col_edges: Vec<f32> = Vec::with_capacity(columns.len() + 1);
|
||||
let min_x = page_items.iter().map(|(_, i)| i.x).reduce(f32::min)?;
|
||||
col_edges.push(min_x - 5.0);
|
||||
for pair in columns.windows(2) {
|
||||
col_edges.push((pair[0] + pair[1]) / 2.0);
|
||||
}
|
||||
let max_x_right = page_items
|
||||
.iter()
|
||||
.map(|(_, i)| i.x + i.width)
|
||||
.reduce(f32::max)?;
|
||||
col_edges.push(max_x_right + 5.0);
|
||||
|
||||
let num_cols = col_edges.len() - 1;
|
||||
let num_rows = row_edges.len() - 1;
|
||||
|
||||
@@ -1617,12 +1711,25 @@ fn detect_row_stripe_table_from_cell_rects(
|
||||
page_items.len()
|
||||
);
|
||||
|
||||
let (cells, item_indices) = assign_items_to_grid(items, &col_edges, &row_edges, page);
|
||||
let (mut cells, item_indices) = assign_items_to_grid(items, &col_edges, &row_edges, page);
|
||||
|
||||
if item_indices.is_empty() {
|
||||
return None;
|
||||
}
|
||||
|
||||
let mut row_edges = row_edges;
|
||||
let (collapsed_cells, collapsed_row_edges, collapsed_rows) =
|
||||
collapse_multiline_description_rows(cells, row_edges, &col_edges);
|
||||
let has_wrapped_description_rows = collapsed_rows > 0;
|
||||
cells = collapsed_cells;
|
||||
row_edges = collapsed_row_edges;
|
||||
if collapsed_rows > 0 {
|
||||
debug!(
|
||||
" cell-rect collapsed {} wrapped description rows",
|
||||
collapsed_rows
|
||||
);
|
||||
}
|
||||
|
||||
// Validate: >=2 non-empty rows, >=25% density
|
||||
let non_empty_rows = cells
|
||||
.iter()
|
||||
@@ -1636,6 +1743,7 @@ fn detect_row_stripe_table_from_cell_rects(
|
||||
return None;
|
||||
}
|
||||
|
||||
let num_rows = cells.len();
|
||||
let total_cells = (num_cols * num_rows) as f32;
|
||||
let non_empty_cells = cells
|
||||
.iter()
|
||||
@@ -1655,17 +1763,21 @@ fn detect_row_stripe_table_from_cell_rects(
|
||||
return None;
|
||||
}
|
||||
|
||||
// Reject tables with paragraph-length cells (layout backgrounds, not tables)
|
||||
// Reject tables with paragraph-length cells — typically layout
|
||||
// backgrounds (sidebars, banners) where a single big rectangle
|
||||
// contains a wall of prose. Spare multi-row key/value tables where
|
||||
// the value column is a multi-bullet description: those pass every
|
||||
// other gate and shouldn't get killed on cell length alone.
|
||||
let max_cell_len = cells
|
||||
.iter()
|
||||
.flat_map(|row| row.iter())
|
||||
.map(|c| c.len())
|
||||
.max()
|
||||
.unwrap_or(0);
|
||||
if max_cell_len > 500 {
|
||||
if max_cell_len > 500 && non_empty_rows < 4 {
|
||||
debug!(
|
||||
" cell-rect rejected: max cell length {} > 500",
|
||||
max_cell_len
|
||||
" cell-rect rejected: max cell length {} > 500 ({} rows, layout background)",
|
||||
max_cell_len, non_empty_rows
|
||||
);
|
||||
return None;
|
||||
}
|
||||
@@ -1681,13 +1793,36 @@ fn detect_row_stripe_table_from_cell_rects(
|
||||
|
||||
// Reject "tables" that are actually prose in a framed region.
|
||||
// Columns here come from text X-position clustering; when prose wraps
|
||||
// inside a bounding-box rect (e.g. chat-transcript figures) the
|
||||
// word-boundary gaps cluster into many spurious columns, and the
|
||||
// resulting cells hold sentence fragments riddled with common English
|
||||
// function words. Count cells with any such word and reject when
|
||||
// 20%+ of non-empty cells match — real tabular data (labels, units,
|
||||
// numbers) rarely contains these words.
|
||||
if num_cols >= 4 {
|
||||
// inside a bounding-box rect (e.g. chat-transcript figures, two-column
|
||||
// legal-text blocks in forms) the word-boundary gaps cluster into
|
||||
// spurious columns, and the resulting cells hold sentence fragments
|
||||
// riddled with common English function words.
|
||||
//
|
||||
// Apply at any column count >= 2. The 2-col case is the bite — a
|
||||
// paragraph wrapped into 2 justified columns produces the same
|
||||
// surface signal as a real "label / value" table in the
|
||||
// well-distributed-cols check (both cols populated), so we need a
|
||||
// content-based signal to tell them apart.
|
||||
//
|
||||
// Layered checks combine after the 20%-of-cells prose-word
|
||||
// trigger fires:
|
||||
// (a) Long-cell content: prose-in-a-frame averages ~70-100 chars
|
||||
// per non-empty cell (sentence fragments); real data tables
|
||||
// are typically <30 chars, occasionally up to ~55 for
|
||||
// descriptive 4-col tables. The 65-char threshold cleanly
|
||||
// separates them on observed fixtures (accessory_building
|
||||
// prose=74 chars, upstage data=53, greencomp=20). This
|
||||
// overrides the well-distributed relaxation — long cells
|
||||
// are the strongest prose signal even when both cols are
|
||||
// populated.
|
||||
// (b) Two-column text-only scaffold: when both columns were inferred
|
||||
// from text starts rather than rect edges, prose fragments can look
|
||||
// perfectly balanced. Require rect evidence for this relaxed shape.
|
||||
// (c) Well-distributed columns: ≥75% of cols hold ≥2 non-empty
|
||||
// cells. Catches the prose-paragraph-as-many-cols shape
|
||||
// while admitting real "label / value / description /
|
||||
// benefit"-style tables.
|
||||
if num_cols >= 2 {
|
||||
const PROSE_WORDS: &[&str] = &[
|
||||
"a", "an", "the", "of", "to", "is", "was", "are", "were", "be", "been", "in", "on",
|
||||
"at", "with", "for", "by", "as", "and", "or", "but", "this", "that", "these", "those",
|
||||
@@ -1697,6 +1832,7 @@ fn detect_row_stripe_table_from_cell_rects(
|
||||
];
|
||||
let mut prose_cells = 0usize;
|
||||
let mut counted = 0usize;
|
||||
let mut total_chars = 0usize;
|
||||
for row in &cells {
|
||||
for cell in row {
|
||||
let t = cell.trim();
|
||||
@@ -1704,6 +1840,7 @@ fn detect_row_stripe_table_from_cell_rects(
|
||||
continue;
|
||||
}
|
||||
counted += 1;
|
||||
total_chars += t.chars().count();
|
||||
let lower = t.to_ascii_lowercase();
|
||||
let has_prose_word = lower
|
||||
.split(|c: char| !c.is_ascii_alphabetic() && c != '\'')
|
||||
@@ -1714,11 +1851,65 @@ fn detect_row_stripe_table_from_cell_rects(
|
||||
}
|
||||
}
|
||||
if counted > 0 && prose_cells * 5 >= counted {
|
||||
// (a) Long-cell content: overrides the well-distributed
|
||||
// relaxation. The 2-col prose-in-a-frame case populates
|
||||
// both cols (passes well-distributed) but every cell
|
||||
// holds a sentence fragment, so mean cell length is the
|
||||
// discriminator.
|
||||
const PROSE_MEAN_CHAR_THRESHOLD: usize = 65;
|
||||
let mean_chars = total_chars / counted;
|
||||
if mean_chars > PROSE_MEAN_CHAR_THRESHOLD && !has_wrapped_description_rows {
|
||||
debug!(
|
||||
" cell-rect rejected: prose-in-frame, mean non-empty cell {} chars > {} (prose words {}/{})",
|
||||
mean_chars, PROSE_MEAN_CHAR_THRESHOLD, prose_cells, counted
|
||||
);
|
||||
return None;
|
||||
} else if mean_chars > PROSE_MEAN_CHAR_THRESHOLD {
|
||||
debug!(
|
||||
" cell-rect prose check relaxed: wrapped description rows, mean {} chars (prose words {}/{})",
|
||||
mean_chars, prose_cells, counted
|
||||
);
|
||||
}
|
||||
|
||||
// (b) Two text-derived columns are not enough vector evidence once
|
||||
// the content looks prose-like. Real 2-col rect tables still pass
|
||||
// when the column scaffold comes from drawn cell geometry.
|
||||
if columns_from_text && num_cols == 2 {
|
||||
debug!(
|
||||
" cell-rect rejected: prose-in-frame with text-derived 2-col scaffold (mean {} chars, prose words {}/{})",
|
||||
mean_chars, prose_cells, counted
|
||||
);
|
||||
return None;
|
||||
}
|
||||
|
||||
// (c) Well-distributed columns.
|
||||
let filled_cols = (0..num_cols)
|
||||
.filter(|&c| {
|
||||
cells
|
||||
.iter()
|
||||
.filter(|row| {
|
||||
!row.get(c)
|
||||
.map(String::as_str)
|
||||
.unwrap_or("")
|
||||
.trim()
|
||||
.is_empty()
|
||||
})
|
||||
.count()
|
||||
>= 2
|
||||
})
|
||||
.count();
|
||||
let well_distributed = filled_cols * 4 >= num_cols * 3;
|
||||
if !well_distributed {
|
||||
debug!(
|
||||
" cell-rect rejected: {}/{} cells contain prose function words — likely prose ({}/{} cols filled, mean {} chars)",
|
||||
prose_cells, counted, filled_cols, num_cols, mean_chars
|
||||
);
|
||||
return None;
|
||||
}
|
||||
debug!(
|
||||
" cell-rect rejected: {}/{} cells contain prose function words — likely prose",
|
||||
prose_cells, counted
|
||||
" cell-rect prose check relaxed: {}/{} cols filled, mean {} chars — table-with-description-col",
|
||||
filled_cols, num_cols, mean_chars
|
||||
);
|
||||
return None;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1739,6 +1930,132 @@ fn detect_row_stripe_table_from_cell_rects(
|
||||
Some(Table::new(column_centers, row_centers, cells, item_indices))
|
||||
}
|
||||
|
||||
/// Merge wrapped description-line bands back into their visual data rows.
|
||||
///
|
||||
/// Some Word/PDF exports draw enough rectangle geometry to prove a table exists
|
||||
/// but expose Y bands per wrapped text line instead of per cell row. In the
|
||||
/// common mapping-table shape, a narrow row-label column precedes one wide
|
||||
/// description column, and wrapped continuation bands have content only in that
|
||||
/// wide column. Merge only that high-confidence shape so framed prose still
|
||||
/// falls through the existing prose guards.
|
||||
fn collapse_multiline_description_rows(
|
||||
cells: Vec<Vec<String>>,
|
||||
row_edges: Vec<f32>,
|
||||
col_edges: &[f32],
|
||||
) -> (Vec<Vec<String>>, Vec<f32>, usize) {
|
||||
let num_rows = cells.len();
|
||||
let num_cols = col_edges.len().saturating_sub(1);
|
||||
if num_rows < 3 || num_cols < 3 || row_edges.len() != num_rows + 1 {
|
||||
return (cells, row_edges, 0);
|
||||
}
|
||||
|
||||
let table_width = col_edges[num_cols] - col_edges[0];
|
||||
if table_width <= 0.0 {
|
||||
return (cells, row_edges, 0);
|
||||
}
|
||||
|
||||
let Some((description_col, description_width)) = (0..num_cols)
|
||||
.map(|c| (c, col_edges[c + 1] - col_edges[c]))
|
||||
.max_by(|a, b| a.1.total_cmp(&b.1))
|
||||
else {
|
||||
return (cells, row_edges, 0);
|
||||
};
|
||||
|
||||
// Require a preceding row-label column. Without it (e.g. a prose frame
|
||||
// split into text-start columns), "one populated wide column" is not enough
|
||||
// evidence to find visual row starts safely.
|
||||
if description_col == 0 || description_width < table_width * 0.35 {
|
||||
return (cells, row_edges, 0);
|
||||
}
|
||||
|
||||
let row_has_left_label = |row: &[String]| {
|
||||
row.iter()
|
||||
.take(description_col)
|
||||
.any(|cell| !cell.trim().is_empty())
|
||||
};
|
||||
let labeled_rows = cells.iter().filter(|row| row_has_left_label(row)).count();
|
||||
if labeled_rows < 2 {
|
||||
return (cells, row_edges, 0);
|
||||
}
|
||||
|
||||
let mut merged_rows = 0usize;
|
||||
let mut wrapped_description_rows = 0usize;
|
||||
let mut new_cells: Vec<Vec<String>> = Vec::with_capacity(num_rows);
|
||||
let mut new_edges = Vec::with_capacity(row_edges.len());
|
||||
new_edges.push(row_edges[0]);
|
||||
|
||||
for (row_idx, row) in cells.into_iter().enumerate() {
|
||||
let desc_text = row
|
||||
.get(description_col)
|
||||
.map(String::as_str)
|
||||
.unwrap_or("")
|
||||
.trim();
|
||||
let left_label = row_has_left_label(&row);
|
||||
let non_desc_non_empty = row
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(col, cell)| *col != description_col && !cell.trim().is_empty())
|
||||
.count();
|
||||
|
||||
// Wrapped continuation bands contain only description-column text.
|
||||
// The preceding label/marker column is empty because the visual row's
|
||||
// label cell spans the whole wrapped block.
|
||||
let is_description_continuation = row_idx > 0
|
||||
&& !desc_text.is_empty()
|
||||
&& !left_label
|
||||
&& non_desc_non_empty == 0
|
||||
&& !new_cells.is_empty();
|
||||
|
||||
// Header cells are often split as "Controls" / "Version" in the first
|
||||
// column while the other header labels sit on the first band.
|
||||
let only_first_col = row
|
||||
.iter()
|
||||
.enumerate()
|
||||
.all(|(col, cell)| col == 0 || cell.trim().is_empty());
|
||||
let is_header_continuation = row_idx > 0
|
||||
&& only_first_col
|
||||
&& row
|
||||
.first()
|
||||
.is_some_and(|cell| !cell.trim().is_empty() && cell.chars().count() <= 24)
|
||||
&& !new_cells.is_empty()
|
||||
&& new_cells
|
||||
.last()
|
||||
.is_some_and(|prev| prev.iter().filter(|c| !c.trim().is_empty()).count() >= 2);
|
||||
|
||||
if is_description_continuation || is_header_continuation {
|
||||
if let Some(prev) = new_cells.last_mut() {
|
||||
for (col, cell) in row.iter().enumerate() {
|
||||
let text = cell.trim();
|
||||
if text.is_empty() {
|
||||
continue;
|
||||
}
|
||||
if !prev[col].trim().is_empty() {
|
||||
prev[col].push(' ');
|
||||
}
|
||||
prev[col].push_str(text);
|
||||
}
|
||||
}
|
||||
merged_rows += 1;
|
||||
if is_description_continuation {
|
||||
wrapped_description_rows += 1;
|
||||
}
|
||||
} else {
|
||||
if !new_cells.is_empty() {
|
||||
new_edges.push(row_edges[row_idx]);
|
||||
}
|
||||
new_cells.push(row);
|
||||
}
|
||||
}
|
||||
|
||||
new_edges.push(*row_edges.last().unwrap());
|
||||
|
||||
if merged_rows == 0 || new_cells.len() < 2 || new_edges.len() != new_cells.len() + 1 {
|
||||
return (new_cells, row_edges, 0);
|
||||
}
|
||||
|
||||
(new_cells, new_edges, wrapped_description_rows)
|
||||
}
|
||||
|
||||
/// Detect a table by merging all cluster rects into one group.
|
||||
///
|
||||
/// This handles clip-path PDFs where each column's cell rects form a separate
|
||||
@@ -1874,18 +2191,20 @@ fn detect_merged_cluster_table(
|
||||
return None;
|
||||
}
|
||||
|
||||
// Reject if any cell has excessive text — layout background rects produce
|
||||
// "cells" containing paragraphs, not short data-table values.
|
||||
// Reject if any cell has excessive text — layout background rects
|
||||
// produce "cells" containing paragraphs, not short data-table values.
|
||||
// Multi-row key/value tables can legitimately have one column of
|
||||
// long descriptive text, so only reject narrow-row layouts here.
|
||||
let max_cell_len = cells
|
||||
.iter()
|
||||
.flat_map(|row| row.iter())
|
||||
.map(|c| c.len())
|
||||
.max()
|
||||
.unwrap_or(0);
|
||||
if max_cell_len > 500 {
|
||||
if max_cell_len > 500 && non_empty_rows < 4 {
|
||||
debug!(
|
||||
" merged-cluster rejected: max cell length {} > 500 (layout background)",
|
||||
max_cell_len
|
||||
" merged-cluster rejected: max cell length {} > 500 ({} rows, layout background)",
|
||||
max_cell_len, non_empty_rows
|
||||
);
|
||||
return None;
|
||||
}
|
||||
@@ -2315,6 +2634,46 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_row_stripe_accepts_multi_row_key_value_long_cells() {
|
||||
// Multi-row 2-column key/value table where one value cell holds
|
||||
// a paragraph (>500 chars). The old `max_cell_len > 500` check
|
||||
// rejected this shape as a "layout background"; with the
|
||||
// multi-row guard, it should be accepted.
|
||||
let mut rects = Vec::new();
|
||||
let row_h = 25.0_f32;
|
||||
let y_top = 700.0_f32;
|
||||
for i in 0..8 {
|
||||
let y = y_top - (i as f32) * row_h;
|
||||
rects.push((40.0, y, 510.0, row_h));
|
||||
}
|
||||
let mut items = Vec::new();
|
||||
for i in 0..8 {
|
||||
let row_center_y = y_top - (i as f32) * row_h + row_h / 2.0;
|
||||
// Left column: short label
|
||||
items.push(make_item(&format!("Field {}", i), 45.0, row_center_y, 10.0));
|
||||
// Right column: short value, except the last row which is a paragraph
|
||||
let value = if i == 7 {
|
||||
"X".repeat(800)
|
||||
} else {
|
||||
"value".to_string()
|
||||
};
|
||||
items.push(make_item(&value, 300.0, row_center_y, 10.0));
|
||||
}
|
||||
let result = detect_row_stripe_table(&items, &rects, 1);
|
||||
assert!(
|
||||
result.is_some(),
|
||||
"multi-row key/value table with one long cell should be accepted"
|
||||
);
|
||||
let t = result.unwrap();
|
||||
assert!(
|
||||
t.cells.len() >= 4,
|
||||
"expected ≥4 rows, got {}",
|
||||
t.cells.len()
|
||||
);
|
||||
assert_eq!(t.cells[0].len(), 2, "expected 2 columns");
|
||||
}
|
||||
|
||||
// --- propagate_merged_cells ---
|
||||
|
||||
#[test]
|
||||
@@ -2929,6 +3288,232 @@ mod tests {
|
||||
// If tables were detected, that's also acceptable
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn text_derived_two_col_prose_is_not_cell_rect_table() {
|
||||
let page = 1;
|
||||
let mut rects = Vec::new();
|
||||
for row in 0..8 {
|
||||
rects.push(PdfRect {
|
||||
x: 50.0,
|
||||
y: 100.0 + row as f32 * 20.0,
|
||||
width: 180.0,
|
||||
height: 18.0,
|
||||
page,
|
||||
});
|
||||
}
|
||||
|
||||
let mut items = Vec::new();
|
||||
let left = [
|
||||
"the annual plan was revised",
|
||||
"and the team noted changes",
|
||||
"this section explains limits",
|
||||
"with additional notes below",
|
||||
"the policy was reviewed",
|
||||
"and results are summarized",
|
||||
"this appendix describes scope",
|
||||
"with examples for reference",
|
||||
];
|
||||
let right = [
|
||||
"for each area in the review",
|
||||
"as part of the assessment",
|
||||
"that were applied in context",
|
||||
"to support the conclusion",
|
||||
"for use by the committee",
|
||||
"as shown in the narrative",
|
||||
"that remain under discussion",
|
||||
"to clarify the method",
|
||||
];
|
||||
for row in 0..8 {
|
||||
let y = 104.0 + row as f32 * 20.0;
|
||||
let mut left_item = make_item(left[row], 60.0, y, 9.0);
|
||||
left_item.width = 50.0;
|
||||
items.push(left_item);
|
||||
let mut right_item = make_item(right[row], 150.0, y, 9.0);
|
||||
right_item.width = 50.0;
|
||||
items.push(right_item);
|
||||
}
|
||||
|
||||
let (tables, _hints) = detect_tables_from_rects(&items, &rects, page);
|
||||
assert!(
|
||||
tables.is_empty(),
|
||||
"text-derived two-column prose must not be accepted as a rect table; got {:?}",
|
||||
tables
|
||||
.iter()
|
||||
.map(|t| (t.rows.len(), t.columns.len()))
|
||||
.collect::<Vec<_>>()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn multiline_indented_description_rows_collapse_to_visual_rows() {
|
||||
let page = 1;
|
||||
let col_edges = [0.0, 60.0, 420.0, 460.0, 500.0, 540.0];
|
||||
let row_edges = [
|
||||
340.0, 320.0, 300.0, 270.0, 250.0, 230.0, 200.0, 180.0, 160.0,
|
||||
];
|
||||
|
||||
let mut rects = Vec::new();
|
||||
for row in 0..row_edges.len() - 1 {
|
||||
let y_top = row_edges[row];
|
||||
let y_bot = row_edges[row + 1];
|
||||
for col in 0..col_edges.len() - 1 {
|
||||
rects.push((
|
||||
col_edges[col],
|
||||
y_bot,
|
||||
col_edges[col + 1] - col_edges[col],
|
||||
y_top - y_bot,
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
let mut items = vec![
|
||||
make_item("Controls", 8.0, 330.0, 9.0),
|
||||
make_item("Control", 70.0, 330.0, 9.0),
|
||||
make_item("IG 1", 428.0, 330.0, 9.0),
|
||||
make_item("IG 2", 468.0, 330.0, 9.0),
|
||||
make_item("IG 3", 508.0, 330.0, 9.0),
|
||||
make_item("Version", 8.0, 310.0, 9.0),
|
||||
make_item("v8", 20.0, 285.0, 9.0),
|
||||
make_item(
|
||||
"4.5 Implement and Manage a Firewall on End-User Devices",
|
||||
70.0,
|
||||
285.0,
|
||||
9.0,
|
||||
),
|
||||
make_item("*", 438.0, 285.0, 9.0),
|
||||
make_item("*", 478.0, 285.0, 9.0),
|
||||
make_item("*", 518.0, 285.0, 9.0),
|
||||
make_item("v7", 20.0, 215.0, 9.0),
|
||||
make_item(
|
||||
"9.4 Apply Host-based Firewalls or Port-Filtering",
|
||||
70.0,
|
||||
215.0,
|
||||
9.0,
|
||||
),
|
||||
make_item("*", 478.0, 215.0, 9.0),
|
||||
make_item("*", 518.0, 215.0, 9.0),
|
||||
];
|
||||
items.push(make_item(
|
||||
"Implement and manage a host-based firewall or port-filtering tool",
|
||||
84.0,
|
||||
260.0,
|
||||
8.0,
|
||||
));
|
||||
items.push(make_item(
|
||||
"on end-user devices with a default-deny rule",
|
||||
84.0,
|
||||
240.0,
|
||||
8.0,
|
||||
));
|
||||
items.push(make_item(
|
||||
"Apply host-based firewalls or port filtering tools on end systems",
|
||||
84.0,
|
||||
190.0,
|
||||
8.0,
|
||||
));
|
||||
items.push(make_item(
|
||||
"and deny unauthorized network communication",
|
||||
84.0,
|
||||
170.0,
|
||||
8.0,
|
||||
));
|
||||
|
||||
let table = detect_row_stripe_table_from_cell_rects(&items, &rects, page)
|
||||
.expect("expected multiline description table");
|
||||
assert_eq!(table.columns.len(), 5);
|
||||
assert_eq!(
|
||||
table.rows.len(),
|
||||
3,
|
||||
"wrapped lines should collapse to header plus two data rows"
|
||||
);
|
||||
assert_eq!(table.cells[0][0], "Controls Version");
|
||||
assert!(table.cells[1][1].contains("host-based firewall"));
|
||||
assert!(table.cells[1][1].contains("default-deny rule"));
|
||||
assert!(table.cells[2][1].contains("deny unauthorized"));
|
||||
}
|
||||
|
||||
/// Wire-bordered 4-column table whose header text is centered/right-aligned
|
||||
/// inside each cell while the data is left-aligned: cluster_x_positions
|
||||
/// merges adjacent columns (data Item→EAN gap is below threshold) and
|
||||
/// drops the header-only x-clusters in the filter pass, leaving only 3
|
||||
/// text-derived columns. Rect borders are 4 columns of ground truth.
|
||||
/// Before the fix the cell-rect path preferred text edges when they were
|
||||
/// the smaller set — losing a column. After the fix, 3+ rect columns
|
||||
/// always win.
|
||||
#[test]
|
||||
fn wired_header_data_misaligned_keeps_all_columns_from_rects() {
|
||||
let page = 1;
|
||||
// 4 cols: Item | EAN | Nombre | Cant
|
||||
let col_xs = [380.0_f32, 410.0, 470.0, 660.0, 700.0];
|
||||
// Header + 9 data rows at 15pt tall each (y descending).
|
||||
let row_ys: Vec<f32> = (0..=10).map(|r| 400.0 - 15.0 * r as f32).collect();
|
||||
|
||||
let mut rects: Vec<(f32, f32, f32, f32)> = Vec::new();
|
||||
for r in 0..10 {
|
||||
let y_top = row_ys[r];
|
||||
let y_bot = row_ys[r + 1];
|
||||
for c in 0..4 {
|
||||
rects.push((col_xs[c], y_bot, col_xs[c + 1] - col_xs[c], y_top - y_bot));
|
||||
}
|
||||
}
|
||||
|
||||
let mut items: Vec<TextItem> = Vec::new();
|
||||
// Header row (y ≈ 392.5): headers sit further to the right than data
|
||||
// because they are centered/right-aligned in the cells.
|
||||
items.push(make_item("Item", 389.0, 392.5, 9.0));
|
||||
items.push(make_item("EAN", 432.0, 392.5, 9.0));
|
||||
items.push(make_item("Nombre", 552.0, 392.5, 9.0));
|
||||
items.push(make_item("Cant", 672.0, 392.5, 9.0));
|
||||
|
||||
let names = [
|
||||
"Arnes Frontal",
|
||||
"Arnes Motor",
|
||||
"Arnes Piso",
|
||||
"Arnes Techo",
|
||||
"Arnes Puerta",
|
||||
"Arnes Tablero",
|
||||
"Arnes Trasero",
|
||||
"Arnes Lateral",
|
||||
"Arnes Sensor",
|
||||
];
|
||||
for r in 0..9 {
|
||||
let y = 377.5 - 15.0 * r as f32;
|
||||
items.push(make_item(&(r + 1).to_string(), 396.0, y, 9.0));
|
||||
items.push(make_item("7701023403016", 410.0, y, 9.0));
|
||||
items.push(make_item(names[r], 480.0, y, 9.0));
|
||||
items.push(make_item("1", 680.0, y, 9.0));
|
||||
}
|
||||
|
||||
let table = detect_row_stripe_table_from_cell_rects(&items, &rects, page)
|
||||
.expect("wired 4-column table with header/data x-misalignment must detect");
|
||||
assert_eq!(
|
||||
table.columns.len(),
|
||||
4,
|
||||
"expected 4 columns from rect borders; cells: {:?}",
|
||||
table.cells
|
||||
);
|
||||
for c in 0..4 {
|
||||
let any_populated = table.cells.iter().any(|row| !row[c].trim().is_empty());
|
||||
assert!(
|
||||
any_populated,
|
||||
"column {} empty across all rows; cells: {:?}",
|
||||
c, table.cells
|
||||
);
|
||||
}
|
||||
// Header row populated in all 4 cells.
|
||||
let header = &table.cells[0];
|
||||
assert_eq!(header[0].trim(), "Item");
|
||||
assert_eq!(header[1].trim(), "EAN");
|
||||
assert_eq!(header[2].trim(), "Nombre");
|
||||
assert_eq!(header[3].trim(), "Cant");
|
||||
// First data row: Item="1", EAN, name, count="1" — no Item↔EAN merge.
|
||||
let data1 = &table.cells[1];
|
||||
assert_eq!(data1[0].trim(), "1");
|
||||
assert_eq!(data1[1].trim(), "7701023403016");
|
||||
assert!(data1[2].trim().contains("Arnes"));
|
||||
assert_eq!(data1[3].trim(), "1");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn failed_cluster_no_hint_without_items() {
|
||||
// Rects with no text items inside → no failed-cluster hint generated.
|
||||
|
||||
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
+1296
-7
File diff suppressed because it is too large
Load Diff
@@ -54,9 +54,14 @@ company other than a life insurance company shall make a return on Form 1120PC.
|
||||
|
||||
annual statement (or a pro forma annual statement), including the underwriting and investment exhibit for the year covered by such return.
|
||||
|
||||
||(3) Foreign insurance companies. The provisions of paragraphs (c)(1) and|
|
||||
|---|---|
|
||||
||(c)(2) of this section concerning the returns and statements of insurance companies subject to tax under section 801 or section 831 also apply to foreign insurance companies subject to tax under those sections, except that the copy of the annual statement required to be submitted with the return shall, in the case of a foreign insurance company that is not required to file an annual statement, be a copy of the pro forma annual statement relating to the United States business of such company. (4) Exception for insurance companies filing their Federal income tax returns electronically. If an insurance company described in paragraph (c)(1), (c)(2), or (c)(3) of this section files its Federal income tax return electronically, it should not include on or with such return its annual statement (or pro forma annual statement), or any portion thereof. Such statement must be available at all times for inspection by authorized Internal Revenue Service officers or employees and retained for so long as such statements may be material in the administration of any internal revenue law. See §1.6001-1(e). (5) Definition. For purposes of this section, the term annual statement means the annual statement, the form of which is approved by the National Association of Insurance Commissioners (NAIC), which is filed by an insurance company for the year with the insurance departments of States, Territories, and the District of|
|
||||
(3) Foreign insurance companies. The provisions of paragraphs (c)(1) and
|
||||
(c)(2) of this section concerning the returns and statements of insurance companies subject to tax under section 801 or section 831 also apply to foreign insurance companies subject to tax under those sections, except that the copy of the annual statement required to be submitted with the return shall, in the case of a foreign insurance company that is not required to file an annual statement, be a copy of the pro forma annual statement relating to the United States business of such company.
|
||||
(4) Exception for insurance companies filing their Federal income tax returns
|
||||
electronically. If an insurance company described in paragraph (c)(1), (c)(2), or
|
||||
|
||||
(c)(3) of this section files its Federal income tax return electronically, it should not include on or with such return its annual statement (or pro forma annual statement), or any portion thereof. Such statement must be available at all times for inspection by authorized Internal Revenue Service officers or employees and retained for so long as such statements may be material in the administration of any internal revenue law. See §1.6001-1(e).
|
||||
(5) Definition. For purposes of this section, the term annual statement means
|
||||
the annual statement, the form of which is approved by the National Association of Insurance Commissioners (NAIC), which is filed by an insurance company for the year with the insurance departments of States, Territories, and the District of
|
||||
|
||||
Columbia. The term annual statement also includes a pro forma annual statement if the insurance company is not required to file the NAIC annual statement.
|
||||
|
||||
@@ -201,3 +206,4 @@ CFR part or section where Current OMB identified or described control No.
|
||||
Deputy Commissioner for Services and Enforcement.
|
||||
|
||||
Approved: May 19, 2006 Eric Solomon Acting Deputy Assistant Secretary of the Treasury (Tax Policy).
|
||||
|
||||
|
||||
Reference in New Issue
Block a user