Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2b5611e9c1 | ||
|
|
41410e61c5 | ||
|
|
e83101f8d7 |
@@ -46,8 +46,6 @@ jobs:
|
||||
target: x86_64-unknown-linux-gnu
|
||||
- os: macos-14
|
||||
target: aarch64-apple-darwin
|
||||
- os: windows-latest
|
||||
target: x86_64-pc-windows-msvc
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
|
||||
+2
-3
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@firecrawl/pdf-inspector",
|
||||
"version": "1.3.0",
|
||||
"version": "1.0.0",
|
||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||
"main": "index.js",
|
||||
"types": "index.d.ts",
|
||||
@@ -38,8 +38,7 @@
|
||||
"binaryName": "pdf-inspector",
|
||||
"targets": [
|
||||
"x86_64-unknown-linux-gnu",
|
||||
"aarch64-apple-darwin",
|
||||
"x86_64-pc-windows-msvc"
|
||||
"aarch64-apple-darwin"
|
||||
],
|
||||
"package": {
|
||||
"name": "@firecrawl/pdf-inspector-js"
|
||||
|
||||
@@ -581,14 +581,8 @@ fn detect_table_in_region(items: &[(usize, &TextItem)], mode: TableDetectionMode
|
||||
// Validation 1: some rows should have content in first column.
|
||||
// Use a lower threshold (25%) for tables with wrapped cells where
|
||||
// continuation lines leave the first column empty.
|
||||
// Skip when cells form a narrow TOC pattern: hierarchical entries indented
|
||||
// across multiple X levels leave the leftmost column sparse (only top-level
|
||||
// chapters land there) but the structure is still a valid TOC. Narrow only
|
||||
// (<=5 cols) — wide multi-column TOCs (e.g. 2-up indices) would render
|
||||
// poorly through format_toc_as_list, which assumes one entry per row.
|
||||
let rows_with_first_col = cells.iter().filter(|row| !row[0].is_empty()).count();
|
||||
let is_narrow_toc = columns.len() <= 5 && is_table_of_contents(&cells);
|
||||
if rows_with_first_col < rows.len() / 4 && !is_narrow_toc {
|
||||
if rows_with_first_col < rows.len() / 4 {
|
||||
log::debug!(
|
||||
" validation 1 fail: {}/{} rows have first col",
|
||||
rows_with_first_col,
|
||||
@@ -659,12 +653,8 @@ fn detect_table_in_region(items: &[(usize, &TextItem)], mode: TableDetectionMode
|
||||
return None;
|
||||
}
|
||||
|
||||
// Validation 8: Reject paragraph-like content falsely detected as tables.
|
||||
// TOC pages with deep indentation (top-level chapters in col 0, subsections
|
||||
// in cols 1-3, page numbers in last col) leave most cells empty and trip
|
||||
// the paragraph heuristic; TOC shape is a safer signal here. Narrow only
|
||||
// — see narrow-TOC rationale at validation 1.
|
||||
if is_paragraph_content(&cells) && !is_narrow_toc {
|
||||
// Validation 8: Reject paragraph-like content falsely detected as tables
|
||||
if is_paragraph_content(&cells) {
|
||||
log::debug!(" validation 9 fail: paragraph content");
|
||||
return None;
|
||||
}
|
||||
@@ -1611,62 +1601,6 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn is_table_of_contents_accepts_hierarchical_indented_toc() {
|
||||
// Mythos system card pages 4-5: top-level chapters indent at col 0,
|
||||
// subsections at cols 1-2, leaving col 0 mostly empty (only ~10% of
|
||||
// rows). Validation 1 was rejecting these even though the structure
|
||||
// is unambiguously a TOC.
|
||||
let cells = vec![
|
||||
vec!["Abstract".to_string(), String::new(), "3".to_string()],
|
||||
vec![
|
||||
"1 Introduction".to_string(),
|
||||
String::new(),
|
||||
"10".to_string(),
|
||||
],
|
||||
vec![
|
||||
String::new(),
|
||||
"1.1 Model training".to_string(),
|
||||
"11".to_string(),
|
||||
],
|
||||
vec![
|
||||
String::new(),
|
||||
"1.1.1 Training data".to_string(),
|
||||
"11".to_string(),
|
||||
],
|
||||
vec![
|
||||
String::new(),
|
||||
"1.1.2 Crowd workers".to_string(),
|
||||
"12".to_string(),
|
||||
],
|
||||
vec![
|
||||
String::new(),
|
||||
"1.2 Release decision".to_string(),
|
||||
"13".to_string(),
|
||||
],
|
||||
vec![
|
||||
"2 RSP evaluations".to_string(),
|
||||
String::new(),
|
||||
"16".to_string(),
|
||||
],
|
||||
vec![
|
||||
String::new(),
|
||||
"2.1 RSP risk assessment".to_string(),
|
||||
"16".to_string(),
|
||||
],
|
||||
vec![String::new(), "2.1.1 Context".to_string(), "16".to_string()],
|
||||
vec![
|
||||
String::new(),
|
||||
"2.2 CB evaluations".to_string(),
|
||||
"20".to_string(),
|
||||
],
|
||||
];
|
||||
assert!(
|
||||
is_table_of_contents(&cells),
|
||||
"hierarchical TOC with sparse col 0 should still be detected"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn is_table_of_contents_rejects_dotless_toc() {
|
||||
// Tabular TOC without leader dots: first column starts with dotted
|
||||
|
||||
@@ -520,18 +520,6 @@ impl ToUnicodeCMap {
|
||||
}
|
||||
}
|
||||
|
||||
/// Get the maximum source CID across all mappings (char_map + ranges).
|
||||
fn max_source_cid(&self) -> Option<u16> {
|
||||
let char_max = self.char_map.keys().copied().max();
|
||||
let range_max = self.ranges.iter().map(|&(_, end, _)| end).max();
|
||||
match (char_max, range_max) {
|
||||
(Some(a), Some(b)) => Some(a.max(b)),
|
||||
(a @ Some(_), None) => a,
|
||||
(None, b @ Some(_)) => b,
|
||||
(None, None) => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Remap a CMap that references pre-subsetting GIDs to sequential post-subsetting GIDs.
|
||||
/// Collects all source CIDs, sorts them, and reassigns to 1, 2, 3, ...
|
||||
pub fn remap_to_sequential(&self) -> ToUnicodeCMap {
|
||||
@@ -669,81 +657,6 @@ fn get_w_array_start_cid(cid_font_dict: &lopdf::Dictionary, doc: &Document) -> O
|
||||
}
|
||||
}
|
||||
|
||||
/// Return true if the CIDFont's W (widths) array explicitly covers the given CID.
|
||||
///
|
||||
/// The W array uses two formats (PDF 32000-1:2008, §9.7.4.3):
|
||||
/// 1. `c [w1 w2 ... wn]` — widths for CIDs c, c+1, ..., c+n-1
|
||||
/// 2. `c_first c_last w` — CIDs c_first..c_last all have width w
|
||||
fn w_array_covers_cid(cid_font_dict: &lopdf::Dictionary, doc: &Document, target: u16) -> bool {
|
||||
let Ok(w_obj) = cid_font_dict.get(b"W") else {
|
||||
return false;
|
||||
};
|
||||
let arr = match w_obj {
|
||||
Object::Array(arr) => arr,
|
||||
Object::Reference(r) => match doc.get_object(*r) {
|
||||
Ok(Object::Array(arr)) => arr,
|
||||
_ => return false,
|
||||
},
|
||||
_ => return false,
|
||||
};
|
||||
|
||||
let resolve_int = |o: &Object| -> Option<i64> {
|
||||
match o {
|
||||
Object::Integer(n) => Some(*n),
|
||||
Object::Reference(r) => match doc.get_object(*r) {
|
||||
Ok(Object::Integer(n)) => Some(*n),
|
||||
_ => None,
|
||||
},
|
||||
_ => None,
|
||||
}
|
||||
};
|
||||
|
||||
let resolve_arr = |o: &Object| -> Option<Vec<Object>> {
|
||||
match o {
|
||||
Object::Array(a) => Some(a.clone()),
|
||||
Object::Reference(r) => match doc.get_object(*r) {
|
||||
Ok(Object::Array(a)) => Some(a.clone()),
|
||||
_ => None,
|
||||
},
|
||||
_ => None,
|
||||
}
|
||||
};
|
||||
|
||||
let target = target as i64;
|
||||
let mut i = 0usize;
|
||||
while i < arr.len() {
|
||||
let Some(first) = resolve_int(&arr[i]) else {
|
||||
break;
|
||||
};
|
||||
i += 1;
|
||||
if i >= arr.len() {
|
||||
break;
|
||||
}
|
||||
// Peek at arr[i] to decide format.
|
||||
if let Some(widths) = resolve_arr(&arr[i]) {
|
||||
// Format 1: c [w1 ... wn]
|
||||
let last = first + widths.len() as i64 - 1;
|
||||
if target >= first && target <= last {
|
||||
return true;
|
||||
}
|
||||
i += 1;
|
||||
} else if let Some(last) = resolve_int(&arr[i]) {
|
||||
// Format 2: c_first c_last w
|
||||
i += 1;
|
||||
if i < arr.len() {
|
||||
i += 1; // skip the width value
|
||||
}
|
||||
if target >= first && target <= last {
|
||||
return true;
|
||||
}
|
||||
} else {
|
||||
// Unknown token — abort parsing safely
|
||||
break;
|
||||
}
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
/// Extract CIDToGIDMap as a vector of GIDs (u16) indexed by CID.
|
||||
fn get_cid_to_gid_map(cid_font_dict: &lopdf::Dictionary, doc: &Document) -> Option<Vec<u16>> {
|
||||
let obj = cid_font_dict.get(b"CIDToGIDMap").ok()?;
|
||||
@@ -839,20 +752,6 @@ fn try_remap_subset_cmap(
|
||||
_ => return (cmap, None),
|
||||
};
|
||||
|
||||
// If the W array actually covers the CMap's max source CID, the CMap is
|
||||
// aligned with the font — no sequential renumbering happened. A sparse W
|
||||
// array starting at CID 0 (for .notdef) with additional high-CID entries
|
||||
// matching the CMap is the normal subset layout, not a mismatch.
|
||||
if let Some(max_cid) = cmap.max_source_cid() {
|
||||
if w_array_covers_cid(cid_font_dict, doc, max_cid) {
|
||||
debug!(
|
||||
"Subset remap skipped for obj={}: W array covers CMap max CID {}",
|
||||
obj_num, max_cid
|
||||
);
|
||||
return (cmap, None);
|
||||
}
|
||||
}
|
||||
|
||||
debug!(
|
||||
"Subset GID mismatch detected for obj={}: W starts at CID {}, CMap min CID {}. Remapping to sequential.",
|
||||
obj_num, w_start, min_cid
|
||||
@@ -2818,187 +2717,4 @@ endbfchar
|
||||
assert_eq!(remapped.unwrap().char_map.len(), 50);
|
||||
assert_eq!(fallback.unwrap().char_map.len(), 10);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_max_source_cid() {
|
||||
let cmap_content = r#"
|
||||
1 begincodespacerange
|
||||
<0000><FFFF>
|
||||
endcodespacerange
|
||||
2 beginbfchar
|
||||
<0003> <0020>
|
||||
<0031> <004E>
|
||||
endbfchar
|
||||
1 beginbfrange
|
||||
<0208> <0227> <0430>
|
||||
endbfrange
|
||||
"#;
|
||||
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
|
||||
assert_eq!(cmap.min_source_cid(), Some(0x0003));
|
||||
assert_eq!(cmap.max_source_cid(), Some(0x0227));
|
||||
}
|
||||
|
||||
/// Helper: build a minimal CIDFont dict with a W array and check coverage.
|
||||
fn cid_font_dict_with_w(w_items: Vec<lopdf::Object>) -> lopdf::Dictionary {
|
||||
let mut d = lopdf::Dictionary::new();
|
||||
d.set("W", lopdf::Object::Array(w_items));
|
||||
d
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_w_array_covers_cid_format1() {
|
||||
// Format 1: `c [w1 w2 ... wn]` — widths for CIDs c..c+n-1.
|
||||
// Mimics the 16.pdf Tahoma W array: 0[1000] 3[313] 5[401] 11[383 383] 16[363 303 382]
|
||||
let doc = Document::new();
|
||||
let d = cid_font_dict_with_w(vec![
|
||||
lopdf::Object::Integer(0),
|
||||
lopdf::Object::Array(vec![lopdf::Object::Integer(1000)]),
|
||||
lopdf::Object::Integer(3),
|
||||
lopdf::Object::Array(vec![lopdf::Object::Integer(313)]),
|
||||
lopdf::Object::Integer(5),
|
||||
lopdf::Object::Array(vec![lopdf::Object::Integer(401)]),
|
||||
lopdf::Object::Integer(11),
|
||||
lopdf::Object::Array(vec![
|
||||
lopdf::Object::Integer(383),
|
||||
lopdf::Object::Integer(383),
|
||||
]),
|
||||
lopdf::Object::Integer(16),
|
||||
lopdf::Object::Array(vec![
|
||||
lopdf::Object::Integer(363),
|
||||
lopdf::Object::Integer(303),
|
||||
lopdf::Object::Integer(382),
|
||||
]),
|
||||
lopdf::Object::Integer(570),
|
||||
lopdf::Object::Array(vec![lopdf::Object::Integer(667); 26]),
|
||||
]);
|
||||
|
||||
assert!(w_array_covers_cid(&d, &doc, 0));
|
||||
assert!(w_array_covers_cid(&d, &doc, 3));
|
||||
assert!(w_array_covers_cid(&d, &doc, 5));
|
||||
assert!(w_array_covers_cid(&d, &doc, 11));
|
||||
assert!(w_array_covers_cid(&d, &doc, 12));
|
||||
assert!(w_array_covers_cid(&d, &doc, 16));
|
||||
assert!(w_array_covers_cid(&d, &doc, 18));
|
||||
assert!(w_array_covers_cid(&d, &doc, 570));
|
||||
assert!(w_array_covers_cid(&d, &doc, 595));
|
||||
// Gaps are NOT covered
|
||||
assert!(!w_array_covers_cid(&d, &doc, 1));
|
||||
assert!(!w_array_covers_cid(&d, &doc, 4));
|
||||
assert!(!w_array_covers_cid(&d, &doc, 19));
|
||||
assert!(!w_array_covers_cid(&d, &doc, 596));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_w_array_covers_cid_format2() {
|
||||
// Format 2: `c_first c_last w` — CIDs c_first..c_last all have width w.
|
||||
let doc = Document::new();
|
||||
let d = cid_font_dict_with_w(vec![
|
||||
lopdf::Object::Integer(100),
|
||||
lopdf::Object::Integer(120),
|
||||
lopdf::Object::Integer(500),
|
||||
]);
|
||||
|
||||
assert!(w_array_covers_cid(&d, &doc, 100));
|
||||
assert!(w_array_covers_cid(&d, &doc, 110));
|
||||
assert!(w_array_covers_cid(&d, &doc, 120));
|
||||
assert!(!w_array_covers_cid(&d, &doc, 99));
|
||||
assert!(!w_array_covers_cid(&d, &doc, 121));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_w_array_covers_cid_missing_w() {
|
||||
let doc = Document::new();
|
||||
let d = lopdf::Dictionary::new();
|
||||
assert!(!w_array_covers_cid(&d, &doc, 3));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_try_remap_skipped_when_w_covers_cmap() {
|
||||
// Simulates 16.pdf: CMap's max source CID (0x0279 = 633) is explicitly
|
||||
// in the W array, so no subset-renumbering happened — remap must NOT fire.
|
||||
let cmap_content = r#"
|
||||
1 begincodespacerange
|
||||
<0000><FFFF>
|
||||
endcodespacerange
|
||||
2 beginbfchar
|
||||
<0003> <0020>
|
||||
<0031> <004E>
|
||||
endbfchar
|
||||
2 beginbfrange
|
||||
<023A> <0253> <0410>
|
||||
<0255> <0279> <042B>
|
||||
endbfrange
|
||||
"#;
|
||||
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
|
||||
|
||||
let mut doc = Document::new();
|
||||
// Build a CIDFont dict with Identity CIDToGIDMap and a W array that
|
||||
// covers CID 633 via `597 [widths...]`.
|
||||
let mut cid_font = lopdf::Dictionary::new();
|
||||
cid_font.set("CIDToGIDMap", lopdf::Object::Name(b"Identity".to_vec()));
|
||||
cid_font.set(
|
||||
"W",
|
||||
lopdf::Object::Array(vec![
|
||||
lopdf::Object::Integer(0),
|
||||
lopdf::Object::Array(vec![lopdf::Object::Integer(750)]),
|
||||
lopdf::Object::Integer(597),
|
||||
lopdf::Object::Array(vec![lopdf::Object::Integer(500); 37]), // 597..633
|
||||
]),
|
||||
);
|
||||
let cid_font_id = doc.add_object(cid_font);
|
||||
|
||||
// Build the Type0 font dict with Identity-H + DescendantFonts ref.
|
||||
let mut font_dict = lopdf::Dictionary::new();
|
||||
font_dict.set("Encoding", lopdf::Object::Name(b"Identity-H".to_vec()));
|
||||
font_dict.set(
|
||||
"DescendantFonts",
|
||||
lopdf::Object::Array(vec![lopdf::Object::Reference(cid_font_id)]),
|
||||
);
|
||||
|
||||
let (primary, remapped) = try_remap_subset_cmap(cmap, &font_dict, &doc, 123);
|
||||
assert!(
|
||||
remapped.is_none(),
|
||||
"Remap must be skipped when W covers CMap max CID (this is 16.pdf)"
|
||||
);
|
||||
assert_eq!(primary.lookup(0x0003), Some(" ".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_try_remap_fires_for_true_subset_mismatch() {
|
||||
// True mismatch: CMap has high CIDs (512-544) but W only lists low sequential CIDs.
|
||||
let cmap_content = r#"
|
||||
1 begincodespacerange
|
||||
<0000><FFFF>
|
||||
endcodespacerange
|
||||
1 beginbfrange
|
||||
<0200> <0220> <0410>
|
||||
endbfrange
|
||||
"#;
|
||||
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
|
||||
|
||||
let mut doc = Document::new();
|
||||
let mut cid_font = lopdf::Dictionary::new();
|
||||
cid_font.set("CIDToGIDMap", lopdf::Object::Name(b"Identity".to_vec()));
|
||||
cid_font.set(
|
||||
"W",
|
||||
lopdf::Object::Array(vec![
|
||||
lopdf::Object::Integer(0),
|
||||
lopdf::Object::Array(vec![lopdf::Object::Integer(500); 34]), // 0..33
|
||||
]),
|
||||
);
|
||||
let cid_font_id = doc.add_object(cid_font);
|
||||
|
||||
let mut font_dict = lopdf::Dictionary::new();
|
||||
font_dict.set("Encoding", lopdf::Object::Name(b"Identity-H".to_vec()));
|
||||
font_dict.set(
|
||||
"DescendantFonts",
|
||||
lopdf::Object::Array(vec![lopdf::Object::Reference(cid_font_id)]),
|
||||
);
|
||||
|
||||
let (_primary, remapped) = try_remap_subset_cmap(cmap, &font_dict, &doc, 456);
|
||||
assert!(
|
||||
remapped.is_some(),
|
||||
"Remap must fire when CMap's CIDs are outside W array coverage"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user