feat(bindings): expose TextItem.mcid and structure-tree element extraction (#346)

Tagged PDFs carry a structure tree with real heading roles (H1..H6), and
the core already parses it (structure_tree::StructTree) and threads MCIDs
onto TextItem — but neither surfaced through the bindings.

- Expose TextItem.mcid (Option<i64>) through the napi and pyo3 bindings,
  matching the core field added with the marked-content extractor.
- Add StructRole::name(), the inverse of from_name, so roles have a
  stable string form.
- Add extract_structure_elements / extract_structure_elements_mem to the
  core: one (page, mcid, role) entry per marked-content reference, sorted
  by (page, mcid), empty for untagged PDFs. Pages are 1-indexed to match
  TextItem.page, so results join directly against
  extract_text_with_positions output.
- Bind it as extractStructureElements (napi) and
  extract_structure_elements / extract_structure_elements_bytes (pyo3),
  with type-stub updates in pdf_inspector.pyi.
- Cover the join in Rust integration tests, napi test.mjs, and pytest,
  using the existing firecrawl_docs_tagged.pdf fixture (tagged) and
  thermo-freon12.pdf (untagged).

Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Abimael Martell
2026-08-10 20:03:07 -07:00
committed by GitHub
co-authored by Claude Fable 5
parent 965dc65f1b
commit a67ee03269
8 changed files with 557 additions and 0 deletions
+71
View File
@@ -1429,6 +1429,77 @@ fn test_firecrawl_tagged_pdf_struct_tree() {
assert_eq!(fence_count % 2, 0, "Code fences should be balanced");
}
#[test]
fn test_tagged_pdf_text_items_carry_mcid() {
let buf = std::fs::read("tests/fixtures/firecrawl_docs_tagged.pdf").unwrap();
let items = pdf_inspector::extractor::extract_text_with_positions_mem(&buf).unwrap();
assert!(
items.iter().any(|i| i.mcid.is_some()),
"Tagged PDF text items should carry Marked Content IDs"
);
}
#[test]
fn test_extract_structure_elements_tagged_pdf() {
let buf = std::fs::read("tests/fixtures/firecrawl_docs_tagged.pdf").unwrap();
let elements = pdf_inspector::extract_structure_elements_mem(&buf, None).unwrap();
assert!(!elements.is_empty(), "Tagged PDF should yield elements");
assert!(
elements.iter().any(|e| e.role == "H1"),
"Should surface H1 heading roles"
);
assert!(
elements.iter().all(|e| !e.role.is_empty()),
"Every element should carry a role name"
);
// Sorted by (page, mcid) for deterministic output
assert!(
elements
.windows(2)
.all(|w| (w[0].page, w[0].mcid) <= (w[1].page, w[1].mcid)),
"Elements should be sorted by (page, mcid)"
);
// The advertised join: (page, mcid) pairs must line up with the
// mcid-carrying TextItems from positioned extraction, and joining the
// H1 entries must recover non-empty heading text.
let items = pdf_inspector::extractor::extract_text_with_positions_mem(&buf).unwrap();
let h1_refs: std::collections::HashSet<(u32, i64)> = elements
.iter()
.filter(|e| e.role == "H1")
.map(|e| (e.page, e.mcid))
.collect();
let h1_text: String = items
.iter()
.filter(|i| i.mcid.is_some_and(|mcid| h1_refs.contains(&(i.page, mcid))))
.map(|i| i.text.as_str())
.collect();
assert!(
!h1_text.trim().is_empty(),
"Joining H1 structure elements to text items should recover heading text"
);
// Page filter is 1-indexed (matching TextItem.page) and equals the
// corresponding subset of the full document result.
let page1 = pdf_inspector::extract_structure_elements_mem(&buf, Some(&[1])).unwrap();
assert!(!page1.is_empty(), "Page 1 should have elements");
assert!(page1.iter().all(|e| e.page == 1));
let full_page1_count = elements.iter().filter(|e| e.page == 1).count();
assert_eq!(page1.len(), full_page1_count);
}
#[test]
fn test_extract_structure_elements_untagged_pdf_empty() {
let buf = std::fs::read("tests/fixtures/thermo-freon12.pdf").unwrap();
let elements = pdf_inspector::extract_structure_elements_mem(&buf, None).unwrap();
assert!(
elements.is_empty(),
"Untagged PDF should yield no structure elements, got {:?}",
elements
);
}
#[test]
fn test_identity_h_no_tounicode_suppresses_garbage() {
// shinagawa_identity_h.pdf uses YuGothic with Identity-H encoding and no