feat(bindings): expose TextItem.mcid and structure-tree element extraction (#346)

Tagged PDFs carry a structure tree with real heading roles (H1..H6), and
the core already parses it (structure_tree::StructTree) and threads MCIDs
onto TextItem — but neither surfaced through the bindings.

- Expose TextItem.mcid (Option<i64>) through the napi and pyo3 bindings,
  matching the core field added with the marked-content extractor.
- Add StructRole::name(), the inverse of from_name, so roles have a
  stable string form.
- Add extract_structure_elements / extract_structure_elements_mem to the
  core: one (page, mcid, role) entry per marked-content reference, sorted
  by (page, mcid), empty for untagged PDFs. Pages are 1-indexed to match
  TextItem.page, so results join directly against
  extract_text_with_positions output.
- Bind it as extractStructureElements (napi) and
  extract_structure_elements / extract_structure_elements_bytes (pyo3),
  with type-stub updates in pdf_inspector.pyi.
- Cover the join in Rust integration tests, napi test.mjs, and pytest,
  using the existing firecrawl_docs_tagged.pdf fixture (tagged) and
  thermo-freon12.pdf (untagged).

Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Abimael Martell
2026-08-10 20:03:07 -07:00
committed by GitHub
co-authored by Claude Fable 5
parent 965dc65f1b
commit a67ee03269
8 changed files with 557 additions and 0 deletions
+73
View File
@@ -203,6 +203,79 @@ class TestExtractTextWithPositions:
assert len(items) > 0
assert all(item.page == 1 for item in items)
def test_mcid(self):
# Untagged fixture: mcid is None or int, never anything else
items = pdf_inspector.extract_text_with_positions(
fixture_path("thermo-freon12.pdf")
)
assert all(item.mcid is None or isinstance(item.mcid, int) for item in items)
# Tagged fixture: marked content carries MCIDs
tagged = pdf_inspector.extract_text_with_positions(
fixture_path("firecrawl_docs_tagged.pdf")
)
assert any(item.mcid is not None for item in tagged)
# ---------------------------------------------------------------------------
# extract_structure_elements / extract_structure_elements_bytes
# ---------------------------------------------------------------------------
class TestExtractStructureElements:
def test_tagged_file(self):
elements = pdf_inspector.extract_structure_elements(
fixture_path("firecrawl_docs_tagged.pdf")
)
assert len(elements) > 0
assert all(isinstance(e.page, int) for e in elements)
assert all(isinstance(e.mcid, int) for e in elements)
assert all(isinstance(e.role, str) and len(e.role) > 0 for e in elements)
assert any(e.role == "H1" for e in elements)
def test_join_with_text_items(self):
# (page, mcid) joins against extract_text_with_positions to recover
# heading text
path = fixture_path("firecrawl_docs_tagged.pdf")
elements = pdf_inspector.extract_structure_elements(path)
items = pdf_inspector.extract_text_with_positions(path)
h1_refs = {(e.page, e.mcid) for e in elements if e.role == "H1"}
h1_text = "".join(
item.text
for item in items
if item.mcid is not None and (item.page, item.mcid) in h1_refs
)
assert len(h1_text.strip()) > 0
def test_with_pages(self):
# pages filter is 1-indexed, matching TextItem.page
elements = pdf_inspector.extract_structure_elements(
fixture_path("firecrawl_docs_tagged.pdf"), pages=[1]
)
assert len(elements) > 0
assert all(e.page == 1 for e in elements)
def test_bytes(self):
data = fixture_bytes("firecrawl_docs_tagged.pdf")
elements = pdf_inspector.extract_structure_elements_bytes(data)
assert len(elements) > 0
assert any(e.role == "H1" for e in elements)
def test_untagged_returns_empty(self):
elements = pdf_inspector.extract_structure_elements(
fixture_path("thermo-freon12.pdf")
)
assert elements == []
def test_repr(self):
elements = pdf_inspector.extract_structure_elements(
fixture_path("firecrawl_docs_tagged.pdf")
)
assert "StructureElement" in repr(elements[0])
def test_not_a_pdf(self):
with pytest.raises(ValueError):
pdf_inspector.extract_structure_elements_bytes(b"not a pdf")
# ---------------------------------------------------------------------------
# extract_text_in_regions / extract_text_in_regions_bytes