Compare commits

..
Author SHA1 Message Date
Abimael Martell 2d4cade1d4 feat(cli): export positioned text item json 2026-07-08 10:06:09 -07:00
Abimael Martell 26ad531857 fix(extractor): harden underline detection 2026-07-08 08:40:41 -07:00
Abimael MartellandCursor 2d97f5aadc fix(extractor): underline rules only from painted rects, normalized extents (review)
Two review fixes: (1) normalize rect extents before the thickness/width
checks — `re` operands pass through the CTM so width/height can be
negative, which missed negative-width rules and let negative-height
bands pass as thin; (2) only feed painted rects to underline detection —
`re` rects now wait in a pending list until a paint operator (S/s, f/F/
f*, B/B*/b/b*) confirms them, and `re W n` clip-only paths are discarded
at `n`, so invisible clip boundaries no longer underline nearby text.
Marking moved into content_stream where paint state lives (pre-rotation,
consistent device space).

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-07 22:38:58 -07:00
Abimael MartellandCursor 28decf15d6 feat(extractor): geometric underline detection on TextItem (ENG-5015)
PDFs carry no underline font flag — underlines are stroked horizontal
lines or thin filled rects drawn under the baseline. Correlate those
graphics (already parsed from the content stream) with text items in a
post-pass: a rule within ~0.35em below the baseline covering >=60% of
an item's width marks is_underline.

Exposed through the napi and python bindings. Verified on real docs:
4/4 underlined sentences flagged on a Japanese report, links/headings
flagged on 8 of 10 underline-bearing eval docs, zero flags on docs
without underlines. Known FP source (table cell borders) documented —
downstream applies inline styling only to plain-text regions.

napi 1.9.8 -> 1.9.9.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-07 22:24:15 -07:00
Abimael Martell 30eddade77 fix(extractor): preserve tagged overlapping text order (#114) 2026-06-24 10:26:52 -07:00
Abimael Martell 1b2e2c76d6 fix(extractor): make trace previews unicode-safe (#113) 2026-06-24 01:27:27 -06:00
Abimael Martell ce49794719 fix(extractor): reduce garbled OCR false positives (#112)
* fix(extractor): reduce garbled OCR false positives

* fix(extractor): tighten garbled text OCR routing

* fix(extractor): decode UTF-16 ToUnicode destinations

* fix(extractor): narrow ToUnicode destination cleanup

* fix(extractor): decode Aptos private ff ligature
2026-06-24 01:13:16 -06:00
Abimael Martell 1a5ba6f1e9 feat(api): expose OCR reason signal (#110) 2026-06-23 15:51:55 -07:00
Abimael Martell 57b98c6a5d fix(extractor): flag garbled text spans for OCR (#108)
* fix(extractor): flag garbled text spans for OCR

* fix(extractor): apply text quality checks to regions

* chore(napi): bump npm package version
2026-06-23 13:12:10 -07:00
Abimael Martell f25808e0a7 fix(extractor): restore CID font state for Chinese text (#106)
* fix Chinese CID text decoding

* bump package versions
2026-06-20 19:43:27 -06:00
Abimael Martell 9360c8464d ci: add crates trusted publishing (#103) 2026-06-05 11:21:55 -07:00
Abimael Martell 252d87ac58 docs(readme): add package badges and install docs (#102)
* docs: add crates.io install instructions

* docs: add npm badge
2026-06-05 11:01:56 -07:00
Abimael Martell 85890648c9 chore: use crates.io lopdf (#101) 2026-06-05 10:50:05 -07:00
Abimael Martell 6e55e38b55 fix(markdown): handle wrapped bold abstracts (#100)
* fix(markdown): handle wrapped bold abstracts

* chore: bump napi package version
2026-06-01 14:59:18 -07:00
Abimael Martell 42befcea57 fix(pdf-inspector): recover wrapped key-value tables (#99)
* fix(pdf-inspector): recover wrapped key-value tables

* fix(pdf-inspector): satisfy clippy
2026-06-01 10:10:55 -07:00
Abimael Martell e547f616f9 fix(pdf-inspector): recover key-value region tables (#98) 2026-05-29 18:55:38 -07:00
Abimael Martell 455dfe5a74 fix(pdf-inspector): recover borderless region tables (#97) 2026-05-28 10:20:06 -07:00
Abimael Martell 839317525b fix(pdf-inspector): relax vector table confidence gates (#96) 2026-05-27 11:45:00 -07:00
36 changed files with 4272 additions and 192 deletions
+87
View File
@@ -0,0 +1,87 @@
name: Publish Rust crate
on:
push:
branches: [main]
paths: ['Cargo.toml']
permissions:
contents: read
env:
CARGO_TERM_COLOR: always
jobs:
check-version:
name: Check version change
runs-on: ubuntu-latest
outputs:
changed: ${{ steps.check.outputs.changed }}
published: ${{ steps.check.outputs.published }}
version: ${{ steps.check.outputs.version }}
steps:
- uses: actions/checkout@v4
with:
fetch-depth: 2
- name: Check if version changed
id: check
run: |
NEW_VERSION=$(python3 -c 'import pathlib, tomllib; print(tomllib.loads(pathlib.Path("Cargo.toml").read_text())["package"]["version"])')
OLD_VERSION=$(git show HEAD~1:Cargo.toml | python3 -c 'import sys, tomllib; print(tomllib.loads(sys.stdin.read())["package"]["version"])')
echo "old=$OLD_VERSION new=$NEW_VERSION"
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
if [ "$NEW_VERSION" = "$OLD_VERSION" ]; then
echo "changed=false" >> "$GITHUB_OUTPUT"
echo "published=false" >> "$GITHUB_OUTPUT"
exit 0
fi
echo "changed=true" >> "$GITHUB_OUTPUT"
HTTP_STATUS=$(curl --silent --show-error --output /tmp/crate-version.json --write-out "%{http_code}" \
-H "User-Agent: firecrawl/pdf-inspector publish workflow (https://github.com/firecrawl/pdf-inspector)" \
"https://crates.io/api/v1/crates/pdf-inspector/$NEW_VERSION")
case "$HTTP_STATUS" in
200)
echo "published=true" >> "$GITHUB_OUTPUT"
echo "pdf-inspector v$NEW_VERSION is already published"
;;
404)
echo "published=false" >> "$GITHUB_OUTPUT"
;;
*)
cat /tmp/crate-version.json
echo "Unexpected crates.io response: $HTTP_STATUS" >&2
exit 1
;;
esac
publish:
name: Publish to crates.io
needs: check-version
if: needs.check-version.outputs.changed == 'true' && needs.check-version.outputs.published == 'false'
runs-on: ubuntu-latest
environment: crates-io
permissions:
contents: read
id-token: write
steps:
- uses: actions/checkout@v4
- name: Install Rust
uses: dtolnay/rust-toolchain@stable
- name: Verify package
run: cargo publish --dry-run
- name: Authenticate with crates.io
id: auth
uses: rust-lang/crates-io-auth-action@v1
- name: Publish crate
run: cargo publish
env:
CARGO_REGISTRY_TOKEN: ${{ steps.auth.outputs.token }}
+2 -2
View File
@@ -1,6 +1,6 @@
[package]
name = "pdf-inspector"
version = "0.1.0"
version = "0.1.3"
edition = "2021"
autobins = false
authors = ["Firecrawl Team"]
@@ -17,7 +17,7 @@ crate-type = ["lib", "cdylib"]
pyo3 = { version = "0.25", features = ["extension-module"], optional = true }
# PDF parsing
lopdf = { git = "https://github.com/J-F-Liu/lopdf", rev = "7a05512d831415b1f2b1ce522391d6beab8a1284", features = ["rayon"] }
lopdf = { version = "0.41.0", features = ["rayon"] }
# Error handling
thiserror = "2.0"
+28 -9
View File
@@ -1,5 +1,8 @@
# pdf-inspector
[![Crates.io](https://img.shields.io/crates/v/pdf-inspector.svg)](https://crates.io/crates/pdf-inspector)
[![npm](https://img.shields.io/npm/v/@firecrawl/pdf-inspector.svg)](https://www.npmjs.com/package/@firecrawl/pdf-inspector)
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Includes bindings for [Python](docs/python.md) and [Node.js](napi/README.md).
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
@@ -71,9 +74,17 @@ console.log(result.markdown); // Markdown string or null
### Rust
Install from [crates.io](https://crates.io/crates/pdf-inspector):
```bash
cargo add pdf-inspector
```
Or add it manually:
```toml
[dependencies]
pdf-inspector = { git = "https://github.com/firecrawl/pdf-inspector" }
pdf-inspector = "0.1"
```
```rust
@@ -91,29 +102,37 @@ if let Some(markdown) = &result.markdown {
### CLI
```bash
# Install the CLI tools
cargo install pdf-inspector
# Convert PDF to Markdown
cargo run --bin pdf2md -- document.pdf
pdf2md document.pdf
# JSON output (for piping)
cargo run --bin pdf2md -- document.pdf --json
pdf2md document.pdf --json
# Positioned TextItem JSON, including is_underline metadata
pdf2md document.pdf --items-json
# Raw markdown only (no headers)
cargo run --bin pdf2md -- document.pdf --raw
pdf2md document.pdf --raw
# Insert page break markers (<!-- Page N -->)
cargo run --bin pdf2md -- document.pdf --pages
pdf2md document.pdf --pages
# Process only specific pages
cargo run --bin pdf2md -- document.pdf --select-pages 1,3,5-10
pdf2md document.pdf --select-pages 1,3,5-10
# Detection only (no extraction)
cargo run --bin detect-pdf -- document.pdf
cargo run --bin detect-pdf -- document.pdf --json
detect-pdf document.pdf
detect-pdf document.pdf --json
# Detection + layout analysis (tables, columns)
cargo run --bin detect-pdf -- document.pdf --analyze --json
detect-pdf document.pdf --analyze --json
```
From a source checkout, use `cargo run --bin pdf2md -- document.pdf` or `cargo run --bin detect-pdf -- document.pdf` instead.
## Architecture
```
+21
View File
@@ -0,0 +1,21 @@
# Publishing
The Rust crate is published to [crates.io](https://crates.io/crates/pdf-inspector) with trusted publishing from GitHub Actions. The first release was published manually; future releases publish from `.github/workflows/publish-crate.yml` when a `Cargo.toml` version change lands on `main`.
## crates.io Trusted Publisher
Configure the trusted publisher for the `pdf-inspector` crate with:
- Repository: `firecrawl/pdf-inspector`
- Workflow: `publish-crate.yml`
- Environment: `crates-io`
The workflow uses `rust-lang/crates-io-auth-action@v1` to exchange GitHub's OIDC token for a short-lived crates.io token, then passes it to `cargo publish`.
## Release Steps
1. Update `version` in `Cargo.toml`.
2. Merge the version bump to `main`.
3. The publish workflow compares the new `Cargo.toml` version with `HEAD~1`, runs `cargo publish --dry-run`, then publishes if that version is not already on crates.io.
If `Cargo.toml` changes without a package version bump, the workflow exits without publishing.
+5 -4
View File
@@ -672,8 +672,9 @@ checksum = "5e5032e24019045c762d3c0f28f5b6b8bbf38563a65908389bf7978758920897"
[[package]]
name = "lopdf"
version = "0.40.0"
source = "git+https://github.com/J-F-Liu/lopdf?rev=7a05512d831415b1f2b1ce522391d6beab8a1284#7a05512d831415b1f2b1ce522391d6beab8a1284"
version = "0.41.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "67513274c50a2b51e5f75d9e682fcf4ab064a8a9c9ae2c3c59309084882bb24d"
dependencies = [
"aes",
"bitflags",
@@ -829,7 +830,7 @@ checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe"
[[package]]
name = "pdf-inspector"
version = "0.1.0"
version = "0.1.3"
dependencies = [
"env_logger",
"log",
@@ -844,7 +845,7 @@ dependencies = [
[[package]]
name = "pdf-inspector-napi"
version = "0.2.0"
version = "0.2.2"
dependencies = [
"napi",
"napi-build",
+1 -1
View File
@@ -1,6 +1,6 @@
[package]
name = "pdf-inspector-napi"
version = "0.2.0"
version = "0.2.2"
edition = "2021"
[lib]
+2 -1
View File
@@ -37,7 +37,7 @@ console.log(result.confidence) // 0.875
Extract text within bounding-box regions from a PDF. Designed for hybrid OCR pipelines where a layout model detects regions in rendered page images, and this function extracts text from the PDF structure for text-based pages — skipping GPU OCR.
Each region result includes a `needsOcr` flag that signals unreliable extraction (empty text, GID-encoded fonts, garbage text, encoding issues).
Each region result includes a `needsOcr` flag that signals unreliable extraction (empty text, GID-encoded fonts, garbage text, encoding issues). When the cause is a suspected garbled text layer, `ocrReason` is set to `"suspected_garbled_text"`.
```typescript
import { extractTextInRegions } from '@firecrawl/pdf-inspector'
@@ -84,6 +84,7 @@ interface PageRegionTexts {
interface RegionText {
text: string
needsOcr: boolean // true when text is unreliable
ocrReason?: string // "suspected_garbled_text" when known
}
```
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "@firecrawl/pdf-inspector",
"version": "1.9.1",
"version": "1.9.9",
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
"main": "index.js",
"types": "index.d.ts",
+35
View File
@@ -40,6 +40,8 @@ pub struct PdfResult {
pub processing_time_ms: u32,
/// 1-indexed page numbers that need OCR.
pub pages_needing_ocr: Vec<u32>,
/// Machine-readable OCR reasons by 1-indexed page.
pub ocr_reasons_by_page: Vec<PageOcrReasons>,
pub title: Option<String>,
pub confidence: f64,
pub is_complex_layout: bool,
@@ -48,6 +50,13 @@ pub struct PdfResult {
pub has_encoding_issues: bool,
}
/// OCR reasons for a single 1-indexed page.
#[napi(object)]
pub struct PageOcrReasons {
pub page: u32,
pub reasons: Vec<String>,
}
/// Lightweight PDF classification result.
#[napi(object)]
pub struct PdfClassification {
@@ -71,6 +80,9 @@ pub struct TextItem {
pub page: u32,
pub is_bold: bool,
pub is_italic: bool,
/// Underline detected geometrically (drawn rule/thin rect under the
/// baseline) — PDFs carry no underline font flag.
pub is_underline: bool,
pub item_type: ItemType,
/// URL for link items, `None` for other types.
pub link_url: Option<String>,
@@ -90,6 +102,8 @@ pub struct RegionText {
pub text: String,
/// `true` when the text should not be trusted (empty, GID fonts, garbage, encoding issues).
pub needs_ocr: bool,
/// Machine-readable OCR reason when the cause is known.
pub ocr_reason: Option<String>,
}
/// Extracted text for one page's regions.
@@ -126,6 +140,7 @@ fn to_napi_result(r: pdf_inspector::PdfProcessResult) -> PdfResult {
page_count: r.page_count,
processing_time_ms: r.processing_time_ms as u32,
pages_needing_ocr: r.pages_needing_ocr,
ocr_reasons_by_page: to_napi_page_ocr_reasons(r.ocr_reasons_by_page),
title: r.title,
confidence: r.confidence as f64,
is_complex_layout: r.layout.is_complex,
@@ -135,6 +150,18 @@ fn to_napi_result(r: pdf_inspector::PdfProcessResult) -> PdfResult {
}
}
fn to_napi_page_ocr_reasons(
reasons: Vec<pdf_inspector::PageOcrReasons>,
) -> Vec<PageOcrReasons> {
reasons
.into_iter()
.map(|reason| PageOcrReasons {
page: reason.page,
reasons: reason.reasons,
})
.collect()
}
fn convert_item_type(t: &pdf_inspector::types::ItemType) -> (ItemType, Option<String>) {
match t {
pdf_inspector::types::ItemType::Text => (ItemType::Text, None),
@@ -266,6 +293,7 @@ pub fn extract_text_with_positions(
page: item.page,
is_bold: item.is_bold,
is_italic: item.is_italic,
is_underline: item.is_underline,
item_type,
link_url,
}
@@ -563,6 +591,8 @@ pub struct PageMarkdownResult {
pub markdown: String,
/// `true` when text on this page is unreliable.
pub needs_ocr: bool,
/// Machine-readable OCR reason when the cause is known.
pub ocr_reason: Option<String>,
}
/// Combined per-page markdown extraction and layout classification result.
@@ -576,6 +606,8 @@ pub struct PagesExtractionResult {
pub pages_with_columns: Vec<u32>,
/// 1-indexed pages that need OCR (scanned/image-based).
pub pages_needing_ocr: Vec<u32>,
/// Machine-readable OCR reasons by 1-indexed page.
pub ocr_reasons_by_page: Vec<PageOcrReasons>,
/// True if any page has tables or columns.
pub is_complex: bool,
}
@@ -607,11 +639,13 @@ pub fn extract_pages_markdown(
page: r.page,
markdown: r.markdown,
needs_ocr: r.needs_ocr,
ocr_reason: r.ocr_reason,
})
.collect(),
pages_with_tables: result.pages_with_tables,
pages_with_columns: result.pages_with_columns,
pages_needing_ocr: result.pages_needing_ocr,
ocr_reasons_by_page: to_napi_page_ocr_reasons(result.ocr_reasons_by_page),
is_complex: result.is_complex,
})
})
@@ -648,6 +682,7 @@ fn to_page_region_texts(results: Vec<pdf_inspector::PageRegionResult>) -> Vec<Pa
.map(|r| RegionText {
text: r.text,
needs_ocr: r.needs_ocr,
ocr_reason: r.ocr_reason,
})
.collect(),
})
+1
View File
@@ -38,6 +38,7 @@ class TextItem:
page: int
is_bold: bool
is_italic: bool
is_underline: bool
item_type: str
class RegionText:
+127 -3
View File
@@ -1,6 +1,10 @@
//! CLI tool for PDF to Markdown conversion
use pdf_inspector::{process_pdf_with_options, LayoutComplexity, PdfOptions, PdfType, ProcessMode};
use pdf_inspector::extractor::ItemType;
use pdf_inspector::{
extract_text_with_positions_pages, process_pdf_with_options, LayoutComplexity, PdfOptions,
PdfType, ProcessMode, TextItem,
};
use std::collections::HashSet;
use std::env;
use std::fmt::Write;
@@ -31,6 +35,108 @@ fn json_escape(s: &str) -> String {
out
}
fn format_ocr_reasons_by_page(reasons: &[pdf_inspector::PageOcrReasons]) -> String {
reasons
.iter()
.map(|entry| {
let reasons_json = entry
.reasons
.iter()
.map(|reason| format!(r#""{}""#, json_escape(reason)))
.collect::<Vec<_>>()
.join(",");
format!(r#"{{"page":{},"reasons":[{}]}}"#, entry.page, reasons_json)
})
.collect::<Vec<_>>()
.join(",")
}
fn item_type_label(item_type: &ItemType) -> &'static str {
match item_type {
ItemType::Text => "text",
ItemType::Image => "image",
ItemType::Link(_) => "link",
ItemType::FormField => "form_field",
}
}
fn format_items_json(items: &[TextItem]) -> String {
let underlined_count = items.iter().filter(|item| item.is_underline).count();
let items_json = items
.iter()
.map(|item| {
let mcid = item
.mcid
.map(|value| value.to_string())
.unwrap_or_else(|| "null".to_string());
let link_url = match &item.item_type {
ItemType::Link(url) => format!(r#","url":"{}""#, json_escape(url)),
_ => String::new(),
};
format!(
r#"{{"text":"{}","page":{},"x":{:.2},"y":{:.2},"width":{:.2},"height":{:.2},"font":"{}","font_size":{:.2},"is_bold":{},"is_italic":{},"is_underline":{},"item_type":"{}","mcid":{}{}}}"#,
json_escape(&item.text),
item.page,
item.x,
item.y,
item.width,
item.height,
json_escape(&item.font),
item.font_size,
item.is_bold,
item.is_italic,
item.is_underline,
item_type_label(&item.item_type),
mcid,
link_url,
)
})
.collect::<Vec<_>>()
.join(",");
format!(
r#"{{"total_items":{},"underlined_count":{},"items":[{}]}}"#,
items.len(),
underlined_count,
items_json
)
}
#[cfg(test)]
mod tests {
use super::format_items_json;
use pdf_inspector::extractor::ItemType;
use pdf_inspector::TextItem;
#[test]
fn items_json_includes_position_and_underline_metadata() {
let items = vec![TextItem {
text: "A \"quoted\" item".to_string(),
x: 12.345,
y: 67.891,
width: 23.456,
height: 9.876,
font: "F1".to_string(),
font_size: 10.0,
page: 2,
is_bold: false,
is_italic: true,
is_underline: true,
item_type: ItemType::Text,
mcid: Some(7),
}];
let json = format_items_json(&items);
assert!(json.contains(r#""text":"A \"quoted\" item""#));
assert!(json.contains(r#""page":2"#));
assert!(json.contains(r#""x":12.35"#));
assert!(json.contains(r#""is_underline":true"#));
assert!(json.contains(r#""item_type":"text""#));
assert!(json.contains(r#""mcid":7"#));
}
}
/// Parse a page specification like "1,3,5-10,20" into a HashSet of page numbers.
fn parse_page_spec(spec: &str) -> Result<HashSet<u32>, String> {
let mut pages = HashSet::new();
@@ -88,6 +194,7 @@ fn main() {
if args.len() < 2 {
eprintln!("Usage: {} <pdf_file> [output_file]", args[0]);
eprintln!(" {} <pdf_file> --json", args[0]);
eprintln!(" {} <pdf_file> --items-json", args[0]);
eprintln!(" {} <pdf_file> --raw", args[0]);
eprintln!();
eprintln!("Converts PDF to Markdown with smart type detection.");
@@ -95,6 +202,7 @@ fn main() {
eprintln!();
eprintln!("Options:");
eprintln!(" --json Output result as JSON");
eprintln!(" --items-json Output positioned TextItem JSON");
eprintln!(" --raw Output only markdown (no headers)");
eprintln!(" --pages Insert page break markers (<!-- Page N -->)");
eprintln!(" --select-pages N Only process specified pages (e.g. 1,3,5-10)");
@@ -105,6 +213,7 @@ fn main() {
let pdf_path = &args[1];
let json_output = args.iter().any(|a| a == "--json");
let items_json_output = args.iter().any(|a| a == "--items-json");
let raw_output = args.iter().any(|a| a == "--raw");
let page_numbers = args.iter().any(|a| a == "--pages");
let detect_only = args.iter().any(|a| a == "--detect-only");
@@ -129,6 +238,17 @@ fn main() {
})
});
if items_json_output {
match extract_text_with_positions_pages(pdf_path, page_filter.as_ref()) {
Ok(items) => println!("{}", format_items_json(&items)),
Err(e) => {
println!(r#"{{"error":"{}"}}"#, json_escape(&e.to_string()));
process::exit(1);
}
}
return;
}
let output_file = args
.get(2)
.filter(|a| !a.starts_with("--"))
@@ -177,12 +297,14 @@ fn main() {
.iter()
.map(|p| p.to_string())
.collect();
let ocr_reasons = format_ocr_reasons_by_page(&result.ocr_reasons_by_page);
println!(
r#"{{"pdf_type":"{}","page_count":{},"processing_time_ms":{},"pages_needing_ocr":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"has_encoding_issues":{}}}"#,
r#"{{"pdf_type":"{}","page_count":{},"processing_time_ms":{},"pages_needing_ocr":[{}],"ocr_reasons_by_page":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"has_encoding_issues":{}}}"#,
pdf_type_str,
result.page_count,
result.processing_time_ms,
ocr_pages.join(","),
ocr_reasons,
result.layout.is_complex,
table_pages.join(","),
col_pages.join(","),
@@ -223,8 +345,9 @@ fn main() {
.iter()
.map(|p| p.to_string())
.collect();
let ocr_reasons = format_ocr_reasons_by_page(&result.ocr_reasons_by_page);
println!(
r#"{{"pdf_type":"{}","page_count":{},"has_text":{},"processing_time_ms":{},"markdown_length":{},"pages_needing_ocr":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"has_encoding_issues":{},"markdown":"{}"}}"#,
r#"{{"pdf_type":"{}","page_count":{},"has_text":{},"processing_time_ms":{},"markdown_length":{},"pages_needing_ocr":[{}],"ocr_reasons_by_page":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"has_encoding_issues":{},"markdown":"{}"}}"#,
match result.pdf_type {
PdfType::TextBased => "text_based",
PdfType::Scanned => "scanned",
@@ -236,6 +359,7 @@ fn main() {
result.processing_time_ms,
result.markdown.as_ref().map(|m| m.len()).unwrap_or(0),
ocr_pages.join(","),
ocr_reasons,
result.layout.is_complex,
table_pages.join(","),
col_pages.join(","),
+321 -19
View File
@@ -17,6 +17,7 @@ use super::fonts::{
build_font_encodings, build_font_widths, compute_string_width_ts, extract_text_from_operand,
get_font_file2_obj_num, get_operand_bytes, CMapDecisionCache,
};
use super::underline::UnderlineLine;
use super::xobjects::{extract_form_xobject_text, get_page_xobjects, XObjectType};
use super::{get_number, image_bbox_from_ctm, multiply_matrices};
@@ -74,6 +75,37 @@ fn strip_pdf_comments(data: &[u8]) -> Vec<u8> {
result
}
fn transform_path_point(x: f32, y: f32, ctm: &[f32; 6]) -> (f32, f32) {
(
x * ctm[0] + y * ctm[2] + ctm[4],
x * ctm[1] + y * ctm[3] + ctm[5],
)
}
fn transformed_stroke_width(
line_width: f32,
ctm: &[f32; 6],
x1: f32,
y1: f32,
x2: f32,
y2: f32,
) -> f32 {
let user_width = line_width.abs();
let dx = x2 - x1;
let dy = y2 - y1;
let len = (dx * dx + dy * dy).sqrt();
if len <= f32::EPSILON {
return user_width;
}
// PDF stroke width scales perpendicular to the path direction.
let nx = -dy / len;
let ny = dx / len;
let ndx = nx * ctm[0] + ny * ctm[2];
let ndy = nx * ctm[1] + ny * ctm[3];
user_width * (ndx * ndx + ndy * ndy).sqrt()
}
/// Returns `(page_extraction, has_gid_fonts)` where `has_gid_fonts` indicates
/// the page uses fonts with unresolvable gid-encoded glyphs.
pub(crate) fn extract_page_text_items(
@@ -89,6 +121,7 @@ pub(crate) fn extract_page_text_items(
let mut rects: Vec<PdfRect> = Vec::new();
let mut clip_rects: Vec<PdfRect> = Vec::new();
let mut lines: Vec<PdfLine> = Vec::new();
let mut underline_lines: Vec<UnderlineLine> = Vec::new();
// Path construction state for m/l/h → S/s line extraction
let mut path_subpath_start: Option<(f32, f32)> = None;
@@ -97,6 +130,12 @@ pub(crate) fn extract_page_text_items(
// Completed subpaths (each a vec of line segments) for f/f* rect extraction
let mut pending_subpaths: Vec<Vec<(f32, f32, f32, f32)>> = Vec::new();
let mut fill_rects: Vec<PdfRect> = Vec::new();
// `re` rects awaiting a paint operator. Underline detection must only
// see painted rects: a `re W n` clip path or `re n` no-op draws nothing
// on the page, so treating every `re` as ink would underline text that
// merely sits near an invisible clip boundary.
let mut pending_re_rects: Vec<PdfRect> = Vec::new();
let mut painted_rects: Vec<PdfRect> = Vec::new();
// Get fonts for encoding
let fonts = doc.get_page_fonts(page_id).unwrap_or_default();
@@ -188,7 +227,19 @@ pub(crate) fn extract_page_text_items(
// Graphics state tracking
let mut ctm = [1.0f32, 0.0, 0.0, 1.0, 0.0, 0.0]; // Current Transformation Matrix
let mut text_rendering_mode: i32 = 0; // 0=fill, 1=stroke, 2=fill+stroke, 3=invisible
let mut gstate_stack: Vec<([f32; 6], i32, f32, f32)> = Vec::new();
let mut line_width: f32 = 1.0;
#[derive(Clone)]
struct SavedGraphicsState {
ctm: [f32; 6],
text_rendering_mode: i32,
line_width: f32,
char_spacing: f32,
word_spacing: f32,
text_leading: f32,
current_font: String,
current_font_size: f32,
}
let mut gstate_stack: Vec<SavedGraphicsState> = Vec::new();
// Text state tracking
let mut current_font = String::new();
@@ -227,15 +278,28 @@ pub(crate) fn extract_page_text_items(
match op.operator.as_str() {
"q" => {
// Save graphics state
gstate_stack.push((ctm, text_rendering_mode, char_spacing, word_spacing));
gstate_stack.push(SavedGraphicsState {
ctm,
text_rendering_mode,
line_width,
char_spacing,
word_spacing,
text_leading,
current_font: current_font.clone(),
current_font_size,
});
}
"Q" => {
// Restore graphics state
if let Some((saved_ctm, saved_tr, saved_tc, saved_tw)) = gstate_stack.pop() {
ctm = saved_ctm;
text_rendering_mode = saved_tr;
char_spacing = saved_tc;
word_spacing = saved_tw;
if let Some(saved) = gstate_stack.pop() {
ctm = saved.ctm;
text_rendering_mode = saved.text_rendering_mode;
line_width = saved.line_width;
char_spacing = saved.char_spacing;
word_spacing = saved.word_spacing;
text_leading = saved.text_leading;
current_font = saved.current_font;
current_font_size = saved.current_font_size;
}
}
"cm" => {
@@ -252,6 +316,11 @@ pub(crate) fn extract_page_text_items(
ctm = multiply_matrices(&new_matrix, &ctm);
}
}
"w" => {
if let Some(width) = op.operands.first().and_then(get_number) {
line_width = width;
}
}
"BT" => {
// Begin text block
in_text_block = true;
@@ -420,6 +489,7 @@ pub(crate) fn extract_page_text_items(
page: page_num,
is_bold: is_bold_font(base_font),
is_italic: is_italic_font(base_font),
is_underline: false,
item_type: ItemType::Text,
mcid: current_mcid(&marked_content_stack),
});
@@ -585,6 +655,7 @@ pub(crate) fn extract_page_text_items(
page: page_num,
is_bold: is_bold_font(base_font),
is_italic: is_italic_font(base_font),
is_underline: false,
item_type: ItemType::Text,
mcid: current_mcid(&marked_content_stack),
});
@@ -648,6 +719,7 @@ pub(crate) fn extract_page_text_items(
page: page_num,
is_bold: is_bold_font(base_font),
is_italic: is_italic_font(base_font),
is_underline: false,
item_type: ItemType::Text,
mcid: current_mcid(&marked_content_stack),
});
@@ -684,6 +756,7 @@ pub(crate) fn extract_page_text_items(
page: page_num,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Image,
mcid: current_mcid(&marked_content_stack),
});
@@ -780,6 +853,7 @@ pub(crate) fn extract_page_text_items(
page: page_num,
is_bold: is_bold_font(base_font),
is_italic: is_italic_font(base_font),
is_underline: false,
item_type: ItemType::Text,
mcid: entry
.mcid
@@ -804,13 +878,19 @@ pub(crate) fn extract_page_text_items(
let y_dev = rx * ctm[1] + ry * ctm[3] + ctm[5];
let w_dev = rw * ctm[0];
let h_dev = rh * ctm[3];
rects.push(PdfRect {
let rect = PdfRect {
x: x_dev,
y: y_dev,
width: w_dev,
height: h_dev,
page: page_num,
});
};
// Underline detection must only see rects that are
// actually painted — a `re` used purely as a clip path
// (`re W n`) or discarded (`re n`) draws nothing. Hold
// the rect as pending until a paint operator confirms it.
pending_re_rects.push(rect.clone());
rects.push(rect);
}
}
// ── Path construction operators ──────────────────────
@@ -860,10 +940,8 @@ pub(crate) fn extract_page_text_items(
}
}
for (x1, y1, x2, y2) in pending_lines.drain(..) {
let x1d = x1 * ctm[0] + y1 * ctm[2] + ctm[4];
let y1d = x1 * ctm[1] + y1 * ctm[3] + ctm[5];
let x2d = x2 * ctm[0] + y2 * ctm[2] + ctm[4];
let y2d = x2 * ctm[1] + y2 * ctm[3] + ctm[5];
let (x1d, y1d) = transform_path_point(x1, y1, &ctm);
let (x2d, y2d) = transform_path_point(x2, y2, &ctm);
lines.push(PdfLine {
x1: x1d,
y1: y1d,
@@ -871,7 +949,16 @@ pub(crate) fn extract_page_text_items(
y2: y2d,
page: page_num,
});
underline_lines.push(UnderlineLine {
x1: x1d,
y1: y1d,
x2: x2d,
y2: y2d,
stroke_width: transformed_stroke_width(line_width, &ctm, x1, y1, x2, y2),
page: page_num,
});
}
painted_rects.append(&mut pending_re_rects);
pending_subpaths.clear();
path_subpath_start = None;
path_current = None;
@@ -887,10 +974,8 @@ pub(crate) fn extract_page_text_items(
}
}
for (x1, y1, x2, y2) in pending_lines.drain(..) {
let x1d = x1 * ctm[0] + y1 * ctm[2] + ctm[4];
let y1d = x1 * ctm[1] + y1 * ctm[3] + ctm[5];
let x2d = x2 * ctm[0] + y2 * ctm[2] + ctm[4];
let y2d = x2 * ctm[1] + y2 * ctm[3] + ctm[5];
let (x1d, y1d) = transform_path_point(x1, y1, &ctm);
let (x2d, y2d) = transform_path_point(x2, y2, &ctm);
lines.push(PdfLine {
x1: x1d,
y1: y1d,
@@ -898,7 +983,16 @@ pub(crate) fn extract_page_text_items(
y2: y2d,
page: page_num,
});
underline_lines.push(UnderlineLine {
x1: x1d,
y1: y1d,
x2: x2d,
y2: y2d,
stroke_width: transformed_stroke_width(line_width, &ctm, x1, y1, x2, y2),
page: page_num,
});
}
painted_rects.append(&mut pending_re_rects);
pending_subpaths.clear();
path_subpath_start = None;
path_current = None;
@@ -956,6 +1050,7 @@ pub(crate) fn extract_page_text_items(
}
}
}
painted_rects.append(&mut pending_re_rects);
pending_lines.clear();
path_subpath_start = None;
path_current = None;
@@ -1021,7 +1116,10 @@ pub(crate) fn extract_page_text_items(
// Do NOT clear pending_lines — the following `n` does that
}
"n" => {
// end path (no-op): discard
// end path (no-op): discard — including any `re` rects that
// were only ever part of a clip path (`re W n`), which draw
// no ink and must not feed underline detection.
pending_re_rects.clear();
pending_lines.clear();
pending_subpaths.clear();
path_subpath_start = None;
@@ -1031,6 +1129,12 @@ pub(crate) fn extract_page_text_items(
}
}
// Underline detection reads only painted ink: `re` rects confirmed by
// a paint operator plus filled-subpath rects — never clip-only rects,
// which draw nothing.
let mut underline_rects = painted_rects;
underline_rects.extend(fill_rects.iter().cloned());
// Only use clip/fill rects when no `re` rects exist on this page.
// Clip rects take priority over fill rects, but first we deduplicate
// them: some PDFs wrap every text block in a full-page W* clip path,
@@ -1060,8 +1164,17 @@ pub(crate) fn extract_page_text_items(
// Some PDFs embed landscape content in portrait pages using a rotated text
// matrix (e.g. [0, b, -b, 0, tx, ty] for 90° CCW). The layout engine
// assumes x=horizontal, y=vertical — so we swap coordinates to match.
let (items, rects, lines, coords_rotated) =
let (mut items, rects, lines, coords_rotated) =
correct_rotated_page(items, rects, lines, &rotation_votes);
if coords_rotated {
rotate_underline_graphics(&mut underline_rects, &mut underline_lines);
}
super::underline::mark_underlined_items(
&mut items,
&underline_rects,
&underline_lines,
page_num,
);
let items = super::merge_text_items(items);
let items = super::merge_subscript_items(items);
@@ -1147,6 +1260,27 @@ fn correct_rotated_page(
(items, rects, lines, true)
}
fn rotate_underline_graphics(rects: &mut [PdfRect], lines: &mut [UnderlineLine]) {
for rect in rects {
let new_x = rect.y;
let new_y = -(rect.x + rect.width.abs());
rect.x = new_x;
rect.y = new_y;
std::mem::swap(&mut rect.width, &mut rect.height);
}
for line in lines {
let new_x1 = line.y1;
let new_y1 = -line.x1;
let new_x2 = line.y2;
let new_y2 = -line.x2;
line.x1 = new_x1;
line.y1 = new_y1;
line.x2 = new_x2;
line.y2 = new_y2;
}
}
/// Remove near-duplicate rects (same coordinates within 0.5 pt tolerance).
/// Some PDFs emit a full-page clip path for every text block, producing
/// thousands of identical rects. After dedup these collapse to one rect,
@@ -1196,6 +1330,57 @@ mod tests {
}
}
fn simple_doc_with_content(content: &[u8]) -> (lopdf::Document, lopdf::ObjectId) {
use lopdf::{dictionary, Object, Stream};
let mut doc = lopdf::Document::new();
let widths: Vec<Object> = (0..=255).map(|_| 600.into()).collect();
let font_id = doc.add_object(dictionary! {
"Type" => "Font",
"Subtype" => "Type1",
"BaseFont" => "Helvetica",
"FirstChar" => 0,
"LastChar" => 255,
"Widths" => Object::Array(widths),
});
let content_id = doc.add_object(Object::Stream(Stream::new(
dictionary! {},
content.to_vec(),
)));
let page_id = doc.add_object(dictionary! {
"Type" => "Page",
"Contents" => Object::Reference(content_id),
"Resources" => dictionary! {
"Font" => dictionary! {
"F1" => Object::Reference(font_id),
},
},
"MediaBox" => vec![0.into(), 0.into(), 612.into(), 792.into()],
});
let pages_id = doc.add_object(dictionary! {
"Type" => "Pages",
"Count" => Object::Integer(1),
"Kids" => vec![Object::Reference(page_id)],
});
let catalog_id = doc.add_object(dictionary! {
"Type" => "Catalog",
"Pages" => Object::Reference(pages_id),
});
doc.trailer.set("Root", Object::Reference(catalog_id));
(doc, page_id)
}
fn extract_simple_items(content: &[u8]) -> Vec<TextItem> {
use crate::tounicode::FontCMaps;
let (doc, page_id) = simple_doc_with_content(content);
let font_cmaps = FontCMaps::from_doc(&doc);
let ((items, _, _), _, _) =
extract_page_text_items(&doc, page_id, 1, &font_cmaps, false).unwrap();
items
}
#[test]
fn test_dedup_rects_identical() {
let mut rects = vec![rect(0.0, 0.0, 612.0, 792.0, 1); 3759];
@@ -1245,6 +1430,38 @@ mod tests {
assert_eq!(single.len(), 1);
}
#[test]
fn thick_stroked_rule_does_not_mark_underline() {
let content = b"BT /F1 12 Tf 1 0 0 1 100 500 Tm (THICK) Tj ET
4 w
100 498 m 170 498 l S
BT /F1 12 Tf 1 0 0 1 100 480 Tm (THIN) Tj ET
1 w
100 478 m 160 478 l S";
let items = extract_simple_items(content);
let thick = items.iter().find(|item| item.text == "THICK").unwrap();
let thin = items.iter().find(|item| item.text == "THIN").unwrap();
assert!(!thick.is_underline);
assert!(thin.is_underline);
}
#[test]
fn rotated_page_underline_is_detected_after_coordinate_correction() {
let content = b"BT /F1 12 Tf 0 1 -1 0 200 100 Tm (HELLO) Tj ET
BT /F1 12 Tf 0 1 -1 0 240 100 Tm (WORLD) Tj ET
1 w
202 100 m 202 170 l S";
let items = extract_simple_items(content);
let hello = items.iter().find(|item| item.text == "HELLO").unwrap();
let world = items.iter().find(|item| item.text == "WORLD").unwrap();
assert!(hello.is_underline);
assert!(!world.is_underline);
}
#[test]
fn test_skip_excessive_operations() {
use crate::tounicode::FontCMaps;
@@ -1286,6 +1503,91 @@ mod tests {
assert!(lines.is_empty());
}
#[test]
fn test_q_restores_current_font_for_text_decoding() {
use crate::tounicode::FontCMaps;
use lopdf::{dictionary, Object, Stream};
fn cmap_stream(dst_hex: &str) -> Stream {
let cmap = format!(
r#"/CIDInit /ProcSet findresource begin
12 dict begin
begincmap
/CIDSystemInfo << /Registry (Adobe) /Ordering (UCS) /Supplement 0 >> def
/CMapName /Test-UCS def
/CMapType 2 def
1 begincodespacerange
<00> <FF>
endcodespacerange
1 beginbfchar
<41> <{dst_hex}>
endbfchar
endcmap
CMapName currentdict /CMap defineresource pop
end
end"#
);
Stream::new(dictionary! {}, cmap.into_bytes())
}
let mut doc = lopdf::Document::new();
let f1_cmap = doc.add_object(Object::Stream(cmap_stream("0058"))); // X
let f2_cmap = doc.add_object(Object::Stream(cmap_stream("0059"))); // Y
let f1 = doc.add_object(dictionary! {
"Type" => "Font",
"Subtype" => "Type1",
"BaseFont" => "Helvetica",
"ToUnicode" => Object::Reference(f1_cmap),
});
let f2 = doc.add_object(dictionary! {
"Type" => "Font",
"Subtype" => "Type1",
"BaseFont" => "Helvetica",
"ToUnicode" => Object::Reference(f2_cmap),
});
let content = b"BT /F1 12 Tf 10 700 Tm <41> Tj ET
q
BT /F2 12 Tf 20 700 Tm <41> Tj ET
Q
BT 30 700 Tm <41> Tj ET";
let content_id = doc.add_object(Object::Stream(Stream::new(
dictionary! {},
content.to_vec(),
)));
let page_id = doc.add_object(dictionary! {
"Type" => "Page",
"Contents" => Object::Reference(content_id),
"Resources" => dictionary! {
"Font" => dictionary! {
"F1" => Object::Reference(f1),
"F2" => Object::Reference(f2),
},
},
"MediaBox" => vec![0.into(), 0.into(), 612.into(), 792.into()],
});
let pages_id = doc.add_object(dictionary! {
"Type" => "Pages",
"Count" => Object::Integer(1),
"Kids" => vec![Object::Reference(page_id)],
});
let catalog_id = doc.add_object(dictionary! {
"Type" => "Catalog",
"Pages" => Object::Reference(pages_id),
});
doc.trailer.set("Root", Object::Reference(catalog_id));
let font_cmaps = FontCMaps::from_doc(&doc);
let ((items, _, _), _, _) =
extract_page_text_items(&doc, page_id, 1, &font_cmaps, false).unwrap();
let text = items
.iter()
.map(|item| item.text.as_str())
.collect::<String>();
assert_eq!(text, "XYX");
}
#[test]
fn test_strip_pdf_comments() {
// Basic comment stripping
+277 -46
View File
@@ -526,6 +526,11 @@ pub(crate) fn parse_font_encoding(
font_dict: &lopdf::Dictionary,
) -> Option<EncodingResult> {
let encoding_obj = font_dict.get(b"Encoding").ok()?;
let base_font_name = font_dict
.get(b"BaseFont")
.ok()
.and_then(|o| o.as_name().ok())
.map(|n| String::from_utf8_lossy(n).to_string());
// Encoding can be a name or a dictionary
match encoding_obj {
@@ -538,12 +543,14 @@ pub(crate) fn parse_font_encoding(
Object::Reference(obj_ref) => {
// Reference to encoding dictionary
if let Ok(enc_dict) = doc.get_dictionary(*obj_ref) {
parse_encoding_dictionary(doc, enc_dict)
parse_encoding_dictionary(doc, enc_dict, base_font_name.as_deref())
} else {
None
}
}
Object::Dictionary(enc_dict) => parse_encoding_dictionary(doc, enc_dict),
Object::Dictionary(enc_dict) => {
parse_encoding_dictionary(doc, enc_dict, base_font_name.as_deref())
}
_ => None,
}
}
@@ -562,6 +569,7 @@ pub(crate) struct EncodingResult {
pub(crate) fn parse_encoding_dictionary(
doc: &Document,
enc_dict: &lopdf::Dictionary,
base_font_name: Option<&str>,
) -> Option<EncodingResult> {
let differences = enc_dict.get(b"Differences").ok()?;
@@ -591,11 +599,9 @@ pub(crate) fn parse_encoding_dictionary(
Object::Name(name) => {
// Map current code to glyph name -> Unicode
let glyph_name = String::from_utf8_lossy(&name).to_string();
if glyph_name == "fi"
|| glyph_name == "fl"
|| glyph_name == "ffi"
|| glyph_name == "ffl"
{
let mapped_char = glyph_to_char(&glyph_name)
.or_else(|| private_glyph_to_char(&glyph_name, base_font_name));
if mapped_char.is_some_and(is_ligature_char) {
debug!(
" Differences: code=0x{:02X} glyph={:?} (ligature)",
current_code, glyph_name
@@ -610,7 +616,7 @@ pub(crate) fn parse_encoding_dictionary(
{
gid_glyph_count += 1;
}
if let Some(ch) = glyph_to_char(&glyph_name) {
if let Some(ch) = mapped_char {
encoding_map.insert(current_code, ch);
} else {
debug!(
@@ -645,6 +651,31 @@ pub(crate) fn parse_encoding_dictionary(
})
}
fn private_glyph_to_char(glyph_name: &str, base_font_name: Option<&str>) -> Option<char> {
let base_font_name = strip_subset_prefix(base_font_name?);
// Aptos CFF subsets from Office PDFs can expose the ff ligature as /g431
// without a ToUnicode map. Keep this font-scoped because /gNNN names are private.
if base_font_name.eq_ignore_ascii_case("Aptos") && glyph_name == "g431" {
Some('\u{FB00}')
} else {
None
}
}
fn strip_subset_prefix(font_name: &str) -> &str {
font_name
.split_once('+')
.map_or(font_name, |(_, stripped)| stripped)
}
fn is_ligature_char(ch: char) -> bool {
matches!(
ch,
'\u{FB00}' | '\u{FB01}' | '\u{FB02}' | '\u{FB03}' | '\u{FB04}'
)
}
/// Get the CMap lookup key for an Identity-H/V CID font without ToUnicode.
/// Returns the object number used by `collect_cmaps_from_fonts` to store the CMap:
/// - FontFile2 or FontFile3 obj_num (for embedded font cmap)
@@ -725,6 +756,8 @@ pub(crate) fn extract_text_from_operand(
let is_type0_cid_font = font_widths
.get(current_font)
.is_some_and(|info| info.is_cid);
let use_cp1252_fallback =
should_use_cp1252_single_byte_fallback(base_font_name, is_type0_cid_font);
let result = (|| -> Option<String> {
if let Object::String(bytes, _) = obj {
let mut decode_with_entry = |entry: &crate::tounicode::CMapEntry| -> Option<String> {
@@ -755,9 +788,12 @@ pub(crate) fn extract_text_from_operand(
return Some(ch.to_string());
}
}
// 4. Printable ASCII/Latin-1 fallback
// 4. Printable single-byte fallback
if b >= 0x20 {
return Some((b as char).to_string());
return Some(
decode_single_byte_fallback_char(b, use_cp1252_fallback)
.to_string(),
);
}
None
})
@@ -858,6 +894,13 @@ pub(crate) fn extract_text_from_operand(
// unmapped. Don't fall through to text-interpretation fallbacks
// (Latin-1, UTF-16, etc.) which would misinterpret CID bytes as
// character codes (e.g. CID 0x01A9 → Latin-1 "©").
if is_type0_cid_font && bytes.iter().any(|&b| b > 0x7F) {
// 2-byte CIDs (Identity-H) are by far the common case; for
// an odd byte count we still emit at least one marker so
// detection downstream fires.
let cid_count = (bytes.len() / 2).max(1);
return Some("\u{FFFD}".repeat(cid_count));
}
// Try our custom encoding map from Differences arrays.
// The Differences array overrides specific codes in a base encoding (typically
@@ -873,8 +916,9 @@ pub(crate) fn extract_text_from_operand(
Some(ch)
} else if b >= 0x20 {
// Base encoding fallback for printable bytes.
// For codes 0x20-0x7E this matches all standard PDF encodings.
Some(b as char)
// Most PDFs with simple fonts use WinAnsi/PDFDocEncoding
// semantics, not ISO-8859-1 C1 controls.
Some(decode_single_byte_fallback_char(b, use_cp1252_fallback))
} else {
None // Skip unmapped control characters
}
@@ -930,6 +974,7 @@ pub(crate) fn extract_text_from_operand(
// Try to decode using cached font encoding from lopdf
if let Some(encoding) = encoding_cache.get(current_font) {
if let Ok(text) = Document::decode_text(encoding, bytes) {
let text = normalize_cp1252_controls(text, use_cp1252_fallback);
if text.contains('\u{FFFD}') {
debug!(
"decode_text produced replacement for font={} bytes_len={}",
@@ -966,37 +1011,119 @@ pub(crate) fn extract_text_from_operand(
return Some(symbol_text);
}
// Latin-1 fallback. Safe ONLY for fonts that use single-byte
// encodings — for these, an unmapped byte is a valid character
// code in Latin-1/WinAnsi space. CID fonts (Type0 / Identity-H)
// emit multi-byte CIDs that aren't characters; per-byte Latin-1
// produces mojibake (e.g. 2-byte CID 0xCDD9 → "ÍÙ" for the
// production scrape_id 019de78c-... samples).
//
// For a CID font (has_cmap is set OR a /ToUnicode reference
// exists) with any non-ASCII bytes, emit a single U+FFFD per
// CID instead. This both replaces the mojibake with a proper
// "decode failed" marker AND keeps `detect_encoding_issues`
// tripping so the page is flagged for OCR — the existing
// garbage-detection path that the high-Latin-1 mojibake used
// to satisfy by accident.
if is_type0_cid_font && bytes.iter().any(|&b| b > 0x7F) {
// 2-byte CIDs (Identity-H) are by far the common case; for
// an odd byte count we still emit at least one marker so
// detection downstream fires.
let cid_count = (bytes.len() / 2).max(1);
return Some("\u{FFFD}".repeat(cid_count));
}
// Pure ASCII bytes round-trip safely (Latin-1 == ASCII for
// 0x00..=0x7F), and non-CID (Type1 / TrueType / Type3) fonts
// use single-byte encodings where Latin-1 fallback is the
// canonical interpretation.
Some(bytes.iter().map(|&b| b as char).collect())
// Non-CID (Type1 / TrueType / Type3) fonts use single-byte
// encodings. In practice the fallback should follow WinAnsi for
// 0x80..=0x9F so bytes like 0x92 become smart punctuation instead
// of C1 controls that look like CID mojibake.
Some(decode_single_byte_fallback(bytes, use_cp1252_fallback))
} else {
None
}
})();
result.map(clean_symbol_pua)
result.map(|text| {
let text = clean_symbol_pua(text);
normalize_cp1252_controls(text, use_cp1252_fallback)
})
}
fn decode_single_byte_fallback(bytes: &[u8], use_cp1252_fallback: bool) -> String {
bytes
.iter()
.map(|&b| decode_single_byte_fallback_char(b, use_cp1252_fallback))
.collect()
}
fn decode_single_byte_fallback_char(byte: u8, use_cp1252_fallback: bool) -> char {
if !use_cp1252_fallback {
return byte as char;
}
match byte {
0x80 => '\u{20AC}',
0x82 => '\u{201A}',
0x83 => '\u{0192}',
0x84 => '\u{201E}',
0x85 => '\u{2026}',
0x86 => '\u{2020}',
0x87 => '\u{2021}',
0x88 => '\u{02C6}',
0x89 => '\u{2030}',
0x8A => '\u{0160}',
0x8B => '\u{2039}',
0x8C => '\u{0152}',
0x8E => '\u{017D}',
0x91 => '\u{2018}',
0x92 => '\u{2019}',
0x93 => '\u{201C}',
0x94 => '\u{201D}',
0x95 => '\u{2022}',
0x96 => '\u{2013}',
0x97 => '\u{2014}',
0x98 => '\u{02DC}',
0x99 => '\u{2122}',
0x9A => '\u{0161}',
0x9B => '\u{203A}',
0x9C => '\u{0153}',
0x9E => '\u{017E}',
0x9F => '\u{0178}',
_ => byte as char,
}
}
fn normalize_cp1252_controls(text: String, use_cp1252_fallback: bool) -> String {
if !use_cp1252_fallback {
return text;
}
if !text
.chars()
.any(|ch| ('\u{0080}'..='\u{009F}').contains(&ch))
{
return text;
}
text.chars()
.map(|ch| {
if ('\u{0080}'..='\u{009F}').contains(&ch) {
decode_single_byte_fallback_char(ch as u8, true)
} else {
ch
}
})
.collect()
}
fn should_use_cp1252_single_byte_fallback(
base_font_name: Option<&str>,
is_type0_cid_font: bool,
) -> bool {
if is_type0_cid_font {
return false;
}
let Some(base_font_name) = base_font_name else {
return true;
};
let font_name = base_font_name
.rsplit_once('+')
.map_or(base_font_name, |(_, stripped)| stripped)
.to_ascii_lowercase();
// TeX/Computer Modern and math/symbol fonts often place ligatures or
// symbols in the C1 byte range. Treating those bytes as Windows-1252 makes
// words like "deficiente" become "de…ciente" and "fluid" become "‡uid".
let non_cp1252_prefixes = [
"cmr", "cmb", "cmmi", "cmsy", "cmex", "cmtt", "cmss", "cmti", "ecrm", "ecbx", "ecti",
"tcrm", "tctt", "msam", "msbm", "ttdc",
];
if non_cp1252_prefixes
.iter()
.any(|prefix| font_name.starts_with(prefix))
{
return false;
}
let non_cp1252_names = ["math", "symbol", "dingbat", "emoji"];
!non_cp1252_names.iter().any(|name| font_name.contains(name))
}
/// Replace PUA characters in the F000-F0FF range with standard Unicode equivalents.
@@ -1118,6 +1245,7 @@ fn score_text(text: &str) -> i32 {
#[cfg(test)]
mod tests {
use super::*;
use lopdf::dictionary;
fn make_font_info(widths: &[(u16, u16)], default_width: u16, is_cid: bool) -> FontWidthInfo {
FontWidthInfo {
@@ -1242,6 +1370,51 @@ mod tests {
assert!(score_text(good) > score_text(bad));
}
fn doc_with_private_differences() -> (Document, lopdf::ObjectId) {
let mut doc = Document::with_version("1.7");
let encoding_id = doc.add_object(dictionary! {
"Differences" => Object::Array(vec![
Object::Integer(0x88),
Object::Name(b"g431".to_vec()),
Object::Name(b"fi".to_vec()),
Object::Integer(0xAD),
Object::Name(b"fl".to_vec()),
]),
});
(doc, encoding_id)
}
#[test]
fn aptos_private_g431_maps_to_ff_ligature() {
let (doc, encoding_id) = doc_with_private_differences();
let font_dict = dictionary! {
"BaseFont" => Object::Name(b"NJEQOD+Aptos".to_vec()),
"Encoding" => Object::Reference(encoding_id),
};
let result = parse_font_encoding(&doc, &font_dict).expect("encoding should parse");
assert_eq!(result.map.get(&0x88u8), Some(&'\u{FB00}'));
assert_eq!(result.map.get(&0x89u8), Some(&'\u{FB01}'));
assert_eq!(result.map.get(&0xADu8), Some(&'\u{FB02}'));
}
#[test]
fn private_g431_does_not_map_for_unrelated_fonts() {
let (doc, encoding_id) = doc_with_private_differences();
let font_dict = dictionary! {
"BaseFont" => Object::Name(b"ABCDEF+OtherFont".to_vec()),
"Encoding" => Object::Reference(encoding_id),
};
let result = parse_font_encoding(&doc, &font_dict).expect("encoding should parse");
assert!(!result.map.contains_key(&0x88u8));
assert_eq!(result.map.get(&0x89u8), Some(&'\u{FB01}'));
assert_eq!(result.map.get(&0xADu8), Some(&'\u{FB02}'));
}
#[test]
fn cid_font_with_unparseable_cmap_does_not_emit_latin1_mojibake() {
// Type0/CID font (font_widths reports `is_cid=true`) where the
@@ -1293,15 +1466,15 @@ mod tests {
}
#[test]
fn simple_font_latin1_fallback_passes_high_bytes_through() {
fn simple_font_single_byte_fallback_passes_high_bytes_through() {
// A Type1/TrueType simple font (is_cid=false) with a `/ToUnicode`
// reference but no usable CMap and no `/Differences` map.
// Per-byte Latin-1 IS the canonical interpretation here — these
// bytes are character codes, not CIDs. The CID guard must NOT
// strip them. Reproduces the false positive that an earlier
// version of the guard introduced for fonts in PDFs like
// pdf-evals/Navigating-Artificial-Intelligence-..., where bytes
// like 0xB6 are legitimate Latin-1 character codes.
// Per-byte fallback is the canonical interpretation here — these
// bytes are character codes, not CIDs. The CID guard must NOT strip
// them. Reproduces the false positive that an earlier version of the
// guard introduced for fonts in PDFs like pdf-evals/Navigating-
// Artificial-Intelligence-..., where bytes like 0xB6 are legitimate
// single-byte character codes.
let bytes = vec![0x24_u8, 0x47, 0xB6, 0x56]; // "$G¶V"
let obj = Object::String(bytes, lopdf::StringFormat::Hexadecimal);
@@ -1334,4 +1507,62 @@ mod tests {
"simple font fallback must not stamp FFFD over legitimate bytes: {text:?}"
);
}
#[test]
fn simple_font_single_byte_fallback_maps_cp1252_punctuation() {
let bytes = vec![b'l', 0x92_u8, b'a', b'c', b'a', b'd'];
let obj = Object::String(bytes, lopdf::StringFormat::Hexadecimal);
let font_cmaps = FontCMaps::default();
let font_tounicode_refs: HashMap<String, u32> = HashMap::new();
let inline_cmaps = HashMap::new();
let font_encodings: PageFontEncodings = HashMap::new();
let encoding_cache: HashMap<String, Encoding<'_>> = HashMap::new();
let mut decisions = CMapDecisionCache::new();
let font_widths: PageFontWidths = HashMap::new();
let text = extract_text_from_operand(
&obj,
"F1",
None,
&font_cmaps,
&font_tounicode_refs,
&inline_cmaps,
&font_encodings,
&encoding_cache,
&mut decisions,
&font_widths,
)
.expect("simple font should decode CP1252 punctuation");
assert_eq!(text, "lacad");
}
#[test]
fn cached_encoding_decode_normalizes_cp1252_controls() {
let text = normalize_cp1252_controls("d\u{92}un \u{96} test".to_string(), true);
assert_eq!(text, "dun test");
}
#[test]
fn tex_font_decode_keeps_c1_ligature_bytes_unmodified() {
let text = normalize_cp1252_controls("de\u{85}ciente \u{87}uid".to_string(), false);
assert_eq!(text, "de\u{85}ciente \u{87}uid");
assert!(!should_use_cp1252_single_byte_fallback(
Some("TTdcr10"),
false
));
assert!(!should_use_cp1252_single_byte_fallback(
Some("cmr10"),
false
));
}
#[test]
fn winansi_text_font_uses_cp1252_fallback() {
assert!(should_use_cp1252_single_byte_fallback(
Some("BJPQNQ+Times-Roman"),
false
));
}
}
+3 -5
View File
@@ -1234,11 +1234,7 @@ pub(crate) fn group_into_lines_with_thresholds(
ci,
item.x,
item.y,
if item.text.len() > 60 {
&item.text[..60]
} else {
&item.text
}
super::trace_text_preview(&item.text, 60)
);
}
}
@@ -1498,6 +1494,7 @@ mod tests {
page,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
}
@@ -1627,6 +1624,7 @@ mod tests {
page,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
});
+2
View File
@@ -78,6 +78,7 @@ pub fn extract_page_links(doc: &Document, page_id: ObjectId, page_num: u32) -> V
page: page_num,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Link(url),
mcid: None,
});
@@ -316,6 +317,7 @@ pub(crate) fn walk_form_fields(
page: page_num,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::FormField,
mcid: None,
});
+370 -16
View File
@@ -6,11 +6,12 @@ pub(crate) mod content_stream;
mod fonts;
mod layout;
mod links;
pub(crate) mod underline;
mod xobjects;
use crate::text_utils::is_rtl_text;
use crate::tounicode::FontCMaps;
use crate::types::{PageExtraction, TextItem};
use crate::types::{PageExtraction, PdfLine, PdfRect, TextItem};
use crate::PdfError;
use log::debug;
use lopdf::{Document, Object, ObjectId};
@@ -33,6 +34,13 @@ pub(crate) use layout::ColumnRegion;
// Public API
// ---------------------------------------------------------------------------
pub(crate) fn trace_text_preview(text: &str, max_chars: usize) -> &str {
match text.char_indices().nth(max_chars) {
Some((idx, _)) => &text[..idx],
None => text,
}
}
/// Extract text from PDF file as plain string
pub fn extract_text<P: AsRef<Path>>(path: P) -> Result<String, PdfError> {
crate::validate_pdf_file(&path)?;
@@ -173,6 +181,7 @@ fn extract_positioned_text_impl(
if threshold > 0.10 {
page_thresholds.insert(*page_num, threshold);
}
suppress_table_underlines(&mut items, &rects, &lines, *page_num);
debug!(
"page {}: {} text items, {} rects, {} lines{}",
page_num,
@@ -195,11 +204,7 @@ fn extract_positioned_text_impl(
item.width,
item.font_size,
item.font,
if item.text.len() > 80 {
&item.text[..80]
} else {
&item.text
}
trace_text_preview(&item.text, 80)
);
}
}
@@ -223,6 +228,38 @@ fn extract_positioned_text_impl(
))
}
fn suppress_table_underlines(
items: &mut [TextItem],
rects: &[PdfRect],
lines: &[PdfLine],
page: u32,
) {
if !items.iter().any(|item| item.is_underline) {
return;
}
let mut table_item_indices: HashSet<usize> = HashSet::new();
if !rects.is_empty() {
let (rect_tables, _) = crate::tables::detect_tables_from_rects(items, rects, page);
for table in rect_tables {
table_item_indices.extend(table.item_indices);
}
}
if !lines.is_empty() {
for table in crate::tables::detect_tables_from_lines(items, lines, page) {
table_item_indices.extend(table.item_indices);
}
}
for index in table_item_indices {
if let Some(item) = items.get_mut(index) {
item.is_underline = false;
}
}
}
// ---------------------------------------------------------------------------
// Shared helpers (used by submodules via `super::`)
// ---------------------------------------------------------------------------
@@ -349,6 +386,133 @@ fn effective_merge_width(item: &TextItem) -> f32 {
}
}
fn is_standalone_bullet_text(text: &str) -> bool {
matches!(text.trim(), "" | "" | "" | "")
}
fn first_text_char(text: &str) -> Option<char> {
text.trim_start().chars().next()
}
fn is_short_alpha_fragment(text: &str) -> bool {
let trimmed = text.trim();
let char_count = trimmed.chars().count();
(1..=4).contains(&char_count) && trimmed.chars().all(char::is_alphabetic)
}
fn has_phrase_continuation_shape(text: &str) -> bool {
let trimmed = text.trim_start();
trimmed
.chars()
.take(24)
.any(|ch| ch.is_whitespace() || matches!(ch, '-'))
}
fn should_preserve_overlapping_stream_order(group: &[&TextItem]) -> bool {
if group.len() < 3 {
return false;
}
let Some(first) = group.iter().find(|item| !item.text.trim().is_empty()) else {
return false;
};
if group.iter().all(|item| item.mcid.is_none()) {
return false;
}
let mut nonempty_count = 0;
let mut saw_backtrack = false;
let mut nonspace_chars = 0;
let mut math_symbol_chars = 0;
let mut max_font_size = first.font_size;
for item in group {
if !item.text.trim().is_empty() {
nonempty_count += 1;
}
if (item.font_size - first.font_size).abs() > first.font_size * 0.25 {
return false;
}
max_font_size = max_font_size.max(item.font_size);
for ch in item.text.chars().filter(|ch| !ch.is_whitespace()) {
nonspace_chars += 1;
if matches!(
ch,
'*' | 'ˆ' | '^' | '=' | '+' | '_' | '[' | ']' | '{' | '}' | '|' | '<' | '>'
) {
math_symbol_chars += 1;
}
}
}
if nonempty_count < 2 {
return false;
}
if nonspace_chars > 0 && math_symbol_chars * 4 > nonspace_chars {
return false;
}
let mut sorted_by_x = group.to_vec();
sorted_by_x.sort_by(|a, b| a.x.total_cmp(&b.x));
let cluster_start = sorted_by_x[0].x;
let mut cluster_end = cluster_start + effective_merge_width(sorted_by_x[0]);
for item in sorted_by_x.iter().skip(1) {
let gap = item.x - cluster_end;
if gap > max_font_size * 2.5 {
return false;
}
cluster_end = cluster_end.max(item.x + effective_merge_width(item));
}
if cluster_end - cluster_start > max_font_size * 36.0 {
return false;
}
for index in 0..group.len() - 1 {
let previous = group[index];
let next = group[index + 1];
let font_size = previous.font_size.max(next.font_size);
let backtrack_threshold = font_size * 0.25;
let previous_start = previous.x;
let next_start = next.x;
let next_end = next.x + effective_merge_width(next);
if next_start < previous_start - backtrack_threshold
&& next_end > previous_start + backtrack_threshold
{
let has_near_prefix = group[..=index].iter().rev().take(4).any(|item| {
is_short_alpha_fragment(&item.text)
&& item.x >= next_start - font_size * 0.5
&& item.x <= next_start + font_size * 4.0
});
let starts_lowercase = first_text_char(&next.text).is_some_and(char::is_lowercase);
let phrase_continuation = has_phrase_continuation_shape(&next.text);
let has_near_bullet = group[..=index]
.iter()
.position(|item| {
is_standalone_bullet_text(&item.text) && next_start <= item.x + font_size * 3.0
})
.is_some_and(|bullet_index| {
if bullet_index >= index {
return false;
}
group[bullet_index + 1..=index]
.iter()
.rev()
.find(|item| !item.text.trim().is_empty())
.is_some_and(|item| {
item.text.trim().chars().count() <= 8
&& has_phrase_continuation_shape(&next.text)
})
});
if (has_near_prefix && starts_lowercase && phrase_continuation) || has_near_bullet {
saw_backtrack = true;
break;
}
}
}
saw_backtrack
}
pub(crate) fn merge_text_items(items: Vec<TextItem>) -> Vec<TextItem> {
if items.is_empty() {
return items;
@@ -369,28 +533,33 @@ pub(crate) fn merge_text_items(items: Vec<TextItem>) -> Vec<TextItem> {
}
}
// Sort each group by X position (direction-aware)
for (_, _, group) in &mut line_groups {
let mut ordered_line_groups: Vec<(u32, f32, Vec<&TextItem>, bool)> = Vec::new();
// Sort each group by X position (direction-aware), except for lines whose
// content stream intentionally backtracks to overlay ActualText fragments.
for (page, y, mut group) in line_groups {
let rtl = is_rtl_text(group.iter().map(|i| &i.text));
let preserve_stream_order = !rtl && should_preserve_overlapping_stream_order(&group);
if rtl {
group.sort_by(|a, b| b.x.total_cmp(&a.x));
} else {
} else if !preserve_stream_order {
group.sort_by(|a, b| a.x.total_cmp(&b.x));
}
ordered_line_groups.push((page, y, group, preserve_stream_order));
}
// Sort groups by page then Y descending (top of page first)
line_groups.sort_by(|a, b| a.0.cmp(&b.0).then_with(|| b.1.total_cmp(&a.1)));
ordered_line_groups.sort_by(|a, b| a.0.cmp(&b.0).then_with(|| b.1.total_cmp(&a.1)));
let mut merged = Vec::new();
for (_, _, group) in &line_groups {
for (_, _, group, preserve_stream_order) in &ordered_line_groups {
let mut i = 0;
while i < group.len() {
let first = group[i];
let mut text = first.text.clone();
let mut end_x = first.x + effective_merge_width(first);
let x_gap_max = first.font_size * 0.5;
let mut is_underline = first.is_underline;
let mut j = i + 1;
while j < group.len() {
@@ -400,10 +569,15 @@ pub(crate) fn merge_text_items(items: Vec<TextItem>) -> Vec<TextItem> {
break;
}
let gap = next.x - end_x;
let x_gap_max = if *preserve_stream_order && is_standalone_bullet_text(&text) {
first.font_size * 1.2
} else {
first.font_size * 0.5
};
if gap > x_gap_max {
break;
}
if gap < -first.font_size * 0.5 {
if gap < -first.font_size * 0.5 && !preserve_stream_order {
break;
}
// Insert space at word boundaries.
@@ -425,11 +599,20 @@ pub(crate) fn merge_text_items(items: Vec<TextItem>) -> Vec<TextItem> {
first.font_size * 0.08
}
};
if gap > threshold {
let needs_bullet_space = *preserve_stream_order
&& is_standalone_bullet_text(&text)
&& !next.text.trim().is_empty();
if needs_bullet_space || gap > threshold {
text.push(' ');
}
text.push_str(&next.text);
end_x = next.x + effective_merge_width(next);
is_underline |= next.is_underline;
let next_end = next.x + effective_merge_width(next);
end_x = if *preserve_stream_order {
end_x.max(next_end)
} else {
next_end
};
j += 1;
}
@@ -444,6 +627,7 @@ pub(crate) fn merge_text_items(items: Vec<TextItem>) -> Vec<TextItem> {
page: first.page,
is_bold: first.is_bold,
is_italic: first.is_italic,
is_underline,
item_type: first.item_type.clone(),
mcid: first.mcid,
});
@@ -554,7 +738,7 @@ pub(crate) fn get_number(obj: &Object) -> Option<f32> {
mod tests {
use super::*;
use crate::text_utils::{is_cjk_char, is_rtl_char, is_rtl_text, sort_line_items};
use crate::types::{ItemType, TextLine};
use crate::types::{ItemType, PdfLine, TextLine};
use layout::{detect_columns, is_newspaper_layout, ColumnRegion};
fn make_merge_item(text: &str, x: f32, width: f32) -> TextItem {
@@ -569,11 +753,37 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
}
}
fn with_mcid(mut item: TextItem) -> TextItem {
item.mcid = Some(1);
item
}
fn make_line(x1: f32, y1: f32, x2: f32, y2: f32) -> PdfLine {
PdfLine {
x1,
y1,
x2,
y2,
page: 1,
}
}
#[test]
fn trace_text_preview_truncates_on_char_boundary() {
let text = format!("{}{}tail", "a".repeat(79), '\u{FFFD}');
let preview = trace_text_preview(&text, 80);
assert_eq!(preview.chars().count(), 80);
assert!(text.is_char_boundary(preview.len()));
assert!(preview.ends_with('\u{FFFD}'));
}
#[test]
fn merge_items_no_space_before_period() {
// Simulate Tc/Tw-adjusted width: "date" width is smaller than the gap
@@ -612,6 +822,127 @@ mod tests {
assert_eq!(merged[0].text, "hello world");
}
#[test]
fn merge_items_preserves_underline_from_later_fragment() {
let mut items = vec![
make_merge_item("pre", 100.0, 18.0),
make_merge_item("fix", 119.0, 18.0),
];
items[1].is_underline = true;
let merged = merge_text_items(items);
assert_eq!(merged.len(), 1);
assert_eq!(merged[0].text, "prefix");
assert!(merged[0].is_underline);
}
#[test]
fn merge_items_preserves_stream_order_for_backtracking_heading() {
// Some tagged PDFs emit first-letter ActualText fragments, then reset
// the text matrix and draw the rest of the word from the line start.
let items = vec![
with_mcid(make_merge_item("F", 79.4, 4.5)),
with_mcid(make_merge_item("r", 83.9, 3.3)),
with_mcid(make_merge_item("om tables to data-", 79.4, 89.7)),
with_mcid(make_merge_item("", 168.9, 33.9)),
with_mcid(make_merge_item("analytics-", 168.9, 75.5)),
with_mcid(make_merge_item("ready content", 210.5, 60.8)),
];
let merged = merge_text_items(items);
assert_eq!(merged.len(), 1);
assert_eq!(
merged[0].text,
"From tables to data-analytics-ready content"
);
}
#[test]
fn merge_items_preserves_stream_order_for_reset_word_prefix() {
let items = vec![
with_mcid(make_merge_item("N", 68.0, 7.0)),
with_mcid(make_merge_item("e", 75.1, 4.0)),
with_mcid(make_merge_item("w fields created", 68.0, 82.0)),
];
let merged = merge_text_items(items);
assert_eq!(merged.len(), 1);
assert_eq!(merged[0].text, "New fields created");
}
#[test]
fn merge_items_uses_x_order_for_untagged_backtracking_text() {
let items = vec![
make_merge_item("N", 68.0, 7.0),
make_merge_item("e", 75.1, 4.0),
make_merge_item("w fields created", 68.2, 82.0),
];
let merged = merge_text_items(items);
let texts: Vec<_> = merged.iter().map(|item| item.text.as_str()).collect();
assert_eq!(texts, vec!["N", "w fields created", "e"]);
}
#[test]
fn merge_items_preserves_bullet_stream_order_with_backtracking() {
let items = vec![
with_mcid(make_merge_item("", 79.4, 5.0)),
with_mcid(make_merge_item("The MS", 91.0, 32.6)),
with_mcid(make_merge_item("A LoS project", 84.4, 70.0)),
];
let merged = merge_text_items(items);
assert_eq!(merged.len(), 1);
assert_eq!(merged[0].text, "• The MSA LoS project");
}
#[test]
fn merge_items_keeps_normal_bullet_gap_limit_without_stream_order() {
let items = vec![
make_merge_item("", 79.4, 5.0),
make_merge_item("Distant item", 91.0, 60.0),
];
let merged = merge_text_items(items);
let texts: Vec<_> = merged.iter().map(|item| item.text.as_str()).collect();
assert_eq!(texts, vec!["", "Distant item"]);
}
#[test]
fn suppress_table_underlines_clears_line_detected_table_items() {
let mut items = vec![
make_merge_item("H1", 125.0, 20.0),
make_merge_item("H2", 225.0, 20.0),
make_merge_item("A", 125.0, 20.0),
make_merge_item("B", 225.0, 20.0),
];
items[0].y = 490.0;
items[1].y = 490.0;
items[2].y = 470.0;
items[3].y = 470.0;
for item in &mut items {
item.is_underline = true;
}
let lines = vec![
make_line(100.0, 500.0, 300.0, 500.0),
make_line(100.0, 480.0, 300.0, 480.0),
make_line(100.0, 460.0, 300.0, 460.0),
make_line(100.0, 460.0, 100.0, 500.0),
make_line(200.0, 460.0, 200.0, 500.0),
make_line(300.0, 460.0, 300.0, 500.0),
];
suppress_table_underlines(&mut items, &[], &lines, 1);
assert!(items.iter().all(|item| !item.is_underline));
}
#[test]
fn test_group_into_lines() {
let items = vec![
@@ -626,6 +957,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
},
@@ -640,6 +972,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
},
@@ -654,6 +987,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
},
@@ -708,6 +1042,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
},
@@ -722,6 +1057,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
},
@@ -736,6 +1072,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
},
@@ -761,6 +1098,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
},
@@ -775,6 +1113,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
},
@@ -789,6 +1128,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
},
@@ -816,6 +1156,7 @@ mod tests {
page: 1,
is_bold: true,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
}
@@ -850,6 +1191,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
}
@@ -885,6 +1227,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
},
@@ -899,6 +1242,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
},
@@ -913,6 +1257,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
},
@@ -935,6 +1280,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
}
@@ -1047,6 +1393,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
},
@@ -1061,6 +1408,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
},
@@ -1085,6 +1433,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
},
@@ -1099,6 +1448,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
},
@@ -1139,6 +1489,7 @@ mod tests {
page,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
}],
@@ -1183,6 +1534,7 @@ mod tests {
page,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
}],
@@ -1227,6 +1579,7 @@ mod tests {
page,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
}],
@@ -1264,6 +1617,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
}
+500
View File
@@ -0,0 +1,500 @@
//! Geometric underline detection.
//!
//! PDFs have no underline font flag — underlines are drawn as separate
//! graphics: stroked horizontal lines (`l`/`S` operators) or thin filled
//! rectangles (`re`/`f`). This pass correlates those graphics with text
//! items after extraction: an item is underlined when a horizontal
//! line/thin rect sits just below its baseline and covers most of its
//! horizontal extent.
//!
//! Repeated same-span rules are treated as table/form rulings rather than
//! underlines, which avoids marking every cell in ruled tables.
use std::collections::HashSet;
use crate::types::{ItemType, PdfRect, TextItem};
/// Max thickness (pt) for a stroked line / filled rect to count as an
/// underline rule rather than a border or decorative band.
const MAX_RULE_THICKNESS: f32 = 2.0;
/// Fraction of the item's width that the rule must cover horizontally.
const MIN_X_OVERLAP: f32 = 0.6;
/// Same-span rules repeated at this many y-levels are usually table/form
/// rulings, not semantic underlines.
const MIN_REPEATED_RULE_LEVELS: usize = 3;
/// Vertical tolerance for considering two rules to be on the same row edge.
const RULE_Y_DEDUP_EPS: f32 = 2.0;
/// Horizontal span similarity required when clustering repeated rulings.
const RULE_SPAN_OVERLAP_RATIO: f32 = 0.8;
const RULE_SPAN_WIDTH_RATIO: f32 = 1.5;
/// Multiple separated rule segments on one row are usually per-column table
/// header/body separators.
const MIN_SEGMENTED_ROW_RULES: usize = 3;
const MIN_SEGMENTED_ROW_GAPS: usize = 2;
const SEGMENTED_ROW_GAP_MIN: f32 = 12.0;
/// A single rule under several widely separated items is usually a table
/// header/body separator, not a sentence underline.
const MIN_TABULAR_RULE_ITEMS: usize = 3;
const MIN_TABULAR_RULE_GAPS: usize = 2;
const TABULAR_RULE_GAP_EM: f32 = 2.0;
#[derive(Clone)]
pub(crate) struct UnderlineLine {
pub(crate) x1: f32,
pub(crate) y1: f32,
pub(crate) x2: f32,
pub(crate) y2: f32,
pub(crate) stroke_width: f32,
pub(crate) page: u32,
}
/// A horizontal rule candidate in page coordinates (PDF y-up).
#[derive(Clone)]
struct Rule {
x1: f32,
x2: f32,
y: f32,
}
impl Rule {
fn width(&self) -> f32 {
self.x2 - self.x1
}
}
fn rules_from_graphics(rects: &[PdfRect], lines: &[UnderlineLine], page: u32) -> Vec<Rule> {
let mut rules: Vec<Rule> = Vec::new();
for l in lines {
if l.page != page {
continue;
}
// Horizontal stroked line (tolerate slight skew).
if l.stroke_width <= MAX_RULE_THICKNESS && (l.y1 - l.y2).abs() <= MAX_RULE_THICKNESS {
let (x1, x2) = if l.x1 <= l.x2 {
(l.x1, l.x2)
} else {
(l.x2, l.x1)
};
if x2 - x1 > 1.0 {
rules.push(Rule {
x1,
x2,
y: (l.y1 + l.y2) / 2.0,
});
}
}
}
for r in rects {
if r.page != page {
continue;
}
// Thin filled rect used as an underline rule. Extents are
// normalized first: `re` operands pass through the CTM, so
// width/height can be negative (flipped axes / negative scale) —
// without normalization negative-width rules are missed and
// negative-height bands sneak past the thickness check.
let (x1, x2) = if r.width >= 0.0 {
(r.x, r.x + r.width)
} else {
(r.x + r.width, r.x)
};
if r.height.abs() <= MAX_RULE_THICKNESS && x2 - x1 > 1.0 {
rules.push(Rule {
x1,
x2,
y: r.y + r.height / 2.0,
});
}
}
rules
}
fn discard_repeated_ruling_rules(rules: Vec<Rule>) -> Vec<Rule> {
if rules.len() < MIN_REPEATED_RULE_LEVELS {
return rules;
}
rules
.iter()
.filter(|rule| {
!is_repeated_ruling_rule(rule, &rules) && !is_segmented_row_ruling_rule(rule, &rules)
})
.cloned()
.collect()
}
fn is_repeated_ruling_rule(rule: &Rule, rules: &[Rule]) -> bool {
let mut y_levels: Vec<f32> = rules
.iter()
.filter(|other| has_similar_span(rule, other))
.map(|other| other.y)
.collect();
y_levels.sort_by(|a, b| a.total_cmp(b));
y_levels.dedup_by(|a, b| (*a - *b).abs() <= RULE_Y_DEDUP_EPS);
y_levels.len() >= MIN_REPEATED_RULE_LEVELS
}
fn is_segmented_row_ruling_rule(rule: &Rule, rules: &[Rule]) -> bool {
let mut row_rules: Vec<&Rule> = rules
.iter()
.filter(|other| (other.y - rule.y).abs() <= RULE_Y_DEDUP_EPS)
.collect();
if row_rules.len() < MIN_SEGMENTED_ROW_RULES {
return false;
}
row_rules.sort_by(|a, b| a.x1.total_cmp(&b.x1));
let large_gaps = row_rules
.windows(2)
.filter(|pair| pair[1].x1 - pair[0].x2 > SEGMENTED_ROW_GAP_MIN)
.count();
large_gaps >= MIN_SEGMENTED_ROW_GAPS
}
fn has_similar_span(a: &Rule, b: &Rule) -> bool {
let a_width = a.width();
let b_width = b.width();
if a_width <= 1.0 || b_width <= 1.0 {
return false;
}
let width_ratio = a_width.max(b_width) / a_width.min(b_width);
if width_ratio > RULE_SPAN_WIDTH_RATIO {
return false;
}
let overlap = a.x2.min(b.x2) - a.x1.max(b.x1);
overlap >= a_width.min(b_width) * RULE_SPAN_OVERLAP_RATIO
}
fn tabular_row_separator_rule_indices(rules: &[Rule], items: &[TextItem]) -> HashSet<usize> {
let mut tabular_rules = HashSet::new();
for (rule_idx, rule) in rules.iter().enumerate() {
let mut matched_items: Vec<&TextItem> = items
.iter()
.filter(|item| is_underline_candidate(item) && rule_matches_item(rule, item))
.collect();
if matched_items.len() < MIN_TABULAR_RULE_ITEMS {
continue;
}
matched_items.sort_by(|a, b| a.x.total_cmp(&b.x));
let large_gaps = matched_items
.windows(2)
.filter(|pair| {
let left = pair[0];
let right = pair[1];
let gap = right.x - (left.x + left.width);
let font_size = left.font_size.max(right.font_size).max(1.0);
gap > font_size * TABULAR_RULE_GAP_EM
})
.count();
if large_gaps >= MIN_TABULAR_RULE_GAPS {
tabular_rules.insert(rule_idx);
}
}
tabular_rules
}
fn is_underline_candidate(item: &TextItem) -> bool {
matches!(item.item_type, ItemType::Text) && !item.text.trim().is_empty() && item.width > 0.0
}
fn rule_matches_item(rule: &Rule, item: &TextItem) -> bool {
// Vertical window: underlines sit at or slightly below the baseline.
// Fonts draw them at roughly 5-15% of the em below; allow up to 35%
// (min 3pt) below and 1pt above for rounding.
let below = (item.font_size * 0.35).max(3.0);
let y_min = item.y - below;
let y_max = item.y + 1.0;
if rule.y < y_min || rule.y > y_max {
return false;
}
let ix1 = item.x;
let ix2 = item.x + item.width;
let min_overlap = item.width * MIN_X_OVERLAP;
let overlap = rule.x2.min(ix2) - rule.x1.max(ix1);
overlap >= min_overlap
}
/// Mark `is_underline` on text items that have a horizontal rule just
/// below their baseline. `items`, `rects`, and `lines` are a single
/// page's extraction output (all in PDF coordinates, y-up, where
/// `TextItem::y` is the text baseline).
pub(crate) fn mark_underlined_items(
items: &mut [TextItem],
rects: &[PdfRect],
lines: &[UnderlineLine],
page: u32,
) {
let rules = discard_repeated_ruling_rules(rules_from_graphics(rects, lines, page));
if rules.is_empty() {
return;
}
let tabular_rules = tabular_row_separator_rule_indices(&rules, items);
for item in items.iter_mut() {
if !is_underline_candidate(item) {
continue;
}
for (rule_idx, rule) in rules.iter().enumerate() {
if tabular_rules.contains(&rule_idx) {
continue;
}
if rule_matches_item(rule, item) {
item.is_underline = true;
break;
}
}
}
}
#[cfg(test)]
mod tests {
use super::*;
use crate::types::ItemType;
fn item(text: &str, x: f32, y: f32, width: f32, font_size: f32) -> TextItem {
TextItem {
text: text.to_string(),
x,
y,
width,
height: font_size,
font: "F1".to_string(),
font_size,
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
}
}
fn hline(x1: f32, x2: f32, y: f32) -> UnderlineLine {
UnderlineLine {
x1,
y1: y,
x2,
y2: y,
stroke_width: 1.0,
page: 1,
}
}
fn thin_rect(x: f32, y: f32, width: f32) -> PdfRect {
PdfRect {
x,
y,
width,
height: 0.8,
page: 1,
}
}
#[test]
fn stroked_line_under_baseline_marks_underline() {
let mut items = vec![item("underlined", 100.0, 500.0, 60.0, 10.0)];
let lines = vec![hline(99.0, 161.0, 498.5)];
mark_underlined_items(&mut items, &[], &lines, 1);
assert!(items[0].is_underline);
}
#[test]
fn thin_filled_rect_under_baseline_marks_underline() {
let mut items = vec![item("underlined", 100.0, 500.0, 60.0, 10.0)];
let rects = vec![thin_rect(100.0, 497.8, 60.0)];
mark_underlined_items(&mut items, &rects, &[], 1);
assert!(items[0].is_underline);
}
#[test]
fn long_rule_under_multiple_items_marks_each() {
// One underline drawn under a whole sentence: every overlapped
// item gets the flag.
let mut items = vec![
item("first", 100.0, 500.0, 40.0, 10.0),
item("second", 145.0, 500.0, 50.0, 10.0),
];
let lines = vec![hline(98.0, 200.0, 498.0)];
mark_underlined_items(&mut items, &[], &lines, 1);
assert!(items[0].is_underline);
assert!(items[1].is_underline);
}
#[test]
fn line_far_below_baseline_is_not_an_underline() {
// A horizontal rule 30pt below (section divider) must not mark.
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
let lines = vec![hline(90.0, 300.0, 470.0)];
mark_underlined_items(&mut items, &[], &lines, 1);
assert!(!items[0].is_underline);
}
#[test]
fn thick_stroked_line_is_not_an_underline() {
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
let mut line = hline(99.0, 161.0, 498.5);
line.stroke_width = 4.0;
mark_underlined_items(&mut items, &[], &[line], 1);
assert!(!items[0].is_underline);
}
#[test]
fn line_above_baseline_is_not_an_underline() {
// Strikethrough / overline geometry must not mark.
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
let lines = vec![hline(90.0, 300.0, 505.0)];
mark_underlined_items(&mut items, &[], &lines, 1);
assert!(!items[0].is_underline);
}
#[test]
fn insufficient_horizontal_overlap_is_not_an_underline() {
// Rule under only a quarter of the item (e.g. neighboring cell
// border) must not mark.
let mut items = vec![item("wide text item", 100.0, 500.0, 100.0, 10.0)];
let lines = vec![hline(100.0, 125.0, 498.5)];
mark_underlined_items(&mut items, &[], &lines, 1);
assert!(!items[0].is_underline);
}
#[test]
fn negative_width_rect_is_normalized_and_marks_underline() {
// A CTM with negative x-scale (or negative `re` operands) produces
// rects whose width is negative; the rule extents must normalize.
let mut items = vec![item("underlined", 100.0, 500.0, 60.0, 10.0)];
let rects = vec![PdfRect {
x: 160.0,
y: 497.8,
width: -60.0,
height: 0.8,
page: 1,
}];
mark_underlined_items(&mut items, &rects, &[], 1);
assert!(items[0].is_underline);
}
#[test]
fn negative_height_band_is_not_an_underline() {
// A 14pt band expressed with negative height must not pass the
// thickness check via sign trickery.
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
let rects = vec![PdfRect {
x: 95.0,
y: 509.0,
width: 80.0,
height: -14.0,
page: 1,
}];
mark_underlined_items(&mut items, &rects, &[], 1);
assert!(!items[0].is_underline);
}
#[test]
fn thick_band_is_not_an_underline() {
// A highlight bar / filled cell background (tall rect) must not mark.
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
let rects = vec![PdfRect {
x: 95.0,
y: 495.0,
width: 80.0,
height: 14.0,
page: 1,
}];
mark_underlined_items(&mut items, &rects, &[], 1);
assert!(!items[0].is_underline);
}
#[test]
fn vertical_line_is_not_an_underline() {
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
let lines = vec![UnderlineLine {
x1: 120.0,
y1: 498.0,
x2: 120.0,
y2: 400.0,
stroke_width: 1.0,
page: 1,
}];
mark_underlined_items(&mut items, &[], &lines, 1);
assert!(!items[0].is_underline);
}
#[test]
fn other_pages_graphics_do_not_mark() {
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
let mut line = hline(99.0, 161.0, 498.5);
line.page = 2;
mark_underlined_items(&mut items, &[], &[line], 1);
assert!(!items[0].is_underline);
}
#[test]
fn repeated_table_row_rules_do_not_mark_cell_text() {
let mut items = vec![
item("A", 110.0, 500.0, 20.0, 10.0),
item("B", 110.0, 480.0, 20.0, 10.0),
item("C", 110.0, 460.0, 20.0, 10.0),
];
let lines = vec![
hline(100.0, 150.0, 498.0),
hline(100.0, 150.0, 478.0),
hline(100.0, 150.0, 458.0),
];
mark_underlined_items(&mut items, &[], &lines, 1);
assert!(items.iter().all(|item| !item.is_underline));
}
#[test]
fn row_separator_under_spaced_column_labels_is_not_an_underline() {
let mut items = vec![
item("Date", 100.0, 500.0, 25.0, 10.0),
item("Rate", 200.0, 500.0, 25.0, 10.0),
item("Yield", 300.0, 500.0, 30.0, 10.0),
];
let lines = vec![hline(90.0, 340.0, 498.0)];
mark_underlined_items(&mut items, &[], &lines, 1);
assert!(items.iter().all(|item| !item.is_underline));
}
#[test]
fn same_row_spaced_rule_segments_do_not_mark_column_labels() {
let mut items = vec![
item("Date", 100.0, 500.0, 25.0, 10.0),
item("Rate", 200.0, 500.0, 25.0, 10.0),
item("Yield", 300.0, 500.0, 30.0, 10.0),
];
let lines = vec![
hline(98.0, 128.0, 498.0),
hline(198.0, 228.0, 498.0),
hline(298.0, 333.0, 498.0),
];
mark_underlined_items(&mut items, &[], &lines, 1);
assert!(items.iter().all(|item| !item.is_underline));
}
}
+3
View File
@@ -294,6 +294,7 @@ fn extract_form_xobject_text_inner(
page: page_num,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Image,
mcid: None,
});
@@ -438,6 +439,7 @@ fn extract_form_xobject_text_inner(
page: page_num,
is_bold: is_bold_font(base_font),
is_italic: is_italic_font(base_font),
is_underline: false,
item_type: ItemType::Text,
mcid: None,
});
@@ -586,6 +588,7 @@ fn extract_form_xobject_text_inner(
page: page_num,
is_bold: is_bold_font(base_font),
is_italic: is_italic_font(base_font),
is_underline: false,
item_type: ItemType::Text,
mcid: None,
});
+682 -56
View File
File diff suppressed because it is too large Load Diff
+233 -2
View File
@@ -149,6 +149,79 @@ fn find_isolated_lines(lines: &[TextLine], base_size: f32, para_threshold: f32)
set
}
/// Pre-scan body-size all-bold runs that are too long to be headings.
///
/// Some academic PDFs use an all-bold abstract/summary paragraph immediately
/// after the author block. A line-local bold heading heuristic sees each
/// wrapped visual line as "standalone" once the first line is misclassified,
/// producing a stack of `##` headings. Multi-line body-size bold runs with a
/// paragraph-sized word count should stay paragraph text.
fn find_wrapped_bold_paragraph_lines(
lines: &[TextLine],
base_size: f32,
para_threshold: f32,
) -> HashSet<usize> {
let mut set = HashSet::new();
let mut i = 0usize;
while i < lines.len() {
if !is_body_size_all_bold_line(&lines[i], base_size) {
i += 1;
continue;
}
let start = i;
let mut end = i;
let mut word_count = lines[i].text().split_whitespace().count();
while end + 1 < lines.len()
&& is_body_size_all_bold_line(&lines[end + 1], base_size)
&& is_wrapped_same_style_line(&lines[end], &lines[end + 1], para_threshold)
{
end += 1;
word_count += lines[end].text().split_whitespace().count();
}
let line_count = end - start + 1;
if line_count >= 3 && word_count > 20 {
for idx in start..=end {
set.insert(idx);
}
}
i = end + 1;
}
set
}
fn is_body_size_all_bold_line(line: &TextLine, base_size: f32) -> bool {
let Some(first) = line.items.first() else {
return false;
};
first.font_size >= base_size * 0.95
&& first.font_size < base_size * 1.2
&& line
.items
.iter()
.all(|item| item.is_bold && (item.font_size - first.font_size).abs() < 0.5)
}
fn is_wrapped_same_style_line(prev: &TextLine, next: &TextLine, para_threshold: f32) -> bool {
if prev.page != next.page {
return false;
}
let y_gap = prev.y - next.y;
if !(y_gap > 0.0 && y_gap <= para_threshold) {
return false;
}
let prev_x = prev.items.first().map(|item| item.x).unwrap_or(0.0);
let next_x = next.items.first().map(|item| item.x).unwrap_or(0.0);
(prev_x - next_x).abs() <= 40.0
}
/// Resolve the dominant structure role for a text line by looking up its items' MCIDs.
///
/// Returns the first non-container role found (skipping Document/Part/Sect/Div/NonStruct/Span).
@@ -397,6 +470,8 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
// between paragraphs at body font size. Inspired by opendataloader's
// lookahead in HeadingProcessor (prevNode/nextNode context).
let isolated_lines = find_isolated_lines(&lines, base_size, para_threshold);
let wrapped_bold_paragraph_lines =
find_wrapped_bold_paragraph_lines(&lines, base_size, para_threshold);
// Detect struct heading levels that are overused (body text mistagged as headings)
let overused_heading_levels = detect_overused_struct_heading_levels(&lines, struct_roles);
@@ -410,6 +485,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
let mut last_list_x: Option<f32> = None;
let mut in_code_block = false;
let mut prev_had_dot_leaders = false;
let mut paragraph_in_wrapped_bold_run = false;
let mut inserted_tables: HashSet<(u32, usize)> = HashSet::new();
let mut inserted_images: HashSet<(u32, usize)> = HashSet::new();
@@ -475,6 +551,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
current_page = line.page;
prev_y = f32::MAX;
prev_x = 0.0;
paragraph_in_wrapped_bold_run = false;
if options.include_page_numbers {
output.push_str(&format!("<!-- Page {} -->\n\n", current_page));
@@ -489,6 +566,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
output.push('\n');
output.push_str(table_md);
@@ -506,6 +584,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
output.push('\n');
output.push_str(image_md);
@@ -527,9 +606,18 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
&& y_gap.abs() <= para_threshold
&& (prev_x - line_x).abs() > 50.0
&& prev_y < f32::MAX;
if (is_para_break || is_band_switch) && in_paragraph {
let line_all_bold = !line.items.is_empty() && line.items.iter().all(|item| item.is_bold);
let line_in_wrapped_bold_run = wrapped_bold_paragraph_lines.contains(&line_idx);
let is_bold_to_regular_break = in_paragraph
&& paragraph_in_wrapped_bold_run
&& !line_in_wrapped_bold_run
&& !line_all_bold
&& y_gap > base_size * 1.2
&& y_gap <= para_threshold;
if (is_para_break || is_band_switch || is_bold_to_regular_break) && in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
// Don't immediately end list on paragraph break
// Let the continuation check below decide if we're still in a list
@@ -572,6 +660,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
output.push_str(trimmed);
output.push_str("\n\n");
@@ -625,6 +714,9 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
if !(1..=15).contains(&word_count) {
return None;
}
if wrapped_bold_paragraph_lines.contains(&line_idx) {
return None;
}
let rarity = font_size_rarity(line_font_size, &font_stats);
let all_bold = !line.items.is_empty() && line.items.iter().all(|i| i.is_bold);
let standalone = !in_paragraph;
@@ -656,6 +748,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
let prefix = "#".repeat(level);
// Use plain text for headers to avoid redundant formatting
@@ -678,6 +771,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
output.push_str(&format!("- {}", trimmed));
output.push('\n');
@@ -691,6 +785,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
let formatted = format_list_item(trimmed);
output.push_str(&formatted);
@@ -737,6 +832,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
output.push_str(&format!("> {}\n", trimmed));
continue;
@@ -747,6 +843,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
if !in_code_block {
output.push_str("```\n");
@@ -767,6 +864,11 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
}
}
output.push_str(trimmed);
paragraph_in_wrapped_bold_run = if in_paragraph {
paragraph_in_wrapped_bold_run || line_in_wrapped_bold_run
} else {
line_in_wrapped_bold_run
};
in_paragraph = true;
prev_had_dot_leaders = cur_dot_leaders;
}
@@ -836,6 +938,8 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
let para_threshold = compute_paragraph_threshold(&lines, base_size);
let isolated_lines = find_isolated_lines(&lines, base_size, para_threshold);
let wrapped_bold_paragraph_lines =
find_wrapped_bold_paragraph_lines(&lines, base_size, para_threshold);
let mut output = String::new();
let mut current_page = 0u32;
@@ -844,6 +948,7 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
let mut in_paragraph = false;
let mut last_list_x: Option<f32> = None;
let mut prev_had_dot_leaders = false;
let mut paragraph_in_wrapped_bold_run = false;
for (line_idx, line) in lines.iter().enumerate() {
// Page break
@@ -860,6 +965,7 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
in_list = false;
last_list_x = None;
prev_had_dot_leaders = false;
paragraph_in_wrapped_bold_run = false;
if options.include_page_numbers {
output.push_str(&format!("<!-- Page {} -->\n\n", current_page));
@@ -870,9 +976,18 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
// (newspaper columns emitted sequentially on the same page).
let y_gap = prev_y - line.y;
let is_para_break = y_gap.abs() > para_threshold;
if is_para_break && in_paragraph {
let line_all_bold = !line.items.is_empty() && line.items.iter().all(|item| item.is_bold);
let line_in_wrapped_bold_run = wrapped_bold_paragraph_lines.contains(&line_idx);
let is_bold_to_regular_break = in_paragraph
&& paragraph_in_wrapped_bold_run
&& !line_in_wrapped_bold_run
&& !line_all_bold
&& y_gap > base_size * 1.2
&& y_gap <= para_threshold;
if (is_para_break || is_bold_to_regular_break) && in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
// Don't immediately end list on paragraph break
// Let the continuation check below decide if we're still in a list
@@ -896,6 +1011,7 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
output.push_str(trimmed);
output.push_str("\n\n");
@@ -918,6 +1034,9 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
if !(1..=15).contains(&word_count) {
return None;
}
if wrapped_bold_paragraph_lines.contains(&line_idx) {
return None;
}
let rarity = font_size_rarity(line_font_size, &font_stats);
let all_bold = !line.items.is_empty() && line.items.iter().all(|i| i.is_bold);
let standalone = !in_paragraph;
@@ -935,6 +1054,7 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
let prefix = "#".repeat(header_level);
// Use plain text for headers to avoid redundant formatting
@@ -949,6 +1069,7 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
let formatted = format_list_item(trimmed);
output.push_str(&formatted);
@@ -993,6 +1114,7 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
if in_paragraph {
output.push_str("\n\n");
in_paragraph = false;
paragraph_in_wrapped_bold_run = false;
}
// Use plain text for code blocks
output.push_str(&format!("```\n{}\n```\n", plain_trimmed));
@@ -1010,6 +1132,11 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
}
}
output.push_str(trimmed);
paragraph_in_wrapped_bold_run = if in_paragraph {
paragraph_in_wrapped_bold_run || line_in_wrapped_bold_run
} else {
line_in_wrapped_bold_run
};
in_paragraph = true;
prev_had_dot_leaders = cur_dot_leaders;
}
@@ -1042,6 +1169,7 @@ mod tests {
page,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: crate::types::ItemType::Text,
mcid,
}
@@ -1356,6 +1484,109 @@ mod tests {
);
}
#[test]
fn test_wrapped_bold_abstract_is_not_split_into_headings() {
// Regression for arXiv 1107.1353: the opening abstract paragraph is
// entirely bold at body size. The first wrapped lines used to become
// separate H2 headings, and the following body paragraph was joined to
// the bold abstract because the paragraph gap is modest.
let make = |text: &str, y: f32, font_size: f32, bold: bool| {
let mut item = make_item(text, 1, None);
item.y = y;
item.font_size = font_size;
item.height = font_size;
item.is_bold = bold;
item
};
let lines = vec![
make_line(vec![make(
"Quantum Nature of Light Measured With a Single Detector",
747.7,
25.0,
true,
)]),
make_line(vec![make(
"Gesine A. Steudle1*, Stefan Schietinger1, David Höckel1",
651.1,
11.0,
false,
)]),
make_line(vec![make(
"Zwiller2, and Oliver Benson1",
638.5,
11.0,
false,
)]),
make_line(vec![make(
"The introduction of light quanta by Einstein in 1905 triggered strong efforts to",
607.5,
11.0,
true,
)]),
make_line(vec![make(
"demonstrate the quantum properties of light directly, without involving matter",
594.8,
11.0,
true,
)]),
make_line(vec![make(
"quantization. It however took more than seven decades for the quantum granularity",
582.2,
11.0,
true,
)]),
make_line(vec![make(
"of light to be observed in the fluorescence of single atoms. Single atoms emit",
569.5,
11.0,
true,
)]),
make_line(vec![make(
"photons one at a time, this is typically demonstrated with a Hanbury-Brown-Twiss",
556.9,
11.0,
true,
)]),
make_line(vec![make(
"Our work significantly simplifies a widely used photon-correlation technique.",
544.2,
11.0,
true,
)]),
make_line(vec![make(
"A photon is a single excitation of a mode of the electromagnetic field.",
528.7,
11.0,
false,
)]),
];
let md = to_markdown_from_lines_with_tables_and_images(
lines,
MarkdownOptions::default(),
HashMap::new(),
HashMap::new(),
&std::collections::HashSet::new(),
None,
);
assert!(
md.contains("# Quantum Nature of Light Measured With a Single Detector"),
"title should remain a heading: {md}"
);
assert!(
!md.contains("## The introduction")
&& !md.contains("## demonstrate")
&& !md.contains("## quantization"),
"bold abstract lines should not become headings: {md}"
);
assert!(
md.contains("technique.**\n\nA photon is a single excitation"),
"body paragraph should be separated from bold abstract: {md}"
);
}
#[test]
fn test_struct_role_code_multiline_accumulation() {
let mut line1 = make_item("fn main() {", 1, Some(0));
+1
View File
@@ -1217,6 +1217,7 @@ mod tests {
page,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: crate::types::ItemType::Text,
mcid: None,
}
+27
View File
@@ -29,6 +29,7 @@ pub(crate) fn clean_markdown(mut text: String, options: &MarkdownOptions) -> Str
// text item, which combine with gap-based space insertion to produce
// double spaces ("Vice President" instead of "Vice President").
collapse_consecutive_spaces(&mut text);
remove_spaces_before_closing_brackets(&mut text);
// Remove excessive newlines (more than 2 in a row)
while text.contains("\n\n\n") {
@@ -71,6 +72,20 @@ fn collapse_consecutive_spaces(text: &mut String) {
*text = result;
}
/// Remove spaces before closing square brackets.
/// Unit markers and markdown links occasionally pick up a gap-inserted space
/// before `]` (e.g. `[kg/m3 ]`), which is cosmetic padding.
fn remove_spaces_before_closing_brackets(text: &mut String) {
let mut result = String::with_capacity(text.len());
for ch in text.chars() {
if ch == ']' && result.ends_with(' ') {
result.pop();
}
result.push(ch);
}
*text = result;
}
/// Collapse dot leaders (runs of 4+ dots) into " ... "
/// Common in tables of contents: "Introduction...............................1" -> "Introduction ... 1"
fn collapse_dot_leaders(text: &str) -> String {
@@ -342,6 +357,18 @@ mod tests {
assert!(result.contains("Chapter 2 ... 20"));
}
// --- remove_spaces_before_closing_brackets ---
#[test]
fn test_remove_spaces_before_closing_brackets() {
let mut input = "Density [kg/m3 ] and [linked text ](https://example.com)".to_string();
remove_spaces_before_closing_brackets(&mut input);
assert_eq!(
input,
"Density [kg/m3] and [linked text](https://example.com)"
);
}
// --- fix_hyphenation ---
#[test]
+1
View File
@@ -542,6 +542,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid,
}
+52
View File
@@ -30,6 +30,9 @@ pub struct PyPdfResult {
/// 1-indexed page numbers that need OCR.
#[pyo3(get)]
pub pages_needing_ocr: Vec<u32>,
/// Machine-readable OCR reasons by 1-indexed page.
#[pyo3(get)]
pub ocr_reasons_by_page: Vec<PyPageOcrReasons>,
/// Title from PDF metadata.
#[pyo3(get)]
pub title: Option<String>,
@@ -60,6 +63,28 @@ impl PyPdfResult {
}
}
/// OCR reasons for a single 1-indexed page.
#[pyclass(name = "PageOcrReasons")]
#[derive(Clone)]
pub struct PyPageOcrReasons {
/// 1-indexed page number.
#[pyo3(get)]
pub page: u32,
/// Machine-readable OCR reason identifiers.
#[pyo3(get)]
pub reasons: Vec<String>,
}
#[pymethods]
impl PyPageOcrReasons {
fn __repr__(&self) -> String {
format!(
"PageOcrReasons(page={}, reasons={:?})",
self.page, self.reasons
)
}
}
// ---------------------------------------------------------------------------
// Classification wrapper (lightweight)
// ---------------------------------------------------------------------------
@@ -106,6 +131,9 @@ pub struct PyRegionText {
/// True when the text should not be trusted (empty, GID fonts, garbage, encoding issues).
#[pyo3(get)]
pub needs_ocr: bool,
/// Machine-readable OCR reason when the cause is known.
#[pyo3(get)]
pub ocr_reason: Option<String>,
}
#[pymethods]
@@ -160,6 +188,9 @@ pub struct PyPageMarkdown {
/// encoding issues, garbage text, or empty extraction).
#[pyo3(get)]
pub needs_ocr: bool,
/// Machine-readable OCR reason when the cause is known.
#[pyo3(get)]
pub ocr_reason: Option<String>,
}
#[pymethods]
@@ -190,6 +221,9 @@ pub struct PyPagesExtractionResult {
/// 1-indexed pages that need OCR (scanned/image-based or unreliable text).
#[pyo3(get)]
pub pages_needing_ocr: Vec<u32>,
/// Machine-readable OCR reasons by 1-indexed page.
#[pyo3(get)]
pub ocr_reasons_by_page: Vec<PyPageOcrReasons>,
/// True if any page has tables or columns.
#[pyo3(get)]
pub is_complex: bool,
@@ -232,6 +266,8 @@ pub struct PyTextItem {
#[pyo3(get)]
pub is_italic: bool,
#[pyo3(get)]
pub is_underline: bool,
#[pyo3(get)]
pub item_type: String,
}
@@ -268,6 +304,7 @@ fn to_py_result(r: crate::PdfProcessResult) -> PyPdfResult {
page_count: r.page_count,
processing_time_ms: r.processing_time_ms,
pages_needing_ocr: r.pages_needing_ocr,
ocr_reasons_by_page: to_py_page_ocr_reasons(r.ocr_reasons_by_page),
title: r.title,
confidence: r.confidence,
is_complex_layout: r.layout.is_complex,
@@ -277,6 +314,16 @@ fn to_py_result(r: crate::PdfProcessResult) -> PyPdfResult {
}
}
fn to_py_page_ocr_reasons(reasons: Vec<crate::PageOcrReasons>) -> Vec<PyPageOcrReasons> {
reasons
.into_iter()
.map(|reason| PyPageOcrReasons {
page: reason.page,
reasons: reason.reasons,
})
.collect()
}
fn to_py_err(e: crate::PdfError) -> PyErr {
PyValueError::new_err(e.to_string())
}
@@ -304,6 +351,7 @@ fn convert_text_items(items: Vec<crate::TextItem>) -> Vec<PyTextItem> {
page: item.page,
is_bold: item.is_bold,
is_italic: item.is_italic,
is_underline: item.is_underline,
item_type: item_type_str(&item.item_type),
})
.collect()
@@ -350,11 +398,13 @@ fn to_py_pages_result(r: crate::PagesExtractionResult) -> PyPagesExtractionResul
page: p.page,
markdown: p.markdown,
needs_ocr: p.needs_ocr,
ocr_reason: p.ocr_reason,
})
.collect(),
pages_with_tables: r.pages_with_tables,
pages_with_columns: r.pages_with_columns,
pages_needing_ocr: r.pages_needing_ocr,
ocr_reasons_by_page: to_py_page_ocr_reasons(r.ocr_reasons_by_page),
is_complex: r.is_complex,
}
}
@@ -370,6 +420,7 @@ fn convert_region_results(results: Vec<crate::PageRegionResult>) -> Vec<PyPageRe
.map(|r| PyRegionText {
text: r.text,
needs_ocr: r.needs_ocr,
ocr_reason: r.ocr_reason,
})
.collect(),
})
@@ -563,6 +614,7 @@ fn extract_pages_markdown_bytes(
#[pymodule]
fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
m.add_class::<PyPdfResult>()?;
m.add_class::<PyPageOcrReasons>()?;
m.add_class::<PyPdfClassification>()?;
m.add_class::<PyTextItem>()?;
m.add_class::<PyRegionText>()?;
+1
View File
@@ -104,6 +104,7 @@ pub(crate) fn merge_adjacent_items(items: &[TextItem]) -> (Vec<TextItem>, Vec<Ve
page: first_item.page,
is_bold: first_item.is_bold,
is_italic: first_item.is_italic,
is_underline: first_item.is_underline,
item_type: first_item.item_type.clone(),
mcid: first_item.mcid,
});
+1
View File
@@ -393,6 +393,7 @@ mod tests {
page,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
}
+3
View File
@@ -2314,6 +2314,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
}
@@ -3272,6 +3273,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: crate::types::ItemType::Text,
mcid: None,
});
@@ -3581,6 +3583,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: crate::types::ItemType::Text,
mcid: None,
});
+1
View File
@@ -586,6 +586,7 @@ mod tests {
page,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid,
}
+1
View File
@@ -108,6 +108,7 @@ pub(crate) fn try_split_financial_item(item: &TextItem) -> Option<Vec<TextItem>>
page: item.page,
is_bold: item.is_bold,
is_italic: item.is_italic,
is_underline: item.is_underline,
item_type: item.item_type.clone(),
mcid: item.mcid,
});
+65 -4
View File
@@ -208,6 +208,24 @@ fn looks_like_compact_entry_label(cell: &str) -> bool {
(1..=6).contains(&words)
}
fn looks_like_plain_section_label(cell: &str) -> bool {
let trimmed = cell.trim();
if trimmed.len() < 4 || trimmed.len() > 40 {
return false;
}
if trimmed.ends_with(['.', ',', ';', ':']) || trimmed.contains(|ch: char| ch.is_ascii_digit()) {
return false;
}
if trimmed.len() <= 4 && trimmed.chars().all(|ch| !ch.is_lowercase()) {
return false;
}
trimmed
.chars()
.all(|ch| ch.is_alphabetic() || ch.is_whitespace() || matches!(ch, '&' | '/' | '-'))
&& starts_with_uppercase_alpha(trimmed)
&& (1..=4).contains(&alpha_word_count(trimmed))
}
fn ends_like_incomplete_phrase(cell: &str) -> bool {
let lower = cell.trim_end().to_ascii_lowercase();
lower.ends_with(" and")
@@ -305,6 +323,10 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
.and_then(|r| r.first())
.map(|c| c.trim())
.unwrap_or("");
let header_filled = cleaned
.first()
.map(|r| r.iter().filter(|c| !c.trim().is_empty()).count())
.unwrap_or(num_cols);
let looks_like_spanning_first_column_row = first_cell.is_empty()
&& row.len() >= 4
&& non_first_cells.len() == row.len().saturating_sub(1)
@@ -328,6 +350,10 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
&& non_first_cells
.iter()
.any(|cell| looks_like_compact_entry_label(cell));
let looks_like_section_label_row = !first_cell.is_empty()
&& filled_cells == 1
&& header_filled >= 3
&& looks_like_plain_section_label(first_cell);
// Classic continuation: first cell empty, content in other cells
let is_classic_continuation = first_cell.is_empty()
&& !non_first_cells.is_empty()
@@ -344,10 +370,6 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
.last()
.map(|r| r.iter().filter(|c| !c.trim().is_empty()).count())
.unwrap_or(0);
let header_filled = cleaned
.first()
.map(|r| r.iter().filter(|c| !c.trim().is_empty()).count())
.unwrap_or(num_cols);
// Merge when the row has significantly fewer filled cells than header.
// For wide tables (5+ cols), require ≤50% of header cells.
// For narrow tables (2-4 cols), require fewer than header cells.
@@ -369,6 +391,7 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
&& !looks_like_spanning_first_column_row
&& !looks_like_hierarchical_subrow
&& !looks_like_new_first_column_entry
&& !looks_like_section_label_row
&& !is_short_subheader;
let is_continuation = is_classic_continuation || is_wrapped_continuation;
@@ -514,6 +537,44 @@ mod tests {
assert!(cleaned[1][1].contains("continued text here"));
}
#[test]
fn test_clean_table_cells_first_column_section_label_not_merged() {
let cells = vec![
vec![
"Properties".into(),
"Conditions".into(),
"Method".into(),
"Typical values".into(),
"Units".into(),
],
vec![
"Melt Flow Rate".into(),
"230 C/2.16 kg".into(),
"ASTM D1238".into(),
"3.0".into(),
"g/10 min".into(),
],
vec![
"Mechanical".into(),
"".into(),
"".into(),
"".into(),
"".into(),
],
vec![
"Tensile Stress at Yield".into(),
"50 mm/min".into(),
"ASTM D638".into(),
"31".into(),
"MPa".into(),
],
];
let (cleaned, _) = clean_table_cells(&cells);
assert_eq!(cleaned.len(), 4);
assert_eq!(cleaned[2][0], "Mechanical");
}
#[test]
fn test_clean_table_cells_short_subheader_not_merged() {
let cells = vec![
+3
View File
@@ -514,6 +514,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
}
@@ -866,6 +867,7 @@ mod tests {
font: String::new(),
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
page: 1,
@@ -902,6 +904,7 @@ mod tests {
font: String::new(),
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
page: 1,
+1237
View File
File diff suppressed because it is too large Load Diff
+3
View File
@@ -883,6 +883,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
}
@@ -1002,6 +1003,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
});
@@ -1078,6 +1080,7 @@ mod tests {
page: 1,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
}
+168 -23
View File
@@ -329,7 +329,7 @@ impl ToUnicodeCMap {
if let (Some(start), Some(end), Some(base)) = (
parse_hex_u16(&start_hex),
parse_hex_u16(&end_hex),
parse_hex_u32(&base_hex),
hex_to_unicode_scalar(&base_hex),
) {
self.ranges.push((start, end, base));
}
@@ -575,32 +575,86 @@ fn parse_hex_u16(hex: &str) -> Option<u16> {
u16::from_str_radix(hex.trim(), 16).ok()
}
/// Parse a hex string to u32
fn parse_hex_u32(hex: &str) -> Option<u32> {
u32::from_str_radix(hex.trim(), 16).ok()
}
/// Convert a hex string to a Unicode string
/// Handles both 2-byte (BMP) and 4-byte (supplementary) codepoints
/// Convert a ToUnicode destination hex string to Unicode.
///
/// PDF ToUnicode destinations are UTF-16BE strings. Supplementary-plane
/// characters are encoded as surrogate pairs, so treating each 4-hex chunk as
/// a scalar drops emoji like D83CDF1F.
fn hex_to_unicode_string(hex: &str) -> Option<String> {
let hex = hex.trim();
let mut result = String::new();
// Process 4 hex digits at a time
let mut i = 0;
while i + 4 <= hex.len() {
if let Ok(cp) = u32::from_str_radix(&hex[i..i + 4], 16) {
if let Some(c) = char::from_u32(cp) {
result.push(c);
}
}
i += 4;
let hex: String = hex.chars().filter(|ch| !ch.is_ascii_whitespace()).collect();
if hex.is_empty() || !hex.len().is_multiple_of(2) {
return None;
}
if result.is_empty() {
None
let bytes: Option<Vec<u8>> = (0..hex.len())
.step_by(2)
.map(|i| u8::from_str_radix(&hex[i..i + 2], 16).ok())
.collect();
let bytes = bytes?;
if bytes.len().is_multiple_of(2) {
let units: Vec<u16> = bytes
.chunks_exact(2)
.map(|chunk| u16::from_be_bytes([chunk[0], chunk[1]]))
.collect();
if let Ok(result) = String::from_utf16(&units) {
if !result.is_empty() {
return Some(normalize_tounicode_destination(result));
}
}
}
// Be permissive for non-standard one-byte destinations.
if bytes.len() == 1 {
let ch = bytes[0] as char;
if !ch.is_control() || ch == '\t' || ch == '\n' {
return Some(ch.to_string());
}
}
None
}
fn normalize_tounicode_destination(text: String) -> String {
let is_multi_char = text.chars().nth(1).is_some();
// Some malformed producer CMaps put a list of alternative whitespace or
// hyphen codepoints into one destination. Keep ordinary multi-character
// mappings intact unless that malformed signature is present.
if is_multi_char
&& text.chars().all(char::is_whitespace)
&& text.chars().any(|ch| matches!(ch, '\t' | '\n' | '\r'))
{
return if text.contains('\t') {
"\t".to_string()
} else {
" ".to_string()
};
}
if is_multi_char
&& text.contains('\u{00ad}')
&& text.chars().all(|ch| {
matches!(
ch,
'-' | '\u{00ad}' | '\u{2010}' | '\u{2011}' | '\u{2012}' | '\u{2013}' | '\u{2212}'
)
})
{
return "-".to_string();
}
text
}
fn hex_to_unicode_scalar(hex: &str) -> Option<u32> {
let text = hex_to_unicode_string(hex)?;
let mut chars = text.chars();
let ch = chars.next()?;
if chars.next().is_none() {
Some(ch as u32)
} else {
Some(result)
None
}
}
@@ -2607,6 +2661,97 @@ endbfrange
assert_eq!(cmap.lookup(0x0005), Some("C".to_string()));
}
#[test]
fn test_parse_bfchar_surrogate_pair_emoji() {
let cmap_content = r#"
1 begincodespacerange
<00> <FF>
endcodespacerange
2 beginbfchar
<16> <D83CDF1F>
<9D> <D83CDFAD>
endbfchar
"#;
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
assert_eq!(cmap.code_byte_length, 1);
assert_eq!(cmap.lookup(0x16), Some("🌟".to_string()));
assert_eq!(cmap.lookup(0x9D), Some("🎭".to_string()));
}
#[test]
fn test_parse_bfrange_surrogate_pair_base() {
let cmap_content = r#"
1 begincodespacerange
<00> <FF>
endcodespacerange
1 beginbfrange
<C8> <C9> <D83CDFD8>
endbfrange
"#;
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
assert_eq!(cmap.code_byte_length, 1);
assert_eq!(cmap.lookup(0xC8), Some("🏘".to_string()));
assert_eq!(cmap.lookup(0xC9), Some("🏙".to_string()));
}
#[test]
fn test_parse_bfrange_preserves_single_hyphen_like_base() {
let cmap_content = r#"
1 begincodespacerange
<00> <FF>
endcodespacerange
1 beginbfrange
<21> <22> <2013>
endbfrange
"#;
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
assert_eq!(cmap.lookup(0x21), Some("".to_string()));
assert_eq!(cmap.lookup(0x22), Some("".to_string()));
}
#[test]
fn test_parse_spaced_destination_hex_without_control_noise() {
let cmap_content = r#"
1 begincodespacerange
<00> <FF>
endcodespacerange
3 beginbfchar
<21> < 0009 000d 0020 00a0 >
<22> < 002d 00ad 2010 >
<23> <00a0>
endbfchar
"#;
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
assert_eq!(cmap.lookup(0x21), Some("\t".to_string()));
assert_eq!(cmap.lookup(0x22), Some("-".to_string()));
assert_eq!(cmap.lookup(0x23), Some("\u{00a0}".to_string()));
}
#[test]
fn test_parse_preserves_valid_multi_character_destinations() {
let cmap_content = r#"
1 begincodespacerange
<00> <FF>
endcodespacerange
4 beginbfchar
<21> <002d002d>
<22> <20132013>
<23> <002000a0>
<24> <00660069>
endbfchar
"#;
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
assert_eq!(cmap.lookup(0x21), Some("--".to_string()));
assert_eq!(cmap.lookup(0x22), Some("––".to_string()));
assert_eq!(cmap.lookup(0x23), Some(" \u{00a0}".to_string()));
assert_eq!(cmap.lookup(0x24), Some("fi".to_string()));
}
#[test]
fn test_remap_to_sequential() {
// Simulate a broken CMap where GIDs are from pre-subsetting:
+4
View File
@@ -116,6 +116,10 @@ pub struct TextItem {
pub is_bold: bool,
/// Whether the font is italic
pub is_italic: bool,
/// Whether the text is underlined (drawn rule/thin rect under the
/// baseline — PDFs have no underline font flag, so this is detected
/// geometrically after extraction; see `extractor::underline`).
pub is_underline: bool,
/// Type of item (text, image, link)
pub item_type: ItemType,
/// Marked Content ID from the content stream's BDC/BMC operator.
+3
View File
@@ -104,6 +104,7 @@ fn make_text_item(text: &str, x: f32, y: f32, font_size: f32, page: u32) -> Text
page,
is_bold: false,
is_italic: false,
is_underline: false,
item_type: ItemType::Text,
mcid: None,
}
@@ -129,6 +130,7 @@ fn make_text_item_with_font(
page,
is_bold: is_bold_font(font),
is_italic: is_italic_font(font),
is_underline: false,
item_type: ItemType::Text,
mcid: None,
}
@@ -1107,6 +1109,7 @@ fn test_pages_needing_ocr_field_accessible() {
page_count: 1,
processing_time_ms: 0,
pages_needing_ocr: vec![1, 3],
ocr_reasons_by_page: Vec::new(),
title: None,
confidence: 1.0,
layout: pdf_inspector::LayoutComplexity::default(),