Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
bb696a45d8 | ||
|
|
4f364789ba |
@@ -1,87 +0,0 @@
|
||||
name: Publish Rust crate
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths: ['Cargo.toml']
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
env:
|
||||
CARGO_TERM_COLOR: always
|
||||
|
||||
jobs:
|
||||
check-version:
|
||||
name: Check version change
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
changed: ${{ steps.check.outputs.changed }}
|
||||
published: ${{ steps.check.outputs.published }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
- name: Check if version changed
|
||||
id: check
|
||||
run: |
|
||||
NEW_VERSION=$(python3 -c 'import pathlib, tomllib; print(tomllib.loads(pathlib.Path("Cargo.toml").read_text())["package"]["version"])')
|
||||
OLD_VERSION=$(git show HEAD~1:Cargo.toml | python3 -c 'import sys, tomllib; print(tomllib.loads(sys.stdin.read())["package"]["version"])')
|
||||
echo "old=$OLD_VERSION new=$NEW_VERSION"
|
||||
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
|
||||
|
||||
if [ "$NEW_VERSION" = "$OLD_VERSION" ]; then
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
|
||||
HTTP_STATUS=$(curl --silent --show-error --output /tmp/crate-version.json --write-out "%{http_code}" \
|
||||
-H "User-Agent: firecrawl/pdf-inspector publish workflow (https://github.com/firecrawl/pdf-inspector)" \
|
||||
"https://crates.io/api/v1/crates/pdf-inspector/$NEW_VERSION")
|
||||
|
||||
case "$HTTP_STATUS" in
|
||||
200)
|
||||
echo "published=true" >> "$GITHUB_OUTPUT"
|
||||
echo "pdf-inspector v$NEW_VERSION is already published"
|
||||
;;
|
||||
404)
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
;;
|
||||
*)
|
||||
cat /tmp/crate-version.json
|
||||
echo "Unexpected crates.io response: $HTTP_STATUS" >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
|
||||
publish:
|
||||
name: Publish to crates.io
|
||||
needs: check-version
|
||||
if: needs.check-version.outputs.changed == 'true' && needs.check-version.outputs.published == 'false'
|
||||
runs-on: ubuntu-latest
|
||||
environment: crates-io
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
|
||||
- name: Verify package
|
||||
run: cargo publish --dry-run
|
||||
|
||||
- name: Authenticate with crates.io
|
||||
id: auth
|
||||
uses: rust-lang/crates-io-auth-action@v1
|
||||
|
||||
- name: Publish crate
|
||||
run: cargo publish
|
||||
env:
|
||||
CARGO_REGISTRY_TOKEN: ${{ steps.auth.outputs.token }}
|
||||
+2
-2
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "pdf-inspector"
|
||||
version = "0.1.4"
|
||||
version = "0.1.0"
|
||||
edition = "2021"
|
||||
autobins = false
|
||||
authors = ["Firecrawl Team"]
|
||||
@@ -17,7 +17,7 @@ crate-type = ["lib", "cdylib"]
|
||||
pyo3 = { version = "0.25", features = ["extension-module"], optional = true }
|
||||
|
||||
# PDF parsing
|
||||
lopdf = { version = "0.41.0", features = ["rayon"] }
|
||||
lopdf = { git = "https://github.com/J-F-Liu/lopdf", rev = "7a05512d831415b1f2b1ce522391d6beab8a1284", features = ["rayon"] }
|
||||
|
||||
# Error handling
|
||||
thiserror = "2.0"
|
||||
|
||||
@@ -1,21 +0,0 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2026 Firecrawl
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
@@ -1,9 +1,5 @@
|
||||
# pdf-inspector
|
||||
|
||||
[](https://crates.io/crates/pdf-inspector)
|
||||
[](https://www.npmjs.com/package/@firecrawl/pdf-inspector)
|
||||
[](LICENSE)
|
||||
|
||||
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Includes bindings for [Python](docs/python.md) and [Node.js](napi/README.md).
|
||||
|
||||
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
|
||||
@@ -75,17 +71,9 @@ console.log(result.markdown); // Markdown string or null
|
||||
|
||||
### Rust
|
||||
|
||||
Install from [crates.io](https://crates.io/crates/pdf-inspector):
|
||||
|
||||
```bash
|
||||
cargo add pdf-inspector
|
||||
```
|
||||
|
||||
Or add it manually:
|
||||
|
||||
```toml
|
||||
[dependencies]
|
||||
pdf-inspector = "0.1"
|
||||
pdf-inspector = { git = "https://github.com/firecrawl/pdf-inspector" }
|
||||
```
|
||||
|
||||
```rust
|
||||
@@ -103,37 +91,29 @@ if let Some(markdown) = &result.markdown {
|
||||
### CLI
|
||||
|
||||
```bash
|
||||
# Install the CLI tools
|
||||
cargo install pdf-inspector
|
||||
|
||||
# Convert PDF to Markdown
|
||||
pdf2md document.pdf
|
||||
cargo run --bin pdf2md -- document.pdf
|
||||
|
||||
# JSON output (for piping)
|
||||
pdf2md document.pdf --json
|
||||
|
||||
# Positioned TextItem JSON, including is_underline metadata
|
||||
pdf2md document.pdf --items-json
|
||||
cargo run --bin pdf2md -- document.pdf --json
|
||||
|
||||
# Raw markdown only (no headers)
|
||||
pdf2md document.pdf --raw
|
||||
cargo run --bin pdf2md -- document.pdf --raw
|
||||
|
||||
# Insert page break markers (<!-- Page N -->)
|
||||
pdf2md document.pdf --pages
|
||||
cargo run --bin pdf2md -- document.pdf --pages
|
||||
|
||||
# Process only specific pages
|
||||
pdf2md document.pdf --select-pages 1,3,5-10
|
||||
cargo run --bin pdf2md -- document.pdf --select-pages 1,3,5-10
|
||||
|
||||
# Detection only (no extraction)
|
||||
detect-pdf document.pdf
|
||||
detect-pdf document.pdf --json
|
||||
cargo run --bin detect-pdf -- document.pdf
|
||||
cargo run --bin detect-pdf -- document.pdf --json
|
||||
|
||||
# Detection + layout analysis (tables, columns)
|
||||
detect-pdf document.pdf --analyze --json
|
||||
cargo run --bin detect-pdf -- document.pdf --analyze --json
|
||||
```
|
||||
|
||||
From a source checkout, use `cargo run --bin pdf2md -- document.pdf` or `cargo run --bin detect-pdf -- document.pdf` instead.
|
||||
|
||||
## Architecture
|
||||
|
||||
```
|
||||
@@ -243,4 +223,4 @@ See [docs/debugging.md](docs/debugging.md) for `RUST_LOG` environment variable u
|
||||
|
||||
## License
|
||||
|
||||
[MIT](LICENSE)
|
||||
MIT
|
||||
|
||||
@@ -1,21 +0,0 @@
|
||||
# Publishing
|
||||
|
||||
The Rust crate is published to [crates.io](https://crates.io/crates/pdf-inspector) with trusted publishing from GitHub Actions. The first release was published manually; future releases publish from `.github/workflows/publish-crate.yml` when a `Cargo.toml` version change lands on `main`.
|
||||
|
||||
## crates.io Trusted Publisher
|
||||
|
||||
Configure the trusted publisher for the `pdf-inspector` crate with:
|
||||
|
||||
- Repository: `firecrawl/pdf-inspector`
|
||||
- Workflow: `publish-crate.yml`
|
||||
- Environment: `crates-io`
|
||||
|
||||
The workflow uses `rust-lang/crates-io-auth-action@v1` to exchange GitHub's OIDC token for a short-lived crates.io token, then passes it to `cargo publish`.
|
||||
|
||||
## Release Steps
|
||||
|
||||
1. Update `version` in `Cargo.toml`.
|
||||
2. Merge the version bump to `main`.
|
||||
3. The publish workflow compares the new `Cargo.toml` version with `HEAD~1`, runs `cargo publish --dry-run`, then publishes if that version is not already on crates.io.
|
||||
|
||||
If `Cargo.toml` changes without a package version bump, the workflow exits without publishing.
|
||||
Generated
+4
-5
@@ -672,9 +672,8 @@ checksum = "5e5032e24019045c762d3c0f28f5b6b8bbf38563a65908389bf7978758920897"
|
||||
|
||||
[[package]]
|
||||
name = "lopdf"
|
||||
version = "0.41.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "67513274c50a2b51e5f75d9e682fcf4ab064a8a9c9ae2c3c59309084882bb24d"
|
||||
version = "0.40.0"
|
||||
source = "git+https://github.com/J-F-Liu/lopdf?rev=7a05512d831415b1f2b1ce522391d6beab8a1284#7a05512d831415b1f2b1ce522391d6beab8a1284"
|
||||
dependencies = [
|
||||
"aes",
|
||||
"bitflags",
|
||||
@@ -830,7 +829,7 @@ checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe"
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector"
|
||||
version = "0.1.4"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"env_logger",
|
||||
"log",
|
||||
@@ -845,7 +844,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector-napi"
|
||||
version = "0.2.2"
|
||||
version = "0.2.0"
|
||||
dependencies = [
|
||||
"napi",
|
||||
"napi-build",
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "pdf-inspector-napi"
|
||||
version = "0.2.2"
|
||||
version = "0.2.0"
|
||||
edition = "2021"
|
||||
|
||||
[lib]
|
||||
|
||||
+1
-2
@@ -37,7 +37,7 @@ console.log(result.confidence) // 0.875
|
||||
|
||||
Extract text within bounding-box regions from a PDF. Designed for hybrid OCR pipelines where a layout model detects regions in rendered page images, and this function extracts text from the PDF structure for text-based pages — skipping GPU OCR.
|
||||
|
||||
Each region result includes a `needsOcr` flag that signals unreliable extraction (empty text, GID-encoded fonts, garbage text, encoding issues). When the cause is a suspected garbled text layer, `ocrReason` is set to `"suspected_garbled_text"`.
|
||||
Each region result includes a `needsOcr` flag that signals unreliable extraction (empty text, GID-encoded fonts, garbage text, encoding issues).
|
||||
|
||||
```typescript
|
||||
import { extractTextInRegions } from '@firecrawl/pdf-inspector'
|
||||
@@ -84,7 +84,6 @@ interface PageRegionTexts {
|
||||
interface RegionText {
|
||||
text: string
|
||||
needsOcr: boolean // true when text is unreliable
|
||||
ocrReason?: string // "suspected_garbled_text" when known
|
||||
}
|
||||
```
|
||||
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@firecrawl/pdf-inspector",
|
||||
"version": "1.10.0",
|
||||
"version": "1.9.4",
|
||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||
"main": "index.js",
|
||||
"types": "index.d.ts",
|
||||
|
||||
@@ -40,8 +40,6 @@ pub struct PdfResult {
|
||||
pub processing_time_ms: u32,
|
||||
/// 1-indexed page numbers that need OCR.
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
/// Machine-readable OCR reasons by 1-indexed page.
|
||||
pub ocr_reasons_by_page: Vec<PageOcrReasons>,
|
||||
pub title: Option<String>,
|
||||
pub confidence: f64,
|
||||
pub is_complex_layout: bool,
|
||||
@@ -50,13 +48,6 @@ pub struct PdfResult {
|
||||
pub has_encoding_issues: bool,
|
||||
}
|
||||
|
||||
/// OCR reasons for a single 1-indexed page.
|
||||
#[napi(object)]
|
||||
pub struct PageOcrReasons {
|
||||
pub page: u32,
|
||||
pub reasons: Vec<String>,
|
||||
}
|
||||
|
||||
/// Lightweight PDF classification result.
|
||||
#[napi(object)]
|
||||
pub struct PdfClassification {
|
||||
@@ -80,12 +71,6 @@ pub struct TextItem {
|
||||
pub page: u32,
|
||||
pub is_bold: bool,
|
||||
pub is_italic: bool,
|
||||
/// Underline detected geometrically (drawn rule/thin rect under the
|
||||
/// baseline) — PDFs carry no underline font flag.
|
||||
pub is_underline: bool,
|
||||
/// Strikeout detected geometrically (rule crossing the glyphs at mid
|
||||
/// x-height).
|
||||
pub is_strikeout: bool,
|
||||
pub item_type: ItemType,
|
||||
/// URL for link items, `None` for other types.
|
||||
pub link_url: Option<String>,
|
||||
@@ -105,8 +90,6 @@ pub struct RegionText {
|
||||
pub text: String,
|
||||
/// `true` when the text should not be trusted (empty, GID fonts, garbage, encoding issues).
|
||||
pub needs_ocr: bool,
|
||||
/// Machine-readable OCR reason when the cause is known.
|
||||
pub ocr_reason: Option<String>,
|
||||
}
|
||||
|
||||
/// Extracted text for one page's regions.
|
||||
@@ -143,7 +126,6 @@ fn to_napi_result(r: pdf_inspector::PdfProcessResult) -> PdfResult {
|
||||
page_count: r.page_count,
|
||||
processing_time_ms: r.processing_time_ms as u32,
|
||||
pages_needing_ocr: r.pages_needing_ocr,
|
||||
ocr_reasons_by_page: to_napi_page_ocr_reasons(r.ocr_reasons_by_page),
|
||||
title: r.title,
|
||||
confidence: r.confidence as f64,
|
||||
is_complex_layout: r.layout.is_complex,
|
||||
@@ -153,18 +135,6 @@ fn to_napi_result(r: pdf_inspector::PdfProcessResult) -> PdfResult {
|
||||
}
|
||||
}
|
||||
|
||||
fn to_napi_page_ocr_reasons(
|
||||
reasons: Vec<pdf_inspector::PageOcrReasons>,
|
||||
) -> Vec<PageOcrReasons> {
|
||||
reasons
|
||||
.into_iter()
|
||||
.map(|reason| PageOcrReasons {
|
||||
page: reason.page,
|
||||
reasons: reason.reasons,
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn convert_item_type(t: &pdf_inspector::types::ItemType) -> (ItemType, Option<String>) {
|
||||
match t {
|
||||
pdf_inspector::types::ItemType::Text => (ItemType::Text, None),
|
||||
@@ -296,8 +266,6 @@ pub fn extract_text_with_positions(
|
||||
page: item.page,
|
||||
is_bold: item.is_bold,
|
||||
is_italic: item.is_italic,
|
||||
is_underline: item.is_underline,
|
||||
is_strikeout: item.is_strikeout,
|
||||
item_type,
|
||||
link_url,
|
||||
}
|
||||
@@ -595,8 +563,6 @@ pub struct PageMarkdownResult {
|
||||
pub markdown: String,
|
||||
/// `true` when text on this page is unreliable.
|
||||
pub needs_ocr: bool,
|
||||
/// Machine-readable OCR reason when the cause is known.
|
||||
pub ocr_reason: Option<String>,
|
||||
}
|
||||
|
||||
/// Combined per-page markdown extraction and layout classification result.
|
||||
@@ -610,8 +576,6 @@ pub struct PagesExtractionResult {
|
||||
pub pages_with_columns: Vec<u32>,
|
||||
/// 1-indexed pages that need OCR (scanned/image-based).
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
/// Machine-readable OCR reasons by 1-indexed page.
|
||||
pub ocr_reasons_by_page: Vec<PageOcrReasons>,
|
||||
/// True if any page has tables or columns.
|
||||
pub is_complex: bool,
|
||||
}
|
||||
@@ -643,13 +607,11 @@ pub fn extract_pages_markdown(
|
||||
page: r.page,
|
||||
markdown: r.markdown,
|
||||
needs_ocr: r.needs_ocr,
|
||||
ocr_reason: r.ocr_reason,
|
||||
})
|
||||
.collect(),
|
||||
pages_with_tables: result.pages_with_tables,
|
||||
pages_with_columns: result.pages_with_columns,
|
||||
pages_needing_ocr: result.pages_needing_ocr,
|
||||
ocr_reasons_by_page: to_napi_page_ocr_reasons(result.ocr_reasons_by_page),
|
||||
is_complex: result.is_complex,
|
||||
})
|
||||
})
|
||||
@@ -686,7 +648,6 @@ fn to_page_region_texts(results: Vec<pdf_inspector::PageRegionResult>) -> Vec<Pa
|
||||
.map(|r| RegionText {
|
||||
text: r.text,
|
||||
needs_ocr: r.needs_ocr,
|
||||
ocr_reason: r.ocr_reason,
|
||||
})
|
||||
.collect(),
|
||||
})
|
||||
|
||||
@@ -38,8 +38,6 @@ class TextItem:
|
||||
page: int
|
||||
is_bold: bool
|
||||
is_italic: bool
|
||||
is_underline: bool
|
||||
is_strikeout: bool
|
||||
item_type: str
|
||||
|
||||
class RegionText:
|
||||
|
||||
+1
-3
@@ -4,9 +4,7 @@ build-backend = "maturin"
|
||||
|
||||
[project]
|
||||
name = "pdf-inspector"
|
||||
# Version is sourced from Cargo.toml [package] version by maturin so the Python
|
||||
# artifact always tracks the crate release instead of drifting on its own.
|
||||
dynamic = ["version"]
|
||||
version = "0.1.0"
|
||||
description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection"
|
||||
license = { text = "MIT" }
|
||||
requires-python = ">=3.8"
|
||||
|
||||
+3
-129
@@ -1,10 +1,6 @@
|
||||
//! CLI tool for PDF to Markdown conversion
|
||||
|
||||
use pdf_inspector::extractor::ItemType;
|
||||
use pdf_inspector::{
|
||||
extract_text_with_positions_pages, process_pdf_with_options, LayoutComplexity, PdfOptions,
|
||||
PdfType, ProcessMode, TextItem,
|
||||
};
|
||||
use pdf_inspector::{process_pdf_with_options, LayoutComplexity, PdfOptions, PdfType, ProcessMode};
|
||||
use std::collections::HashSet;
|
||||
use std::env;
|
||||
use std::fmt::Write;
|
||||
@@ -35,110 +31,6 @@ fn json_escape(s: &str) -> String {
|
||||
out
|
||||
}
|
||||
|
||||
fn format_ocr_reasons_by_page(reasons: &[pdf_inspector::PageOcrReasons]) -> String {
|
||||
reasons
|
||||
.iter()
|
||||
.map(|entry| {
|
||||
let reasons_json = entry
|
||||
.reasons
|
||||
.iter()
|
||||
.map(|reason| format!(r#""{}""#, json_escape(reason)))
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
format!(r#"{{"page":{},"reasons":[{}]}}"#, entry.page, reasons_json)
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join(",")
|
||||
}
|
||||
|
||||
fn item_type_label(item_type: &ItemType) -> &'static str {
|
||||
match item_type {
|
||||
ItemType::Text => "text",
|
||||
ItemType::Image => "image",
|
||||
ItemType::Link(_) => "link",
|
||||
ItemType::FormField => "form_field",
|
||||
}
|
||||
}
|
||||
|
||||
fn format_items_json(items: &[TextItem]) -> String {
|
||||
let underlined_count = items.iter().filter(|item| item.is_underline).count();
|
||||
let items_json = items
|
||||
.iter()
|
||||
.map(|item| {
|
||||
let mcid = item
|
||||
.mcid
|
||||
.map(|value| value.to_string())
|
||||
.unwrap_or_else(|| "null".to_string());
|
||||
let link_url = match &item.item_type {
|
||||
ItemType::Link(url) => format!(r#","url":"{}""#, json_escape(url)),
|
||||
_ => String::new(),
|
||||
};
|
||||
format!(
|
||||
r#"{{"text":"{}","page":{},"x":{:.2},"y":{:.2},"width":{:.2},"height":{:.2},"font":"{}","font_size":{:.2},"is_bold":{},"is_italic":{},"is_underline":{},"is_strikeout":{},"item_type":"{}","mcid":{}{}}}"#,
|
||||
json_escape(&item.text),
|
||||
item.page,
|
||||
item.x,
|
||||
item.y,
|
||||
item.width,
|
||||
item.height,
|
||||
json_escape(&item.font),
|
||||
item.font_size,
|
||||
item.is_bold,
|
||||
item.is_italic,
|
||||
item.is_underline,
|
||||
item.is_strikeout,
|
||||
item_type_label(&item.item_type),
|
||||
mcid,
|
||||
link_url,
|
||||
)
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
|
||||
format!(
|
||||
r#"{{"total_items":{},"underlined_count":{},"items":[{}]}}"#,
|
||||
items.len(),
|
||||
underlined_count,
|
||||
items_json
|
||||
)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::format_items_json;
|
||||
use pdf_inspector::extractor::ItemType;
|
||||
use pdf_inspector::TextItem;
|
||||
|
||||
#[test]
|
||||
fn items_json_includes_position_and_underline_metadata() {
|
||||
let items = vec![TextItem {
|
||||
text: "A \"quoted\" item".to_string(),
|
||||
x: 12.345,
|
||||
y: 67.891,
|
||||
width: 23.456,
|
||||
height: 9.876,
|
||||
font: "F1".to_string(),
|
||||
font_size: 10.0,
|
||||
page: 2,
|
||||
is_bold: false,
|
||||
is_italic: true,
|
||||
is_underline: true,
|
||||
is_strikeout: true,
|
||||
item_type: ItemType::Text,
|
||||
mcid: Some(7),
|
||||
}];
|
||||
|
||||
let json = format_items_json(&items);
|
||||
|
||||
assert!(json.contains(r#""text":"A \"quoted\" item""#));
|
||||
assert!(json.contains(r#""page":2"#));
|
||||
assert!(json.contains(r#""x":12.35"#));
|
||||
assert!(json.contains(r#""is_underline":true"#));
|
||||
assert!(json.contains(r#""item_type":"text""#));
|
||||
assert!(json.contains(r#""mcid":7"#));
|
||||
}
|
||||
}
|
||||
|
||||
/// Parse a page specification like "1,3,5-10,20" into a HashSet of page numbers.
|
||||
fn parse_page_spec(spec: &str) -> Result<HashSet<u32>, String> {
|
||||
let mut pages = HashSet::new();
|
||||
@@ -196,7 +88,6 @@ fn main() {
|
||||
if args.len() < 2 {
|
||||
eprintln!("Usage: {} <pdf_file> [output_file]", args[0]);
|
||||
eprintln!(" {} <pdf_file> --json", args[0]);
|
||||
eprintln!(" {} <pdf_file> --items-json", args[0]);
|
||||
eprintln!(" {} <pdf_file> --raw", args[0]);
|
||||
eprintln!();
|
||||
eprintln!("Converts PDF to Markdown with smart type detection.");
|
||||
@@ -204,7 +95,6 @@ fn main() {
|
||||
eprintln!();
|
||||
eprintln!("Options:");
|
||||
eprintln!(" --json Output result as JSON");
|
||||
eprintln!(" --items-json Output positioned TextItem JSON");
|
||||
eprintln!(" --raw Output only markdown (no headers)");
|
||||
eprintln!(" --pages Insert page break markers (<!-- Page N -->)");
|
||||
eprintln!(" --select-pages N Only process specified pages (e.g. 1,3,5-10)");
|
||||
@@ -215,7 +105,6 @@ fn main() {
|
||||
|
||||
let pdf_path = &args[1];
|
||||
let json_output = args.iter().any(|a| a == "--json");
|
||||
let items_json_output = args.iter().any(|a| a == "--items-json");
|
||||
let raw_output = args.iter().any(|a| a == "--raw");
|
||||
let page_numbers = args.iter().any(|a| a == "--pages");
|
||||
let detect_only = args.iter().any(|a| a == "--detect-only");
|
||||
@@ -240,17 +129,6 @@ fn main() {
|
||||
})
|
||||
});
|
||||
|
||||
if items_json_output {
|
||||
match extract_text_with_positions_pages(pdf_path, page_filter.as_ref()) {
|
||||
Ok(items) => println!("{}", format_items_json(&items)),
|
||||
Err(e) => {
|
||||
println!(r#"{{"error":"{}"}}"#, json_escape(&e.to_string()));
|
||||
process::exit(1);
|
||||
}
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
let output_file = args
|
||||
.get(2)
|
||||
.filter(|a| !a.starts_with("--"))
|
||||
@@ -299,14 +177,12 @@ fn main() {
|
||||
.iter()
|
||||
.map(|p| p.to_string())
|
||||
.collect();
|
||||
let ocr_reasons = format_ocr_reasons_by_page(&result.ocr_reasons_by_page);
|
||||
println!(
|
||||
r#"{{"pdf_type":"{}","page_count":{},"processing_time_ms":{},"pages_needing_ocr":[{}],"ocr_reasons_by_page":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"has_encoding_issues":{}}}"#,
|
||||
r#"{{"pdf_type":"{}","page_count":{},"processing_time_ms":{},"pages_needing_ocr":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"has_encoding_issues":{}}}"#,
|
||||
pdf_type_str,
|
||||
result.page_count,
|
||||
result.processing_time_ms,
|
||||
ocr_pages.join(","),
|
||||
ocr_reasons,
|
||||
result.layout.is_complex,
|
||||
table_pages.join(","),
|
||||
col_pages.join(","),
|
||||
@@ -347,9 +223,8 @@ fn main() {
|
||||
.iter()
|
||||
.map(|p| p.to_string())
|
||||
.collect();
|
||||
let ocr_reasons = format_ocr_reasons_by_page(&result.ocr_reasons_by_page);
|
||||
println!(
|
||||
r#"{{"pdf_type":"{}","page_count":{},"has_text":{},"processing_time_ms":{},"markdown_length":{},"pages_needing_ocr":[{}],"ocr_reasons_by_page":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"has_encoding_issues":{},"markdown":"{}"}}"#,
|
||||
r#"{{"pdf_type":"{}","page_count":{},"has_text":{},"processing_time_ms":{},"markdown_length":{},"pages_needing_ocr":[{}],"is_complex":{},"pages_with_tables":[{}],"pages_with_columns":[{}],"has_encoding_issues":{},"markdown":"{}"}}"#,
|
||||
match result.pdf_type {
|
||||
PdfType::TextBased => "text_based",
|
||||
PdfType::Scanned => "scanned",
|
||||
@@ -361,7 +236,6 @@ fn main() {
|
||||
result.processing_time_ms,
|
||||
result.markdown.as_ref().map(|m| m.len()).unwrap_or(0),
|
||||
ocr_pages.join(","),
|
||||
ocr_reasons,
|
||||
result.layout.is_complex,
|
||||
table_pages.join(","),
|
||||
col_pages.join(","),
|
||||
|
||||
+35
-567
@@ -14,11 +14,9 @@ use lopdf::{Document, Encoding, Object, ObjectId};
|
||||
use std::collections::HashMap;
|
||||
|
||||
use super::fonts::{
|
||||
build_font_encodings, build_font_widths, compute_string_width_ts, descriptor_style_flags,
|
||||
extract_text_from_operand, get_font_file2_obj_num, get_operand_bytes, CMapDecisionCache,
|
||||
FontStyleCache,
|
||||
build_font_encodings, build_font_widths, compute_string_width_ts, extract_text_from_operand,
|
||||
get_font_file2_obj_num, get_operand_bytes, CMapDecisionCache,
|
||||
};
|
||||
use super::underline::UnderlineLine;
|
||||
use super::xobjects::{extract_form_xobject_text, get_page_xobjects, XObjectType};
|
||||
use super::{get_number, image_bbox_from_ctm, multiply_matrices};
|
||||
|
||||
@@ -76,56 +74,6 @@ fn strip_pdf_comments(data: &[u8]) -> Vec<u8> {
|
||||
result
|
||||
}
|
||||
|
||||
fn transform_path_point(x: f32, y: f32, ctm: &[f32; 6]) -> (f32, f32) {
|
||||
(
|
||||
x * ctm[0] + y * ctm[2] + ctm[4],
|
||||
x * ctm[1] + y * ctm[3] + ctm[5],
|
||||
)
|
||||
}
|
||||
|
||||
fn transformed_stroke_width(
|
||||
line_width: f32,
|
||||
ctm: &[f32; 6],
|
||||
x1: f32,
|
||||
y1: f32,
|
||||
x2: f32,
|
||||
y2: f32,
|
||||
) -> f32 {
|
||||
let user_width = line_width.abs();
|
||||
let dx = x2 - x1;
|
||||
let dy = y2 - y1;
|
||||
let len = (dx * dx + dy * dy).sqrt();
|
||||
if len <= f32::EPSILON {
|
||||
return user_width;
|
||||
}
|
||||
|
||||
// PDF stroke width scales perpendicular to the path direction.
|
||||
let nx = -dy / len;
|
||||
let ny = dx / len;
|
||||
let ndx = nx * ctm[0] + ny * ctm[2];
|
||||
let ndy = nx * ctm[1] + ny * ctm[3];
|
||||
user_width * (ndx * ndx + ndy * ndy).sqrt()
|
||||
}
|
||||
|
||||
/// Text rise (Ts) displaces the glyph origin by (0, rise) in unscaled text
|
||||
/// space — per the rendering-matrix definition it sits left of Tm, so the
|
||||
/// offset maps through the text matrix's y column. Rise never contributes
|
||||
/// to the advance, so callers apply it only to the rendering position and
|
||||
/// keep advancing the unshifted text matrix.
|
||||
fn rise_adjusted(tm: &[f32; 6], rise: f32) -> [f32; 6] {
|
||||
if rise == 0.0 {
|
||||
return *tm;
|
||||
}
|
||||
[
|
||||
tm[0],
|
||||
tm[1],
|
||||
tm[2],
|
||||
tm[3],
|
||||
tm[4] + rise * tm[2],
|
||||
tm[5] + rise * tm[3],
|
||||
]
|
||||
}
|
||||
|
||||
/// Returns `(page_extraction, has_gid_fonts)` where `has_gid_fonts` indicates
|
||||
/// the page uses fonts with unresolvable gid-encoded glyphs.
|
||||
pub(crate) fn extract_page_text_items(
|
||||
@@ -134,7 +82,6 @@ pub(crate) fn extract_page_text_items(
|
||||
page_num: u32,
|
||||
font_cmaps: &FontCMaps,
|
||||
include_invisible: bool,
|
||||
style_cache: &mut FontStyleCache,
|
||||
) -> Result<(PageExtraction, bool, bool), PdfError> {
|
||||
use lopdf::content::Content;
|
||||
|
||||
@@ -142,7 +89,6 @@ pub(crate) fn extract_page_text_items(
|
||||
let mut rects: Vec<PdfRect> = Vec::new();
|
||||
let mut clip_rects: Vec<PdfRect> = Vec::new();
|
||||
let mut lines: Vec<PdfLine> = Vec::new();
|
||||
let mut underline_lines: Vec<UnderlineLine> = Vec::new();
|
||||
|
||||
// Path construction state for m/l/h → S/s line extraction
|
||||
let mut path_subpath_start: Option<(f32, f32)> = None;
|
||||
@@ -151,12 +97,6 @@ pub(crate) fn extract_page_text_items(
|
||||
// Completed subpaths (each a vec of line segments) for f/f* rect extraction
|
||||
let mut pending_subpaths: Vec<Vec<(f32, f32, f32, f32)>> = Vec::new();
|
||||
let mut fill_rects: Vec<PdfRect> = Vec::new();
|
||||
// `re` rects awaiting a paint operator. Underline detection must only
|
||||
// see painted rects: a `re W n` clip path or `re n` no-op draws nothing
|
||||
// on the page, so treating every `re` as ink would underline text that
|
||||
// merely sits near an invisible clip boundary.
|
||||
let mut pending_re_rects: Vec<PdfRect> = Vec::new();
|
||||
let mut painted_rects: Vec<PdfRect> = Vec::new();
|
||||
|
||||
// Get fonts for encoding
|
||||
let fonts = doc.get_page_fonts(page_id).unwrap_or_default();
|
||||
@@ -174,8 +114,6 @@ pub(crate) fn extract_page_text_items(
|
||||
std::collections::HashMap::new();
|
||||
let mut inline_cmaps: std::collections::HashMap<String, crate::tounicode::CMapEntry> =
|
||||
std::collections::HashMap::new();
|
||||
let mut font_style_flags: std::collections::HashMap<String, (bool, bool)> =
|
||||
std::collections::HashMap::new();
|
||||
for (font_name, font_dict) in &fonts {
|
||||
let resource_name = String::from_utf8_lossy(font_name).to_string();
|
||||
if let Ok(base_font) = font_dict.get(b"BaseFont") {
|
||||
@@ -184,12 +122,6 @@ pub(crate) fn extract_page_text_items(
|
||||
font_base_names.insert(resource_name.clone(), base_name);
|
||||
}
|
||||
}
|
||||
// Descriptor style flags rescue subset fonts whose BaseFont names
|
||||
// are opaque tags the name heuristics can't read.
|
||||
let style = descriptor_style_flags(doc, font_dict, style_cache);
|
||||
if style != (false, false) {
|
||||
font_style_flags.insert(resource_name.clone(), style);
|
||||
}
|
||||
// Track ToUnicode object reference, with FontFile2 fallback for Identity-H/V.
|
||||
// Also handle inline ToUnicode streams.
|
||||
match font_dict.get(b"ToUnicode") {
|
||||
@@ -256,20 +188,7 @@ pub(crate) fn extract_page_text_items(
|
||||
// Graphics state tracking
|
||||
let mut ctm = [1.0f32, 0.0, 0.0, 1.0, 0.0, 0.0]; // Current Transformation Matrix
|
||||
let mut text_rendering_mode: i32 = 0; // 0=fill, 1=stroke, 2=fill+stroke, 3=invisible
|
||||
let mut line_width: f32 = 1.0;
|
||||
#[derive(Clone)]
|
||||
struct SavedGraphicsState {
|
||||
ctm: [f32; 6],
|
||||
text_rendering_mode: i32,
|
||||
line_width: f32,
|
||||
char_spacing: f32,
|
||||
word_spacing: f32,
|
||||
text_rise: f32,
|
||||
text_leading: f32,
|
||||
current_font: String,
|
||||
current_font_size: f32,
|
||||
}
|
||||
let mut gstate_stack: Vec<SavedGraphicsState> = Vec::new();
|
||||
let mut gstate_stack: Vec<([f32; 6], i32, f32, f32)> = Vec::new();
|
||||
|
||||
// Text state tracking
|
||||
let mut current_font = String::new();
|
||||
@@ -277,7 +196,6 @@ pub(crate) fn extract_page_text_items(
|
||||
let mut text_leading: f32 = 0.0; // TL parameter (in text-space units)
|
||||
let mut char_spacing: f32 = 0.0; // Tc parameter (extra spacing per character, unscaled)
|
||||
let mut word_spacing: f32 = 0.0; // Tw parameter (extra spacing per space char, unscaled)
|
||||
let mut text_rise: f32 = 0.0; // Ts parameter (baseline shift for super/subscripts, unscaled)
|
||||
let mut text_matrix = [1.0f32, 0.0, 0.0, 1.0, 0.0, 0.0];
|
||||
let mut line_matrix = [1.0f32, 0.0, 0.0, 1.0, 0.0, 0.0];
|
||||
let mut in_text_block = false;
|
||||
@@ -299,10 +217,6 @@ pub(crate) fn extract_page_text_items(
|
||||
let mut suppress_glyph_extraction = false;
|
||||
let mut actual_text_start_tm: Option<[f32; 6]> = None; // text matrix at BDC entry
|
||||
let mut actual_text_glyph_tm: Option<[f32; 6]> = None; // text matrix at first glyph inside BDC
|
||||
// Text rise in effect at each captured matrix — the item must render at
|
||||
// the rise of its GLYPHS, not whatever rise is set by EMC time.
|
||||
let mut actual_text_start_rise: f32 = 0.0;
|
||||
let mut actual_text_glyph_rise: Option<f32> = None;
|
||||
/// Get the innermost MCID from the marked content stack.
|
||||
fn current_mcid(stack: &[MarkedContentEntry]) -> Option<i64> {
|
||||
stack.iter().rev().find_map(|e| e.mcid)
|
||||
@@ -313,30 +227,15 @@ pub(crate) fn extract_page_text_items(
|
||||
match op.operator.as_str() {
|
||||
"q" => {
|
||||
// Save graphics state
|
||||
gstate_stack.push(SavedGraphicsState {
|
||||
ctm,
|
||||
text_rendering_mode,
|
||||
line_width,
|
||||
char_spacing,
|
||||
word_spacing,
|
||||
text_rise,
|
||||
text_leading,
|
||||
current_font: current_font.clone(),
|
||||
current_font_size,
|
||||
});
|
||||
gstate_stack.push((ctm, text_rendering_mode, char_spacing, word_spacing));
|
||||
}
|
||||
"Q" => {
|
||||
// Restore graphics state
|
||||
if let Some(saved) = gstate_stack.pop() {
|
||||
ctm = saved.ctm;
|
||||
text_rendering_mode = saved.text_rendering_mode;
|
||||
line_width = saved.line_width;
|
||||
char_spacing = saved.char_spacing;
|
||||
word_spacing = saved.word_spacing;
|
||||
text_rise = saved.text_rise;
|
||||
text_leading = saved.text_leading;
|
||||
current_font = saved.current_font;
|
||||
current_font_size = saved.current_font_size;
|
||||
if let Some((saved_ctm, saved_tr, saved_tc, saved_tw)) = gstate_stack.pop() {
|
||||
ctm = saved_ctm;
|
||||
text_rendering_mode = saved_tr;
|
||||
char_spacing = saved_tc;
|
||||
word_spacing = saved_tw;
|
||||
}
|
||||
}
|
||||
"cm" => {
|
||||
@@ -353,11 +252,6 @@ pub(crate) fn extract_page_text_items(
|
||||
ctm = multiply_matrices(&new_matrix, &ctm);
|
||||
}
|
||||
}
|
||||
"w" => {
|
||||
if let Some(width) = op.operands.first().and_then(get_number) {
|
||||
line_width = width;
|
||||
}
|
||||
}
|
||||
"BT" => {
|
||||
// Begin text block
|
||||
in_text_block = true;
|
||||
@@ -406,12 +300,6 @@ pub(crate) fn extract_page_text_items(
|
||||
word_spacing = tw;
|
||||
}
|
||||
}
|
||||
"Ts" => {
|
||||
// Set text rise (baseline shift for superscripts/subscripts)
|
||||
if let Some(ts) = op.operands.first().and_then(get_number) {
|
||||
text_rise = ts;
|
||||
}
|
||||
}
|
||||
"Td" | "TD" => {
|
||||
// Move text position: TLM = T(tx,ty) × TLM; Tm = TLM
|
||||
// tx,ty are in text space — must be scaled by the text line matrix
|
||||
@@ -470,7 +358,6 @@ pub(crate) fn extract_page_text_items(
|
||||
if suppress_glyph_extraction {
|
||||
if actual_text_glyph_tm.is_none() {
|
||||
actual_text_glyph_tm = Some(text_matrix);
|
||||
actual_text_glyph_rise = Some(text_rise);
|
||||
}
|
||||
if let Some(w_ts) = w_ts_opt {
|
||||
text_matrix[4] += w_ts * text_matrix[0];
|
||||
@@ -500,8 +387,7 @@ pub(crate) fn extract_page_text_items(
|
||||
&mut cmap_decisions,
|
||||
&font_widths,
|
||||
) {
|
||||
let combined =
|
||||
multiply_matrices(&rise_adjusted(&text_matrix, text_rise), &ctm);
|
||||
let combined = multiply_matrices(&text_matrix, &ctm);
|
||||
let rendered_size = effective_font_size(current_font_size, &combined);
|
||||
let (x, y) = (combined[4], combined[5]);
|
||||
if combined[0].abs() >= combined[1].abs() {
|
||||
@@ -523,10 +409,6 @@ pub(crate) fn extract_page_text_items(
|
||||
.get(¤t_font)
|
||||
.map(|s| s.as_str())
|
||||
.unwrap_or(¤t_font);
|
||||
let (desc_italic, desc_bold) = font_style_flags
|
||||
.get(¤t_font)
|
||||
.copied()
|
||||
.unwrap_or((false, false));
|
||||
items.push(TextItem {
|
||||
text: expand_ligatures(&text),
|
||||
x,
|
||||
@@ -536,10 +418,8 @@ pub(crate) fn extract_page_text_items(
|
||||
font: current_font.clone(),
|
||||
font_size: rendered_size,
|
||||
page: page_num,
|
||||
is_bold: is_bold_font(base_font) || desc_bold,
|
||||
is_italic: is_italic_font(base_font) || desc_italic,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
is_bold: is_bold_font(base_font),
|
||||
is_italic: is_italic_font(base_font),
|
||||
item_type: ItemType::Text,
|
||||
mcid: current_mcid(&marked_content_stack),
|
||||
});
|
||||
@@ -557,7 +437,6 @@ pub(crate) fn extract_page_text_items(
|
||||
// Capture first-glyph position for ActualText
|
||||
if suppress_glyph_extraction && actual_text_glyph_tm.is_none() {
|
||||
actual_text_glyph_tm = Some(text_matrix);
|
||||
actual_text_glyph_rise = Some(text_rise);
|
||||
}
|
||||
|
||||
// Compute space threshold based on font metrics when available
|
||||
@@ -678,10 +557,6 @@ pub(crate) fn extract_page_text_items(
|
||||
.get(¤t_font)
|
||||
.map(|s| s.as_str())
|
||||
.unwrap_or(¤t_font);
|
||||
let (desc_italic, desc_bold) = font_style_flags
|
||||
.get(¤t_font)
|
||||
.copied()
|
||||
.unwrap_or((false, false));
|
||||
let scale_x = text_matrix[0] * ctm[0] + text_matrix[1] * ctm[2];
|
||||
for (text, start_w, end_w) in &sub_items {
|
||||
let offset_tm = [
|
||||
@@ -692,8 +567,7 @@ pub(crate) fn extract_page_text_items(
|
||||
text_matrix[4] + start_w * text_matrix[0],
|
||||
text_matrix[5] + start_w * text_matrix[1],
|
||||
];
|
||||
let combined =
|
||||
multiply_matrices(&rise_adjusted(&offset_tm, text_rise), &ctm);
|
||||
let combined = multiply_matrices(&offset_tm, &ctm);
|
||||
let (x, y) = (combined[4], combined[5]);
|
||||
let width = if font_info.is_some() {
|
||||
((end_w - start_w) * scale_x).abs()
|
||||
@@ -709,10 +583,8 @@ pub(crate) fn extract_page_text_items(
|
||||
font: current_font.clone(),
|
||||
font_size: rendered_size,
|
||||
page: page_num,
|
||||
is_bold: is_bold_font(base_font) || desc_bold,
|
||||
is_italic: is_italic_font(base_font) || desc_italic,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
is_bold: is_bold_font(base_font),
|
||||
is_italic: is_italic_font(base_font),
|
||||
item_type: ItemType::Text,
|
||||
mcid: current_mcid(&marked_content_stack),
|
||||
});
|
||||
@@ -736,26 +608,6 @@ pub(crate) fn extract_page_text_items(
|
||||
line_matrix[4] += (-tl) * line_matrix[2];
|
||||
line_matrix[5] += (-tl) * line_matrix[3];
|
||||
text_matrix = line_matrix;
|
||||
// Capture first-glyph position for ActualText AFTER the
|
||||
// line move — the BDC-entry matrix is on the previous line.
|
||||
if suppress_glyph_extraction && actual_text_glyph_tm.is_none() {
|
||||
actual_text_glyph_tm = Some(text_matrix);
|
||||
actual_text_glyph_rise = Some(text_rise);
|
||||
}
|
||||
// Advance width, as for Tj — without it the item stays
|
||||
// zero-width and geometric underline/strikeout detection
|
||||
// rejects it (`is_underline_candidate` needs width > 0).
|
||||
let w_ts_opt = font_widths.get(¤t_font).and_then(|fi| {
|
||||
op.operands.first().and_then(get_operand_bytes).map(|raw| {
|
||||
compute_string_width_ts(
|
||||
raw,
|
||||
fi,
|
||||
current_font_size,
|
||||
char_spacing,
|
||||
word_spacing,
|
||||
)
|
||||
})
|
||||
});
|
||||
if !((text_rendering_mode == 3 && !include_invisible)
|
||||
|| suppress_glyph_extraction
|
||||
|| op.operands.is_empty())
|
||||
@@ -773,8 +625,7 @@ pub(crate) fn extract_page_text_items(
|
||||
&font_widths,
|
||||
) {
|
||||
if !text.trim().is_empty() {
|
||||
let combined =
|
||||
multiply_matrices(&rise_adjusted(&text_matrix, text_rise), &ctm);
|
||||
let combined = multiply_matrices(&text_matrix, &ctm);
|
||||
if combined[0].abs() >= combined[1].abs() {
|
||||
rotation_votes.horizontal += 1;
|
||||
} else {
|
||||
@@ -782,45 +633,27 @@ pub(crate) fn extract_page_text_items(
|
||||
}
|
||||
let rendered_size = effective_font_size(current_font_size, &combined);
|
||||
let (x, y) = (combined[4], combined[5]);
|
||||
let width = w_ts_opt
|
||||
.map(|w_ts| {
|
||||
(w_ts * (text_matrix[0] * ctm[0] + text_matrix[1] * ctm[2]))
|
||||
.abs()
|
||||
})
|
||||
.unwrap_or(0.0);
|
||||
let base_font = font_base_names
|
||||
.get(¤t_font)
|
||||
.map(|s| s.as_str())
|
||||
.unwrap_or(¤t_font);
|
||||
let (desc_italic, desc_bold) = font_style_flags
|
||||
.get(¤t_font)
|
||||
.copied()
|
||||
.unwrap_or((false, false));
|
||||
items.push(TextItem {
|
||||
text: expand_ligatures(&text),
|
||||
x,
|
||||
y,
|
||||
width,
|
||||
width: 0.0,
|
||||
height: rendered_size,
|
||||
font: current_font.clone(),
|
||||
font_size: rendered_size,
|
||||
page: page_num,
|
||||
is_bold: is_bold_font(base_font) || desc_bold,
|
||||
is_italic: is_italic_font(base_font) || desc_italic,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
is_bold: is_bold_font(base_font),
|
||||
is_italic: is_italic_font(base_font),
|
||||
item_type: ItemType::Text,
|
||||
mcid: current_mcid(&marked_content_stack),
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
// Advance regardless of visibility so later show-text
|
||||
// operators on the same line stay positioned (as for Tj).
|
||||
if let Some(w_ts) = w_ts_opt {
|
||||
text_matrix[4] += w_ts * text_matrix[0];
|
||||
text_matrix[5] += w_ts * text_matrix[1];
|
||||
}
|
||||
}
|
||||
"Do" => {
|
||||
// XObject invocation - could be an image or form
|
||||
@@ -851,8 +684,6 @@ pub(crate) fn extract_page_text_items(
|
||||
page: page_num,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Image,
|
||||
mcid: current_mcid(&marked_content_stack),
|
||||
});
|
||||
@@ -866,7 +697,6 @@ pub(crate) fn extract_page_text_items(
|
||||
font_cmaps,
|
||||
&ctm,
|
||||
&mut cmap_decisions,
|
||||
style_cache,
|
||||
);
|
||||
items.extend(form_items);
|
||||
}
|
||||
@@ -907,9 +737,7 @@ pub(crate) fn extract_page_text_items(
|
||||
if actual_text.is_some() {
|
||||
suppress_glyph_extraction = true;
|
||||
actual_text_start_tm = Some(text_matrix);
|
||||
actual_text_start_rise = text_rise;
|
||||
actual_text_glyph_tm = None; // reset — will be captured at first Tj/TJ
|
||||
actual_text_glyph_rise = None;
|
||||
}
|
||||
marked_content_stack.push(MarkedContentEntry { actual_text, mcid });
|
||||
}
|
||||
@@ -922,11 +750,9 @@ pub(crate) fn extract_page_text_items(
|
||||
// Tj may have moved the text position to the correct line —
|
||||
// the BDC-entry position can be on the previous line.
|
||||
let glyph_tm = actual_text_glyph_tm.take();
|
||||
let glyph_rise = actual_text_glyph_rise.take();
|
||||
let entry_tm = actual_text_start_tm.take();
|
||||
if let Some(start_tm) = glyph_tm.or(entry_tm) {
|
||||
let rise = glyph_rise.unwrap_or(actual_text_start_rise);
|
||||
let combined = multiply_matrices(&rise_adjusted(&start_tm, rise), &ctm);
|
||||
let combined = multiply_matrices(&start_tm, &ctm);
|
||||
if combined[0].abs() >= combined[1].abs() {
|
||||
rotation_votes.horizontal += 1;
|
||||
} else {
|
||||
@@ -943,10 +769,6 @@ pub(crate) fn extract_page_text_items(
|
||||
.get(¤t_font)
|
||||
.map(|s| s.as_str())
|
||||
.unwrap_or(¤t_font);
|
||||
let (desc_italic, desc_bold) = font_style_flags
|
||||
.get(¤t_font)
|
||||
.copied()
|
||||
.unwrap_or((false, false));
|
||||
items.push(TextItem {
|
||||
text: expand_ligatures(&at),
|
||||
x,
|
||||
@@ -956,10 +778,8 @@ pub(crate) fn extract_page_text_items(
|
||||
font: current_font.clone(),
|
||||
font_size: rendered_size,
|
||||
page: page_num,
|
||||
is_bold: is_bold_font(base_font) || desc_bold,
|
||||
is_italic: is_italic_font(base_font) || desc_italic,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
is_bold: is_bold_font(base_font),
|
||||
is_italic: is_italic_font(base_font),
|
||||
item_type: ItemType::Text,
|
||||
mcid: entry
|
||||
.mcid
|
||||
@@ -984,19 +804,13 @@ pub(crate) fn extract_page_text_items(
|
||||
let y_dev = rx * ctm[1] + ry * ctm[3] + ctm[5];
|
||||
let w_dev = rw * ctm[0];
|
||||
let h_dev = rh * ctm[3];
|
||||
let rect = PdfRect {
|
||||
rects.push(PdfRect {
|
||||
x: x_dev,
|
||||
y: y_dev,
|
||||
width: w_dev,
|
||||
height: h_dev,
|
||||
page: page_num,
|
||||
};
|
||||
// Underline detection must only see rects that are
|
||||
// actually painted — a `re` used purely as a clip path
|
||||
// (`re W n`) or discarded (`re n`) draws nothing. Hold
|
||||
// the rect as pending until a paint operator confirms it.
|
||||
pending_re_rects.push(rect.clone());
|
||||
rects.push(rect);
|
||||
});
|
||||
}
|
||||
}
|
||||
// ── Path construction operators ──────────────────────
|
||||
@@ -1046,8 +860,10 @@ pub(crate) fn extract_page_text_items(
|
||||
}
|
||||
}
|
||||
for (x1, y1, x2, y2) in pending_lines.drain(..) {
|
||||
let (x1d, y1d) = transform_path_point(x1, y1, &ctm);
|
||||
let (x2d, y2d) = transform_path_point(x2, y2, &ctm);
|
||||
let x1d = x1 * ctm[0] + y1 * ctm[2] + ctm[4];
|
||||
let y1d = x1 * ctm[1] + y1 * ctm[3] + ctm[5];
|
||||
let x2d = x2 * ctm[0] + y2 * ctm[2] + ctm[4];
|
||||
let y2d = x2 * ctm[1] + y2 * ctm[3] + ctm[5];
|
||||
lines.push(PdfLine {
|
||||
x1: x1d,
|
||||
y1: y1d,
|
||||
@@ -1055,16 +871,7 @@ pub(crate) fn extract_page_text_items(
|
||||
y2: y2d,
|
||||
page: page_num,
|
||||
});
|
||||
underline_lines.push(UnderlineLine {
|
||||
x1: x1d,
|
||||
y1: y1d,
|
||||
x2: x2d,
|
||||
y2: y2d,
|
||||
stroke_width: transformed_stroke_width(line_width, &ctm, x1, y1, x2, y2),
|
||||
page: page_num,
|
||||
});
|
||||
}
|
||||
painted_rects.append(&mut pending_re_rects);
|
||||
pending_subpaths.clear();
|
||||
path_subpath_start = None;
|
||||
path_current = None;
|
||||
@@ -1080,8 +887,10 @@ pub(crate) fn extract_page_text_items(
|
||||
}
|
||||
}
|
||||
for (x1, y1, x2, y2) in pending_lines.drain(..) {
|
||||
let (x1d, y1d) = transform_path_point(x1, y1, &ctm);
|
||||
let (x2d, y2d) = transform_path_point(x2, y2, &ctm);
|
||||
let x1d = x1 * ctm[0] + y1 * ctm[2] + ctm[4];
|
||||
let y1d = x1 * ctm[1] + y1 * ctm[3] + ctm[5];
|
||||
let x2d = x2 * ctm[0] + y2 * ctm[2] + ctm[4];
|
||||
let y2d = x2 * ctm[1] + y2 * ctm[3] + ctm[5];
|
||||
lines.push(PdfLine {
|
||||
x1: x1d,
|
||||
y1: y1d,
|
||||
@@ -1089,16 +898,7 @@ pub(crate) fn extract_page_text_items(
|
||||
y2: y2d,
|
||||
page: page_num,
|
||||
});
|
||||
underline_lines.push(UnderlineLine {
|
||||
x1: x1d,
|
||||
y1: y1d,
|
||||
x2: x2d,
|
||||
y2: y2d,
|
||||
stroke_width: transformed_stroke_width(line_width, &ctm, x1, y1, x2, y2),
|
||||
page: page_num,
|
||||
});
|
||||
}
|
||||
painted_rects.append(&mut pending_re_rects);
|
||||
pending_subpaths.clear();
|
||||
path_subpath_start = None;
|
||||
path_current = None;
|
||||
@@ -1156,7 +956,6 @@ pub(crate) fn extract_page_text_items(
|
||||
}
|
||||
}
|
||||
}
|
||||
painted_rects.append(&mut pending_re_rects);
|
||||
pending_lines.clear();
|
||||
path_subpath_start = None;
|
||||
path_current = None;
|
||||
@@ -1222,10 +1021,7 @@ pub(crate) fn extract_page_text_items(
|
||||
// Do NOT clear pending_lines — the following `n` does that
|
||||
}
|
||||
"n" => {
|
||||
// end path (no-op): discard — including any `re` rects that
|
||||
// were only ever part of a clip path (`re W n`), which draw
|
||||
// no ink and must not feed underline detection.
|
||||
pending_re_rects.clear();
|
||||
// end path (no-op): discard
|
||||
pending_lines.clear();
|
||||
pending_subpaths.clear();
|
||||
path_subpath_start = None;
|
||||
@@ -1235,12 +1031,6 @@ pub(crate) fn extract_page_text_items(
|
||||
}
|
||||
}
|
||||
|
||||
// Underline detection reads only painted ink: `re` rects confirmed by
|
||||
// a paint operator plus filled-subpath rects — never clip-only rects,
|
||||
// which draw nothing.
|
||||
let mut underline_rects = painted_rects;
|
||||
underline_rects.extend(fill_rects.iter().cloned());
|
||||
|
||||
// Only use clip/fill rects when no `re` rects exist on this page.
|
||||
// Clip rects take priority over fill rects, but first we deduplicate
|
||||
// them: some PDFs wrap every text block in a full-page W* clip path,
|
||||
@@ -1270,17 +1060,8 @@ pub(crate) fn extract_page_text_items(
|
||||
// Some PDFs embed landscape content in portrait pages using a rotated text
|
||||
// matrix (e.g. [0, b, -b, 0, tx, ty] for 90° CCW). The layout engine
|
||||
// assumes x=horizontal, y=vertical — so we swap coordinates to match.
|
||||
let (mut items, rects, lines, coords_rotated) =
|
||||
let (items, rects, lines, coords_rotated) =
|
||||
correct_rotated_page(items, rects, lines, &rotation_votes);
|
||||
if coords_rotated {
|
||||
rotate_underline_graphics(&mut underline_rects, &mut underline_lines);
|
||||
}
|
||||
super::underline::mark_underlined_items(
|
||||
&mut items,
|
||||
&underline_rects,
|
||||
&underline_lines,
|
||||
page_num,
|
||||
);
|
||||
|
||||
let items = super::merge_text_items(items);
|
||||
let items = super::merge_subscript_items(items);
|
||||
@@ -1366,27 +1147,6 @@ fn correct_rotated_page(
|
||||
(items, rects, lines, true)
|
||||
}
|
||||
|
||||
fn rotate_underline_graphics(rects: &mut [PdfRect], lines: &mut [UnderlineLine]) {
|
||||
for rect in rects {
|
||||
let new_x = rect.y;
|
||||
let new_y = -(rect.x + rect.width.abs());
|
||||
rect.x = new_x;
|
||||
rect.y = new_y;
|
||||
std::mem::swap(&mut rect.width, &mut rect.height);
|
||||
}
|
||||
|
||||
for line in lines {
|
||||
let new_x1 = line.y1;
|
||||
let new_y1 = -line.x1;
|
||||
let new_x2 = line.y2;
|
||||
let new_y2 = -line.x2;
|
||||
line.x1 = new_x1;
|
||||
line.y1 = new_y1;
|
||||
line.x2 = new_x2;
|
||||
line.y2 = new_y2;
|
||||
}
|
||||
}
|
||||
|
||||
/// Remove near-duplicate rects (same coordinates within 0.5 pt tolerance).
|
||||
/// Some PDFs emit a full-page clip path for every text block, producing
|
||||
/// thousands of identical rects. After dedup these collapse to one rect,
|
||||
@@ -1436,64 +1196,6 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
fn simple_doc_with_content(content: &[u8]) -> (lopdf::Document, lopdf::ObjectId) {
|
||||
use lopdf::{dictionary, Object, Stream};
|
||||
|
||||
let mut doc = lopdf::Document::new();
|
||||
let widths: Vec<Object> = (0..=255).map(|_| 600.into()).collect();
|
||||
let font_id = doc.add_object(dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "Type1",
|
||||
"BaseFont" => "Helvetica",
|
||||
"FirstChar" => 0,
|
||||
"LastChar" => 255,
|
||||
"Widths" => Object::Array(widths),
|
||||
});
|
||||
let content_id = doc.add_object(Object::Stream(Stream::new(
|
||||
dictionary! {},
|
||||
content.to_vec(),
|
||||
)));
|
||||
let page_id = doc.add_object(dictionary! {
|
||||
"Type" => "Page",
|
||||
"Contents" => Object::Reference(content_id),
|
||||
"Resources" => dictionary! {
|
||||
"Font" => dictionary! {
|
||||
"F1" => Object::Reference(font_id),
|
||||
},
|
||||
},
|
||||
"MediaBox" => vec![0.into(), 0.into(), 612.into(), 792.into()],
|
||||
});
|
||||
let pages_id = doc.add_object(dictionary! {
|
||||
"Type" => "Pages",
|
||||
"Count" => Object::Integer(1),
|
||||
"Kids" => vec![Object::Reference(page_id)],
|
||||
});
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"Pages" => Object::Reference(pages_id),
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
|
||||
(doc, page_id)
|
||||
}
|
||||
|
||||
fn extract_simple_items(content: &[u8]) -> Vec<TextItem> {
|
||||
use crate::tounicode::FontCMaps;
|
||||
|
||||
let (doc, page_id) = simple_doc_with_content(content);
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let ((items, _, _), _, _) = extract_page_text_items(
|
||||
&doc,
|
||||
page_id,
|
||||
1,
|
||||
&font_cmaps,
|
||||
false,
|
||||
&mut FontStyleCache::new(),
|
||||
)
|
||||
.unwrap();
|
||||
items
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_dedup_rects_identical() {
|
||||
let mut rects = vec![rect(0.0, 0.0, 612.0, 792.0, 1); 3759];
|
||||
@@ -1543,140 +1245,6 @@ mod tests {
|
||||
assert_eq!(single.len(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn thick_stroked_rule_does_not_mark_underline() {
|
||||
let content = b"BT /F1 12 Tf 1 0 0 1 100 500 Tm (THICK) Tj ET
|
||||
4 w
|
||||
100 498 m 170 498 l S
|
||||
BT /F1 12 Tf 1 0 0 1 100 480 Tm (THIN) Tj ET
|
||||
1 w
|
||||
100 478 m 160 478 l S";
|
||||
|
||||
let items = extract_simple_items(content);
|
||||
let thick = items.iter().find(|item| item.text == "THICK").unwrap();
|
||||
let thin = items.iter().find(|item| item.text == "THIN").unwrap();
|
||||
|
||||
assert!(!thick.is_underline);
|
||||
assert!(thin.is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rotated_page_underline_is_detected_after_coordinate_correction() {
|
||||
let content = b"BT /F1 12 Tf 0 1 -1 0 200 100 Tm (HELLO) Tj ET
|
||||
BT /F1 12 Tf 0 1 -1 0 240 100 Tm (WORLD) Tj ET
|
||||
1 w
|
||||
202 100 m 202 170 l S";
|
||||
|
||||
let items = extract_simple_items(content);
|
||||
let hello = items.iter().find(|item| item.text == "HELLO").unwrap();
|
||||
let world = items.iter().find(|item| item.text == "WORLD").unwrap();
|
||||
|
||||
assert!(hello.is_underline);
|
||||
assert!(!world.is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn quote_operator_text_carries_advance_width() {
|
||||
// `'` (move-to-next-line-and-show-text) must retain the string's
|
||||
// advance width like Tj — zero-width items are invisible to
|
||||
// geometric underline/strikeout detection.
|
||||
let content = b"BT /F1 12 Tf 12 TL 1 0 0 1 100 512 Tm (first) Tj (struck) ' ET
|
||||
1 w
|
||||
99 503 m 145 503 l S";
|
||||
|
||||
let items = extract_simple_items(content);
|
||||
let struck = items.iter().find(|item| item.text == "struck").unwrap();
|
||||
|
||||
// 6 glyphs x 600/1000 x 12pt = 43.2pt, drawn one leading below Tm.
|
||||
assert!((struck.width - 43.2).abs() < 0.1);
|
||||
assert!((struck.y - 500.0).abs() < 0.1);
|
||||
assert!(struck.is_strikeout);
|
||||
assert!(!struck.is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn quote_operator_advances_text_matrix() {
|
||||
// Text shown after `'` on the same line must start past the shown
|
||||
// string: "CD" lands at x=114.4 (2 glyphs x 600/1000 x 12pt after
|
||||
// x=100), flush against "AB", so the merge pass joins them. Without
|
||||
// the advance "CD" overlaps "AB" at x=100 and the items stay apart.
|
||||
let content = b"BT /F1 12 Tf 12 TL 1 0 0 1 100 512 Tm (AB) ' (CD) Tj ET";
|
||||
|
||||
let items = extract_simple_items(content);
|
||||
let merged = items.iter().find(|item| item.text == "ABCD").unwrap();
|
||||
|
||||
assert!((merged.x - 100.0).abs() < 0.1);
|
||||
assert!((merged.width - 28.8).abs() < 0.1);
|
||||
assert!((merged.y - 500.0).abs() < 0.1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn text_rise_shifts_item_baseline() {
|
||||
// Ts displaces the glyph origin vertically without touching the
|
||||
// advance; the next run at rise 0 must return to the original
|
||||
// baseline and follow the raised run horizontally.
|
||||
let content =
|
||||
b"BT /F1 12 Tf 1 0 0 1 100 500 Tm (base) Tj 5 Ts (super) Tj 0 Ts (after) Tj ET";
|
||||
|
||||
let items = extract_simple_items(content);
|
||||
let base = items.iter().find(|item| item.text == "base").unwrap();
|
||||
let raised = items.iter().find(|item| item.text == "super").unwrap();
|
||||
let after = items.iter().find(|item| item.text == "after").unwrap();
|
||||
|
||||
assert!((base.y - 500.0).abs() < 0.1);
|
||||
assert!((raised.y - 505.0).abs() < 0.1);
|
||||
assert!((after.y - 500.0).abs() < 0.1);
|
||||
assert!(after.x > raised.x);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn actual_text_item_uses_glyph_rise() {
|
||||
// The ActualText replacement item must render at the rise in
|
||||
// effect when its glyphs were drawn — not the unshifted BDC
|
||||
// baseline, and not whatever rise is set by EMC time.
|
||||
let content = b"BT /F1 12 Tf 1 0 0 1 100 500 Tm \
|
||||
/Span <</ActualText (super) >> BDC 5 Ts (sup) Tj 0 Ts EMC (after) Tj ET";
|
||||
|
||||
let items = extract_simple_items(content);
|
||||
let sup = items.iter().find(|item| item.text == "super").unwrap();
|
||||
let after = items.iter().find(|item| item.text == "after").unwrap();
|
||||
|
||||
assert!((sup.y - 505.0).abs() < 0.1);
|
||||
assert!((after.y - 500.0).abs() < 0.1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn actual_text_shown_with_quote_op_uses_moved_risen_baseline() {
|
||||
// When the tagged span's show op is `'`, the glyph position is
|
||||
// only known AFTER its line move — falling back to the BDC-entry
|
||||
// matrix would place the item on the previous line, unrisen.
|
||||
let content = b"BT /F1 12 Tf 14 TL 1 0 0 1 100 500 Tm \
|
||||
/Span <</ActualText (replaced) >> BDC 3 Ts (raw) ' 0 Ts EMC ET";
|
||||
|
||||
let items = extract_simple_items(content);
|
||||
let item = items.iter().find(|item| item.text == "replaced").unwrap();
|
||||
|
||||
// Line move: 500 - 14 = 486; rise: +3 -> 489.
|
||||
assert!((item.y - 489.0).abs() < 0.1);
|
||||
assert!(item.width > 0.0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strikeout_detected_on_risen_text() {
|
||||
// The rule crosses the glyphs at their risen position; without the
|
||||
// rise in item.y the strike window sits 4pt too low and misses.
|
||||
let content = b"BT /F1 12 Tf 1 0 0 1 100 500 Tm 4 Ts (struck) Tj ET
|
||||
1 w
|
||||
99 507 m 145 507 l S";
|
||||
|
||||
let items = extract_simple_items(content);
|
||||
let struck = items.iter().find(|item| item.text == "struck").unwrap();
|
||||
|
||||
assert!((struck.y - 504.0).abs() < 0.1);
|
||||
assert!(struck.is_strikeout);
|
||||
assert!(!struck.is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_skip_excessive_operations() {
|
||||
use crate::tounicode::FontCMaps;
|
||||
@@ -1711,113 +1279,13 @@ BT /F1 12 Tf 0 1 -1 0 240 100 Tm (WORLD) Tj ET
|
||||
doc.add_object(catalog);
|
||||
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let result = extract_page_text_items(
|
||||
&doc,
|
||||
page_id,
|
||||
1,
|
||||
&font_cmaps,
|
||||
false,
|
||||
&mut FontStyleCache::new(),
|
||||
)
|
||||
.unwrap();
|
||||
let result = extract_page_text_items(&doc, page_id, 1, &font_cmaps, false).unwrap();
|
||||
let ((items, rects, lines), _has_gid, _coords_rotated) = result;
|
||||
assert!(items.is_empty());
|
||||
assert!(rects.is_empty());
|
||||
assert!(lines.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_q_restores_current_font_for_text_decoding() {
|
||||
use crate::tounicode::FontCMaps;
|
||||
use lopdf::{dictionary, Object, Stream};
|
||||
|
||||
fn cmap_stream(dst_hex: &str) -> Stream {
|
||||
let cmap = format!(
|
||||
r#"/CIDInit /ProcSet findresource begin
|
||||
12 dict begin
|
||||
begincmap
|
||||
/CIDSystemInfo << /Registry (Adobe) /Ordering (UCS) /Supplement 0 >> def
|
||||
/CMapName /Test-UCS def
|
||||
/CMapType 2 def
|
||||
1 begincodespacerange
|
||||
<00> <FF>
|
||||
endcodespacerange
|
||||
1 beginbfchar
|
||||
<41> <{dst_hex}>
|
||||
endbfchar
|
||||
endcmap
|
||||
CMapName currentdict /CMap defineresource pop
|
||||
end
|
||||
end"#
|
||||
);
|
||||
Stream::new(dictionary! {}, cmap.into_bytes())
|
||||
}
|
||||
|
||||
let mut doc = lopdf::Document::new();
|
||||
let f1_cmap = doc.add_object(Object::Stream(cmap_stream("0058"))); // X
|
||||
let f2_cmap = doc.add_object(Object::Stream(cmap_stream("0059"))); // Y
|
||||
let f1 = doc.add_object(dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "Type1",
|
||||
"BaseFont" => "Helvetica",
|
||||
"ToUnicode" => Object::Reference(f1_cmap),
|
||||
});
|
||||
let f2 = doc.add_object(dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "Type1",
|
||||
"BaseFont" => "Helvetica",
|
||||
"ToUnicode" => Object::Reference(f2_cmap),
|
||||
});
|
||||
|
||||
let content = b"BT /F1 12 Tf 10 700 Tm <41> Tj ET
|
||||
q
|
||||
BT /F2 12 Tf 20 700 Tm <41> Tj ET
|
||||
Q
|
||||
BT 30 700 Tm <41> Tj ET";
|
||||
let content_id = doc.add_object(Object::Stream(Stream::new(
|
||||
dictionary! {},
|
||||
content.to_vec(),
|
||||
)));
|
||||
let page_id = doc.add_object(dictionary! {
|
||||
"Type" => "Page",
|
||||
"Contents" => Object::Reference(content_id),
|
||||
"Resources" => dictionary! {
|
||||
"Font" => dictionary! {
|
||||
"F1" => Object::Reference(f1),
|
||||
"F2" => Object::Reference(f2),
|
||||
},
|
||||
},
|
||||
"MediaBox" => vec![0.into(), 0.into(), 612.into(), 792.into()],
|
||||
});
|
||||
let pages_id = doc.add_object(dictionary! {
|
||||
"Type" => "Pages",
|
||||
"Count" => Object::Integer(1),
|
||||
"Kids" => vec![Object::Reference(page_id)],
|
||||
});
|
||||
let catalog_id = doc.add_object(dictionary! {
|
||||
"Type" => "Catalog",
|
||||
"Pages" => Object::Reference(pages_id),
|
||||
});
|
||||
doc.trailer.set("Root", Object::Reference(catalog_id));
|
||||
|
||||
let font_cmaps = FontCMaps::from_doc(&doc);
|
||||
let ((items, _, _), _, _) = extract_page_text_items(
|
||||
&doc,
|
||||
page_id,
|
||||
1,
|
||||
&font_cmaps,
|
||||
false,
|
||||
&mut FontStyleCache::new(),
|
||||
)
|
||||
.unwrap();
|
||||
let text = items
|
||||
.iter()
|
||||
.map(|item| item.text.as_str())
|
||||
.collect::<String>();
|
||||
|
||||
assert_eq!(text, "XYX");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_strip_pdf_comments() {
|
||||
// Basic comment stripping
|
||||
|
||||
+47
-609
@@ -4,7 +4,7 @@ use crate::glyph_names::glyph_to_char;
|
||||
use crate::tounicode::FontCMaps;
|
||||
use crate::types::{FontEncodingMap, FontWidthInfo, PageFontEncodings, PageFontWidths};
|
||||
use log::debug;
|
||||
use lopdf::{Document, Encoding, Object, ObjectId};
|
||||
use lopdf::{Document, Encoding, Object};
|
||||
use std::collections::HashMap;
|
||||
|
||||
#[derive(Debug, Copy, Clone, PartialEq, Eq)]
|
||||
@@ -526,11 +526,6 @@ pub(crate) fn parse_font_encoding(
|
||||
font_dict: &lopdf::Dictionary,
|
||||
) -> Option<EncodingResult> {
|
||||
let encoding_obj = font_dict.get(b"Encoding").ok()?;
|
||||
let base_font_name = font_dict
|
||||
.get(b"BaseFont")
|
||||
.ok()
|
||||
.and_then(|o| o.as_name().ok())
|
||||
.map(|n| String::from_utf8_lossy(n).to_string());
|
||||
|
||||
// Encoding can be a name or a dictionary
|
||||
match encoding_obj {
|
||||
@@ -543,14 +538,12 @@ pub(crate) fn parse_font_encoding(
|
||||
Object::Reference(obj_ref) => {
|
||||
// Reference to encoding dictionary
|
||||
if let Ok(enc_dict) = doc.get_dictionary(*obj_ref) {
|
||||
parse_encoding_dictionary(doc, enc_dict, base_font_name.as_deref())
|
||||
parse_encoding_dictionary(doc, enc_dict)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
Object::Dictionary(enc_dict) => {
|
||||
parse_encoding_dictionary(doc, enc_dict, base_font_name.as_deref())
|
||||
}
|
||||
Object::Dictionary(enc_dict) => parse_encoding_dictionary(doc, enc_dict),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
@@ -569,7 +562,6 @@ pub(crate) struct EncodingResult {
|
||||
pub(crate) fn parse_encoding_dictionary(
|
||||
doc: &Document,
|
||||
enc_dict: &lopdf::Dictionary,
|
||||
base_font_name: Option<&str>,
|
||||
) -> Option<EncodingResult> {
|
||||
let differences = enc_dict.get(b"Differences").ok()?;
|
||||
|
||||
@@ -599,9 +591,11 @@ pub(crate) fn parse_encoding_dictionary(
|
||||
Object::Name(name) => {
|
||||
// Map current code to glyph name -> Unicode
|
||||
let glyph_name = String::from_utf8_lossy(&name).to_string();
|
||||
let mapped_char = glyph_to_char(&glyph_name)
|
||||
.or_else(|| private_glyph_to_char(&glyph_name, base_font_name));
|
||||
if mapped_char.is_some_and(is_ligature_char) {
|
||||
if glyph_name == "fi"
|
||||
|| glyph_name == "fl"
|
||||
|| glyph_name == "ffi"
|
||||
|| glyph_name == "ffl"
|
||||
{
|
||||
debug!(
|
||||
" Differences: code=0x{:02X} glyph={:?} (ligature)",
|
||||
current_code, glyph_name
|
||||
@@ -616,7 +610,7 @@ pub(crate) fn parse_encoding_dictionary(
|
||||
{
|
||||
gid_glyph_count += 1;
|
||||
}
|
||||
if let Some(ch) = mapped_char {
|
||||
if let Some(ch) = glyph_to_char(&glyph_name) {
|
||||
encoding_map.insert(current_code, ch);
|
||||
} else {
|
||||
debug!(
|
||||
@@ -651,31 +645,6 @@ pub(crate) fn parse_encoding_dictionary(
|
||||
})
|
||||
}
|
||||
|
||||
fn private_glyph_to_char(glyph_name: &str, base_font_name: Option<&str>) -> Option<char> {
|
||||
let base_font_name = strip_subset_prefix(base_font_name?);
|
||||
|
||||
// Aptos CFF subsets from Office PDFs can expose the ff ligature as /g431
|
||||
// without a ToUnicode map. Keep this font-scoped because /gNNN names are private.
|
||||
if base_font_name.eq_ignore_ascii_case("Aptos") && glyph_name == "g431" {
|
||||
Some('\u{FB00}')
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
fn strip_subset_prefix(font_name: &str) -> &str {
|
||||
font_name
|
||||
.split_once('+')
|
||||
.map_or(font_name, |(_, stripped)| stripped)
|
||||
}
|
||||
|
||||
fn is_ligature_char(ch: char) -> bool {
|
||||
matches!(
|
||||
ch,
|
||||
'\u{FB00}' | '\u{FB01}' | '\u{FB02}' | '\u{FB03}' | '\u{FB04}'
|
||||
)
|
||||
}
|
||||
|
||||
/// Get the CMap lookup key for an Identity-H/V CID font without ToUnicode.
|
||||
/// Returns the object number used by `collect_cmaps_from_fonts` to store the CMap:
|
||||
/// - FontFile2 or FontFile3 obj_num (for embedded font cmap)
|
||||
@@ -739,171 +708,6 @@ pub(crate) fn get_font_file2_obj_num(doc: &Document, font_dict: &lopdf::Dictiona
|
||||
.map(|r| r.0)
|
||||
}
|
||||
|
||||
/// Document-scoped memo of embedded-font style flags, keyed by the
|
||||
/// FontFile2/FontFile3 stream's object id. The same font program is
|
||||
/// referenced from every page that uses the font, and decompressing +
|
||||
/// parsing it dominates `descriptor_style_flags` — without the memo that
|
||||
/// cost repeats per page whenever the descriptor leaves a flag unset
|
||||
/// (the common case: regular fonts report neither italic nor bold).
|
||||
#[derive(Debug, Default)]
|
||||
pub(crate) struct FontStyleCache {
|
||||
by_font_file: HashMap<ObjectId, (bool, bool)>,
|
||||
}
|
||||
|
||||
impl FontStyleCache {
|
||||
pub(crate) fn new() -> Self {
|
||||
Self::default()
|
||||
}
|
||||
}
|
||||
|
||||
/// Style flags from the FontDescriptor, which survive subset fonts whose
|
||||
/// BaseFont names are opaque tags ("Tc1", "ABCDEF+F1") that defeat the
|
||||
/// name-based bold/italic heuristics.
|
||||
///
|
||||
/// Italic: `ItalicAngle` beyond a few degrees, or Flags bit 7 (Italic,
|
||||
/// value 64). Bold: Flags bit 19 (ForceBold, value 1<<18). The small
|
||||
/// ItalicAngle threshold skips fonts that declare a token slant.
|
||||
pub(crate) fn descriptor_style_flags(
|
||||
doc: &Document,
|
||||
font_dict: &lopdf::Dictionary,
|
||||
style_cache: &mut FontStyleCache,
|
||||
) -> (bool, bool) {
|
||||
let descriptor = font_dict
|
||||
.get(b"FontDescriptor")
|
||||
.ok()
|
||||
.and_then(|obj| resolve_dict(doc, obj))
|
||||
.or_else(|| {
|
||||
// Type0 fonts hang the descriptor off DescendantFonts[0].
|
||||
let desc_fonts = font_dict.get(b"DescendantFonts").ok()?;
|
||||
let desc_fonts = resolve_array(doc, desc_fonts)?;
|
||||
let cid_font_dict = resolve_dict(doc, desc_fonts.first()?)?;
|
||||
resolve_dict(doc, cid_font_dict.get(b"FontDescriptor").ok()?)
|
||||
});
|
||||
let Some(descriptor) = descriptor else {
|
||||
return (false, false);
|
||||
};
|
||||
|
||||
let italic_angle = descriptor
|
||||
.get(b"ItalicAngle")
|
||||
.ok()
|
||||
.and_then(|obj| match obj {
|
||||
Object::Integer(i) => Some(*i as f32),
|
||||
Object::Real(r) => Some(*r),
|
||||
_ => None,
|
||||
})
|
||||
.unwrap_or(0.0);
|
||||
let flags = descriptor
|
||||
.get(b"Flags")
|
||||
.ok()
|
||||
.and_then(|obj| obj.as_i64().ok())
|
||||
.unwrap_or(0);
|
||||
|
||||
let mut italic = italic_angle.abs() >= 4.0 || flags & (1 << 6) != 0;
|
||||
let mut bold = flags & (1 << 18) != 0;
|
||||
|
||||
// Descriptors lie: subset generators write ItalicAngle 0 for genuinely
|
||||
// italic faces. The embedded font file keeps the truth — OS/2
|
||||
// fsSelection (via `Face::is_italic`) and the post table's italicAngle.
|
||||
if !italic || !bold {
|
||||
if let Some(ff_ref) = font_file_ref(descriptor) {
|
||||
let (emb_italic, emb_bold) = *style_cache
|
||||
.by_font_file
|
||||
.entry(ff_ref)
|
||||
.or_insert_with(|| embedded_style_flags(doc, ff_ref));
|
||||
italic = italic || emb_italic;
|
||||
bold = bold || emb_bold;
|
||||
}
|
||||
}
|
||||
(italic, bold)
|
||||
}
|
||||
|
||||
/// Style flags parsed from an embedded font program stream.
|
||||
fn embedded_style_flags(doc: &Document, ff_ref: ObjectId) -> (bool, bool) {
|
||||
let Some(data) = font_file_data(doc, ff_ref) else {
|
||||
return (false, false);
|
||||
};
|
||||
if let Ok(face) = ttf_parser::Face::parse(&data, 0) {
|
||||
(
|
||||
face.is_italic() || face.italic_angle().abs() >= 4.0,
|
||||
face.is_bold(),
|
||||
)
|
||||
} else if let Some(name) = cff_font_name(&data) {
|
||||
// FontFile3 is bare CFF (no sfnt container) — ttf_parser
|
||||
// can't open it, but the CFF Name INDEX keeps the real
|
||||
// PostScript name ("XXXXXX+Amplitude-LightItalic") even
|
||||
// when the descriptor was rewritten to claim upright.
|
||||
(
|
||||
crate::text_utils::is_italic_font(&name),
|
||||
crate::text_utils::is_bold_font(&name),
|
||||
)
|
||||
} else {
|
||||
(false, false)
|
||||
}
|
||||
}
|
||||
|
||||
/// First PostScript name from a bare CFF font's Name INDEX (CFF spec §7).
|
||||
fn cff_font_name(data: &[u8]) -> Option<String> {
|
||||
// Header: major(1) minor(1) hdrSize(1) offSize(1); major must be 1.
|
||||
if data.len() < 4 || data[0] != 1 {
|
||||
return None;
|
||||
}
|
||||
let hdr_size = data[2] as usize;
|
||||
// Name INDEX: count(u16) offSize(u8) offsets[count+1] data
|
||||
let count = u16::from_be_bytes([*data.get(hdr_size)?, *data.get(hdr_size + 1)?]) as usize;
|
||||
if count == 0 {
|
||||
return None;
|
||||
}
|
||||
let off_size = *data.get(hdr_size + 2)? as usize;
|
||||
if !(1..=4).contains(&off_size) {
|
||||
return None;
|
||||
}
|
||||
let read_offset = |idx: usize| -> Option<usize> {
|
||||
let at = hdr_size + 3 + idx * off_size;
|
||||
let bytes = data.get(at..at + off_size)?;
|
||||
let mut v = 0usize;
|
||||
for b in bytes {
|
||||
v = (v << 8) | *b as usize;
|
||||
}
|
||||
Some(v)
|
||||
};
|
||||
let start = read_offset(0)?;
|
||||
let end = read_offset(1)?;
|
||||
if start == 0 || end < start {
|
||||
return None;
|
||||
}
|
||||
// Offsets are 1-based from the byte before the object data.
|
||||
let objects_base = hdr_size + 3 + (count + 1) * off_size - 1;
|
||||
let name = data.get(objects_base + start..objects_base + end)?;
|
||||
Some(String::from_utf8_lossy(name).to_string())
|
||||
}
|
||||
|
||||
/// FontFile2/FontFile3 stream reference from a FontDescriptor.
|
||||
fn font_file_ref(descriptor: &lopdf::Dictionary) -> Option<ObjectId> {
|
||||
descriptor
|
||||
.get(b"FontFile2")
|
||||
.ok()
|
||||
.and_then(|o| o.as_reference().ok())
|
||||
.or_else(|| {
|
||||
descriptor
|
||||
.get(b"FontFile3")
|
||||
.ok()
|
||||
.and_then(|o| o.as_reference().ok())
|
||||
})
|
||||
}
|
||||
|
||||
/// Decompressed embedded font program bytes.
|
||||
fn font_file_data(doc: &Document, ff_ref: ObjectId) -> Option<Vec<u8>> {
|
||||
let stream = doc
|
||||
.get_object(ff_ref)
|
||||
.and_then(lopdf::Object::as_stream)
|
||||
.ok()?;
|
||||
Some(
|
||||
stream
|
||||
.decompressed_content()
|
||||
.unwrap_or_else(|_| stream.content.clone()),
|
||||
)
|
||||
}
|
||||
|
||||
/// Decode text from a PDF string operand using font CMaps, encodings, and fallbacks.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub(crate) fn extract_text_from_operand(
|
||||
@@ -921,8 +725,6 @@ pub(crate) fn extract_text_from_operand(
|
||||
let is_type0_cid_font = font_widths
|
||||
.get(current_font)
|
||||
.is_some_and(|info| info.is_cid);
|
||||
let use_cp1252_fallback =
|
||||
should_use_cp1252_single_byte_fallback(base_font_name, is_type0_cid_font);
|
||||
let result = (|| -> Option<String> {
|
||||
if let Object::String(bytes, _) = obj {
|
||||
let mut decode_with_entry = |entry: &crate::tounicode::CMapEntry| -> Option<String> {
|
||||
@@ -953,12 +755,9 @@ pub(crate) fn extract_text_from_operand(
|
||||
return Some(ch.to_string());
|
||||
}
|
||||
}
|
||||
// 4. Printable single-byte fallback
|
||||
// 4. Printable ASCII/Latin-1 fallback
|
||||
if b >= 0x20 {
|
||||
return Some(
|
||||
decode_single_byte_fallback_char(b, use_cp1252_fallback)
|
||||
.to_string(),
|
||||
);
|
||||
return Some((b as char).to_string());
|
||||
}
|
||||
None
|
||||
})
|
||||
@@ -1059,13 +858,6 @@ pub(crate) fn extract_text_from_operand(
|
||||
// unmapped. Don't fall through to text-interpretation fallbacks
|
||||
// (Latin-1, UTF-16, etc.) which would misinterpret CID bytes as
|
||||
// character codes (e.g. CID 0x01A9 → Latin-1 "©").
|
||||
if is_type0_cid_font && bytes.iter().any(|&b| b > 0x7F) {
|
||||
// 2-byte CIDs (Identity-H) are by far the common case; for
|
||||
// an odd byte count we still emit at least one marker so
|
||||
// detection downstream fires.
|
||||
let cid_count = (bytes.len() / 2).max(1);
|
||||
return Some("\u{FFFD}".repeat(cid_count));
|
||||
}
|
||||
|
||||
// Try our custom encoding map from Differences arrays.
|
||||
// The Differences array overrides specific codes in a base encoding (typically
|
||||
@@ -1081,9 +873,8 @@ pub(crate) fn extract_text_from_operand(
|
||||
Some(ch)
|
||||
} else if b >= 0x20 {
|
||||
// Base encoding fallback for printable bytes.
|
||||
// Most PDFs with simple fonts use WinAnsi/PDFDocEncoding
|
||||
// semantics, not ISO-8859-1 C1 controls.
|
||||
Some(decode_single_byte_fallback_char(b, use_cp1252_fallback))
|
||||
// For codes 0x20-0x7E this matches all standard PDF encodings.
|
||||
Some(b as char)
|
||||
} else {
|
||||
None // Skip unmapped control characters
|
||||
}
|
||||
@@ -1139,7 +930,6 @@ pub(crate) fn extract_text_from_operand(
|
||||
// Try to decode using cached font encoding from lopdf
|
||||
if let Some(encoding) = encoding_cache.get(current_font) {
|
||||
if let Ok(text) = Document::decode_text(encoding, bytes) {
|
||||
let text = normalize_cp1252_controls(text, use_cp1252_fallback);
|
||||
if text.contains('\u{FFFD}') {
|
||||
debug!(
|
||||
"decode_text produced replacement for font={} bytes_len={}",
|
||||
@@ -1176,119 +966,37 @@ pub(crate) fn extract_text_from_operand(
|
||||
return Some(symbol_text);
|
||||
}
|
||||
|
||||
// Non-CID (Type1 / TrueType / Type3) fonts use single-byte
|
||||
// encodings. In practice the fallback should follow WinAnsi for
|
||||
// 0x80..=0x9F so bytes like 0x92 become smart punctuation instead
|
||||
// of C1 controls that look like CID mojibake.
|
||||
Some(decode_single_byte_fallback(bytes, use_cp1252_fallback))
|
||||
// Latin-1 fallback. Safe ONLY for fonts that use single-byte
|
||||
// encodings — for these, an unmapped byte is a valid character
|
||||
// code in Latin-1/WinAnsi space. CID fonts (Type0 / Identity-H)
|
||||
// emit multi-byte CIDs that aren't characters; per-byte Latin-1
|
||||
// produces mojibake (e.g. 2-byte CID 0xCDD9 → "ÍÙ" for the
|
||||
// production scrape_id 019de78c-... samples).
|
||||
//
|
||||
// For a CID font (has_cmap is set OR a /ToUnicode reference
|
||||
// exists) with any non-ASCII bytes, emit a single U+FFFD per
|
||||
// CID instead. This both replaces the mojibake with a proper
|
||||
// "decode failed" marker AND keeps `detect_encoding_issues`
|
||||
// tripping so the page is flagged for OCR — the existing
|
||||
// garbage-detection path that the high-Latin-1 mojibake used
|
||||
// to satisfy by accident.
|
||||
if is_type0_cid_font && bytes.iter().any(|&b| b > 0x7F) {
|
||||
// 2-byte CIDs (Identity-H) are by far the common case; for
|
||||
// an odd byte count we still emit at least one marker so
|
||||
// detection downstream fires.
|
||||
let cid_count = (bytes.len() / 2).max(1);
|
||||
return Some("\u{FFFD}".repeat(cid_count));
|
||||
}
|
||||
// Pure ASCII bytes round-trip safely (Latin-1 == ASCII for
|
||||
// 0x00..=0x7F), and non-CID (Type1 / TrueType / Type3) fonts
|
||||
// use single-byte encodings where Latin-1 fallback is the
|
||||
// canonical interpretation.
|
||||
Some(bytes.iter().map(|&b| b as char).collect())
|
||||
} else {
|
||||
None
|
||||
}
|
||||
})();
|
||||
result.map(|text| {
|
||||
let text = clean_symbol_pua(text);
|
||||
normalize_cp1252_controls(text, use_cp1252_fallback)
|
||||
})
|
||||
}
|
||||
|
||||
fn decode_single_byte_fallback(bytes: &[u8], use_cp1252_fallback: bool) -> String {
|
||||
bytes
|
||||
.iter()
|
||||
.map(|&b| decode_single_byte_fallback_char(b, use_cp1252_fallback))
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn decode_single_byte_fallback_char(byte: u8, use_cp1252_fallback: bool) -> char {
|
||||
if !use_cp1252_fallback {
|
||||
return byte as char;
|
||||
}
|
||||
|
||||
match byte {
|
||||
0x80 => '\u{20AC}',
|
||||
0x82 => '\u{201A}',
|
||||
0x83 => '\u{0192}',
|
||||
0x84 => '\u{201E}',
|
||||
0x85 => '\u{2026}',
|
||||
0x86 => '\u{2020}',
|
||||
0x87 => '\u{2021}',
|
||||
0x88 => '\u{02C6}',
|
||||
0x89 => '\u{2030}',
|
||||
0x8A => '\u{0160}',
|
||||
0x8B => '\u{2039}',
|
||||
0x8C => '\u{0152}',
|
||||
0x8E => '\u{017D}',
|
||||
0x91 => '\u{2018}',
|
||||
0x92 => '\u{2019}',
|
||||
0x93 => '\u{201C}',
|
||||
0x94 => '\u{201D}',
|
||||
0x95 => '\u{2022}',
|
||||
0x96 => '\u{2013}',
|
||||
0x97 => '\u{2014}',
|
||||
0x98 => '\u{02DC}',
|
||||
0x99 => '\u{2122}',
|
||||
0x9A => '\u{0161}',
|
||||
0x9B => '\u{203A}',
|
||||
0x9C => '\u{0153}',
|
||||
0x9E => '\u{017E}',
|
||||
0x9F => '\u{0178}',
|
||||
_ => byte as char,
|
||||
}
|
||||
}
|
||||
|
||||
fn normalize_cp1252_controls(text: String, use_cp1252_fallback: bool) -> String {
|
||||
if !use_cp1252_fallback {
|
||||
return text;
|
||||
}
|
||||
if !text
|
||||
.chars()
|
||||
.any(|ch| ('\u{0080}'..='\u{009F}').contains(&ch))
|
||||
{
|
||||
return text;
|
||||
}
|
||||
|
||||
text.chars()
|
||||
.map(|ch| {
|
||||
if ('\u{0080}'..='\u{009F}').contains(&ch) {
|
||||
decode_single_byte_fallback_char(ch as u8, true)
|
||||
} else {
|
||||
ch
|
||||
}
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn should_use_cp1252_single_byte_fallback(
|
||||
base_font_name: Option<&str>,
|
||||
is_type0_cid_font: bool,
|
||||
) -> bool {
|
||||
if is_type0_cid_font {
|
||||
return false;
|
||||
}
|
||||
|
||||
let Some(base_font_name) = base_font_name else {
|
||||
return true;
|
||||
};
|
||||
let font_name = base_font_name
|
||||
.rsplit_once('+')
|
||||
.map_or(base_font_name, |(_, stripped)| stripped)
|
||||
.to_ascii_lowercase();
|
||||
|
||||
// TeX/Computer Modern and math/symbol fonts often place ligatures or
|
||||
// symbols in the C1 byte range. Treating those bytes as Windows-1252 makes
|
||||
// words like "deficiente" become "de…ciente" and "fluid" become "‡uid".
|
||||
let non_cp1252_prefixes = [
|
||||
"cmr", "cmb", "cmmi", "cmsy", "cmex", "cmtt", "cmss", "cmti", "ecrm", "ecbx", "ecti",
|
||||
"tcrm", "tctt", "msam", "msbm", "ttdc",
|
||||
];
|
||||
if non_cp1252_prefixes
|
||||
.iter()
|
||||
.any(|prefix| font_name.starts_with(prefix))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
let non_cp1252_names = ["math", "symbol", "dingbat", "emoji"];
|
||||
!non_cp1252_names.iter().any(|name| font_name.contains(name))
|
||||
result.map(clean_symbol_pua)
|
||||
}
|
||||
|
||||
/// Replace PUA characters in the F000-F0FF range with standard Unicode equivalents.
|
||||
@@ -1410,7 +1118,6 @@ fn score_text(text: &str) -> i32 {
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use lopdf::dictionary;
|
||||
|
||||
fn make_font_info(widths: &[(u16, u16)], default_width: u16, is_cid: bool) -> FontWidthInfo {
|
||||
FontWidthInfo {
|
||||
@@ -1427,172 +1134,6 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
fn doc_with_descriptor(descriptor: lopdf::Dictionary) -> (Document, lopdf::Dictionary) {
|
||||
let mut doc = Document::with_version("1.4");
|
||||
let desc_id = doc.add_object(descriptor);
|
||||
let font_dict = dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "TrueType",
|
||||
"BaseFont" => "Tc1",
|
||||
"FontDescriptor" => desc_id,
|
||||
};
|
||||
(doc, font_dict)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn descriptor_italic_angle_sets_italic() {
|
||||
// Subset font with an opaque BaseFont name ("Tc1") — the name
|
||||
// heuristic sees nothing, the descriptor carries the truth.
|
||||
let (doc, font_dict) = doc_with_descriptor(dictionary! {
|
||||
"Type" => "FontDescriptor",
|
||||
"FontName" => "Tc1",
|
||||
"ItalicAngle" => -12,
|
||||
"Flags" => 32,
|
||||
});
|
||||
assert_eq!(
|
||||
descriptor_style_flags(&doc, &font_dict, &mut FontStyleCache::new()),
|
||||
(true, false)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn descriptor_italic_flag_bit_sets_italic() {
|
||||
let (doc, font_dict) = doc_with_descriptor(dictionary! {
|
||||
"Type" => "FontDescriptor",
|
||||
"FontName" => "Tc1",
|
||||
"ItalicAngle" => 0,
|
||||
"Flags" => 64, // bit 7: Italic
|
||||
});
|
||||
assert_eq!(
|
||||
descriptor_style_flags(&doc, &font_dict, &mut FontStyleCache::new()),
|
||||
(true, false)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn descriptor_force_bold_flag_sets_bold() {
|
||||
let (doc, font_dict) = doc_with_descriptor(dictionary! {
|
||||
"Type" => "FontDescriptor",
|
||||
"FontName" => "Tc1",
|
||||
"ItalicAngle" => 0,
|
||||
"Flags" => 1 << 18, // ForceBold
|
||||
});
|
||||
assert_eq!(
|
||||
descriptor_style_flags(&doc, &font_dict, &mut FontStyleCache::new()),
|
||||
(false, true)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tiny_italic_angle_is_not_italic() {
|
||||
// A token 1-degree slant is optical correction, not italic.
|
||||
let (doc, font_dict) = doc_with_descriptor(dictionary! {
|
||||
"Type" => "FontDescriptor",
|
||||
"FontName" => "Tc1",
|
||||
"ItalicAngle" => lopdf::Object::Real(-1.0),
|
||||
"Flags" => 32,
|
||||
});
|
||||
assert_eq!(
|
||||
descriptor_style_flags(&doc, &font_dict, &mut FontStyleCache::new()),
|
||||
(false, false)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn missing_descriptor_yields_no_flags() {
|
||||
let doc = Document::with_version("1.4");
|
||||
let font_dict = dictionary! { "Type" => "Font", "BaseFont" => "Tc1" };
|
||||
assert_eq!(
|
||||
descriptor_style_flags(&doc, &font_dict, &mut FontStyleCache::new()),
|
||||
(false, false)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn type0_descendant_descriptor_is_resolved() {
|
||||
let mut doc = Document::with_version("1.4");
|
||||
let desc_id = doc.add_object(dictionary! {
|
||||
"Type" => "FontDescriptor",
|
||||
"FontName" => "ABCDEF+F1",
|
||||
"ItalicAngle" => -15,
|
||||
});
|
||||
let cid_id = doc.add_object(dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "CIDFontType2",
|
||||
"FontDescriptor" => desc_id,
|
||||
});
|
||||
let font_dict = dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "Type0",
|
||||
"BaseFont" => "ABCDEF+F1",
|
||||
"DescendantFonts" => vec![lopdf::Object::Reference(cid_id)],
|
||||
};
|
||||
assert_eq!(
|
||||
descriptor_style_flags(&doc, &font_dict, &mut FontStyleCache::new()),
|
||||
(true, false)
|
||||
);
|
||||
}
|
||||
|
||||
/// Bare CFF: header + Name INDEX only — enough for `cff_font_name`.
|
||||
fn bare_cff_with_name(name: &str) -> Vec<u8> {
|
||||
let mut data = vec![1, 0, 4, 1]; // major, minor, hdrSize, offSize
|
||||
data.extend_from_slice(&1u16.to_be_bytes()); // Name INDEX count
|
||||
data.push(1); // offSize
|
||||
data.push(1); // offset of first name
|
||||
data.push(1 + name.len() as u8); // offset past last name
|
||||
data.extend_from_slice(name.as_bytes());
|
||||
data
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn embedded_font_style_is_cached_by_font_file_object() {
|
||||
use lopdf::{Object, Stream};
|
||||
|
||||
let mut doc = Document::with_version("1.4");
|
||||
let ff_id = doc.add_object(Object::Stream(Stream::new(
|
||||
dictionary! {},
|
||||
bare_cff_with_name("ABCDEF+Test-BoldItalic"),
|
||||
)));
|
||||
let desc_id = doc.add_object(dictionary! {
|
||||
"Type" => "FontDescriptor",
|
||||
"FontName" => "ABCDEF+Test-BoldItalic",
|
||||
"ItalicAngle" => 0,
|
||||
"Flags" => 32,
|
||||
"FontFile3" => ff_id,
|
||||
});
|
||||
let font_dict = dictionary! {
|
||||
"Type" => "Font",
|
||||
"Subtype" => "Type1",
|
||||
"BaseFont" => "Tc1",
|
||||
"FontDescriptor" => desc_id,
|
||||
};
|
||||
|
||||
let mut cache = FontStyleCache::new();
|
||||
assert_eq!(
|
||||
descriptor_style_flags(&doc, &font_dict, &mut cache),
|
||||
(true, true)
|
||||
);
|
||||
assert_eq!(cache.by_font_file.len(), 1);
|
||||
|
||||
// Replace the font program with garbage: a repeat call must serve
|
||||
// the memo instead of re-reading the stream — repeated per-page
|
||||
// decompression is exactly what the cache exists to avoid.
|
||||
doc.objects.insert(
|
||||
ff_id,
|
||||
Object::Stream(Stream::new(dictionary! {}, vec![0u8; 4])),
|
||||
);
|
||||
assert_eq!(
|
||||
descriptor_style_flags(&doc, &font_dict, &mut cache),
|
||||
(true, true)
|
||||
);
|
||||
// A cold cache parses the (now garbage) stream, proving the warm
|
||||
// call above answered from the memo.
|
||||
assert_eq!(
|
||||
descriptor_style_flags(&doc, &font_dict, &mut FontStyleCache::new()),
|
||||
(false, false)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn compute_string_width_ts_no_tc_tw() {
|
||||
// Without Tc/Tw (both 0), width = glyph widths only
|
||||
@@ -1701,51 +1242,6 @@ mod tests {
|
||||
assert!(score_text(good) > score_text(bad));
|
||||
}
|
||||
|
||||
fn doc_with_private_differences() -> (Document, lopdf::ObjectId) {
|
||||
let mut doc = Document::with_version("1.7");
|
||||
let encoding_id = doc.add_object(dictionary! {
|
||||
"Differences" => Object::Array(vec![
|
||||
Object::Integer(0x88),
|
||||
Object::Name(b"g431".to_vec()),
|
||||
Object::Name(b"fi".to_vec()),
|
||||
Object::Integer(0xAD),
|
||||
Object::Name(b"fl".to_vec()),
|
||||
]),
|
||||
});
|
||||
|
||||
(doc, encoding_id)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn aptos_private_g431_maps_to_ff_ligature() {
|
||||
let (doc, encoding_id) = doc_with_private_differences();
|
||||
let font_dict = dictionary! {
|
||||
"BaseFont" => Object::Name(b"NJEQOD+Aptos".to_vec()),
|
||||
"Encoding" => Object::Reference(encoding_id),
|
||||
};
|
||||
|
||||
let result = parse_font_encoding(&doc, &font_dict).expect("encoding should parse");
|
||||
|
||||
assert_eq!(result.map.get(&0x88u8), Some(&'\u{FB00}'));
|
||||
assert_eq!(result.map.get(&0x89u8), Some(&'\u{FB01}'));
|
||||
assert_eq!(result.map.get(&0xADu8), Some(&'\u{FB02}'));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn private_g431_does_not_map_for_unrelated_fonts() {
|
||||
let (doc, encoding_id) = doc_with_private_differences();
|
||||
let font_dict = dictionary! {
|
||||
"BaseFont" => Object::Name(b"ABCDEF+OtherFont".to_vec()),
|
||||
"Encoding" => Object::Reference(encoding_id),
|
||||
};
|
||||
|
||||
let result = parse_font_encoding(&doc, &font_dict).expect("encoding should parse");
|
||||
|
||||
assert!(!result.map.contains_key(&0x88u8));
|
||||
assert_eq!(result.map.get(&0x89u8), Some(&'\u{FB01}'));
|
||||
assert_eq!(result.map.get(&0xADu8), Some(&'\u{FB02}'));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cid_font_with_unparseable_cmap_does_not_emit_latin1_mojibake() {
|
||||
// Type0/CID font (font_widths reports `is_cid=true`) where the
|
||||
@@ -1797,15 +1293,15 @@ mod tests {
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn simple_font_single_byte_fallback_passes_high_bytes_through() {
|
||||
fn simple_font_latin1_fallback_passes_high_bytes_through() {
|
||||
// A Type1/TrueType simple font (is_cid=false) with a `/ToUnicode`
|
||||
// reference but no usable CMap and no `/Differences` map.
|
||||
// Per-byte fallback is the canonical interpretation here — these
|
||||
// bytes are character codes, not CIDs. The CID guard must NOT strip
|
||||
// them. Reproduces the false positive that an earlier version of the
|
||||
// guard introduced for fonts in PDFs like pdf-evals/Navigating-
|
||||
// Artificial-Intelligence-..., where bytes like 0xB6 are legitimate
|
||||
// single-byte character codes.
|
||||
// Per-byte Latin-1 IS the canonical interpretation here — these
|
||||
// bytes are character codes, not CIDs. The CID guard must NOT
|
||||
// strip them. Reproduces the false positive that an earlier
|
||||
// version of the guard introduced for fonts in PDFs like
|
||||
// pdf-evals/Navigating-Artificial-Intelligence-..., where bytes
|
||||
// like 0xB6 are legitimate Latin-1 character codes.
|
||||
let bytes = vec![0x24_u8, 0x47, 0xB6, 0x56]; // "$G¶V"
|
||||
let obj = Object::String(bytes, lopdf::StringFormat::Hexadecimal);
|
||||
|
||||
@@ -1838,62 +1334,4 @@ mod tests {
|
||||
"simple font fallback must not stamp FFFD over legitimate bytes: {text:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn simple_font_single_byte_fallback_maps_cp1252_punctuation() {
|
||||
let bytes = vec![b'l', 0x92_u8, b'a', b'c', b'a', b'd'];
|
||||
let obj = Object::String(bytes, lopdf::StringFormat::Hexadecimal);
|
||||
|
||||
let font_cmaps = FontCMaps::default();
|
||||
let font_tounicode_refs: HashMap<String, u32> = HashMap::new();
|
||||
let inline_cmaps = HashMap::new();
|
||||
let font_encodings: PageFontEncodings = HashMap::new();
|
||||
let encoding_cache: HashMap<String, Encoding<'_>> = HashMap::new();
|
||||
let mut decisions = CMapDecisionCache::new();
|
||||
let font_widths: PageFontWidths = HashMap::new();
|
||||
|
||||
let text = extract_text_from_operand(
|
||||
&obj,
|
||||
"F1",
|
||||
None,
|
||||
&font_cmaps,
|
||||
&font_tounicode_refs,
|
||||
&inline_cmaps,
|
||||
&font_encodings,
|
||||
&encoding_cache,
|
||||
&mut decisions,
|
||||
&font_widths,
|
||||
)
|
||||
.expect("simple font should decode CP1252 punctuation");
|
||||
|
||||
assert_eq!(text, "l’acad");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cached_encoding_decode_normalizes_cp1252_controls() {
|
||||
let text = normalize_cp1252_controls("d\u{92}un \u{96} test".to_string(), true);
|
||||
assert_eq!(text, "d’un – test");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tex_font_decode_keeps_c1_ligature_bytes_unmodified() {
|
||||
let text = normalize_cp1252_controls("de\u{85}ciente \u{87}uid".to_string(), false);
|
||||
assert_eq!(text, "de\u{85}ciente \u{87}uid");
|
||||
assert!(!should_use_cp1252_single_byte_fallback(
|
||||
Some("TTdcr10"),
|
||||
false
|
||||
));
|
||||
assert!(!should_use_cp1252_single_byte_fallback(
|
||||
Some("cmr10"),
|
||||
false
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn winansi_text_font_uses_cp1252_fallback() {
|
||||
assert!(should_use_cp1252_single_byte_fallback(
|
||||
Some("BJPQNQ+Times-Roman"),
|
||||
false
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1234,7 +1234,11 @@ pub(crate) fn group_into_lines_with_thresholds(
|
||||
ci,
|
||||
item.x,
|
||||
item.y,
|
||||
super::trace_text_preview(&item.text, 60)
|
||||
if item.text.len() > 60 {
|
||||
&item.text[..60]
|
||||
} else {
|
||||
&item.text
|
||||
}
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -1494,8 +1498,6 @@ mod tests {
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
@@ -1625,8 +1627,6 @@ mod tests {
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
});
|
||||
|
||||
@@ -78,8 +78,6 @@ pub fn extract_page_links(doc: &Document, page_id: ObjectId, page_num: u32) -> V
|
||||
page: page_num,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Link(url),
|
||||
mcid: None,
|
||||
});
|
||||
@@ -318,8 +316,6 @@ pub(crate) fn walk_form_fields(
|
||||
page: page_num,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::FormField,
|
||||
mcid: None,
|
||||
});
|
||||
|
||||
+22
-518
@@ -6,12 +6,11 @@ pub(crate) mod content_stream;
|
||||
mod fonts;
|
||||
mod layout;
|
||||
mod links;
|
||||
pub(crate) mod underline;
|
||||
mod xobjects;
|
||||
|
||||
use crate::text_utils::is_rtl_text;
|
||||
use crate::tounicode::FontCMaps;
|
||||
use crate::types::{PageExtraction, PdfLine, PdfRect, TextItem};
|
||||
use crate::types::{PageExtraction, TextItem};
|
||||
use crate::PdfError;
|
||||
use log::debug;
|
||||
use lopdf::{Document, Object, ObjectId};
|
||||
@@ -24,7 +23,6 @@ use links::{extract_form_fields, extract_page_links};
|
||||
// Re-export public types so existing `crate::extractor::X` paths keep working.
|
||||
pub use crate::text_utils::{is_bold_font, is_italic_font};
|
||||
pub use crate::types::{ItemType, TextLine};
|
||||
pub(crate) use fonts::FontStyleCache;
|
||||
pub(crate) use layout::detect_columns;
|
||||
pub use layout::group_into_lines;
|
||||
pub(crate) use layout::group_into_lines_with_thresholds;
|
||||
@@ -35,13 +33,6 @@ pub(crate) use layout::ColumnRegion;
|
||||
// Public API
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub(crate) fn trace_text_preview(text: &str, max_chars: usize) -> &str {
|
||||
match text.char_indices().nth(max_chars) {
|
||||
Some((idx, _)) => &text[..idx],
|
||||
None => text,
|
||||
}
|
||||
}
|
||||
|
||||
/// Extract text from PDF file as plain string
|
||||
pub fn extract_text<P: AsRef<Path>>(path: P) -> Result<String, PdfError> {
|
||||
crate::validate_pdf_file(&path)?;
|
||||
@@ -162,9 +153,6 @@ fn extract_positioned_text_impl(
|
||||
let mut all_lines = Vec::new();
|
||||
let mut page_thresholds: PageThresholds = HashMap::new();
|
||||
let mut gid_encoded_pages: HashSet<u32> = HashSet::new();
|
||||
// Embedded-font style flags are document-scoped: the same font program
|
||||
// is shared across pages, so parse it once, not once per page.
|
||||
let mut style_cache = FontStyleCache::new();
|
||||
|
||||
// Build page ObjectId → page number map for form field extraction
|
||||
let page_id_to_num: HashMap<ObjectId, u32> =
|
||||
@@ -176,14 +164,8 @@ fn extract_positioned_text_impl(
|
||||
continue;
|
||||
}
|
||||
}
|
||||
let ((mut items, rects, lines), has_gid_fonts, _coords_rotated) = extract_page_text_items(
|
||||
doc,
|
||||
page_id,
|
||||
*page_num,
|
||||
font_cmaps,
|
||||
include_invisible,
|
||||
&mut style_cache,
|
||||
)?;
|
||||
let ((mut items, rects, lines), has_gid_fonts, _coords_rotated) =
|
||||
extract_page_text_items(doc, page_id, *page_num, font_cmaps, include_invisible)?;
|
||||
if has_gid_fonts {
|
||||
gid_encoded_pages.insert(*page_num);
|
||||
}
|
||||
@@ -191,7 +173,6 @@ fn extract_positioned_text_impl(
|
||||
if threshold > 0.10 {
|
||||
page_thresholds.insert(*page_num, threshold);
|
||||
}
|
||||
suppress_table_underlines(&mut items, &rects, &lines, *page_num);
|
||||
debug!(
|
||||
"page {}: {} text items, {} rects, {} lines{}",
|
||||
page_num,
|
||||
@@ -214,7 +195,11 @@ fn extract_positioned_text_impl(
|
||||
item.width,
|
||||
item.font_size,
|
||||
item.font,
|
||||
trace_text_preview(&item.text, 80)
|
||||
if item.text.len() > 80 {
|
||||
&item.text[..80]
|
||||
} else {
|
||||
&item.text
|
||||
}
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -238,42 +223,6 @@ fn extract_positioned_text_impl(
|
||||
))
|
||||
}
|
||||
|
||||
fn suppress_table_underlines(
|
||||
items: &mut [TextItem],
|
||||
rects: &[PdfRect],
|
||||
lines: &[PdfLine],
|
||||
page: u32,
|
||||
) {
|
||||
if !items
|
||||
.iter()
|
||||
.any(|item| item.is_underline || item.is_strikeout)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
let mut table_item_indices: HashSet<usize> = HashSet::new();
|
||||
|
||||
if !rects.is_empty() {
|
||||
let (rect_tables, _) = crate::tables::detect_tables_from_rects(items, rects, page);
|
||||
for table in rect_tables {
|
||||
table_item_indices.extend(table.item_indices);
|
||||
}
|
||||
}
|
||||
|
||||
if !lines.is_empty() {
|
||||
for table in crate::tables::detect_tables_from_lines(items, lines, page) {
|
||||
table_item_indices.extend(table.item_indices);
|
||||
}
|
||||
}
|
||||
|
||||
for index in table_item_indices {
|
||||
if let Some(item) = items.get_mut(index) {
|
||||
item.is_underline = false;
|
||||
item.is_strikeout = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Shared helpers (used by submodules via `super::`)
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -400,133 +349,6 @@ fn effective_merge_width(item: &TextItem) -> f32 {
|
||||
}
|
||||
}
|
||||
|
||||
fn is_standalone_bullet_text(text: &str) -> bool {
|
||||
matches!(text.trim(), "•" | "○" | "●" | "◦")
|
||||
}
|
||||
|
||||
fn first_text_char(text: &str) -> Option<char> {
|
||||
text.trim_start().chars().next()
|
||||
}
|
||||
|
||||
fn is_short_alpha_fragment(text: &str) -> bool {
|
||||
let trimmed = text.trim();
|
||||
let char_count = trimmed.chars().count();
|
||||
(1..=4).contains(&char_count) && trimmed.chars().all(char::is_alphabetic)
|
||||
}
|
||||
|
||||
fn has_phrase_continuation_shape(text: &str) -> bool {
|
||||
let trimmed = text.trim_start();
|
||||
trimmed
|
||||
.chars()
|
||||
.take(24)
|
||||
.any(|ch| ch.is_whitespace() || matches!(ch, '-'))
|
||||
}
|
||||
|
||||
fn should_preserve_overlapping_stream_order(group: &[&TextItem]) -> bool {
|
||||
if group.len() < 3 {
|
||||
return false;
|
||||
}
|
||||
|
||||
let Some(first) = group.iter().find(|item| !item.text.trim().is_empty()) else {
|
||||
return false;
|
||||
};
|
||||
if group.iter().all(|item| item.mcid.is_none()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
let mut nonempty_count = 0;
|
||||
let mut saw_backtrack = false;
|
||||
let mut nonspace_chars = 0;
|
||||
let mut math_symbol_chars = 0;
|
||||
let mut max_font_size = first.font_size;
|
||||
|
||||
for item in group {
|
||||
if !item.text.trim().is_empty() {
|
||||
nonempty_count += 1;
|
||||
}
|
||||
if (item.font_size - first.font_size).abs() > first.font_size * 0.25 {
|
||||
return false;
|
||||
}
|
||||
max_font_size = max_font_size.max(item.font_size);
|
||||
for ch in item.text.chars().filter(|ch| !ch.is_whitespace()) {
|
||||
nonspace_chars += 1;
|
||||
if matches!(
|
||||
ch,
|
||||
'*' | 'ˆ' | '^' | '=' | '+' | '_' | '[' | ']' | '{' | '}' | '|' | '<' | '>'
|
||||
) {
|
||||
math_symbol_chars += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if nonempty_count < 2 {
|
||||
return false;
|
||||
}
|
||||
if nonspace_chars > 0 && math_symbol_chars * 4 > nonspace_chars {
|
||||
return false;
|
||||
}
|
||||
|
||||
let mut sorted_by_x = group.to_vec();
|
||||
sorted_by_x.sort_by(|a, b| a.x.total_cmp(&b.x));
|
||||
let cluster_start = sorted_by_x[0].x;
|
||||
let mut cluster_end = cluster_start + effective_merge_width(sorted_by_x[0]);
|
||||
for item in sorted_by_x.iter().skip(1) {
|
||||
let gap = item.x - cluster_end;
|
||||
if gap > max_font_size * 2.5 {
|
||||
return false;
|
||||
}
|
||||
cluster_end = cluster_end.max(item.x + effective_merge_width(item));
|
||||
}
|
||||
if cluster_end - cluster_start > max_font_size * 36.0 {
|
||||
return false;
|
||||
}
|
||||
|
||||
for index in 0..group.len() - 1 {
|
||||
let previous = group[index];
|
||||
let next = group[index + 1];
|
||||
let font_size = previous.font_size.max(next.font_size);
|
||||
let backtrack_threshold = font_size * 0.25;
|
||||
let previous_start = previous.x;
|
||||
let next_start = next.x;
|
||||
let next_end = next.x + effective_merge_width(next);
|
||||
if next_start < previous_start - backtrack_threshold
|
||||
&& next_end > previous_start + backtrack_threshold
|
||||
{
|
||||
let has_near_prefix = group[..=index].iter().rev().take(4).any(|item| {
|
||||
is_short_alpha_fragment(&item.text)
|
||||
&& item.x >= next_start - font_size * 0.5
|
||||
&& item.x <= next_start + font_size * 4.0
|
||||
});
|
||||
let starts_lowercase = first_text_char(&next.text).is_some_and(char::is_lowercase);
|
||||
let phrase_continuation = has_phrase_continuation_shape(&next.text);
|
||||
let has_near_bullet = group[..=index]
|
||||
.iter()
|
||||
.position(|item| {
|
||||
is_standalone_bullet_text(&item.text) && next_start <= item.x + font_size * 3.0
|
||||
})
|
||||
.is_some_and(|bullet_index| {
|
||||
if bullet_index >= index {
|
||||
return false;
|
||||
}
|
||||
group[bullet_index + 1..=index]
|
||||
.iter()
|
||||
.rev()
|
||||
.find(|item| !item.text.trim().is_empty())
|
||||
.is_some_and(|item| {
|
||||
item.text.trim().chars().count() <= 8
|
||||
&& has_phrase_continuation_shape(&next.text)
|
||||
})
|
||||
});
|
||||
if (has_near_prefix && starts_lowercase && phrase_continuation) || has_near_bullet {
|
||||
saw_backtrack = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
saw_backtrack
|
||||
}
|
||||
|
||||
pub(crate) fn merge_text_items(items: Vec<TextItem>) -> Vec<TextItem> {
|
||||
if items.is_empty() {
|
||||
return items;
|
||||
@@ -547,32 +369,28 @@ pub(crate) fn merge_text_items(items: Vec<TextItem>) -> Vec<TextItem> {
|
||||
}
|
||||
}
|
||||
|
||||
let mut ordered_line_groups: Vec<(u32, f32, Vec<&TextItem>, bool)> = Vec::new();
|
||||
|
||||
// Sort each group by X position (direction-aware), except for lines whose
|
||||
// content stream intentionally backtracks to overlay ActualText fragments.
|
||||
for (page, y, mut group) in line_groups {
|
||||
// Sort each group by X position (direction-aware)
|
||||
for (_, _, group) in &mut line_groups {
|
||||
let rtl = is_rtl_text(group.iter().map(|i| &i.text));
|
||||
let preserve_stream_order = !rtl && should_preserve_overlapping_stream_order(&group);
|
||||
if rtl {
|
||||
group.sort_by(|a, b| b.x.total_cmp(&a.x));
|
||||
} else if !preserve_stream_order {
|
||||
} else {
|
||||
group.sort_by(|a, b| a.x.total_cmp(&b.x));
|
||||
}
|
||||
ordered_line_groups.push((page, y, group, preserve_stream_order));
|
||||
}
|
||||
|
||||
// Sort groups by page then Y descending (top of page first)
|
||||
ordered_line_groups.sort_by(|a, b| a.0.cmp(&b.0).then_with(|| b.1.total_cmp(&a.1)));
|
||||
line_groups.sort_by(|a, b| a.0.cmp(&b.0).then_with(|| b.1.total_cmp(&a.1)));
|
||||
|
||||
let mut merged = Vec::new();
|
||||
|
||||
for (_, _, group, preserve_stream_order) in &ordered_line_groups {
|
||||
for (_, _, group) in &line_groups {
|
||||
let mut i = 0;
|
||||
while i < group.len() {
|
||||
let first = group[i];
|
||||
let mut text = first.text.clone();
|
||||
let mut end_x = first.x + effective_merge_width(first);
|
||||
let x_gap_max = first.font_size * 0.5;
|
||||
|
||||
let mut j = i + 1;
|
||||
while j < group.len() {
|
||||
@@ -581,29 +399,11 @@ pub(crate) fn merge_text_items(items: Vec<TextItem>) -> Vec<TextItem> {
|
||||
if (next.font_size - first.font_size).abs() > first.font_size * 0.20 {
|
||||
break;
|
||||
}
|
||||
// Never merge across style boundaries: the merged item
|
||||
// carries `first`'s flags, so absorbing a styled run into a
|
||||
// plain neighbor (or vice versa) silently erases the styling
|
||||
// that markdown emission and downstream inline-styling need —
|
||||
// and OR-ing underline instead would stretch `<u>` spans over
|
||||
// neighboring plain text.
|
||||
if next.is_bold != first.is_bold
|
||||
|| next.is_italic != first.is_italic
|
||||
|| next.is_underline != first.is_underline
|
||||
|| next.is_strikeout != first.is_strikeout
|
||||
{
|
||||
break;
|
||||
}
|
||||
let gap = next.x - end_x;
|
||||
let x_gap_max = if *preserve_stream_order && is_standalone_bullet_text(&text) {
|
||||
first.font_size * 1.2
|
||||
} else {
|
||||
first.font_size * 0.5
|
||||
};
|
||||
if gap > x_gap_max {
|
||||
break;
|
||||
}
|
||||
if gap < -first.font_size * 0.5 && !preserve_stream_order {
|
||||
if gap < -first.font_size * 0.5 {
|
||||
break;
|
||||
}
|
||||
// Insert space at word boundaries.
|
||||
@@ -625,19 +425,11 @@ pub(crate) fn merge_text_items(items: Vec<TextItem>) -> Vec<TextItem> {
|
||||
first.font_size * 0.08
|
||||
}
|
||||
};
|
||||
let needs_bullet_space = *preserve_stream_order
|
||||
&& is_standalone_bullet_text(&text)
|
||||
&& !next.text.trim().is_empty();
|
||||
if needs_bullet_space || gap > threshold {
|
||||
if gap > threshold {
|
||||
text.push(' ');
|
||||
}
|
||||
text.push_str(&next.text);
|
||||
let next_end = next.x + effective_merge_width(next);
|
||||
end_x = if *preserve_stream_order {
|
||||
end_x.max(next_end)
|
||||
} else {
|
||||
next_end
|
||||
};
|
||||
end_x = next.x + effective_merge_width(next);
|
||||
j += 1;
|
||||
}
|
||||
|
||||
@@ -652,8 +444,6 @@ pub(crate) fn merge_text_items(items: Vec<TextItem>) -> Vec<TextItem> {
|
||||
page: first.page,
|
||||
is_bold: first.is_bold,
|
||||
is_italic: first.is_italic,
|
||||
is_underline: first.is_underline,
|
||||
is_strikeout: first.is_strikeout,
|
||||
item_type: first.item_type.clone(),
|
||||
mcid: first.mcid,
|
||||
});
|
||||
@@ -731,24 +521,12 @@ pub(crate) fn merge_subscript_items(items: Vec<TextItem>) -> Vec<TextItem> {
|
||||
.chars()
|
||||
.last()
|
||||
.is_some_and(|c| c.is_alphabetic());
|
||||
let same_marks = parent.is_underline == item.is_underline
|
||||
&& parent.is_strikeout == item.is_strikeout;
|
||||
if parent.font_size >= sub_threshold && ends_with_letter && same_marks {
|
||||
if parent.font_size >= sub_threshold && ends_with_letter {
|
||||
let parent_right = parent.x + parent.width;
|
||||
let gap = item.x - parent_right;
|
||||
// Subscripts must be tightly adjacent (within ~1pt)
|
||||
if gap < parent.font_size * 0.2 && gap > -parent.font_size * 0.3 {
|
||||
// Preserve the script when absorbing it: map the
|
||||
// digits to Unicode sub/superscript forms so the
|
||||
// raised/lowered rendering survives in extracted
|
||||
// text ("H"+"2" → "H₂", "word"+"2" → "word²").
|
||||
// NFKC/NFKD normalization folds these back to
|
||||
// plain digits, so text matching downstream is
|
||||
// unaffected. Direction from the baseline offset
|
||||
// (y-up here): raised → superscript (footnote
|
||||
// refs), lowered/level → subscript (chemistry).
|
||||
let raised = item.y > parent.y + parent.font_size * 0.1;
|
||||
parent.text.push_str(&map_script_digits(&item.text, raised));
|
||||
parent.text.push_str(&item.text);
|
||||
parent.width = (item.x + item.width) - parent.x;
|
||||
continue;
|
||||
}
|
||||
@@ -763,21 +541,6 @@ pub(crate) fn merge_subscript_items(items: Vec<TextItem>) -> Vec<TextItem> {
|
||||
result
|
||||
}
|
||||
|
||||
/// Map ASCII digits to their Unicode superscript (`raised`) or subscript
|
||||
/// forms. Callers guarantee digit-only input (see `merge_subscript_items`);
|
||||
/// anything else passes through unchanged.
|
||||
fn map_script_digits(text: &str, raised: bool) -> String {
|
||||
const SUP: [char; 10] = ['⁰', '¹', '²', '³', '⁴', '⁵', '⁶', '⁷', '⁸', '⁹'];
|
||||
const SUB: [char; 10] = ['₀', '₁', '₂', '₃', '₄', '₅', '₆', '₇', '₈', '₉'];
|
||||
text.chars()
|
||||
.map(|c| match c.to_digit(10) {
|
||||
Some(d) if raised => SUP[d as usize],
|
||||
Some(d) => SUB[d as usize],
|
||||
None => c,
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Helper to get f32 from Object
|
||||
pub(crate) fn get_number(obj: &Object) -> Option<f32> {
|
||||
match obj {
|
||||
@@ -791,7 +554,7 @@ pub(crate) fn get_number(obj: &Object) -> Option<f32> {
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::text_utils::{is_cjk_char, is_rtl_char, is_rtl_text, sort_line_items};
|
||||
use crate::types::{ItemType, PdfLine, TextLine};
|
||||
use crate::types::{ItemType, TextLine};
|
||||
use layout::{detect_columns, is_newspaper_layout, ColumnRegion};
|
||||
|
||||
fn make_merge_item(text: &str, x: f32, width: f32) -> TextItem {
|
||||
@@ -806,60 +569,11 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
}
|
||||
|
||||
fn with_mcid(mut item: TextItem) -> TextItem {
|
||||
item.mcid = Some(1);
|
||||
item
|
||||
}
|
||||
|
||||
fn make_line(x1: f32, y1: f32, x2: f32, y2: f32) -> PdfLine {
|
||||
PdfLine {
|
||||
x1,
|
||||
y1,
|
||||
x2,
|
||||
y2,
|
||||
page: 1,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn trace_text_preview_truncates_on_char_boundary() {
|
||||
let text = format!("{}{}tail", "a".repeat(79), '\u{FFFD}');
|
||||
let preview = trace_text_preview(&text, 80);
|
||||
|
||||
assert_eq!(preview.chars().count(), 80);
|
||||
assert!(text.is_char_boundary(preview.len()));
|
||||
assert!(preview.ends_with('\u{FFFD}'));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_items_breaks_at_style_boundaries() {
|
||||
// A styled run adjacent to plain text must stay a separate item —
|
||||
// merging would erase the flags (italic) or stretch the span
|
||||
// (underline) before markdown emission sees them.
|
||||
let mut italic = make_merge_item("emphasis", 150.0, 40.0);
|
||||
italic.is_italic = true;
|
||||
let mut underlined = make_merge_item("term", 195.0, 20.0);
|
||||
underlined.is_underline = true;
|
||||
let items = vec![
|
||||
make_merge_item("plain lead", 100.0, 48.0),
|
||||
italic,
|
||||
underlined,
|
||||
make_merge_item("plain tail", 218.0, 45.0),
|
||||
];
|
||||
let merged = merge_text_items(items);
|
||||
assert_eq!(merged.len(), 4);
|
||||
assert!(merged[1].is_italic && !merged[1].is_underline);
|
||||
assert!(merged[2].is_underline && !merged[2].is_italic);
|
||||
assert!(!merged[3].is_underline && !merged[3].is_italic);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_items_no_space_before_period() {
|
||||
// Simulate Tc/Tw-adjusted width: "date" width is smaller than the gap
|
||||
@@ -898,158 +612,6 @@ mod tests {
|
||||
assert_eq!(merged[0].text, "hello world");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_items_preserves_underline_from_later_fragment() {
|
||||
// Fragments with differing underline stay separate items — OR-merging
|
||||
// would stretch the eventual `<u>` span over the plain fragment.
|
||||
// Line-level text assembly still joins them without a space (tight
|
||||
// gap), so the rendered word is unchanged: `pre<u>fix</u>`.
|
||||
let mut items = vec![
|
||||
make_merge_item("pre", 100.0, 18.0),
|
||||
make_merge_item("fix", 119.0, 18.0),
|
||||
];
|
||||
items[1].is_underline = true;
|
||||
|
||||
let merged = merge_text_items(items);
|
||||
|
||||
assert_eq!(merged.len(), 2);
|
||||
assert_eq!(merged[0].text, "pre");
|
||||
assert!(!merged[0].is_underline);
|
||||
assert_eq!(merged[1].text, "fix");
|
||||
assert!(merged[1].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_items_preserves_stream_order_for_backtracking_heading() {
|
||||
// Some tagged PDFs emit first-letter ActualText fragments, then reset
|
||||
// the text matrix and draw the rest of the word from the line start.
|
||||
let items = vec![
|
||||
with_mcid(make_merge_item("F", 79.4, 4.5)),
|
||||
with_mcid(make_merge_item("r", 83.9, 3.3)),
|
||||
with_mcid(make_merge_item("om tables to data-", 79.4, 89.7)),
|
||||
with_mcid(make_merge_item("", 168.9, 33.9)),
|
||||
with_mcid(make_merge_item("analytics-", 168.9, 75.5)),
|
||||
with_mcid(make_merge_item("ready content", 210.5, 60.8)),
|
||||
];
|
||||
|
||||
let merged = merge_text_items(items);
|
||||
|
||||
assert_eq!(merged.len(), 1);
|
||||
assert_eq!(
|
||||
merged[0].text,
|
||||
"From tables to data-analytics-ready content"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_items_preserves_stream_order_for_reset_word_prefix() {
|
||||
let items = vec![
|
||||
with_mcid(make_merge_item("N", 68.0, 7.0)),
|
||||
with_mcid(make_merge_item("e", 75.1, 4.0)),
|
||||
with_mcid(make_merge_item("w fields created", 68.0, 82.0)),
|
||||
];
|
||||
|
||||
let merged = merge_text_items(items);
|
||||
|
||||
assert_eq!(merged.len(), 1);
|
||||
assert_eq!(merged[0].text, "New fields created");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_items_uses_x_order_for_untagged_backtracking_text() {
|
||||
let items = vec![
|
||||
make_merge_item("N", 68.0, 7.0),
|
||||
make_merge_item("e", 75.1, 4.0),
|
||||
make_merge_item("w fields created", 68.2, 82.0),
|
||||
];
|
||||
|
||||
let merged = merge_text_items(items);
|
||||
|
||||
let texts: Vec<_> = merged.iter().map(|item| item.text.as_str()).collect();
|
||||
assert_eq!(texts, vec!["N", "w fields created", "e"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_items_preserves_bullet_stream_order_with_backtracking() {
|
||||
let items = vec![
|
||||
with_mcid(make_merge_item("•", 79.4, 5.0)),
|
||||
with_mcid(make_merge_item("The MS", 91.0, 32.6)),
|
||||
with_mcid(make_merge_item("A LoS project", 84.4, 70.0)),
|
||||
];
|
||||
|
||||
let merged = merge_text_items(items);
|
||||
|
||||
assert_eq!(merged.len(), 1);
|
||||
assert_eq!(merged[0].text, "• The MSA LoS project");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_items_keeps_normal_bullet_gap_limit_without_stream_order() {
|
||||
let items = vec![
|
||||
make_merge_item("•", 79.4, 5.0),
|
||||
make_merge_item("Distant item", 91.0, 60.0),
|
||||
];
|
||||
|
||||
let merged = merge_text_items(items);
|
||||
|
||||
let texts: Vec<_> = merged.iter().map(|item| item.text.as_str()).collect();
|
||||
assert_eq!(texts, vec!["•", "Distant item"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn suppress_table_underlines_clears_line_detected_table_items() {
|
||||
let mut items = vec![
|
||||
make_merge_item("H1", 125.0, 20.0),
|
||||
make_merge_item("H2", 225.0, 20.0),
|
||||
make_merge_item("A", 125.0, 20.0),
|
||||
make_merge_item("B", 225.0, 20.0),
|
||||
];
|
||||
items[0].y = 490.0;
|
||||
items[1].y = 490.0;
|
||||
items[2].y = 470.0;
|
||||
items[3].y = 470.0;
|
||||
for item in &mut items {
|
||||
item.is_underline = true;
|
||||
item.is_strikeout = true;
|
||||
}
|
||||
let lines = vec![
|
||||
make_line(100.0, 500.0, 300.0, 500.0),
|
||||
make_line(100.0, 480.0, 300.0, 480.0),
|
||||
make_line(100.0, 460.0, 300.0, 460.0),
|
||||
make_line(100.0, 460.0, 100.0, 500.0),
|
||||
make_line(200.0, 460.0, 200.0, 500.0),
|
||||
make_line(300.0, 460.0, 300.0, 500.0),
|
||||
];
|
||||
|
||||
suppress_table_underlines(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| !item.is_underline));
|
||||
assert!(items.iter().all(|item| !item.is_strikeout));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn subscript_digit_with_different_marks_is_not_absorbed() {
|
||||
// A struck-out word followed by an unmarked footnote digit: merging
|
||||
// would widen the parent's strikeout claim over the digit (and the
|
||||
// reverse would drop the digit's own mark). Style boundaries break
|
||||
// the merge, as in merge_text_items.
|
||||
let mut word = make_merge_item("word", 100.0, 24.0);
|
||||
word.font_size = 10.0;
|
||||
word.is_strikeout = true;
|
||||
let mut digit = make_merge_item("2", 124.5, 4.0);
|
||||
digit.font_size = 6.0;
|
||||
digit.y = word.y + 3.0;
|
||||
|
||||
let merged = merge_subscript_items(vec![word.clone(), digit.clone()]);
|
||||
assert_eq!(merged.len(), 2);
|
||||
|
||||
// Same marks still merge (footnote ref inside the strike).
|
||||
digit.is_strikeout = true;
|
||||
let merged = merge_subscript_items(vec![word, digit]);
|
||||
assert_eq!(merged.len(), 1);
|
||||
assert!(merged[0].text.starts_with("word"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_group_into_lines() {
|
||||
let items = vec![
|
||||
@@ -1064,8 +626,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -1080,8 +640,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -1096,8 +654,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -1152,8 +708,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -1168,8 +722,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -1184,8 +736,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -1211,8 +761,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -1227,8 +775,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -1243,8 +789,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -1272,8 +816,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: true,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
@@ -1308,8 +850,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
@@ -1345,8 +885,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -1361,8 +899,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -1377,8 +913,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -1401,8 +935,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
@@ -1515,8 +1047,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -1531,8 +1061,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -1557,8 +1085,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -1573,8 +1099,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
},
|
||||
@@ -1615,8 +1139,6 @@ mod tests {
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}],
|
||||
@@ -1661,8 +1183,6 @@ mod tests {
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}],
|
||||
@@ -1707,8 +1227,6 @@ mod tests {
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}],
|
||||
@@ -1746,8 +1264,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
@@ -1763,8 +1279,7 @@ mod tests {
|
||||
];
|
||||
let merged = merge_subscript_items(items);
|
||||
assert_eq!(merged.len(), 2);
|
||||
// Lowered baseline → Unicode subscript form (NFKC folds back to "NH3")
|
||||
assert_eq!(merged[0].text, "NH₃");
|
||||
assert_eq!(merged[0].text, "NH3");
|
||||
assert_eq!(merged[1].text, "Cl");
|
||||
}
|
||||
|
||||
@@ -1778,21 +1293,10 @@ mod tests {
|
||||
];
|
||||
let merged = merge_subscript_items(items);
|
||||
assert_eq!(merged.len(), 2);
|
||||
assert_eq!(merged[0].text, "H₂");
|
||||
assert_eq!(merged[0].text, "H2");
|
||||
assert_eq!(merged[1].text, "O");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_merge_subscript_items_raised_marker_becomes_superscript() {
|
||||
// Footnote reference: "word" followed by a RAISED small "2" → word²
|
||||
let mut marker = make_item_fs("2", 90.0, 502.5, 2.3, 4.7);
|
||||
marker.y = 502.5; // raised above the 499.0 parent baseline
|
||||
let items = vec![make_item_fs("word", 78.0, 499.0, 12.0, 8.0), marker];
|
||||
let merged = merge_subscript_items(items);
|
||||
assert_eq!(merged.len(), 1);
|
||||
assert_eq!(merged[0].text, "word²");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_merge_subscript_items_no_merge_far_gap() {
|
||||
// Subscript-sized item that's far from the parent should NOT merge
|
||||
|
||||
@@ -1,563 +0,0 @@
|
||||
//! Geometric underline detection.
|
||||
//!
|
||||
//! PDFs have no underline font flag — underlines are drawn as separate
|
||||
//! graphics: stroked horizontal lines (`l`/`S` operators) or thin filled
|
||||
//! rectangles (`re`/`f`). This pass correlates those graphics with text
|
||||
//! items after extraction: an item is underlined when a horizontal
|
||||
//! line/thin rect sits just below its baseline and covers most of its
|
||||
//! horizontal extent.
|
||||
//!
|
||||
//! Repeated same-span rules are treated as table/form rulings rather than
|
||||
//! underlines, which avoids marking every cell in ruled tables.
|
||||
|
||||
use std::collections::HashSet;
|
||||
|
||||
use crate::types::{ItemType, PdfRect, TextItem};
|
||||
|
||||
/// Max thickness (pt) for a stroked line / filled rect to count as an
|
||||
/// underline rule rather than a border or decorative band.
|
||||
const MAX_RULE_THICKNESS: f32 = 2.0;
|
||||
|
||||
/// Fraction of the item's width that the rule must cover horizontally.
|
||||
const MIN_X_OVERLAP: f32 = 0.6;
|
||||
|
||||
/// Same-span rules repeated at this many y-levels are usually table/form
|
||||
/// rulings, not semantic underlines.
|
||||
const MIN_REPEATED_RULE_LEVELS: usize = 3;
|
||||
|
||||
/// Vertical tolerance for considering two rules to be on the same row edge.
|
||||
const RULE_Y_DEDUP_EPS: f32 = 2.0;
|
||||
|
||||
/// Horizontal span similarity required when clustering repeated rulings.
|
||||
const RULE_SPAN_OVERLAP_RATIO: f32 = 0.8;
|
||||
const RULE_SPAN_WIDTH_RATIO: f32 = 1.5;
|
||||
|
||||
/// Multiple separated rule segments on one row are usually per-column table
|
||||
/// header/body separators.
|
||||
const MIN_SEGMENTED_ROW_RULES: usize = 3;
|
||||
const MIN_SEGMENTED_ROW_GAPS: usize = 2;
|
||||
const SEGMENTED_ROW_GAP_MIN: f32 = 12.0;
|
||||
|
||||
/// A single rule under several widely separated items is usually a table
|
||||
/// header/body separator, not a sentence underline.
|
||||
const MIN_TABULAR_RULE_ITEMS: usize = 3;
|
||||
const MIN_TABULAR_RULE_GAPS: usize = 2;
|
||||
const TABULAR_RULE_GAP_EM: f32 = 2.0;
|
||||
|
||||
#[derive(Clone)]
|
||||
pub(crate) struct UnderlineLine {
|
||||
pub(crate) x1: f32,
|
||||
pub(crate) y1: f32,
|
||||
pub(crate) x2: f32,
|
||||
pub(crate) y2: f32,
|
||||
pub(crate) stroke_width: f32,
|
||||
pub(crate) page: u32,
|
||||
}
|
||||
|
||||
/// A horizontal rule candidate in page coordinates (PDF y-up).
|
||||
#[derive(Clone)]
|
||||
struct Rule {
|
||||
x1: f32,
|
||||
x2: f32,
|
||||
y: f32,
|
||||
}
|
||||
|
||||
impl Rule {
|
||||
fn width(&self) -> f32 {
|
||||
self.x2 - self.x1
|
||||
}
|
||||
}
|
||||
|
||||
fn rules_from_graphics(rects: &[PdfRect], lines: &[UnderlineLine], page: u32) -> Vec<Rule> {
|
||||
let mut rules: Vec<Rule> = Vec::new();
|
||||
for l in lines {
|
||||
if l.page != page {
|
||||
continue;
|
||||
}
|
||||
// Horizontal stroked line (tolerate slight skew).
|
||||
if l.stroke_width <= MAX_RULE_THICKNESS && (l.y1 - l.y2).abs() <= MAX_RULE_THICKNESS {
|
||||
let (x1, x2) = if l.x1 <= l.x2 {
|
||||
(l.x1, l.x2)
|
||||
} else {
|
||||
(l.x2, l.x1)
|
||||
};
|
||||
if x2 - x1 > 1.0 {
|
||||
rules.push(Rule {
|
||||
x1,
|
||||
x2,
|
||||
y: (l.y1 + l.y2) / 2.0,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
for r in rects {
|
||||
if r.page != page {
|
||||
continue;
|
||||
}
|
||||
// Thin filled rect used as an underline rule. Extents are
|
||||
// normalized first: `re` operands pass through the CTM, so
|
||||
// width/height can be negative (flipped axes / negative scale) —
|
||||
// without normalization negative-width rules are missed and
|
||||
// negative-height bands sneak past the thickness check.
|
||||
let (x1, x2) = if r.width >= 0.0 {
|
||||
(r.x, r.x + r.width)
|
||||
} else {
|
||||
(r.x + r.width, r.x)
|
||||
};
|
||||
if r.height.abs() <= MAX_RULE_THICKNESS && x2 - x1 > 1.0 {
|
||||
rules.push(Rule {
|
||||
x1,
|
||||
x2,
|
||||
y: r.y + r.height / 2.0,
|
||||
});
|
||||
}
|
||||
}
|
||||
rules
|
||||
}
|
||||
|
||||
fn discard_repeated_ruling_rules(rules: Vec<Rule>) -> Vec<Rule> {
|
||||
if rules.len() < MIN_REPEATED_RULE_LEVELS {
|
||||
return rules;
|
||||
}
|
||||
|
||||
rules
|
||||
.iter()
|
||||
.filter(|rule| {
|
||||
!is_repeated_ruling_rule(rule, &rules) && !is_segmented_row_ruling_rule(rule, &rules)
|
||||
})
|
||||
.cloned()
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn is_repeated_ruling_rule(rule: &Rule, rules: &[Rule]) -> bool {
|
||||
let mut y_levels: Vec<f32> = rules
|
||||
.iter()
|
||||
.filter(|other| has_similar_span(rule, other))
|
||||
.map(|other| other.y)
|
||||
.collect();
|
||||
|
||||
y_levels.sort_by(|a, b| a.total_cmp(b));
|
||||
y_levels.dedup_by(|a, b| (*a - *b).abs() <= RULE_Y_DEDUP_EPS);
|
||||
y_levels.len() >= MIN_REPEATED_RULE_LEVELS
|
||||
}
|
||||
|
||||
fn is_segmented_row_ruling_rule(rule: &Rule, rules: &[Rule]) -> bool {
|
||||
let mut row_rules: Vec<&Rule> = rules
|
||||
.iter()
|
||||
.filter(|other| (other.y - rule.y).abs() <= RULE_Y_DEDUP_EPS)
|
||||
.collect();
|
||||
|
||||
if row_rules.len() < MIN_SEGMENTED_ROW_RULES {
|
||||
return false;
|
||||
}
|
||||
|
||||
row_rules.sort_by(|a, b| a.x1.total_cmp(&b.x1));
|
||||
let large_gaps = row_rules
|
||||
.windows(2)
|
||||
.filter(|pair| pair[1].x1 - pair[0].x2 > SEGMENTED_ROW_GAP_MIN)
|
||||
.count();
|
||||
|
||||
large_gaps >= MIN_SEGMENTED_ROW_GAPS
|
||||
}
|
||||
|
||||
fn has_similar_span(a: &Rule, b: &Rule) -> bool {
|
||||
let a_width = a.width();
|
||||
let b_width = b.width();
|
||||
if a_width <= 1.0 || b_width <= 1.0 {
|
||||
return false;
|
||||
}
|
||||
|
||||
let width_ratio = a_width.max(b_width) / a_width.min(b_width);
|
||||
if width_ratio > RULE_SPAN_WIDTH_RATIO {
|
||||
return false;
|
||||
}
|
||||
|
||||
let overlap = a.x2.min(b.x2) - a.x1.max(b.x1);
|
||||
overlap >= a_width.min(b_width) * RULE_SPAN_OVERLAP_RATIO
|
||||
}
|
||||
|
||||
fn tabular_row_separator_rule_indices(rules: &[Rule], items: &[TextItem]) -> HashSet<usize> {
|
||||
let mut tabular_rules = HashSet::new();
|
||||
|
||||
for (rule_idx, rule) in rules.iter().enumerate() {
|
||||
let mut matched_items: Vec<&TextItem> = items
|
||||
.iter()
|
||||
.filter(|item| is_underline_candidate(item) && rule_matches_item(rule, item))
|
||||
.collect();
|
||||
|
||||
if matched_items.len() < MIN_TABULAR_RULE_ITEMS {
|
||||
continue;
|
||||
}
|
||||
|
||||
matched_items.sort_by(|a, b| a.x.total_cmp(&b.x));
|
||||
let large_gaps = matched_items
|
||||
.windows(2)
|
||||
.filter(|pair| {
|
||||
let left = pair[0];
|
||||
let right = pair[1];
|
||||
let gap = right.x - (left.x + left.width);
|
||||
let font_size = left.font_size.max(right.font_size).max(1.0);
|
||||
gap > font_size * TABULAR_RULE_GAP_EM
|
||||
})
|
||||
.count();
|
||||
|
||||
if large_gaps >= MIN_TABULAR_RULE_GAPS {
|
||||
tabular_rules.insert(rule_idx);
|
||||
}
|
||||
}
|
||||
|
||||
tabular_rules
|
||||
}
|
||||
|
||||
fn is_underline_candidate(item: &TextItem) -> bool {
|
||||
matches!(item.item_type, ItemType::Text) && !item.text.trim().is_empty() && item.width > 0.0
|
||||
}
|
||||
|
||||
fn rule_matches_item(rule: &Rule, item: &TextItem) -> bool {
|
||||
// Vertical window: underlines sit at or slightly below the baseline.
|
||||
// Fonts draw them at roughly 5-15% of the em below; allow up to 35%
|
||||
// (min 3pt) below and 1pt above for rounding.
|
||||
let below = (item.font_size * 0.35).max(3.0);
|
||||
let y_min = item.y - below;
|
||||
let y_max = item.y + 1.0;
|
||||
if rule.y < y_min || rule.y > y_max {
|
||||
return false;
|
||||
}
|
||||
|
||||
let ix1 = item.x;
|
||||
let ix2 = item.x + item.width;
|
||||
let min_overlap = item.width * MIN_X_OVERLAP;
|
||||
let overlap = rule.x2.min(ix2) - rule.x1.max(ix1);
|
||||
overlap >= min_overlap
|
||||
}
|
||||
|
||||
/// Strikeout window: a rule crossing the glyphs. Strikethroughs sit at
|
||||
/// roughly 20-35% of the em above the baseline (about half the x-height);
|
||||
/// accept a band well inside the glyph body so baseline underlines and
|
||||
/// overlines never qualify.
|
||||
fn rule_strikes_item(rule: &Rule, item: &TextItem) -> bool {
|
||||
let y_min = item.y + item.font_size * 0.12;
|
||||
let y_max = item.y + item.font_size * 0.55;
|
||||
if rule.y < y_min || rule.y > y_max {
|
||||
return false;
|
||||
}
|
||||
|
||||
let ix1 = item.x;
|
||||
let ix2 = item.x + item.width;
|
||||
let min_overlap = item.width * MIN_X_OVERLAP;
|
||||
let overlap = rule.x2.min(ix2) - rule.x1.max(ix1);
|
||||
overlap >= min_overlap
|
||||
}
|
||||
|
||||
/// Mark `is_underline` on text items that have a horizontal rule just
|
||||
/// below their baseline, and `is_strikeout` on items whose glyphs a rule
|
||||
/// crosses at mid x-height. `items`, `rects`, and `lines` are a single
|
||||
/// page's extraction output (all in PDF coordinates, y-up, where
|
||||
/// `TextItem::y` is the text baseline).
|
||||
pub(crate) fn mark_underlined_items(
|
||||
items: &mut [TextItem],
|
||||
rects: &[PdfRect],
|
||||
lines: &[UnderlineLine],
|
||||
page: u32,
|
||||
) {
|
||||
let rules = discard_repeated_ruling_rules(rules_from_graphics(rects, lines, page));
|
||||
if rules.is_empty() {
|
||||
return;
|
||||
}
|
||||
let tabular_rules = tabular_row_separator_rule_indices(&rules, items);
|
||||
|
||||
for item in items.iter_mut() {
|
||||
if !is_underline_candidate(item) {
|
||||
continue;
|
||||
}
|
||||
|
||||
for (rule_idx, rule) in rules.iter().enumerate() {
|
||||
if tabular_rules.contains(&rule_idx) {
|
||||
continue;
|
||||
}
|
||||
if rule_matches_item(rule, item) {
|
||||
item.is_underline = true;
|
||||
}
|
||||
if rule_strikes_item(rule, item) {
|
||||
item.is_strikeout = true;
|
||||
}
|
||||
if item.is_underline && item.is_strikeout {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::types::ItemType;
|
||||
|
||||
fn item(text: &str, x: f32, y: f32, width: f32, font_size: f32) -> TextItem {
|
||||
TextItem {
|
||||
text: text.to_string(),
|
||||
x,
|
||||
y,
|
||||
width,
|
||||
height: font_size,
|
||||
font: "F1".to_string(),
|
||||
font_size,
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
}
|
||||
|
||||
fn hline(x1: f32, x2: f32, y: f32) -> UnderlineLine {
|
||||
UnderlineLine {
|
||||
x1,
|
||||
y1: y,
|
||||
x2,
|
||||
y2: y,
|
||||
stroke_width: 1.0,
|
||||
page: 1,
|
||||
}
|
||||
}
|
||||
|
||||
fn thin_rect(x: f32, y: f32, width: f32) -> PdfRect {
|
||||
PdfRect {
|
||||
x,
|
||||
y,
|
||||
width,
|
||||
height: 0.8,
|
||||
page: 1,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stroked_line_under_baseline_marks_underline() {
|
||||
let mut items = vec![item("underlined", 100.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![hline(99.0, 161.0, 498.5)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn thin_filled_rect_under_baseline_marks_underline() {
|
||||
let mut items = vec![item("underlined", 100.0, 500.0, 60.0, 10.0)];
|
||||
let rects = vec![thin_rect(100.0, 497.8, 60.0)];
|
||||
mark_underlined_items(&mut items, &rects, &[], 1);
|
||||
assert!(items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn long_rule_under_multiple_items_marks_each() {
|
||||
// One underline drawn under a whole sentence: every overlapped
|
||||
// item gets the flag.
|
||||
let mut items = vec![
|
||||
item("first", 100.0, 500.0, 40.0, 10.0),
|
||||
item("second", 145.0, 500.0, 50.0, 10.0),
|
||||
];
|
||||
let lines = vec![hline(98.0, 200.0, 498.0)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(items[0].is_underline);
|
||||
assert!(items[1].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn line_far_below_baseline_is_not_an_underline() {
|
||||
// A horizontal rule 30pt below (section divider) must not mark.
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![hline(90.0, 300.0, 470.0)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn thick_stroked_line_is_not_an_underline() {
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let mut line = hline(99.0, 161.0, 498.5);
|
||||
line.stroke_width = 4.0;
|
||||
|
||||
mark_underlined_items(&mut items, &[], &[line], 1);
|
||||
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mid_glyph_rule_marks_strikeout_not_underline() {
|
||||
// Rule at ~30% of the em above the baseline crosses the glyphs.
|
||||
let mut items = vec![item("struck out", 100.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![hline(99.0, 161.0, 503.0)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(items[0].is_strikeout);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn baseline_rule_marks_underline_not_strikeout() {
|
||||
let mut items = vec![item("underlined", 100.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![hline(99.0, 161.0, 498.5)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(items[0].is_underline);
|
||||
assert!(!items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn overline_is_neither_underline_nor_strikeout() {
|
||||
// Rule just above the cap height (overline / next line's rule).
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![hline(99.0, 161.0, 507.0)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(!items[0].is_underline);
|
||||
assert!(!items[0].is_strikeout);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn thin_filled_rect_at_mid_glyph_marks_strikeout() {
|
||||
let mut items = vec![item("struck out", 100.0, 500.0, 60.0, 10.0)];
|
||||
let rects = vec![thin_rect(100.0, 502.6, 60.0)];
|
||||
mark_underlined_items(&mut items, &rects, &[], 1);
|
||||
assert!(items[0].is_strikeout);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn line_above_baseline_is_not_an_underline() {
|
||||
// Strikethrough / overline geometry must not mark.
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![hline(90.0, 300.0, 505.0)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn insufficient_horizontal_overlap_is_not_an_underline() {
|
||||
// Rule under only a quarter of the item (e.g. neighboring cell
|
||||
// border) must not mark.
|
||||
let mut items = vec![item("wide text item", 100.0, 500.0, 100.0, 10.0)];
|
||||
let lines = vec![hline(100.0, 125.0, 498.5)];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn negative_width_rect_is_normalized_and_marks_underline() {
|
||||
// A CTM with negative x-scale (or negative `re` operands) produces
|
||||
// rects whose width is negative; the rule extents must normalize.
|
||||
let mut items = vec![item("underlined", 100.0, 500.0, 60.0, 10.0)];
|
||||
let rects = vec![PdfRect {
|
||||
x: 160.0,
|
||||
y: 497.8,
|
||||
width: -60.0,
|
||||
height: 0.8,
|
||||
page: 1,
|
||||
}];
|
||||
mark_underlined_items(&mut items, &rects, &[], 1);
|
||||
assert!(items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn negative_height_band_is_not_an_underline() {
|
||||
// A 14pt band expressed with negative height must not pass the
|
||||
// thickness check via sign trickery.
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let rects = vec![PdfRect {
|
||||
x: 95.0,
|
||||
y: 509.0,
|
||||
width: 80.0,
|
||||
height: -14.0,
|
||||
page: 1,
|
||||
}];
|
||||
mark_underlined_items(&mut items, &rects, &[], 1);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn thick_band_is_not_an_underline() {
|
||||
// A highlight bar / filled cell background (tall rect) must not mark.
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let rects = vec![PdfRect {
|
||||
x: 95.0,
|
||||
y: 495.0,
|
||||
width: 80.0,
|
||||
height: 14.0,
|
||||
page: 1,
|
||||
}];
|
||||
mark_underlined_items(&mut items, &rects, &[], 1);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn vertical_line_is_not_an_underline() {
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let lines = vec![UnderlineLine {
|
||||
x1: 120.0,
|
||||
y1: 498.0,
|
||||
x2: 120.0,
|
||||
y2: 400.0,
|
||||
stroke_width: 1.0,
|
||||
page: 1,
|
||||
}];
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn other_pages_graphics_do_not_mark() {
|
||||
let mut items = vec![item("text", 100.0, 500.0, 60.0, 10.0)];
|
||||
let mut line = hline(99.0, 161.0, 498.5);
|
||||
line.page = 2;
|
||||
mark_underlined_items(&mut items, &[], &[line], 1);
|
||||
assert!(!items[0].is_underline);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repeated_table_row_rules_do_not_mark_cell_text() {
|
||||
let mut items = vec![
|
||||
item("A", 110.0, 500.0, 20.0, 10.0),
|
||||
item("B", 110.0, 480.0, 20.0, 10.0),
|
||||
item("C", 110.0, 460.0, 20.0, 10.0),
|
||||
];
|
||||
let lines = vec![
|
||||
hline(100.0, 150.0, 498.0),
|
||||
hline(100.0, 150.0, 478.0),
|
||||
hline(100.0, 150.0, 458.0),
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| !item.is_underline));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn row_separator_under_spaced_column_labels_is_not_an_underline() {
|
||||
let mut items = vec![
|
||||
item("Date", 100.0, 500.0, 25.0, 10.0),
|
||||
item("Rate", 200.0, 500.0, 25.0, 10.0),
|
||||
item("Yield", 300.0, 500.0, 30.0, 10.0),
|
||||
];
|
||||
let lines = vec![hline(90.0, 340.0, 498.0)];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| !item.is_underline));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn same_row_spaced_rule_segments_do_not_mark_column_labels() {
|
||||
let mut items = vec![
|
||||
item("Date", 100.0, 500.0, 25.0, 10.0),
|
||||
item("Rate", 200.0, 500.0, 25.0, 10.0),
|
||||
item("Yield", 300.0, 500.0, 30.0, 10.0),
|
||||
];
|
||||
let lines = vec![
|
||||
hline(98.0, 128.0, 498.0),
|
||||
hline(198.0, 228.0, 498.0),
|
||||
hline(298.0, 333.0, 498.0),
|
||||
];
|
||||
|
||||
mark_underlined_items(&mut items, &[], &lines, 1);
|
||||
|
||||
assert!(items.iter().all(|item| !item.is_underline));
|
||||
}
|
||||
}
|
||||
@@ -1,6 +1,5 @@
|
||||
//! Form XObject and image XObject extraction.
|
||||
|
||||
use super::fonts::descriptor_style_flags;
|
||||
use crate::text_utils::{effective_font_size, expand_ligatures, is_bold_font, is_italic_font};
|
||||
use crate::tounicode::FontCMaps;
|
||||
use crate::types::{ItemType, TextItem};
|
||||
@@ -9,7 +8,7 @@ use std::collections::HashMap;
|
||||
|
||||
use super::fonts::{
|
||||
build_font_encodings, build_font_widths, compute_string_width_ts, extract_text_from_operand,
|
||||
get_font_file2_obj_num, get_operand_bytes, CMapDecisionCache, FontStyleCache,
|
||||
get_font_file2_obj_num, get_operand_bytes, CMapDecisionCache,
|
||||
};
|
||||
use super::{get_number, image_bbox_from_ctm, multiply_matrices};
|
||||
|
||||
@@ -115,7 +114,6 @@ pub(crate) fn extract_form_xobject_text(
|
||||
font_cmaps: &FontCMaps,
|
||||
parent_ctm: &[f32; 6],
|
||||
cmap_decisions: &mut CMapDecisionCache,
|
||||
style_cache: &mut FontStyleCache,
|
||||
) -> Vec<TextItem> {
|
||||
extract_form_xobject_text_inner(
|
||||
doc,
|
||||
@@ -124,12 +122,10 @@ pub(crate) fn extract_form_xobject_text(
|
||||
font_cmaps,
|
||||
parent_ctm,
|
||||
cmap_decisions,
|
||||
style_cache,
|
||||
0,
|
||||
)
|
||||
}
|
||||
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn extract_form_xobject_text_inner(
|
||||
doc: &Document,
|
||||
form_id: ObjectId,
|
||||
@@ -137,7 +133,6 @@ fn extract_form_xobject_text_inner(
|
||||
font_cmaps: &FontCMaps,
|
||||
parent_ctm: &[f32; 6],
|
||||
cmap_decisions: &mut CMapDecisionCache,
|
||||
style_cache: &mut FontStyleCache,
|
||||
depth: u8,
|
||||
) -> Vec<TextItem> {
|
||||
use lopdf::content::Content;
|
||||
@@ -172,7 +167,6 @@ fn extract_form_xobject_text_inner(
|
||||
let mut font_tounicode_refs: HashMap<String, u32> = HashMap::new();
|
||||
let mut inline_cmaps: HashMap<String, crate::tounicode::CMapEntry> = HashMap::new();
|
||||
|
||||
let mut font_style_flags: HashMap<String, (bool, bool)> = HashMap::new();
|
||||
for (font_name, font_dict) in &form_fonts {
|
||||
let resource_name = String::from_utf8_lossy(font_name).to_string();
|
||||
if let Ok(base_font) = font_dict.get(b"BaseFont") {
|
||||
@@ -181,10 +175,6 @@ fn extract_form_xobject_text_inner(
|
||||
font_base_names.insert(resource_name.clone(), base_name);
|
||||
}
|
||||
}
|
||||
let style = descriptor_style_flags(doc, font_dict, style_cache);
|
||||
if style != (false, false) {
|
||||
font_style_flags.insert(resource_name.clone(), style);
|
||||
}
|
||||
match font_dict.get(b"ToUnicode") {
|
||||
Ok(tounicode) => {
|
||||
if let Ok(obj_ref) = tounicode.as_reference() {
|
||||
@@ -282,7 +272,6 @@ fn extract_form_xobject_text_inner(
|
||||
font_cmaps,
|
||||
&ctm,
|
||||
cmap_decisions,
|
||||
style_cache,
|
||||
depth + 1,
|
||||
);
|
||||
items.extend(nested_items);
|
||||
@@ -305,8 +294,6 @@ fn extract_form_xobject_text_inner(
|
||||
page: page_num,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Image,
|
||||
mcid: None,
|
||||
});
|
||||
@@ -440,10 +427,6 @@ fn extract_form_xobject_text_inner(
|
||||
.get(¤t_font)
|
||||
.map(|s| s.as_str())
|
||||
.unwrap_or(¤t_font);
|
||||
let (desc_italic, desc_bold) = font_style_flags
|
||||
.get(¤t_font)
|
||||
.copied()
|
||||
.unwrap_or((false, false));
|
||||
items.push(TextItem {
|
||||
text: expand_ligatures(&text),
|
||||
x,
|
||||
@@ -453,10 +436,8 @@ fn extract_form_xobject_text_inner(
|
||||
font: current_font.clone(),
|
||||
font_size: rendered_size,
|
||||
page: page_num,
|
||||
is_bold: is_bold_font(base_font) || desc_bold,
|
||||
is_italic: is_italic_font(base_font) || desc_italic,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
is_bold: is_bold_font(base_font),
|
||||
is_italic: is_italic_font(base_font),
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
});
|
||||
@@ -577,10 +558,6 @@ fn extract_form_xobject_text_inner(
|
||||
.get(¤t_font)
|
||||
.map(|s| s.as_str())
|
||||
.unwrap_or(¤t_font);
|
||||
let (desc_italic, desc_bold) = font_style_flags
|
||||
.get(¤t_font)
|
||||
.copied()
|
||||
.unwrap_or((false, false));
|
||||
let scale_x = text_matrix[0] * ctm[0] + text_matrix[1] * ctm[2];
|
||||
for (text, start_w, end_w) in &sub_items {
|
||||
let offset_tm = [
|
||||
@@ -607,10 +584,8 @@ fn extract_form_xobject_text_inner(
|
||||
font: current_font.clone(),
|
||||
font_size: rendered_size,
|
||||
page: page_num,
|
||||
is_bold: is_bold_font(base_font) || desc_bold,
|
||||
is_italic: is_italic_font(base_font) || desc_italic,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
is_bold: is_bold_font(base_font),
|
||||
is_italic: is_italic_font(base_font),
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
});
|
||||
|
||||
+52
-955
File diff suppressed because it is too large
Load Diff
@@ -130,108 +130,6 @@ pub(crate) fn has_dot_leaders(text: &str) -> bool {
|
||||
dot_groups >= 2
|
||||
}
|
||||
|
||||
/// Detect a table-of-contents entry: a line ending in a page number preceded by
|
||||
/// a dot-leader group (e.g. "Measurement Lab worksheet ... 3"). `has_dot_leaders`
|
||||
/// misses single-group leaders ("..."), but a trailing "<dots> <number>" is a
|
||||
/// strong TOC signal on its own. Such lines must never be promoted to headings.
|
||||
pub(crate) fn is_toc_entry_line(text: &str) -> bool {
|
||||
let trimmed = text.trim_end();
|
||||
let digits = trimmed
|
||||
.chars()
|
||||
.rev()
|
||||
.take_while(|c| c.is_ascii_digit())
|
||||
.count();
|
||||
if digits == 0 || digits > 4 {
|
||||
return false;
|
||||
}
|
||||
let before_number = trimmed[..trimmed.len() - digits].trim_end();
|
||||
let dots = before_number
|
||||
.chars()
|
||||
.rev()
|
||||
.take_while(|c| *c == '.')
|
||||
.count();
|
||||
dots >= 3
|
||||
}
|
||||
|
||||
/// A heading that announces a table of contents ("Contents", "Table of
|
||||
/// Contents"). Lines after it on the same page are ToC entries — section
|
||||
/// titles that look exactly like headings but must not be promoted.
|
||||
pub(crate) fn is_toc_marker_heading(text: &str) -> bool {
|
||||
let t = text.trim().trim_end_matches(':').trim().to_lowercase();
|
||||
matches!(t.as_str(), "contents" | "table of contents")
|
||||
}
|
||||
|
||||
/// Lines that resemble headings structurally but are display-math fragments:
|
||||
/// equations ending in an equation number ("S = kB ln W, (2)") or equation
|
||||
/// lead-ins ("Rearranging Equation (8) gives:"). Both carry an "(N)" equation
|
||||
/// reference — but a trailing "(N)" alone is not enough: real headings end
|
||||
/// with parenthesized numbers too ("Nicaea (325)", appendix numbering), so
|
||||
/// the suffix form additionally requires math evidence — an "=" in the line
|
||||
/// or a comma immediately before the number, both present in every display
|
||||
/// equation and absent from name-plus-number headings. A bare trailing colon
|
||||
/// is NOT a fragment signal either: real headings frequently end with colons
|
||||
/// ("Procedure:", "Steps for Using the Microscope:").
|
||||
pub(crate) fn is_heading_fragment(text: &str) -> bool {
|
||||
let t = text.trim_end();
|
||||
|
||||
fn is_equation_number(s: &str) -> bool {
|
||||
s.strip_prefix('(')
|
||||
.and_then(|r| r.strip_suffix(')'))
|
||||
.is_some_and(|inner| {
|
||||
!inner.is_empty() && inner.len() <= 3 && inner.chars().all(|c| c.is_ascii_digit())
|
||||
})
|
||||
}
|
||||
|
||||
// Equation-number suffix with math evidence: "S = kB ln W, (2)"
|
||||
let mut rev = t.rsplit(' ');
|
||||
let last = rev.next().unwrap_or("");
|
||||
if is_equation_number(last) {
|
||||
// Page-of-total running headers: "LIVSMEDELSVERKET PM 2 (10)"
|
||||
if let Some(prev_word) = t.rsplit(' ').nth(1) {
|
||||
if let (Ok(page), Some(total)) = (
|
||||
prev_word.parse::<u32>(),
|
||||
last.trim_start_matches('(')
|
||||
.trim_end_matches(')')
|
||||
.parse::<u32>()
|
||||
.ok(),
|
||||
) {
|
||||
if page <= total {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
let punct_before = rev
|
||||
.next()
|
||||
.is_some_and(|w| w.ends_with(',') || w.ends_with(':'));
|
||||
let has_math_op = t.chars().any(|c| {
|
||||
matches!(
|
||||
c,
|
||||
'=' | '<'
|
||||
| '>'
|
||||
| '≤'
|
||||
| '≥'
|
||||
| '≪'
|
||||
| '≫'
|
||||
| '≈'
|
||||
| '≠'
|
||||
| '±'
|
||||
| '∑'
|
||||
| '∫'
|
||||
| '√'
|
||||
| '∝'
|
||||
)
|
||||
});
|
||||
if punct_before || has_math_op {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
// Lead-in: ends with a colon AND references an equation number inline
|
||||
if t.ends_with(':') && t.split_whitespace().any(is_equation_number) {
|
||||
return true;
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
/// Compute the Y-gap threshold for paragraph break detection.
|
||||
///
|
||||
/// Instead of using a fixed multiple of base_size (which fails for double-spaced
|
||||
@@ -422,65 +320,3 @@ pub(crate) fn detect_header_level(
|
||||
Some(4)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn toc_entry_with_single_dot_group() {
|
||||
assert!(is_toc_entry_line("Measurement Lab worksheet ... 3"));
|
||||
assert!(is_toc_entry_line("Results ........ 12"));
|
||||
assert!(is_toc_entry_line("Appendix B...42"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn non_toc_lines_pass() {
|
||||
assert!(!is_toc_entry_line(
|
||||
"6.2. Expectations for Re-Hiring Employees"
|
||||
));
|
||||
assert!(!is_toc_entry_line("What happened in 2020"));
|
||||
assert!(!is_toc_entry_line("IMPLEMENTATION"));
|
||||
// Ellipsis without a trailing page number
|
||||
assert!(!is_toc_entry_line("and so it goes ..."));
|
||||
// Long numbers are data, not page refs
|
||||
assert!(!is_toc_entry_line("ISBN ... 97814"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn toc_marker_headings() {
|
||||
assert!(is_toc_marker_heading("Contents"));
|
||||
assert!(is_toc_marker_heading("CONTENTS"));
|
||||
assert!(is_toc_marker_heading("Table of Contents"));
|
||||
assert!(is_toc_marker_heading("Table of contents:"));
|
||||
assert!(!is_toc_marker_heading("Contents of the Shipment"));
|
||||
assert!(!is_toc_marker_heading("Introduction"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn heading_fragments() {
|
||||
// Equation lead-ins: colon ending + inline equation reference
|
||||
assert!(is_heading_fragment("Rearranging Equation (8) gives:"));
|
||||
// Display-equation neighbours ending in an equation number
|
||||
assert!(is_heading_fragment("S = kB ln W, (2)"));
|
||||
assert!(is_heading_fragment("E = mc2 (12)"));
|
||||
assert!(is_heading_fragment("x + y = z, (3)"));
|
||||
// Page-of-total running headers
|
||||
assert!(is_heading_fragment("LIVSMEDELSVERKET PM 2 (10)"));
|
||||
// Comparison-operator evidence and colon-before-number
|
||||
assert!(is_heading_fragment(
|
||||
"PLL\u{fe} PHH\u{226a} PLH\u{fe} PHL: (12)"
|
||||
));
|
||||
// Real headings pass — including name-plus-number and colon-ended ones
|
||||
assert!(!is_heading_fragment("Nicaea (325)"));
|
||||
assert!(!is_heading_fragment(
|
||||
"\u{627}\u{644}\u{645}\u{644}\u{62d}\u{642} \u{631}\u{642}\u{645} (1)"
|
||||
));
|
||||
assert!(!is_heading_fragment("4. Entropy"));
|
||||
assert!(!is_heading_fragment("Procedure:"));
|
||||
assert!(!is_heading_fragment("Steps for Using the Microscope:"));
|
||||
assert!(!is_heading_fragment("Changing objectives:"));
|
||||
assert!(!is_heading_fragment("Sales by Region (2024)"));
|
||||
assert!(!is_heading_fragment("Results (preliminary)"));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -131,11 +131,10 @@ pub(crate) fn format_list_item(text: &str) -> String {
|
||||
if let Some(rest) = trimmed.strip_prefix(*bullet) {
|
||||
return format!("- {}", rest.trim_start());
|
||||
}
|
||||
// Bullet inside a leading style run (e.g. "**● Label:** rest" or
|
||||
// "<u>● Label</u>"). The run wraps both the marker and the following
|
||||
// label because both carry the style in the PDF. The marker must move
|
||||
// outside the wrapper so markdown still sees a list item.
|
||||
for wrapper in ["**", "*", "<u>"] {
|
||||
// Bullet inside a leading bold/italic run (e.g. "**● Label:** rest").
|
||||
// The run wraps both the marker and the following label because both
|
||||
// use a bold font in the PDF.
|
||||
for wrapper in ["**", "*"] {
|
||||
if let Some(after_open) = trimmed.strip_prefix(wrapper) {
|
||||
if let Some(rest) = after_open.strip_prefix(*bullet) {
|
||||
return format!("- {}{}", wrapper, rest.trim_start());
|
||||
@@ -236,13 +235,6 @@ mod tests {
|
||||
assert_eq!(format_list_item("• Item"), "- Item");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn format_list_item_bullet_inside_underline() {
|
||||
// Fully-underlined bullet line: the marker must move outside the
|
||||
// <u> wrapper so markdown still renders a list item.
|
||||
assert_eq!(format_list_item("<u>● Item text</u>"), "- <u>Item text</u>");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn format_list_item_bullet_inside_bold() {
|
||||
// PDF that uses bold font for both the marker and the label produces
|
||||
|
||||
+13
-333
@@ -7,8 +7,7 @@ use crate::types::TextLine;
|
||||
|
||||
use super::analysis::{
|
||||
bold_heading_level, calculate_font_stats, compute_heading_tiers, compute_paragraph_threshold,
|
||||
detect_header_level, font_size_rarity, has_dot_leaders, is_heading_fragment, is_toc_entry_line,
|
||||
is_toc_marker_heading,
|
||||
detect_header_level, font_size_rarity, has_dot_leaders,
|
||||
};
|
||||
use super::classify::{
|
||||
format_list_item, is_caption_line, is_list_item, is_monospace_font, starts_with_bullet_marker,
|
||||
@@ -141,11 +140,8 @@ fn find_isolated_lines(lines: &[TextLine], base_size: f32, para_threshold: f32)
|
||||
}
|
||||
}
|
||||
for (&page, &(total, isolated)) in &page_line_counts {
|
||||
// The ratio only means something on pages dense enough for a
|
||||
// multi-column misfire; on sparse pages (covers, ToC pages with a
|
||||
// lone title) one isolated line is 25%+ of the page and exactly the
|
||||
// line isolation exists to find.
|
||||
if total >= 10 && isolated as f32 / total as f32 > 0.25 {
|
||||
if total > 0 && isolated as f32 / total as f32 > 0.25 {
|
||||
// Too many isolated lines on this page — remove them all
|
||||
set.retain(|&i| lines[i].page != page);
|
||||
}
|
||||
}
|
||||
@@ -153,79 +149,6 @@ fn find_isolated_lines(lines: &[TextLine], base_size: f32, para_threshold: f32)
|
||||
set
|
||||
}
|
||||
|
||||
/// Pre-scan body-size all-bold runs that are too long to be headings.
|
||||
///
|
||||
/// Some academic PDFs use an all-bold abstract/summary paragraph immediately
|
||||
/// after the author block. A line-local bold heading heuristic sees each
|
||||
/// wrapped visual line as "standalone" once the first line is misclassified,
|
||||
/// producing a stack of `##` headings. Multi-line body-size bold runs with a
|
||||
/// paragraph-sized word count should stay paragraph text.
|
||||
fn find_wrapped_bold_paragraph_lines(
|
||||
lines: &[TextLine],
|
||||
base_size: f32,
|
||||
para_threshold: f32,
|
||||
) -> HashSet<usize> {
|
||||
let mut set = HashSet::new();
|
||||
let mut i = 0usize;
|
||||
|
||||
while i < lines.len() {
|
||||
if !is_body_size_all_bold_line(&lines[i], base_size) {
|
||||
i += 1;
|
||||
continue;
|
||||
}
|
||||
|
||||
let start = i;
|
||||
let mut end = i;
|
||||
let mut word_count = lines[i].text().split_whitespace().count();
|
||||
|
||||
while end + 1 < lines.len()
|
||||
&& is_body_size_all_bold_line(&lines[end + 1], base_size)
|
||||
&& is_wrapped_same_style_line(&lines[end], &lines[end + 1], para_threshold)
|
||||
{
|
||||
end += 1;
|
||||
word_count += lines[end].text().split_whitespace().count();
|
||||
}
|
||||
|
||||
let line_count = end - start + 1;
|
||||
if line_count >= 3 && word_count > 20 {
|
||||
for idx in start..=end {
|
||||
set.insert(idx);
|
||||
}
|
||||
}
|
||||
|
||||
i = end + 1;
|
||||
}
|
||||
|
||||
set
|
||||
}
|
||||
|
||||
fn is_body_size_all_bold_line(line: &TextLine, base_size: f32) -> bool {
|
||||
let Some(first) = line.items.first() else {
|
||||
return false;
|
||||
};
|
||||
first.font_size >= base_size * 0.95
|
||||
&& first.font_size < base_size * 1.2
|
||||
&& line
|
||||
.items
|
||||
.iter()
|
||||
.all(|item| item.is_bold && (item.font_size - first.font_size).abs() < 0.5)
|
||||
}
|
||||
|
||||
fn is_wrapped_same_style_line(prev: &TextLine, next: &TextLine, para_threshold: f32) -> bool {
|
||||
if prev.page != next.page {
|
||||
return false;
|
||||
}
|
||||
|
||||
let y_gap = prev.y - next.y;
|
||||
if !(y_gap > 0.0 && y_gap <= para_threshold) {
|
||||
return false;
|
||||
}
|
||||
|
||||
let prev_x = prev.items.first().map(|item| item.x).unwrap_or(0.0);
|
||||
let next_x = next.items.first().map(|item| item.x).unwrap_or(0.0);
|
||||
(prev_x - next_x).abs() <= 40.0
|
||||
}
|
||||
|
||||
/// Resolve the dominant structure role for a text line by looking up its items' MCIDs.
|
||||
///
|
||||
/// Returns the first non-container role found (skipping Document/Part/Sect/Div/NonStruct/Span).
|
||||
@@ -474,8 +397,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
// between paragraphs at body font size. Inspired by opendataloader's
|
||||
// lookahead in HeadingProcessor (prevNode/nextNode context).
|
||||
let isolated_lines = find_isolated_lines(&lines, base_size, para_threshold);
|
||||
let wrapped_bold_paragraph_lines =
|
||||
find_wrapped_bold_paragraph_lines(&lines, base_size, para_threshold);
|
||||
|
||||
// Detect struct heading levels that are overused (body text mistagged as headings)
|
||||
let overused_heading_levels = detect_overused_struct_heading_levels(&lines, struct_roles);
|
||||
@@ -489,8 +410,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
let mut last_list_x: Option<f32> = None;
|
||||
let mut in_code_block = false;
|
||||
let mut prev_had_dot_leaders = false;
|
||||
let mut paragraph_in_wrapped_bold_run = false;
|
||||
let mut toc_suppress_page: Option<u32> = None;
|
||||
let mut inserted_tables: HashSet<(u32, usize)> = HashSet::new();
|
||||
let mut inserted_images: HashSet<(u32, usize)> = HashSet::new();
|
||||
|
||||
@@ -556,7 +475,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
current_page = line.page;
|
||||
prev_y = f32::MAX;
|
||||
prev_x = 0.0;
|
||||
paragraph_in_wrapped_bold_run = false;
|
||||
|
||||
if options.include_page_numbers {
|
||||
output.push_str(&format!("<!-- Page {} -->\n\n", current_page));
|
||||
@@ -571,7 +489,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
in_paragraph = false;
|
||||
paragraph_in_wrapped_bold_run = false;
|
||||
}
|
||||
output.push('\n');
|
||||
output.push_str(table_md);
|
||||
@@ -589,7 +506,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
in_paragraph = false;
|
||||
paragraph_in_wrapped_bold_run = false;
|
||||
}
|
||||
output.push('\n');
|
||||
output.push_str(image_md);
|
||||
@@ -611,18 +527,9 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
&& y_gap.abs() <= para_threshold
|
||||
&& (prev_x - line_x).abs() > 50.0
|
||||
&& prev_y < f32::MAX;
|
||||
let line_all_bold = !line.items.is_empty() && line.items.iter().all(|item| item.is_bold);
|
||||
let line_in_wrapped_bold_run = wrapped_bold_paragraph_lines.contains(&line_idx);
|
||||
let is_bold_to_regular_break = in_paragraph
|
||||
&& paragraph_in_wrapped_bold_run
|
||||
&& !line_in_wrapped_bold_run
|
||||
&& !line_all_bold
|
||||
&& y_gap > base_size * 1.2
|
||||
&& y_gap <= para_threshold;
|
||||
if (is_para_break || is_band_switch || is_bold_to_regular_break) && in_paragraph {
|
||||
if (is_para_break || is_band_switch) && in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
in_paragraph = false;
|
||||
paragraph_in_wrapped_bold_run = false;
|
||||
}
|
||||
// Don't immediately end list on paragraph break
|
||||
// Let the continuation check below decide if we're still in a list
|
||||
@@ -630,11 +537,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
prev_x = line_x;
|
||||
|
||||
// Get text with optional bold/italic formatting
|
||||
let text = line.text_with_formatting(
|
||||
options.detect_bold,
|
||||
options.detect_italic,
|
||||
options.detect_underline,
|
||||
);
|
||||
let text = line.text_with_formatting(options.detect_bold, options.detect_italic);
|
||||
let trimmed = text.trim();
|
||||
|
||||
// Also get plain text for pattern matching (list detection, captions, etc.)
|
||||
@@ -669,7 +572,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
in_paragraph = false;
|
||||
paragraph_in_wrapped_bold_run = false;
|
||||
}
|
||||
output.push_str(trimmed);
|
||||
output.push_str("\n\n");
|
||||
@@ -703,22 +605,11 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
_ => false,
|
||||
};
|
||||
|
||||
// Lines explicitly tagged with a non-heading content role must never
|
||||
// be promoted by the visual heuristic — a tagged list item, quote, or
|
||||
// code line can look exactly like a heading (short, isolated).
|
||||
let non_heading_role = struct_role
|
||||
.as_ref()
|
||||
.is_some_and(StructRole::is_non_heading_content);
|
||||
let heuristic_heading = if options.detect_headers
|
||||
&& !non_heading_role
|
||||
&& !is_code_line
|
||||
&& !looks_like_list_continuation
|
||||
&& plain_trimmed.len() > 3
|
||||
&& plain_trimmed.split_whitespace().count() <= 15
|
||||
&& !starts_with_bullet_marker(plain_trimmed)
|
||||
&& !is_toc_entry_line(plain_trimmed)
|
||||
&& !is_heading_fragment(plain_trimmed)
|
||||
&& toc_suppress_page != Some(line.page)
|
||||
{
|
||||
let line_font_size = line.items.first().map(|i| i.font_size).unwrap_or(base_size);
|
||||
detect_header_level(line_font_size, base_size, &heading_tiers).or_else(|| {
|
||||
@@ -734,9 +625,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
if !(1..=15).contains(&word_count) {
|
||||
return None;
|
||||
}
|
||||
if wrapped_bold_paragraph_lines.contains(&line_idx) {
|
||||
return None;
|
||||
}
|
||||
let rarity = font_size_rarity(line_font_size, &font_stats);
|
||||
let all_bold = !line.items.is_empty() && line.items.iter().all(|i| i.is_bold);
|
||||
let standalone = !in_paragraph;
|
||||
@@ -754,11 +642,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
// paragraph continuity and minor font-size variation
|
||||
// inflates rarity scores.
|
||||
let has_strong_signal = all_bold || isolated || (rarity >= 0.97 && word_count <= 8);
|
||||
// Single-word headings ("IMPLEMENTATION", "CONTENTS") are common;
|
||||
// accept them only with the strongest signal combination.
|
||||
let enough_words =
|
||||
word_count >= 2 || (all_bold && isolated && plain_trimmed.len() >= 4);
|
||||
if score >= 0.5 && standalone && enough_words && has_strong_signal {
|
||||
if score >= 0.5 && standalone && word_count >= 2 && has_strong_signal {
|
||||
Some(bold_heading_level(&heading_tiers))
|
||||
} else {
|
||||
None
|
||||
@@ -772,20 +656,10 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
in_paragraph = false;
|
||||
paragraph_in_wrapped_bold_run = false;
|
||||
}
|
||||
let prefix = "#".repeat(level);
|
||||
// Plain text for headers (no redundant bold/italic inside `#`),
|
||||
// but underline is preserved: `<u>` carries meaning `#` doesn't.
|
||||
let heading_text = if options.detect_underline {
|
||||
line.text_with_formatting(false, false, true)
|
||||
} else {
|
||||
plain_text.clone()
|
||||
};
|
||||
output.push_str(&format!("{} {}\n\n", prefix, heading_text.trim()));
|
||||
if is_toc_marker_heading(plain_trimmed) {
|
||||
toc_suppress_page = Some(line.page);
|
||||
}
|
||||
// Use plain text for headers to avoid redundant formatting
|
||||
output.push_str(&format!("{} {}\n\n", prefix, plain_trimmed));
|
||||
in_list = false;
|
||||
continue;
|
||||
}
|
||||
@@ -804,7 +678,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
in_paragraph = false;
|
||||
paragraph_in_wrapped_bold_run = false;
|
||||
}
|
||||
output.push_str(&format!("- {}", trimmed));
|
||||
output.push('\n');
|
||||
@@ -818,7 +691,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
in_paragraph = false;
|
||||
paragraph_in_wrapped_bold_run = false;
|
||||
}
|
||||
let formatted = format_list_item(trimmed);
|
||||
output.push_str(&formatted);
|
||||
@@ -865,7 +737,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
in_paragraph = false;
|
||||
paragraph_in_wrapped_bold_run = false;
|
||||
}
|
||||
output.push_str(&format!("> {}\n", trimmed));
|
||||
continue;
|
||||
@@ -876,7 +747,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
in_paragraph = false;
|
||||
paragraph_in_wrapped_bold_run = false;
|
||||
}
|
||||
if !in_code_block {
|
||||
output.push_str("```\n");
|
||||
@@ -897,11 +767,6 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
}
|
||||
}
|
||||
output.push_str(trimmed);
|
||||
paragraph_in_wrapped_bold_run = if in_paragraph {
|
||||
paragraph_in_wrapped_bold_run || line_in_wrapped_bold_run
|
||||
} else {
|
||||
line_in_wrapped_bold_run
|
||||
};
|
||||
in_paragraph = true;
|
||||
prev_had_dot_leaders = cur_dot_leaders;
|
||||
}
|
||||
@@ -971,8 +836,6 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
let para_threshold = compute_paragraph_threshold(&lines, base_size);
|
||||
|
||||
let isolated_lines = find_isolated_lines(&lines, base_size, para_threshold);
|
||||
let wrapped_bold_paragraph_lines =
|
||||
find_wrapped_bold_paragraph_lines(&lines, base_size, para_threshold);
|
||||
|
||||
let mut output = String::new();
|
||||
let mut current_page = 0u32;
|
||||
@@ -981,8 +844,6 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
let mut in_paragraph = false;
|
||||
let mut last_list_x: Option<f32> = None;
|
||||
let mut prev_had_dot_leaders = false;
|
||||
let mut paragraph_in_wrapped_bold_run = false;
|
||||
let mut toc_suppress_page: Option<u32> = None;
|
||||
|
||||
for (line_idx, line) in lines.iter().enumerate() {
|
||||
// Page break
|
||||
@@ -999,7 +860,6 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
in_list = false;
|
||||
last_list_x = None;
|
||||
prev_had_dot_leaders = false;
|
||||
paragraph_in_wrapped_bold_run = false;
|
||||
|
||||
if options.include_page_numbers {
|
||||
output.push_str(&format!("<!-- Page {} -->\n\n", current_page));
|
||||
@@ -1010,29 +870,16 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
// (newspaper columns emitted sequentially on the same page).
|
||||
let y_gap = prev_y - line.y;
|
||||
let is_para_break = y_gap.abs() > para_threshold;
|
||||
let line_all_bold = !line.items.is_empty() && line.items.iter().all(|item| item.is_bold);
|
||||
let line_in_wrapped_bold_run = wrapped_bold_paragraph_lines.contains(&line_idx);
|
||||
let is_bold_to_regular_break = in_paragraph
|
||||
&& paragraph_in_wrapped_bold_run
|
||||
&& !line_in_wrapped_bold_run
|
||||
&& !line_all_bold
|
||||
&& y_gap > base_size * 1.2
|
||||
&& y_gap <= para_threshold;
|
||||
if (is_para_break || is_bold_to_regular_break) && in_paragraph {
|
||||
if is_para_break && in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
in_paragraph = false;
|
||||
paragraph_in_wrapped_bold_run = false;
|
||||
}
|
||||
// Don't immediately end list on paragraph break
|
||||
// Let the continuation check below decide if we're still in a list
|
||||
prev_y = line.y;
|
||||
|
||||
// Get text with optional bold/italic formatting
|
||||
let text = line.text_with_formatting(
|
||||
options.detect_bold,
|
||||
options.detect_italic,
|
||||
options.detect_underline,
|
||||
);
|
||||
let text = line.text_with_formatting(options.detect_bold, options.detect_italic);
|
||||
let trimmed = text.trim();
|
||||
|
||||
// Also get plain text for pattern matching
|
||||
@@ -1049,7 +896,6 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
in_paragraph = false;
|
||||
paragraph_in_wrapped_bold_run = false;
|
||||
}
|
||||
output.push_str(trimmed);
|
||||
output.push_str("\n\n");
|
||||
@@ -1061,10 +907,6 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
if options.detect_headers
|
||||
&& plain_trimmed.len() > 3
|
||||
&& plain_trimmed.split_whitespace().count() <= 15
|
||||
&& !is_toc_entry_line(plain_trimmed)
|
||||
&& !is_heading_fragment(plain_trimmed)
|
||||
&& toc_suppress_page != Some(line.page)
|
||||
&& !(options.detect_code && line.items.iter().any(|i| is_monospace_font(&i.font)))
|
||||
{
|
||||
let line_font_size = line.items.first().map(|i| i.font_size).unwrap_or(base_size);
|
||||
if let Some(header_level) =
|
||||
@@ -1076,9 +918,6 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
if !(1..=15).contains(&word_count) {
|
||||
return None;
|
||||
}
|
||||
if wrapped_bold_paragraph_lines.contains(&line_idx) {
|
||||
return None;
|
||||
}
|
||||
let rarity = font_size_rarity(line_font_size, &font_stats);
|
||||
let all_bold = !line.items.is_empty() && line.items.iter().all(|i| i.is_bold);
|
||||
let standalone = !in_paragraph;
|
||||
@@ -1087,9 +926,7 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
+ if all_bold { 0.3 } else { 0.0 }
|
||||
+ if standalone { 0.2 } else { 0.0 }
|
||||
+ if isolated { 0.3 } else { 0.0 };
|
||||
let enough_words =
|
||||
word_count >= 2 || (all_bold && isolated && plain_trimmed.len() >= 4);
|
||||
if score >= 0.5 && standalone && enough_words {
|
||||
if score >= 0.5 && standalone && word_count >= 2 {
|
||||
return Some(bold_heading_level(&heading_tiers));
|
||||
}
|
||||
None
|
||||
@@ -1098,19 +935,10 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
in_paragraph = false;
|
||||
paragraph_in_wrapped_bold_run = false;
|
||||
}
|
||||
let prefix = "#".repeat(header_level);
|
||||
// Plain text for headers, except underline (see above).
|
||||
let heading_text = if options.detect_underline {
|
||||
line.text_with_formatting(false, false, true)
|
||||
} else {
|
||||
plain_text.clone()
|
||||
};
|
||||
output.push_str(&format!("{} {}\n\n", prefix, heading_text.trim()));
|
||||
if is_toc_marker_heading(plain_trimmed) {
|
||||
toc_suppress_page = Some(line.page);
|
||||
}
|
||||
// Use plain text for headers to avoid redundant formatting
|
||||
output.push_str(&format!("{} {}\n\n", prefix, plain_trimmed));
|
||||
in_list = false;
|
||||
continue;
|
||||
}
|
||||
@@ -1121,7 +949,6 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
in_paragraph = false;
|
||||
paragraph_in_wrapped_bold_run = false;
|
||||
}
|
||||
let formatted = format_list_item(trimmed);
|
||||
output.push_str(&formatted);
|
||||
@@ -1166,7 +993,6 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
in_paragraph = false;
|
||||
paragraph_in_wrapped_bold_run = false;
|
||||
}
|
||||
// Use plain text for code blocks
|
||||
output.push_str(&format!("```\n{}\n```\n", plain_trimmed));
|
||||
@@ -1184,11 +1010,6 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
}
|
||||
}
|
||||
output.push_str(trimmed);
|
||||
paragraph_in_wrapped_bold_run = if in_paragraph {
|
||||
paragraph_in_wrapped_bold_run || line_in_wrapped_bold_run
|
||||
} else {
|
||||
line_in_wrapped_bold_run
|
||||
};
|
||||
in_paragraph = true;
|
||||
prev_had_dot_leaders = cur_dot_leaders;
|
||||
}
|
||||
@@ -1221,8 +1042,6 @@ mod tests {
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: crate::types::ItemType::Text,
|
||||
mcid,
|
||||
}
|
||||
@@ -1239,42 +1058,6 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
fn line_at(text: &str, page: u32, y: f32) -> TextLine {
|
||||
let mut item = make_item(text, page, None);
|
||||
item.y = y;
|
||||
make_line(vec![item])
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn isolated_lines_kept_on_sparse_pages() {
|
||||
// A ToC page with a lone title and one entry far below: the density
|
||||
// ratio is 50% but the page is too sparse for the multi-column
|
||||
// misfire the guard targets — the title must stay isolated.
|
||||
let lines = vec![
|
||||
line_at("CONTENTS", 1, 700.0),
|
||||
line_at("Chapter One 5", 1, 500.0),
|
||||
];
|
||||
let isolated = find_isolated_lines(&lines, 12.0, 20.0);
|
||||
assert!(
|
||||
isolated.contains(&0),
|
||||
"sparse-page title must stay isolated"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn isolated_lines_wiped_on_dense_pages() {
|
||||
// 12 short lines all with paragraph gaps — the multi-column misfire
|
||||
// shape. The guard must clear them all.
|
||||
let lines: Vec<TextLine> = (0..12)
|
||||
.map(|i| line_at("Short column line", 1, 700.0 - i as f32 * 50.0))
|
||||
.collect();
|
||||
let isolated = find_isolated_lines(&lines, 12.0, 20.0);
|
||||
assert!(
|
||||
isolated.is_empty(),
|
||||
"dense page of isolated lines must be wiped"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_struct_role_heading() {
|
||||
let lines = vec![
|
||||
@@ -1573,109 +1356,6 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_wrapped_bold_abstract_is_not_split_into_headings() {
|
||||
// Regression for arXiv 1107.1353: the opening abstract paragraph is
|
||||
// entirely bold at body size. The first wrapped lines used to become
|
||||
// separate H2 headings, and the following body paragraph was joined to
|
||||
// the bold abstract because the paragraph gap is modest.
|
||||
let make = |text: &str, y: f32, font_size: f32, bold: bool| {
|
||||
let mut item = make_item(text, 1, None);
|
||||
item.y = y;
|
||||
item.font_size = font_size;
|
||||
item.height = font_size;
|
||||
item.is_bold = bold;
|
||||
item
|
||||
};
|
||||
|
||||
let lines = vec![
|
||||
make_line(vec![make(
|
||||
"Quantum Nature of Light Measured With a Single Detector",
|
||||
747.7,
|
||||
25.0,
|
||||
true,
|
||||
)]),
|
||||
make_line(vec![make(
|
||||
"Gesine A. Steudle1*, Stefan Schietinger1, David Höckel1",
|
||||
651.1,
|
||||
11.0,
|
||||
false,
|
||||
)]),
|
||||
make_line(vec![make(
|
||||
"Zwiller2, and Oliver Benson1",
|
||||
638.5,
|
||||
11.0,
|
||||
false,
|
||||
)]),
|
||||
make_line(vec![make(
|
||||
"The introduction of light quanta by Einstein in 1905 triggered strong efforts to",
|
||||
607.5,
|
||||
11.0,
|
||||
true,
|
||||
)]),
|
||||
make_line(vec![make(
|
||||
"demonstrate the quantum properties of light directly, without involving matter",
|
||||
594.8,
|
||||
11.0,
|
||||
true,
|
||||
)]),
|
||||
make_line(vec![make(
|
||||
"quantization. It however took more than seven decades for the quantum granularity",
|
||||
582.2,
|
||||
11.0,
|
||||
true,
|
||||
)]),
|
||||
make_line(vec![make(
|
||||
"of light to be observed in the fluorescence of single atoms. Single atoms emit",
|
||||
569.5,
|
||||
11.0,
|
||||
true,
|
||||
)]),
|
||||
make_line(vec![make(
|
||||
"photons one at a time, this is typically demonstrated with a Hanbury-Brown-Twiss",
|
||||
556.9,
|
||||
11.0,
|
||||
true,
|
||||
)]),
|
||||
make_line(vec![make(
|
||||
"Our work significantly simplifies a widely used photon-correlation technique.",
|
||||
544.2,
|
||||
11.0,
|
||||
true,
|
||||
)]),
|
||||
make_line(vec![make(
|
||||
"A photon is a single excitation of a mode of the electromagnetic field.",
|
||||
528.7,
|
||||
11.0,
|
||||
false,
|
||||
)]),
|
||||
];
|
||||
|
||||
let md = to_markdown_from_lines_with_tables_and_images(
|
||||
lines,
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
None,
|
||||
);
|
||||
|
||||
assert!(
|
||||
md.contains("# Quantum Nature of Light Measured With a Single Detector"),
|
||||
"title should remain a heading: {md}"
|
||||
);
|
||||
assert!(
|
||||
!md.contains("## The introduction")
|
||||
&& !md.contains("## demonstrate")
|
||||
&& !md.contains("## quantization"),
|
||||
"bold abstract lines should not become headings: {md}"
|
||||
);
|
||||
assert!(
|
||||
md.contains("technique.**\n\nA photon is a single excitation"),
|
||||
"body paragraph should be separated from bold abstract: {md}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_struct_role_code_multiline_accumulation() {
|
||||
let mut line1 = make_item("fn main() {", 1, Some(0));
|
||||
|
||||
@@ -400,8 +400,6 @@ pub struct MarkdownOptions {
|
||||
pub detect_bold: bool,
|
||||
/// Detect and format italic text from font names
|
||||
pub detect_italic: bool,
|
||||
/// Emit `<u>` runs for text with a geometrically-detected underline
|
||||
pub detect_underline: bool,
|
||||
/// Include image placeholders in output
|
||||
pub include_images: bool,
|
||||
/// Include extracted hyperlinks
|
||||
@@ -424,7 +422,6 @@ impl Default for MarkdownOptions {
|
||||
fix_hyphenation: true,
|
||||
detect_bold: true,
|
||||
detect_italic: true,
|
||||
detect_underline: true,
|
||||
// `include_images: false` is intentional. The content-stream walker
|
||||
// now emits `ItemType::Image` `TextItem`s for every Image XObject
|
||||
// it encounters (see `extractor/content_stream.rs`). If we rendered
|
||||
@@ -1220,8 +1217,6 @@ mod tests {
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: crate::types::ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
|
||||
@@ -29,8 +29,6 @@ pub(crate) fn clean_markdown(mut text: String, options: &MarkdownOptions) -> Str
|
||||
// text item, which combine with gap-based space insertion to produce
|
||||
// double spaces ("Vice President" instead of "Vice President").
|
||||
collapse_consecutive_spaces(&mut text);
|
||||
remove_spaces_before_closing_brackets(&mut text);
|
||||
remove_spaces_before_sentence_punctuation(&mut text);
|
||||
|
||||
// Remove excessive newlines (more than 2 in a row)
|
||||
while text.contains("\n\n\n") {
|
||||
@@ -73,46 +71,6 @@ fn collapse_consecutive_spaces(text: &mut String) {
|
||||
*text = result;
|
||||
}
|
||||
|
||||
/// Remove spaces before closing square brackets.
|
||||
/// Unit markers and markdown links occasionally pick up a gap-inserted space
|
||||
/// before `]` (e.g. `[kg/m3 ]`), which is cosmetic padding.
|
||||
fn remove_spaces_before_closing_brackets(text: &mut String) {
|
||||
let mut result = String::with_capacity(text.len());
|
||||
for ch in text.chars() {
|
||||
if ch == ']' && result.ends_with(' ') {
|
||||
result.pop();
|
||||
}
|
||||
result.push(ch);
|
||||
}
|
||||
*text = result;
|
||||
}
|
||||
|
||||
/// Remove a stray space before sentence punctuation ("word ." → "word.").
|
||||
/// Style-boundary item splits (bold/italic/underline runs) can strand a
|
||||
/// trailing period or comma in its own fragment, and several assembly paths
|
||||
/// join fragments with spaces. Only fires when the punctuation ends the
|
||||
/// token (followed by whitespace or end of text), so decimals ("3 .14" stays
|
||||
/// untouched — no such input exists, but the guard is cheap) and dot leaders
|
||||
/// (" ... ") are unaffected.
|
||||
fn remove_spaces_before_sentence_punctuation(text: &mut String) {
|
||||
let chars: Vec<char> = text.chars().collect();
|
||||
let mut result = String::with_capacity(text.len());
|
||||
for (i, &ch) in chars.iter().enumerate() {
|
||||
if matches!(ch, '.' | ',' | ';') && result.ends_with(' ') {
|
||||
let next = chars.get(i + 1);
|
||||
// `|` counts as a token end so table cells get the same fix.
|
||||
let token_ends = next.is_none_or(|c| c.is_whitespace() || *c == '|');
|
||||
// Never touch runs of dots (ellipsis / dot leaders).
|
||||
let in_dot_run = ch == '.' && next == Some(&'.');
|
||||
if token_ends && !in_dot_run {
|
||||
result.pop();
|
||||
}
|
||||
}
|
||||
result.push(ch);
|
||||
}
|
||||
*text = result;
|
||||
}
|
||||
|
||||
/// Collapse dot leaders (runs of 4+ dots) into " ... "
|
||||
/// Common in tables of contents: "Introduction...............................1" -> "Introduction ... 1"
|
||||
fn collapse_dot_leaders(text: &str) -> String {
|
||||
@@ -384,48 +342,6 @@ mod tests {
|
||||
assert!(result.contains("Chapter 2 ... 20"));
|
||||
}
|
||||
|
||||
// --- remove_spaces_before_closing_brackets ---
|
||||
|
||||
#[test]
|
||||
fn test_remove_spaces_before_closing_brackets() {
|
||||
let mut input = "Density [kg/m3 ] and [linked text ](https://example.com)".to_string();
|
||||
remove_spaces_before_closing_brackets(&mut input);
|
||||
assert_eq!(
|
||||
input,
|
||||
"Density [kg/m3] and [linked text](https://example.com)"
|
||||
);
|
||||
}
|
||||
|
||||
// --- remove_spaces_before_sentence_punctuation ---
|
||||
|
||||
#[test]
|
||||
fn strips_space_before_trailing_period() {
|
||||
let mut t = "Foreign insurance companies . The provisions".to_string();
|
||||
remove_spaces_before_sentence_punctuation(&mut t);
|
||||
assert_eq!(t, "Foreign insurance companies. The provisions");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strips_space_before_period_at_cell_boundary() {
|
||||
let mut t = "|Applicability date .|This section|".to_string();
|
||||
remove_spaces_before_sentence_punctuation(&mut t);
|
||||
assert_eq!(t, "|Applicability date.|This section|");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn keeps_dot_leaders_and_ellipses() {
|
||||
let mut t = "Introduction ... 1".to_string();
|
||||
remove_spaces_before_sentence_punctuation(&mut t);
|
||||
assert_eq!(t, "Introduction ... 1");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn keeps_mid_token_periods() {
|
||||
let mut t = "version 3 .14 released".to_string();
|
||||
remove_spaces_before_sentence_punctuation(&mut t);
|
||||
assert_eq!(t, "version 3 .14 released");
|
||||
}
|
||||
|
||||
// --- fix_hyphenation ---
|
||||
|
||||
#[test]
|
||||
|
||||
+1
-102
@@ -3,7 +3,7 @@
|
||||
use std::collections::{HashMap, HashSet};
|
||||
|
||||
use crate::structure_tree::StructRole;
|
||||
use crate::types::{TextItem, TextLine};
|
||||
use crate::types::TextLine;
|
||||
|
||||
use super::analysis::detect_header_level;
|
||||
|
||||
@@ -87,41 +87,6 @@ pub(crate) fn merge_heading_lines(
|
||||
false
|
||||
};
|
||||
|
||||
// Bold headings at body font size never reach a tier, so wrapped ones
|
||||
// split into two output headings ("…of wood pellets and cost" /
|
||||
// "structure in Japan"). Merge a fully-bold line into the previous
|
||||
// fully-bold line when it reads as a wrap continuation: starts
|
||||
// lowercase, tiny Y gap, and the previous line has no terminal
|
||||
// punctuation. Kept deliberately narrow — bold list labels and bold
|
||||
// sentences start with markers or capitals and are unaffected.
|
||||
let should_merge = should_merge
|
||||
|| if let Some(prev) = result.last() {
|
||||
let all_bold = |l: &TextLine| {
|
||||
!l.items.is_empty() && l.items.iter().all(|i: &TextItem| i.is_bold)
|
||||
};
|
||||
let prev_text = prev.text();
|
||||
let prev_trim = prev_text.trim_end();
|
||||
let curr_text = line.text();
|
||||
let curr_trim = curr_text.trim();
|
||||
let y_gap = prev.y - line.y;
|
||||
// Both lines must be tier-less: a tiered/tagged bold heading
|
||||
// followed by bold body text must not absorb it.
|
||||
line_level.is_none()
|
||||
&& effective_heading_level(prev, base_size, heading_tiers, struct_roles)
|
||||
.is_none()
|
||||
&& prev.page == line.page
|
||||
&& all_bold(prev)
|
||||
&& all_bold(&line)
|
||||
&& y_gap > 0.0
|
||||
&& y_gap < line_font * 1.6
|
||||
&& curr_trim.chars().next().is_some_and(|c| c.is_lowercase())
|
||||
&& !prev_trim.ends_with(['.', ':', ';', '!', '?'])
|
||||
&& prev_trim.split_whitespace().count() + curr_trim.split_whitespace().count()
|
||||
<= 20
|
||||
} else {
|
||||
false
|
||||
};
|
||||
|
||||
if should_merge {
|
||||
// Append this line's items to the previous line
|
||||
let prev = result.last_mut().unwrap();
|
||||
@@ -577,8 +542,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid,
|
||||
}
|
||||
@@ -720,68 +683,4 @@ mod tests {
|
||||
.unwrap();
|
||||
assert_eq!(first_header.page, 1, "first occurrence should be on page 1");
|
||||
}
|
||||
|
||||
fn make_bold_line(text: &str, page: u32, y: f32) -> TextLine {
|
||||
let mut item = make_item(text, 12.0, None);
|
||||
item.is_bold = true;
|
||||
TextLine {
|
||||
items: vec![item],
|
||||
y,
|
||||
page,
|
||||
adaptive_threshold: 0.10,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_wrapped_bold_heading_lowercase_continuation() {
|
||||
// Bold-at-body-size heading wrapped across two lines: the second line
|
||||
// starts lowercase and must merge into the first.
|
||||
let lines = vec![
|
||||
make_bold_line(
|
||||
"3. Perspective of supply and demand balance and cost",
|
||||
1,
|
||||
700.0,
|
||||
),
|
||||
make_bold_line("structure in Japan", 1, 686.0),
|
||||
make_line("Body text paragraph follows here.", 12.0, 1, 660.0, None),
|
||||
];
|
||||
let result = merge_heading_lines(lines, 12.0, &[], None);
|
||||
assert_eq!(result.len(), 2, "wrapped bold heading should merge");
|
||||
assert!(result[0].text().contains("cost structure in Japan"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn no_merge_for_bold_sentences_or_new_headings() {
|
||||
// Second bold line starts with a capital — a new heading or label,
|
||||
// not a wrap continuation.
|
||||
let lines = vec![
|
||||
make_bold_line("Replace", 1, 700.0),
|
||||
make_bold_line("Trash", 1, 686.0),
|
||||
];
|
||||
let result = merge_heading_lines(lines, 12.0, &[], None);
|
||||
assert_eq!(result.len(), 2, "distinct bold lines must not merge");
|
||||
|
||||
// Previous line ends a sentence — continuation must not merge.
|
||||
let lines = vec![
|
||||
make_bold_line("This is a bold sentence.", 1, 700.0),
|
||||
make_bold_line("another bold line", 1, 686.0),
|
||||
];
|
||||
let result = merge_heading_lines(lines, 12.0, &[], None);
|
||||
assert_eq!(result.len(), 2, "sentence-final bold line must not merge");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tiered_bold_heading_does_not_absorb_bold_body() {
|
||||
// Previous line is a tier-level bold heading (16pt vs 12pt body);
|
||||
// a following lowercase bold body line must NOT merge into it.
|
||||
let mut heading = make_bold_line("Section Title", 1, 700.0);
|
||||
heading.items[0].font_size = 16.0;
|
||||
heading.items[0].height = 16.0;
|
||||
let lines = vec![
|
||||
heading,
|
||||
make_bold_line("emphasized body text continues here", 1, 686.0),
|
||||
];
|
||||
let result = merge_heading_lines(lines, 12.0, &[16.0], None);
|
||||
assert_eq!(result.len(), 2, "tiered heading must not absorb bold body");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -30,9 +30,6 @@ pub struct PyPdfResult {
|
||||
/// 1-indexed page numbers that need OCR.
|
||||
#[pyo3(get)]
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
/// Machine-readable OCR reasons by 1-indexed page.
|
||||
#[pyo3(get)]
|
||||
pub ocr_reasons_by_page: Vec<PyPageOcrReasons>,
|
||||
/// Title from PDF metadata.
|
||||
#[pyo3(get)]
|
||||
pub title: Option<String>,
|
||||
@@ -63,28 +60,6 @@ impl PyPdfResult {
|
||||
}
|
||||
}
|
||||
|
||||
/// OCR reasons for a single 1-indexed page.
|
||||
#[pyclass(name = "PageOcrReasons")]
|
||||
#[derive(Clone)]
|
||||
pub struct PyPageOcrReasons {
|
||||
/// 1-indexed page number.
|
||||
#[pyo3(get)]
|
||||
pub page: u32,
|
||||
/// Machine-readable OCR reason identifiers.
|
||||
#[pyo3(get)]
|
||||
pub reasons: Vec<String>,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
impl PyPageOcrReasons {
|
||||
fn __repr__(&self) -> String {
|
||||
format!(
|
||||
"PageOcrReasons(page={}, reasons={:?})",
|
||||
self.page, self.reasons
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Classification wrapper (lightweight)
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -131,9 +106,6 @@ pub struct PyRegionText {
|
||||
/// True when the text should not be trusted (empty, GID fonts, garbage, encoding issues).
|
||||
#[pyo3(get)]
|
||||
pub needs_ocr: bool,
|
||||
/// Machine-readable OCR reason when the cause is known.
|
||||
#[pyo3(get)]
|
||||
pub ocr_reason: Option<String>,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
@@ -188,9 +160,6 @@ pub struct PyPageMarkdown {
|
||||
/// encoding issues, garbage text, or empty extraction).
|
||||
#[pyo3(get)]
|
||||
pub needs_ocr: bool,
|
||||
/// Machine-readable OCR reason when the cause is known.
|
||||
#[pyo3(get)]
|
||||
pub ocr_reason: Option<String>,
|
||||
}
|
||||
|
||||
#[pymethods]
|
||||
@@ -221,9 +190,6 @@ pub struct PyPagesExtractionResult {
|
||||
/// 1-indexed pages that need OCR (scanned/image-based or unreliable text).
|
||||
#[pyo3(get)]
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
/// Machine-readable OCR reasons by 1-indexed page.
|
||||
#[pyo3(get)]
|
||||
pub ocr_reasons_by_page: Vec<PyPageOcrReasons>,
|
||||
/// True if any page has tables or columns.
|
||||
#[pyo3(get)]
|
||||
pub is_complex: bool,
|
||||
@@ -266,10 +232,6 @@ pub struct PyTextItem {
|
||||
#[pyo3(get)]
|
||||
pub is_italic: bool,
|
||||
#[pyo3(get)]
|
||||
pub is_underline: bool,
|
||||
#[pyo3(get)]
|
||||
pub is_strikeout: bool,
|
||||
#[pyo3(get)]
|
||||
pub item_type: String,
|
||||
}
|
||||
|
||||
@@ -306,7 +268,6 @@ fn to_py_result(r: crate::PdfProcessResult) -> PyPdfResult {
|
||||
page_count: r.page_count,
|
||||
processing_time_ms: r.processing_time_ms,
|
||||
pages_needing_ocr: r.pages_needing_ocr,
|
||||
ocr_reasons_by_page: to_py_page_ocr_reasons(r.ocr_reasons_by_page),
|
||||
title: r.title,
|
||||
confidence: r.confidence,
|
||||
is_complex_layout: r.layout.is_complex,
|
||||
@@ -316,16 +277,6 @@ fn to_py_result(r: crate::PdfProcessResult) -> PyPdfResult {
|
||||
}
|
||||
}
|
||||
|
||||
fn to_py_page_ocr_reasons(reasons: Vec<crate::PageOcrReasons>) -> Vec<PyPageOcrReasons> {
|
||||
reasons
|
||||
.into_iter()
|
||||
.map(|reason| PyPageOcrReasons {
|
||||
page: reason.page,
|
||||
reasons: reason.reasons,
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn to_py_err(e: crate::PdfError) -> PyErr {
|
||||
PyValueError::new_err(e.to_string())
|
||||
}
|
||||
@@ -353,8 +304,6 @@ fn convert_text_items(items: Vec<crate::TextItem>) -> Vec<PyTextItem> {
|
||||
page: item.page,
|
||||
is_bold: item.is_bold,
|
||||
is_italic: item.is_italic,
|
||||
is_underline: item.is_underline,
|
||||
is_strikeout: item.is_strikeout,
|
||||
item_type: item_type_str(&item.item_type),
|
||||
})
|
||||
.collect()
|
||||
@@ -401,13 +350,11 @@ fn to_py_pages_result(r: crate::PagesExtractionResult) -> PyPagesExtractionResul
|
||||
page: p.page,
|
||||
markdown: p.markdown,
|
||||
needs_ocr: p.needs_ocr,
|
||||
ocr_reason: p.ocr_reason,
|
||||
})
|
||||
.collect(),
|
||||
pages_with_tables: r.pages_with_tables,
|
||||
pages_with_columns: r.pages_with_columns,
|
||||
pages_needing_ocr: r.pages_needing_ocr,
|
||||
ocr_reasons_by_page: to_py_page_ocr_reasons(r.ocr_reasons_by_page),
|
||||
is_complex: r.is_complex,
|
||||
}
|
||||
}
|
||||
@@ -423,7 +370,6 @@ fn convert_region_results(results: Vec<crate::PageRegionResult>) -> Vec<PyPageRe
|
||||
.map(|r| PyRegionText {
|
||||
text: r.text,
|
||||
needs_ocr: r.needs_ocr,
|
||||
ocr_reason: r.ocr_reason,
|
||||
})
|
||||
.collect(),
|
||||
})
|
||||
@@ -617,7 +563,6 @@ fn extract_pages_markdown_bytes(
|
||||
#[pymodule]
|
||||
fn pdf_inspector(m: &Bound<'_, PyModule>) -> PyResult<()> {
|
||||
m.add_class::<PyPdfResult>()?;
|
||||
m.add_class::<PyPageOcrReasons>()?;
|
||||
m.add_class::<PyPdfClassification>()?;
|
||||
m.add_class::<PyTextItem>()?;
|
||||
m.add_class::<PyRegionText>()?;
|
||||
|
||||
@@ -76,52 +76,6 @@ pub enum StructRole {
|
||||
}
|
||||
|
||||
impl StructRole {
|
||||
/// Content roles whose text must never be promoted to a heading by the
|
||||
/// visual heuristic. These carry an explicit non-heading meaning in the
|
||||
/// struct tree (lists, quotes, notes, references, captions, formulas,
|
||||
/// forms, ToC entries), yet their text is often short and visually
|
||||
/// isolated — exactly what the heuristic keys on. Heading roles (H, H1–H6)
|
||||
/// and generic container/flow roles (P, Div, Sect, Span, …) are excluded
|
||||
/// so the heuristic can still fire there.
|
||||
///
|
||||
/// `Figure` is deliberately NOT in this set: cover/banner pages routinely
|
||||
/// tag the document title inside a Figure (alongside a seal or logo), and
|
||||
/// that title is a real heading. `Formula` and `Form` stay — a line
|
||||
/// explicitly tagged as an equation or form field is never a heading.
|
||||
///
|
||||
/// Table roles (Table/TR/TH/TD/THead/TBody/TFoot) are included so that
|
||||
/// when table reconstruction falls back and cells reach the line loop as
|
||||
/// plain text, a short isolated cell — a `TH` column header especially —
|
||||
/// is not promoted to a heading.
|
||||
pub(crate) fn is_non_heading_content(&self) -> bool {
|
||||
matches!(
|
||||
self,
|
||||
Self::L
|
||||
| Self::LI
|
||||
| Self::Lbl
|
||||
| Self::LBody
|
||||
| Self::BlockQuote
|
||||
| Self::Quote
|
||||
| Self::Caption
|
||||
| Self::TOC
|
||||
| Self::TOCI
|
||||
| Self::Index
|
||||
| Self::Note
|
||||
| Self::Reference
|
||||
| Self::BibEntry
|
||||
| Self::Code
|
||||
| Self::Formula
|
||||
| Self::Form
|
||||
| Self::Table
|
||||
| Self::TR
|
||||
| Self::TH
|
||||
| Self::TD
|
||||
| Self::THead
|
||||
| Self::TBody
|
||||
| Self::TFoot
|
||||
)
|
||||
}
|
||||
|
||||
fn from_name(name: &str) -> Self {
|
||||
match name {
|
||||
"Document" => Self::Document,
|
||||
@@ -902,54 +856,6 @@ fn contains_bytes(haystack: &[u8], needle: &[u8]) -> bool {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn non_heading_content_roles() {
|
||||
for r in [
|
||||
StructRole::L,
|
||||
StructRole::LI,
|
||||
StructRole::BlockQuote,
|
||||
StructRole::Quote,
|
||||
StructRole::Caption,
|
||||
StructRole::TOC,
|
||||
StructRole::TOCI,
|
||||
StructRole::Index,
|
||||
StructRole::Note,
|
||||
StructRole::Reference,
|
||||
StructRole::BibEntry,
|
||||
StructRole::Code,
|
||||
StructRole::Formula,
|
||||
StructRole::Form,
|
||||
StructRole::Table,
|
||||
StructRole::TR,
|
||||
StructRole::TH,
|
||||
StructRole::TD,
|
||||
StructRole::THead,
|
||||
StructRole::TBody,
|
||||
StructRole::TFoot,
|
||||
] {
|
||||
assert!(
|
||||
r.is_non_heading_content(),
|
||||
"{r:?} should block heading promotion"
|
||||
);
|
||||
}
|
||||
// Heading and generic container/flow roles must NOT block promotion
|
||||
for r in [
|
||||
StructRole::H,
|
||||
StructRole::H1,
|
||||
StructRole::H3,
|
||||
StructRole::P,
|
||||
StructRole::Div,
|
||||
StructRole::Sect,
|
||||
StructRole::Span,
|
||||
StructRole::Figure,
|
||||
] {
|
||||
assert!(
|
||||
!r.is_non_heading_content(),
|
||||
"{r:?} should allow heading promotion"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_struct_role_from_name() {
|
||||
assert_eq!(StructRole::from_name("H1"), StructRole::H1);
|
||||
|
||||
@@ -104,8 +104,6 @@ pub(crate) fn merge_adjacent_items(items: &[TextItem]) -> (Vec<TextItem>, Vec<Ve
|
||||
page: first_item.page,
|
||||
is_bold: first_item.is_bold,
|
||||
is_italic: first_item.is_italic,
|
||||
is_underline: first_item.is_underline,
|
||||
is_strikeout: first_item.is_strikeout,
|
||||
item_type: first_item.item_type.clone(),
|
||||
mcid: first_item.mcid,
|
||||
});
|
||||
|
||||
@@ -393,8 +393,6 @@ mod tests {
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
|
||||
+1
-220
@@ -1124,13 +1124,12 @@ pub(crate) fn assign_items_to_grid(
|
||||
.unwrap_or(std::cmp::Ordering::Equal)
|
||||
})
|
||||
});
|
||||
let text = col_items
|
||||
let text: String = col_items
|
||||
.iter()
|
||||
.map(|(_, item)| item.text.trim())
|
||||
.filter(|t| !t.is_empty())
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
let text = remove_inner_delimiter_spaces(&text);
|
||||
row_cells.push(text);
|
||||
}
|
||||
cells.push(row_cells);
|
||||
@@ -1139,27 +1138,6 @@ pub(crate) fn assign_items_to_grid(
|
||||
(cells, indices)
|
||||
}
|
||||
|
||||
fn remove_inner_delimiter_spaces(text: &str) -> String {
|
||||
let chars: Vec<char> = text.chars().collect();
|
||||
let mut result = String::with_capacity(text.len());
|
||||
|
||||
for (i, &ch) in chars.iter().enumerate() {
|
||||
if ch == ' ' {
|
||||
let after_open =
|
||||
result.ends_with('(') || result.ends_with('[') || result.ends_with('{');
|
||||
let before_close = chars
|
||||
.get(i + 1)
|
||||
.is_some_and(|next| matches!(next, ')' | ']' | '}'));
|
||||
if after_open || before_close {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
result.push(ch);
|
||||
}
|
||||
|
||||
result
|
||||
}
|
||||
|
||||
/// Consolidate text in vertically-merged cells.
|
||||
///
|
||||
/// When a single rect spans multiple grid rows (e.g. a "Classification" label
|
||||
@@ -1473,14 +1451,6 @@ fn detect_row_stripe_table(
|
||||
(col_edges, cells)
|
||||
};
|
||||
let num_cols = col_edges.len() - 1;
|
||||
if row_stripe_is_sparse_prose_outline(&cells) {
|
||||
debug!(" row-stripe rejected: sparse outline/prose continuation shape");
|
||||
return None;
|
||||
}
|
||||
if has_dominant_prose_cell(&cells) {
|
||||
debug!(" row-stripe rejected: dominant prose cell (chart/figure region over body text)");
|
||||
return None;
|
||||
}
|
||||
|
||||
let column_centers: Vec<f32> = (0..num_cols)
|
||||
.map(|c| (col_edges[c] + col_edges[c + 1]) / 2.0)
|
||||
@@ -1499,85 +1469,6 @@ fn detect_row_stripe_table(
|
||||
Some(Table::new(column_centers, row_centers, cells, item_indices))
|
||||
}
|
||||
|
||||
/// Detect a grid that swallowed body text instead of tabular data.
|
||||
///
|
||||
/// Charts (bar graphs, axis gridlines) emit fields of drawing rects that can
|
||||
/// pass the row-stripe shape test; the resulting "table" then captures the
|
||||
/// page's prose. The signature: one cell holds an entire paragraph — ≥60 words
|
||||
/// AND at least a third of all words in the table.
|
||||
///
|
||||
/// There is deliberately no row-count exemption. A small table whose single
|
||||
/// long cell dominates its word count is indistinguishable by content from a
|
||||
/// phantom grid over body text, and across the regression corpora every such
|
||||
/// grid observed has been swallowed prose, never a real note table. The costs
|
||||
/// are also asymmetric: rejecting a real table degrades it to readable prose,
|
||||
/// while accepting a phantom scrambles the page into Y-interleaved cells.
|
||||
/// Larger legitimate tables are safe because the one-third-of-total threshold
|
||||
/// scales with table size.
|
||||
fn has_dominant_prose_cell(cells: &[Vec<String>]) -> bool {
|
||||
let mut total_words = 0usize;
|
||||
let mut max_cell_words = 0usize;
|
||||
for row in cells {
|
||||
for cell in row {
|
||||
let words = cell.split_whitespace().count();
|
||||
total_words += words;
|
||||
max_cell_words = max_cell_words.max(words);
|
||||
}
|
||||
}
|
||||
max_cell_words >= 60 && max_cell_words * 3 >= total_words
|
||||
}
|
||||
|
||||
fn row_stripe_is_sparse_prose_outline(cells: &[Vec<String>]) -> bool {
|
||||
let Some(num_cols) = cells.first().map(|row| row.len()) else {
|
||||
return false;
|
||||
};
|
||||
if num_cols != 2 || cells.len() < 4 {
|
||||
return false;
|
||||
}
|
||||
|
||||
let non_empty_rows = cells
|
||||
.iter()
|
||||
.filter(|row| row.iter().any(|cell| !cell.trim().is_empty()))
|
||||
.count();
|
||||
if non_empty_rows < 4 {
|
||||
return false;
|
||||
}
|
||||
|
||||
let mut col_counts = [0usize; 2];
|
||||
for row in cells {
|
||||
for (idx, cell) in row.iter().enumerate() {
|
||||
if !cell.trim().is_empty() {
|
||||
col_counts[idx] += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let (sparse_col, dense_col) = if col_counts[0] <= col_counts[1] {
|
||||
(0usize, 1usize)
|
||||
} else {
|
||||
(1usize, 0usize)
|
||||
};
|
||||
let sparse_count = col_counts[sparse_col];
|
||||
let dense_count = col_counts[dense_col];
|
||||
if sparse_count * 2 >= non_empty_rows || dense_count * 3 < non_empty_rows * 2 {
|
||||
return false;
|
||||
}
|
||||
|
||||
let blank_sparse_dense_rows = cells
|
||||
.iter()
|
||||
.filter(|row| row[sparse_col].trim().is_empty() && !row[dense_col].trim().is_empty())
|
||||
.count();
|
||||
if blank_sparse_dense_rows * 2 < non_empty_rows {
|
||||
return false;
|
||||
}
|
||||
|
||||
let long_dense_cells = cells
|
||||
.iter()
|
||||
.filter(|row| row[dense_col].split_whitespace().count() >= 6)
|
||||
.count();
|
||||
long_dense_cells * 2 >= dense_count
|
||||
}
|
||||
|
||||
/// Detect a table from cell-background rects that failed grid detection.
|
||||
///
|
||||
/// Uses rect Y-edges for row boundaries and text X-position clustering for
|
||||
@@ -2326,12 +2217,6 @@ fn detect_merged_cluster_table(
|
||||
);
|
||||
return None;
|
||||
}
|
||||
if has_dominant_prose_cell(&cells) {
|
||||
debug!(
|
||||
" merged-cluster rejected: dominant prose cell (chart/figure region over body text)"
|
||||
);
|
||||
return None;
|
||||
}
|
||||
|
||||
// No empty columns
|
||||
for col in 0..num_cols {
|
||||
@@ -2429,96 +2314,11 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
}
|
||||
|
||||
// --- has_dominant_prose_cell ---
|
||||
|
||||
fn cells_of(rows: &[&[&str]]) -> Vec<Vec<String>> {
|
||||
rows.iter()
|
||||
.map(|r| r.iter().map(|c| c.to_string()).collect())
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dominant_prose_cell_rejects_swallowed_paragraph() {
|
||||
// Two cells hold paragraphs (the shape every observed phantom grid
|
||||
// has: swallowed body text spans multiple cells), rest are chart labels
|
||||
let para = ["word"; 70].join(" ");
|
||||
let para2 = ["word"; 35].join(" ");
|
||||
let cells = cells_of(&[
|
||||
&[para.as_str(), "81", "76"],
|
||||
&[para2.as_str(), "56", "9"],
|
||||
&["2019", "2020", ""],
|
||||
]);
|
||||
assert!(has_dominant_prose_cell(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dominant_prose_cell_rejects_small_table_dominated_by_one_cell() {
|
||||
// Boundary case, documented as INTENDED: a small grid whose single
|
||||
// long cell dominates the word count is rejected even at 4+ rows.
|
||||
// By content alone this shape is indistinguishable from a phantom
|
||||
// grid over body text, and every observed instance in the regression
|
||||
// corpora was swallowed prose (chart/figure regions), not a real
|
||||
// note table. Rejection degrades gracefully — the text is still
|
||||
// extracted as prose — while accepting a phantom scrambles reading
|
||||
// order.
|
||||
let note = ["word"; 70].join(" ");
|
||||
let cells = cells_of(&[
|
||||
&["Purpose", note.as_str()],
|
||||
&["Owner", "Facilities team"],
|
||||
&["Date", "2024-06-01"],
|
||||
&["Status", "Active"],
|
||||
]);
|
||||
assert!(has_dominant_prose_cell(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dominant_prose_cell_allows_description_column() {
|
||||
// Long-ish description cells, but text is spread across the table
|
||||
let desc = ["word"; 25].join(" ");
|
||||
let cells = cells_of(&[
|
||||
&["Item A", desc.as_str(), "100"],
|
||||
&["Item B", desc.as_str(), "200"],
|
||||
&["Item C", desc.as_str(), "300"],
|
||||
&["Item D", desc.as_str(), "400"],
|
||||
]);
|
||||
assert!(!has_dominant_prose_cell(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dominant_prose_cell_allows_short_tables() {
|
||||
let cells = cells_of(&[&["Name", "Value"], &["Total", "42"]]);
|
||||
assert!(!has_dominant_prose_cell(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dominant_prose_cell_allows_data_table_with_long_note() {
|
||||
// A real 4+ row table with one verbose remark cell: the note is ≥60
|
||||
// words but the table's other content carries more than 2× its word
|
||||
// count, so concentration stays below the 1/3 threshold. The
|
||||
// denominator scales with table size — this is what keeps large
|
||||
// legitimate tables safe where a bare length cap would not.
|
||||
let note = ["word"; 60].join(" ");
|
||||
let row_text = ["data"; 12].join(" ");
|
||||
let mut rows: Vec<Vec<String>> = (0..11)
|
||||
.map(|i| {
|
||||
vec![
|
||||
format!("Item {i}"),
|
||||
row_text.clone(),
|
||||
format!("{}", i * 100),
|
||||
]
|
||||
})
|
||||
.collect();
|
||||
rows.push(vec!["Note".into(), note, String::new()]);
|
||||
assert!(!has_dominant_prose_cell(&rows));
|
||||
}
|
||||
|
||||
// --- rects_overlap ---
|
||||
|
||||
#[test]
|
||||
@@ -2723,21 +2523,6 @@ mod tests {
|
||||
assert!(cells[0][0].contains("World"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_assign_items_parenthetical_no_inner_spaces() {
|
||||
let items = vec![
|
||||
make_item("The first sentence", 15.0, 85.0, 10.0),
|
||||
make_item("(", 90.0, 85.0, 10.0),
|
||||
make_item("twice", 95.0, 85.0, 10.0),
|
||||
make_item(")", 120.0, 85.0, 10.0),
|
||||
];
|
||||
let col_edges = vec![10.0, 150.0];
|
||||
let row_edges = vec![90.0, 70.0];
|
||||
let (cells, indices) = assign_items_to_grid(&items, &col_edges, &row_edges, 1);
|
||||
assert_eq!(indices.len(), 4);
|
||||
assert_eq!(cells[0][0], "The first sentence (twice)");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_assign_items_boundary_tolerance() {
|
||||
// Item right at edge with ±2pt tolerance
|
||||
@@ -3487,8 +3272,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: crate::types::ItemType::Text,
|
||||
mcid: None,
|
||||
});
|
||||
@@ -3798,8 +3581,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: crate::types::ItemType::Text,
|
||||
mcid: None,
|
||||
});
|
||||
|
||||
@@ -586,8 +586,6 @@ mod tests {
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid,
|
||||
}
|
||||
|
||||
@@ -108,8 +108,6 @@ pub(crate) fn try_split_financial_item(item: &TextItem) -> Option<Vec<TextItem>>
|
||||
page: item.page,
|
||||
is_bold: item.is_bold,
|
||||
is_italic: item.is_italic,
|
||||
is_underline: item.is_underline,
|
||||
is_strikeout: item.is_strikeout,
|
||||
item_type: item.item_type.clone(),
|
||||
mcid: item.mcid,
|
||||
});
|
||||
|
||||
@@ -369,10 +369,6 @@ pub(crate) fn join_cell_items(items: &[&TextItem]) -> String {
|
||||
let prev_ends_with_hyphen = result.ends_with('-');
|
||||
let curr_is_hyphen = text == "-";
|
||||
let curr_starts_with_hyphen = text.starts_with('-');
|
||||
let prev_ends_with_open_delimiter =
|
||||
result.ends_with('(') || result.ends_with('[') || result.ends_with('{');
|
||||
let curr_starts_with_close_delimiter =
|
||||
text.starts_with(')') || text.starts_with(']') || text.starts_with('}');
|
||||
|
||||
// Detect subscript/superscript: smaller font size and/or Y offset
|
||||
let font_ratio = item.font_size / prev_item.font_size;
|
||||
@@ -389,8 +385,6 @@ pub(crate) fn join_cell_items(items: &[&TextItem]) -> String {
|
||||
|| curr_starts_with_hyphen
|
||||
|| is_sub_super
|
||||
|| was_sub_super
|
||||
|| prev_ends_with_open_delimiter
|
||||
|| curr_starts_with_close_delimiter
|
||||
{
|
||||
result.push_str(text);
|
||||
} else {
|
||||
@@ -520,8 +514,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
@@ -735,18 +727,6 @@ mod tests {
|
||||
assert_eq!(join_cell_items(&[&a, &b, &c]), "pre-fix");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_join_cell_items_parenthetical_no_inner_spaces() {
|
||||
let a = make_item("The first sentence", 100.0, 500.0, 10.0);
|
||||
let b = make_item("(", 190.0, 500.0, 10.0);
|
||||
let c = make_item("twice", 195.0, 500.0, 10.0);
|
||||
let d = make_item(")", 220.0, 500.0, 10.0);
|
||||
assert_eq!(
|
||||
join_cell_items(&[&a, &b, &c, &d]),
|
||||
"The first sentence (twice)"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_join_cell_items_subscript_no_space() {
|
||||
let a = make_item("H", 100.0, 500.0, 12.0);
|
||||
@@ -886,8 +866,6 @@ mod tests {
|
||||
font: String::new(),
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
page: 1,
|
||||
@@ -924,8 +902,6 @@ mod tests {
|
||||
font: String::new(),
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
page: 1,
|
||||
|
||||
@@ -235,8 +235,6 @@ fn split_merged_numbers(item: &TextItem, col_boundaries: &[f32]) -> Vec<TextItem
|
||||
page: item.page,
|
||||
is_bold: item.is_bold,
|
||||
is_italic: item.is_italic,
|
||||
is_underline: item.is_underline,
|
||||
is_strikeout: item.is_strikeout,
|
||||
item_type: item.item_type.clone(),
|
||||
mcid: item.mcid,
|
||||
});
|
||||
@@ -257,8 +255,6 @@ fn split_merged_numbers(item: &TextItem, col_boundaries: &[f32]) -> Vec<TextItem
|
||||
page: item.page,
|
||||
is_bold: item.is_bold,
|
||||
is_italic: item.is_italic,
|
||||
is_underline: item.is_underline,
|
||||
is_strikeout: item.is_strikeout,
|
||||
item_type: item.item_type.clone(),
|
||||
mcid: item.mcid,
|
||||
});
|
||||
@@ -1433,8 +1429,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
@@ -1452,8 +1446,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
|
||||
@@ -93,9 +93,6 @@ pub fn is_bold_font(font_name: &str) -> bool {
|
||||
|| lower.contains("extrabold")
|
||||
|| lower.contains("ultrabold")
|
||||
|| lower.contains("medium") && !lower.contains("mediumitalic") // Some fonts use Medium for semi-bold
|
||||
// URW Type 1 fonts abbreviate Medium as "Medi" (e.g. NimbusRomNo9L-Medi,
|
||||
// the Times-Bold substitute in LaTeX documents; -MediItal is bold italic).
|
||||
|| lower.contains("-medi") && !lower.contains("mediumital")
|
||||
}
|
||||
|
||||
/// Detect if a font name indicates italic/oblique style
|
||||
@@ -765,17 +762,6 @@ mod tests {
|
||||
use super::*;
|
||||
use crate::types::ItemType;
|
||||
|
||||
#[test]
|
||||
fn bold_font_urw_medi_abbreviation() {
|
||||
// URW Type 1 fonts (LaTeX default Times) abbreviate Medium as "Medi"
|
||||
assert!(is_bold_font("NROFIU+NimbusRomNo9L-Medi"));
|
||||
assert!(is_bold_font("NimbusRomNo9L-MediItal"));
|
||||
assert!(!is_bold_font("DSSZWN+NimbusRomNo9L-Regu"));
|
||||
assert!(!is_bold_font("NimbusRomNo9L-ReguItal"));
|
||||
// Medium-Italic exclusion still holds
|
||||
assert!(!is_bold_font("Foo-MediumItalic"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strip_soft_hyphen() {
|
||||
assert_eq!(expand_ligatures("con\u{00AD}tent"), "content");
|
||||
@@ -897,8 +883,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
@@ -1018,8 +1002,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
});
|
||||
@@ -1096,8 +1078,6 @@ mod tests {
|
||||
page: 1,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
|
||||
+20
-165
@@ -329,7 +329,7 @@ impl ToUnicodeCMap {
|
||||
if let (Some(start), Some(end), Some(base)) = (
|
||||
parse_hex_u16(&start_hex),
|
||||
parse_hex_u16(&end_hex),
|
||||
hex_to_unicode_scalar(&base_hex),
|
||||
parse_hex_u32(&base_hex),
|
||||
) {
|
||||
self.ranges.push((start, end, base));
|
||||
}
|
||||
@@ -575,86 +575,32 @@ fn parse_hex_u16(hex: &str) -> Option<u16> {
|
||||
u16::from_str_radix(hex.trim(), 16).ok()
|
||||
}
|
||||
|
||||
/// Convert a ToUnicode destination hex string to Unicode.
|
||||
///
|
||||
/// PDF ToUnicode destinations are UTF-16BE strings. Supplementary-plane
|
||||
/// characters are encoded as surrogate pairs, so treating each 4-hex chunk as
|
||||
/// a scalar drops emoji like D83CDF1F.
|
||||
/// Parse a hex string to u32
|
||||
fn parse_hex_u32(hex: &str) -> Option<u32> {
|
||||
u32::from_str_radix(hex.trim(), 16).ok()
|
||||
}
|
||||
|
||||
/// Convert a hex string to a Unicode string
|
||||
/// Handles both 2-byte (BMP) and 4-byte (supplementary) codepoints
|
||||
fn hex_to_unicode_string(hex: &str) -> Option<String> {
|
||||
let hex: String = hex.chars().filter(|ch| !ch.is_ascii_whitespace()).collect();
|
||||
if hex.is_empty() || !hex.len().is_multiple_of(2) {
|
||||
return None;
|
||||
}
|
||||
let hex = hex.trim();
|
||||
let mut result = String::new();
|
||||
|
||||
let bytes: Option<Vec<u8>> = (0..hex.len())
|
||||
.step_by(2)
|
||||
.map(|i| u8::from_str_radix(&hex[i..i + 2], 16).ok())
|
||||
.collect();
|
||||
let bytes = bytes?;
|
||||
|
||||
if bytes.len().is_multiple_of(2) {
|
||||
let units: Vec<u16> = bytes
|
||||
.chunks_exact(2)
|
||||
.map(|chunk| u16::from_be_bytes([chunk[0], chunk[1]]))
|
||||
.collect();
|
||||
if let Ok(result) = String::from_utf16(&units) {
|
||||
if !result.is_empty() {
|
||||
return Some(normalize_tounicode_destination(result));
|
||||
// Process 4 hex digits at a time
|
||||
let mut i = 0;
|
||||
while i + 4 <= hex.len() {
|
||||
if let Ok(cp) = u32::from_str_radix(&hex[i..i + 4], 16) {
|
||||
if let Some(c) = char::from_u32(cp) {
|
||||
result.push(c);
|
||||
}
|
||||
}
|
||||
i += 4;
|
||||
}
|
||||
|
||||
// Be permissive for non-standard one-byte destinations.
|
||||
if bytes.len() == 1 {
|
||||
let ch = bytes[0] as char;
|
||||
if !ch.is_control() || ch == '\t' || ch == '\n' {
|
||||
return Some(ch.to_string());
|
||||
}
|
||||
}
|
||||
|
||||
None
|
||||
}
|
||||
|
||||
fn normalize_tounicode_destination(text: String) -> String {
|
||||
let is_multi_char = text.chars().nth(1).is_some();
|
||||
|
||||
// Some malformed producer CMaps put a list of alternative whitespace or
|
||||
// hyphen codepoints into one destination. Keep ordinary multi-character
|
||||
// mappings intact unless that malformed signature is present.
|
||||
if is_multi_char
|
||||
&& text.chars().all(char::is_whitespace)
|
||||
&& text.chars().any(|ch| matches!(ch, '\t' | '\n' | '\r'))
|
||||
{
|
||||
return if text.contains('\t') {
|
||||
"\t".to_string()
|
||||
} else {
|
||||
" ".to_string()
|
||||
};
|
||||
}
|
||||
|
||||
if is_multi_char
|
||||
&& text.contains('\u{00ad}')
|
||||
&& text.chars().all(|ch| {
|
||||
matches!(
|
||||
ch,
|
||||
'-' | '\u{00ad}' | '\u{2010}' | '\u{2011}' | '\u{2012}' | '\u{2013}' | '\u{2212}'
|
||||
)
|
||||
})
|
||||
{
|
||||
return "-".to_string();
|
||||
}
|
||||
|
||||
text
|
||||
}
|
||||
|
||||
fn hex_to_unicode_scalar(hex: &str) -> Option<u32> {
|
||||
let text = hex_to_unicode_string(hex)?;
|
||||
let mut chars = text.chars();
|
||||
let ch = chars.next()?;
|
||||
if chars.next().is_none() {
|
||||
Some(ch as u32)
|
||||
} else {
|
||||
if result.is_empty() {
|
||||
None
|
||||
} else {
|
||||
Some(result)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2661,97 +2607,6 @@ endbfrange
|
||||
assert_eq!(cmap.lookup(0x0005), Some("C".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_bfchar_surrogate_pair_emoji() {
|
||||
let cmap_content = r#"
|
||||
1 begincodespacerange
|
||||
<00> <FF>
|
||||
endcodespacerange
|
||||
2 beginbfchar
|
||||
<16> <D83CDF1F>
|
||||
<9D> <D83CDFAD>
|
||||
endbfchar
|
||||
"#;
|
||||
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
|
||||
|
||||
assert_eq!(cmap.code_byte_length, 1);
|
||||
assert_eq!(cmap.lookup(0x16), Some("🌟".to_string()));
|
||||
assert_eq!(cmap.lookup(0x9D), Some("🎭".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_bfrange_surrogate_pair_base() {
|
||||
let cmap_content = r#"
|
||||
1 begincodespacerange
|
||||
<00> <FF>
|
||||
endcodespacerange
|
||||
1 beginbfrange
|
||||
<C8> <C9> <D83CDFD8>
|
||||
endbfrange
|
||||
"#;
|
||||
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
|
||||
|
||||
assert_eq!(cmap.code_byte_length, 1);
|
||||
assert_eq!(cmap.lookup(0xC8), Some("🏘".to_string()));
|
||||
assert_eq!(cmap.lookup(0xC9), Some("🏙".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_bfrange_preserves_single_hyphen_like_base() {
|
||||
let cmap_content = r#"
|
||||
1 begincodespacerange
|
||||
<00> <FF>
|
||||
endcodespacerange
|
||||
1 beginbfrange
|
||||
<21> <22> <2013>
|
||||
endbfrange
|
||||
"#;
|
||||
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
|
||||
|
||||
assert_eq!(cmap.lookup(0x21), Some("–".to_string()));
|
||||
assert_eq!(cmap.lookup(0x22), Some("—".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_spaced_destination_hex_without_control_noise() {
|
||||
let cmap_content = r#"
|
||||
1 begincodespacerange
|
||||
<00> <FF>
|
||||
endcodespacerange
|
||||
3 beginbfchar
|
||||
<21> < 0009 000d 0020 00a0 >
|
||||
<22> < 002d 00ad 2010 >
|
||||
<23> <00a0>
|
||||
endbfchar
|
||||
"#;
|
||||
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
|
||||
|
||||
assert_eq!(cmap.lookup(0x21), Some("\t".to_string()));
|
||||
assert_eq!(cmap.lookup(0x22), Some("-".to_string()));
|
||||
assert_eq!(cmap.lookup(0x23), Some("\u{00a0}".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_preserves_valid_multi_character_destinations() {
|
||||
let cmap_content = r#"
|
||||
1 begincodespacerange
|
||||
<00> <FF>
|
||||
endcodespacerange
|
||||
4 beginbfchar
|
||||
<21> <002d002d>
|
||||
<22> <20132013>
|
||||
<23> <002000a0>
|
||||
<24> <00660069>
|
||||
endbfchar
|
||||
"#;
|
||||
let cmap = ToUnicodeCMap::parse(cmap_content.as_bytes()).unwrap();
|
||||
|
||||
assert_eq!(cmap.lookup(0x21), Some("--".to_string()));
|
||||
assert_eq!(cmap.lookup(0x22), Some("––".to_string()));
|
||||
assert_eq!(cmap.lookup(0x23), Some(" \u{00a0}".to_string()));
|
||||
assert_eq!(cmap.lookup(0x24), Some("fi".to_string()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_remap_to_sequential() {
|
||||
// Simulate a broken CMap where GIDs are from pre-subsetting:
|
||||
|
||||
+7
-36
@@ -116,14 +116,6 @@ pub struct TextItem {
|
||||
pub is_bold: bool,
|
||||
/// Whether the font is italic
|
||||
pub is_italic: bool,
|
||||
/// Whether the text is underlined (drawn rule/thin rect under the
|
||||
/// baseline — PDFs have no underline font flag, so this is detected
|
||||
/// geometrically after extraction; see `extractor::underline`).
|
||||
pub is_underline: bool,
|
||||
/// Whether the text is struck out (drawn rule/thin rect crossing the
|
||||
/// glyphs at mid x-height). Same geometric detection as underline,
|
||||
/// different vertical window; see `extractor::underline`.
|
||||
pub is_strikeout: bool,
|
||||
/// Type of item (text, image, link)
|
||||
pub item_type: ItemType,
|
||||
/// Marked Content ID from the content stream's BDC/BMC operator.
|
||||
@@ -145,17 +137,12 @@ pub struct TextLine {
|
||||
|
||||
impl TextLine {
|
||||
pub fn text(&self) -> String {
|
||||
self.text_with_formatting(false, false, false)
|
||||
self.text_with_formatting(false, false)
|
||||
}
|
||||
|
||||
/// Get text with optional bold/italic/underline markdown formatting
|
||||
pub fn text_with_formatting(
|
||||
&self,
|
||||
format_bold: bool,
|
||||
format_italic: bool,
|
||||
format_underline: bool,
|
||||
) -> String {
|
||||
if !format_bold && !format_italic && !format_underline {
|
||||
/// Get text with optional bold/italic markdown formatting
|
||||
pub fn text_with_formatting(&self, format_bold: bool, format_italic: bool) -> String {
|
||||
if !format_bold && !format_italic {
|
||||
return self.text_plain();
|
||||
}
|
||||
|
||||
@@ -164,7 +151,6 @@ impl TextLine {
|
||||
let mut result = String::new();
|
||||
let mut current_bold = false;
|
||||
let mut current_italic = false;
|
||||
let mut current_underline = false;
|
||||
|
||||
for (i, item) in self.items.iter().enumerate() {
|
||||
let text = item.text.as_str();
|
||||
@@ -190,13 +176,9 @@ impl TextLine {
|
||||
// we push text_trimmed below (which strips it).
|
||||
let has_leading_space = text.starts_with(' ');
|
||||
|
||||
// Check for style changes. Underline is exclusive: `<u>` content
|
||||
// stays free of `**`/`*` markers — consumers (and the eval
|
||||
// harnesses this feeds) match the tag content literally, and
|
||||
// mixed `<u>**x**</u>` nesting breaks that.
|
||||
let item_underline = format_underline && item.is_underline;
|
||||
let item_bold = format_bold && item.is_bold && !item_underline;
|
||||
let item_italic = format_italic && item.is_italic && !item_underline;
|
||||
// Check for style changes
|
||||
let item_bold = format_bold && item.is_bold;
|
||||
let item_italic = format_italic && item.is_italic;
|
||||
|
||||
// Close previous styles if they change
|
||||
if current_italic && !item_italic {
|
||||
@@ -207,10 +189,6 @@ impl TextLine {
|
||||
result.push_str("**");
|
||||
current_bold = false;
|
||||
}
|
||||
if current_underline && !item_underline {
|
||||
result.push_str("</u>");
|
||||
current_underline = false;
|
||||
}
|
||||
|
||||
// Add space: either from spacing logic or preserved from item text
|
||||
if needs_space || (has_leading_space && !result.is_empty() && !result.ends_with(' ')) {
|
||||
@@ -218,10 +196,6 @@ impl TextLine {
|
||||
}
|
||||
|
||||
// Open new styles
|
||||
if item_underline && !current_underline {
|
||||
result.push_str("<u>");
|
||||
current_underline = true;
|
||||
}
|
||||
if item_bold && !current_bold {
|
||||
result.push_str("**");
|
||||
current_bold = true;
|
||||
@@ -241,9 +215,6 @@ impl TextLine {
|
||||
if current_bold {
|
||||
result.push_str("**");
|
||||
}
|
||||
if current_underline {
|
||||
result.push_str("</u>");
|
||||
}
|
||||
|
||||
result
|
||||
}
|
||||
|
||||
BIN
Binary file not shown.
@@ -104,8 +104,6 @@ fn make_text_item(text: &str, x: f32, y: f32, font_size: f32, page: u32) -> Text
|
||||
page,
|
||||
is_bold: false,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
@@ -131,8 +129,6 @@ fn make_text_item_with_font(
|
||||
page,
|
||||
is_bold: is_bold_font(font),
|
||||
is_italic: is_italic_font(font),
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: ItemType::Text,
|
||||
mcid: None,
|
||||
}
|
||||
@@ -1111,7 +1107,6 @@ fn test_pages_needing_ocr_field_accessible() {
|
||||
page_count: 1,
|
||||
processing_time_ms: 0,
|
||||
pages_needing_ocr: vec![1, 3],
|
||||
ocr_reasons_by_page: Vec::new(),
|
||||
title: None,
|
||||
confidence: 1.0,
|
||||
layout: pdf_inspector::LayoutComplexity::default(),
|
||||
@@ -1358,32 +1353,6 @@ fn test_extract_regions_mem_identity_h_needs_ocr() {
|
||||
);
|
||||
}
|
||||
|
||||
/// ParseBench `text_simple__att10k.pdf` (issue #118): the producer authored a
|
||||
/// broken ToUnicode CMap that shifts every character by a per-range constant,
|
||||
/// and the embedded subset font has no `cmap` table to recover from. The
|
||||
/// resulting ciphertext is 100% printable ASCII, so it must be caught by the
|
||||
/// substitution-cipher statistics and routed to OCR instead of served silently.
|
||||
#[test]
|
||||
fn test_extract_pages_mem_shifted_cipher_tounicode_needs_ocr() {
|
||||
let buf = std::fs::read("tests/fixtures/shifted_cipher_tounicode.pdf").unwrap();
|
||||
let result = extract_pages_markdown_mem(&buf, None).unwrap();
|
||||
|
||||
assert_eq!(result.pages.len(), 1);
|
||||
assert!(
|
||||
result.pages[0].needs_ocr,
|
||||
"shifted-cipher garbled page should be flagged needs_ocr"
|
||||
);
|
||||
assert!(
|
||||
result.pages[0].markdown.is_empty(),
|
||||
"garbled markdown should be suppressed"
|
||||
);
|
||||
assert_eq!(result.pages_needing_ocr, vec![1]);
|
||||
assert_eq!(
|
||||
result.pages[0].ocr_reason.as_deref(),
|
||||
Some("suspected_garbled_text")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_regions_mem_multiple_regions_per_page() {
|
||||
let buf = std::fs::read("tests/fixtures/nexo-price-en.pdf").unwrap();
|
||||
|
||||
@@ -188,7 +188,7 @@
|
||||
|156|23/7|General renovation works to Block B at Belonie Secondary School|MOE|Belvedere Builders|SR869,505.75|
|
||||
|157|30/7|Procurement of Engine Block and Crankshaft for Engine A11|PUC|Ras Tek Pvt Ltd|Euro798,650.00|
|
||||
|158|30/7|procurement of Wartsila Engine spares|PUC|Wartsila Eastern Africa ltd|Euro158,424.00|
|
||||
|159|30/7|Proposed walkway, Drain, rock armoring, road and Bridge widening at Anse Talbot( Ex-Golden Egg)|SLTA|G&S Enterpise|SR1,113,010.00|
|
||||
|159|30/7|Proposed walkway, Drain, rock armoring , road and Bridge widening at Anse Talbot( Ex-Golden Egg)|SLTA|G&S Enterpise|SR1,113,010.00|
|
||||
|160|30/7|Procurement of transfer pump control panel|PUC|CA Engineering Consultancy Pte Ltd|SGD14,600.00|
|
||||
|161|30/7|Consultancy service for North to South Victoria Bye- Pass road and utilities organisation|MLUH|Sonnel Seychelles LTD|SR1,332,000.00|
|
||||
|162 AUG|30/7|Procurement of the supply of sodium cardonate|PUC|HPL Chemical LTD|USD42,600.00|
|
||||
@@ -237,7 +237,7 @@
|
||||
|201|24/9|Procurement of vehicle x 2|SLTA|Abhaye Valabhji Pty Ltd|SR1000.000.00|
|
||||
||OCT|||||
|
||||
|202|1/10|Proposed new traffic lane to 5th June Avenue|SLTA|Divy Constrution|SR2,864,589.00|
|
||||
|203|1/10||Proposed Walkway, Drain, rock armoring, road and Bridge widening at Anse Talbot(Ex-Golden Egg) - Variations SLTA|G & S Enterprise|SR200,448.00|
|
||||
|203|1/10||Proposed Walkway, Drain, rock armoring , road and Bridge widening at Anse Talbot( Ex-Golden Egg) - Variations SLTA|G & S Enterprise|SR200,448.00|
|
||||
|204|1/10|Proposed Reconstrcution of Burnt House-Au Cap|MLUH|Furui Construction|SR946,130.00|
|
||||
|205|1/10|Variation on the project associated with the procurement of seven 100m3/day containerised plant|PUC|Tornado Group (UAE)|USD172,500.00|
|
||||
|206|1/10|Works on the breaker system at Bel Omber desalination plant|PUC|United Concrete Products (Sey)Ltd|SR1,998,993.11|
|
||||
|
||||
@@ -10,7 +10,7 @@ Department of the Treasury **Internal Revenue Service**
|
||||
|
||||
### This publication contains:
|
||||
|
||||
**Form 4070A,** Employee’s Daily Record of Tips **Form 4070,** Employee’s Report of Tips to Employer
|
||||
**Form 4070A, Employee’s Daily Record of** Tips **Form 4070, Employee’s Report of Tips to** Employer
|
||||
|
||||
For the period
|
||||
|
||||
@@ -22,7 +22,7 @@ Name and address of employee
|
||||
|
||||
**Publication 1244 (Rev. 7-96)** Cat. No. 44472W
|
||||
|
||||
**Instructions** You must keep sufficient proof to show the amount of your tip income for the year. A daily record of your tip income is considered sufficient proof. Keep a daily record for each workday showing the amount of cash and credit card tips received directly from customers or other employees. Also keep a record of the amount of tips, if any, you paid to other employees through tip sharing, tip pooling or other arrangements, and the names of employees to whom you paid tips. Show the date that each entry is made. This date should be on or near the date you received the tip income. You may use **Form 4070A**, Employee’s Daily Record of Tips, or any other daily record to record your tips. **Reporting Tips to Your Employer.—**If you receive tips that total $20 or more for any month while working for one employer, you must report the tips to your employer. Tips include cash left by customers, tips customers add to credit card charges, and tips you receive from other employees. You must report your tips for any one month by the 10th day of the next month. If the 10th day falls on a Saturday, Sunday, or legal holiday, you may give the report to your employer on the next business day that is not a Saturday, Sunday, or legal holiday. You must report tips that total $20 or more every month regardless of your total wages and tips for the year. You may use **Form 4070**, Employee’s Report of Tips to Employer, to report your tips to your employer. See the instructions on the back of Form 4070. You must include all tips, including tips not reported to your employer, as wages on your income tax return. You may use the last page of this publication to total your tips for the year. Your employer must withhold income, social security, and Medicare (or railroad retirement) taxes on tips you report. Your employer usually deducts the withholding due on tips from your regular wages.
|
||||
**Instructions** You must keep sufficient proof to show the amount of your tip income for the year. A daily record of your tip income is considered sufficient proof. Keep a daily record for each workday showing the amount of cash and credit card tips received directly from customers or other employees. Also keep a record of the amount of tips, if any, you paid to other employees through tip sharing, tip pooling or other arrangements, and the names of employees to whom you paid tips. Show the date that each entry is made. This date should be on or near the date you received the tip income. You may use Form 4070A, Employee’s Daily Record of Tips, or any other daily record to record your tips. **Reporting Tips to Your Employer.—If you** receive tips that total $20 or more for any month while working for one employer, you must report the tips to your employer. Tips include cash left by customers, tips customers add to credit card charges, and tips you receive from other employees. You must report your tips for any one month by the 10th day of the next month. If the 10th day falls on a Saturday, Sunday, or legal holiday, you may give the report to your employer on the next business day that is not a Saturday, Sunday, or legal holiday. You must report tips that total $20 or more every month regardless of your total wages and tips for the year. You may use Form 4070, Employee’s Report of Tips to Employer, to report your tips to your employer. See the instructions on the back of Form 4070. You must include all tips, including tips not reported to your employer, as wages on your income tax return. You may use the last page of this publication to total your tips for the year. Your employer must withhold income, social security, and Medicare (or railroad retirement) taxes on tips you report. Your employer usually deducts the withholding due on tips from your regular wages.
|
||||
|
||||
*(continued on inside of back cover)*
|
||||
|
||||
@@ -30,14 +30,14 @@ Form **4070A** Employee’s Daily Record of Tips (Rev. July 1996) **This is a vo
|
||||
|
||||
Establishment name (if different)
|
||||
|
||||
Date Date **a.** Tips received
|
||||
Date Date **a. Tips received**
|
||||
|
||||
**b.** Credit card tips **c.** Tips paid out to **d.** Names of employees to whom you
|
||||
**b. Credit card tips c. Tips paid out to d. Names of employees to whom you**
|
||||
tips of directly from customers received other employees paid tips rec’d. entry and other employees 1 2 3 4 5 **Subtotals** **For Paperwork Reduction Act Notice, see Instructions on the back of Form 4070. Page 1**
|
||||
|
||||
Date Date **a.** Tips received
|
||||
Date Date **a. Tips received**
|
||||
|
||||
**b.** Credit card tips **c.** Tips paid out to **d.** Names of employees to whom you
|
||||
**b. Credit card tips c. Tips paid out to d. Names of employees to whom you**
|
||||
tips of directly from customers received other employees paid tips rec’d. entry and other employees
|
||||
|
||||
7 8 9 10 11 12 13 14 15 **Subtotals**
|
||||
@@ -48,11 +48,11 @@ tips of directly from customers received other employees paid tips rec’d. entr
|
||||
|
||||
**Page 3**
|
||||
|
||||
27 28 29 30 31 **Subtotals from pages** **1, 2, and 3** **Totals**
|
||||
27 28 29 30 31 **Subtotals** **from pages** **1, 2, and 3** **Totals**
|
||||
|
||||
**1.** Report total cash tips (col. **a**) on Form 4070, line **1.**
|
||||
**2.** Report total credit card tips (col. **b**) on Form 4070, line **2.**
|
||||
**3.** Report total tips paid out (col. **c**) on Form 4070, line **3.** **Page 4**
|
||||
**1.** Report total cash tips (col. a) on Form 4070, line 1.
|
||||
**2.** Report total credit card tips (col. b) on Form 4070, line 2.
|
||||
**3.** Report total tips paid out (col. c) on Form 4070, line 3. **Page 4**
|
||||
|
||||
Form Employee’s Report (Rev. July 1996)
|
||||
|
||||
@@ -66,16 +66,17 @@ Employer’s name and address (include establishment name, if different) **1** C
|
||||
|
||||
**3** Tips paid out
|
||||
|
||||
Month or shorter period in which tips were received **4** Net tips (lines **1 + 2 - 3**) from, 19, to, 19 Signature Date
|
||||
Month or shorter period in which tips were received **4** Net tips (lines 1 + 2 - 3) from, 19, to, 19 Signature Date
|
||||
|
||||
**Paperwork Reduction Act Notice.—**We ask for the information on these forms to carry out the Internal Revenue laws of the United States. You are required to give us the information. We need it to ensure that you are complying with these laws and to allow us to figure and collect the right amount of tax. You are not required to provide the information requested on a form that is subject to the Paperwork Reduction Act unless the form displays a valid OMB control number. Books or records relating to a form or its instructions must be retained as long as their contents may become material in the administration of any Internal Revenue law. Generally, tax returns and return information are confidential, as required by Code section 6103. The time needed to complete Forms 4070 and 4070A will vary depending on individual circumstances. The estimated average times are: **Recordkeeping**—Form 4070, 7 min.; Form 4070A, 3 hr. and 23 min.; **Learning** **about the law**—each form, 2 min.; **Preparing** Form 4070, 13 min.; Form 4070A, 55 min.; and **Copying and** **providing** Form 4070, 10 min.; Form 4070A, 14 min. If you have comments concerning the accuracy of these time estimates or suggestions for making these
|
||||
**Paperwork Reduction Act Notice.—We ask for the** information on these forms to carry out the Internal Revenue laws of the United States. You are required to give us the information. We need it to ensure that you are complying with these laws and to allow us to figure and collect the right amount of tax. You are not required to provide the information requested on a form that is subject to the Paperwork Reduction Act unless the form displays a valid OMB control number. Books or records relating to a form or its instructions must be retained as long as their contents may become material in the administration of any Internal Revenue law. Generally, tax returns and return information are confidential, as required by Code section 6103. The time needed to complete Forms 4070 and 4070A will vary depending on individual circumstances. The estimated average times are: Recordkeeping—Form 4070, 7 min.; Form 4070A, 3 hr. and 23 min.; Learning **about the law—each form, 2 min.; Preparing Form 4070,** 13 min.; Form 4070A, 55 min.; and Copying and **providing Form 4070, 10 min.; Form 4070A, 14 min.** If you have comments concerning the accuracy of these time estimates or suggestions for making these
|
||||
|
||||
forms simpler, we would be happy to hear from you. You can write to the Tax Forms Committee, Western Area Distribution Center, Rancho Cordova, CA 95743-0001. **Purpose.—**Use this form to report tips you receive to your employer. This includes cash tips, tips you receive from other employees, and credit card tips. You must report tips every month regardless of your total wages and tips for the year. However, you do not have to report tips to your employer for any month you received less than $20 in tips while working for that employer. Report tips by the 10th day of the month following the month that you receive them. If the 10th day is a Saturday, Sunday, or legal holiday, report tips by the next day that is not a Saturday, Sunday, or legal holiday. See **Pub. 531**, Reporting Tip Income, for more information. You can get additional copies of **Pub. 1244**, Employee’s Daily Record of Tips and Report to Employer, which contains both Forms 4070A and 4070, by calling 1-800-TAX-FORM (1-800-829-3676).
|
||||
forms simpler, we would be happy to hear from you. You can write to the Tax Forms Committee, Western Area Distribution Center, Rancho Cordova, CA 95743-0001. **Purpose.—Use this form to report tips you receive to** your employer. This includes cash tips, tips you receive from other employees, and credit card tips. You must report tips every month regardless of your total wages and tips for the year. However, you do not have to report tips to your employer for any month you received less than $20 in tips while working for that employer. Report tips by the 10th day of the month following the month that you receive them. If the 10th day is a Saturday, Sunday, or legal holiday, report tips by the next day that is not a Saturday, Sunday, or legal holiday. See Pub. 531, Reporting Tip Income, for more information. You can get additional copies of Pub. 1244, Employee’s Daily Record of Tips and Report to Employer, which contains both Forms 4070A and 4070, by calling 1-800-TAX-FORM (1-800-829-3676).
|
||||
|
||||
**Instructions** *(continued)*
|
||||
**Instructions (continued)**
|
||||
|
||||
**Unreported Tips.—**If you received tips of $20 or more for any month while working for one employer but did not report them to your employer, you must figure and pay social security and Medicare taxes on the unreported tips when you file your tax return. If you have unreported tips, you **must** use Form 1040 and **Form 4137,** Social Security and Medicare Tax on Unreported Tip Income, to report them. You may **not** use Form 1040A or 1040EZ. Employees subject to the Railroad Retirement Tax Act **cannot** use Form 4137 to pay railroad retirement tax on unreported tips. To get railroad retirement credit, you must report tips to your employer. If you do not report tips to your employer as required, you may be charged a penalty of 50% of the social security and Medicare taxes (or railroad retirement tax) due on the unreported tips unless there was reasonable cause for not reporting them. **Additional Information.—**Get **Pub. 531,** Reporting Tip Income, and Form 4137 for more information on tips. If you are an employee of certain large food or beverage establishments, see Pub. 531 for tip allocation rules. **Recordkeeping.—**If you do not keep a daily record of tips, you must keep other reliable proof of the tip income you received. This proof includes copies of restaurant bills and credit card charges that show amounts customers added as tips. Keep your tip income records for as long as the information on them may be needed in the administration of any Internal Revenue law.
|
||||
**Unreported Tips.—If you received tips of $20 or** more for any month while working for one employer but did not report them to your employer, you must figure and pay social security and Medicare taxes on the unreported tips when you file your tax return. If you have unreported tips, you must use Form 1040 and Form 4137, Social Security and Medicare Tax on Unreported Tip Income, to report them. You may not use Form 1040A or 1040EZ. Employees subject to the Railroad Retirement Tax Act cannot use Form 4137 to pay railroad retirement tax on unreported tips. To get railroad retirement credit, you must report tips to your employer. If you do not report tips to your employer as required, you may be charged a penalty of 50% of the social security and Medicare taxes (or railroad retirement tax) due on the unreported tips unless there was reasonable cause for not reporting them. **Additional Information.—Get Pub. 531, Reporting** Tip Income, and Form 4137 for more information on tips. If you are an employee of certain large food or beverage establishments, see Pub. 531 for tip allocation rules. **Recordkeeping.—If you do not keep a daily** record of tips, you must keep other reliable proof of the tip income you received. This proof includes copies of restaurant bills and credit card charges that show amounts customers added as tips. Keep your tip income records for as long as the information on them may be needed in the administration of any Internal Revenue law.
|
||||
|
||||
### Instructions (continued)
|
||||
|
||||
Use this space to total your tips for the year
|
||||
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
|
||||
8 4 Z E L L / L U R I E R E A L E S T A T E C E N T E R
|
||||
|
||||
**Table I:** Cap rate correlations **Cap Rate Correlation With:*** **BBB Corp** **10-Year Bond Yield S&P Dividend** **Treasury (10-15 yr) Yield** Multifamily 0.187 0.771 0.068 Industrial-0.221 0.748-0.307 CBD Office-0.449 0.694-0.458 Retail-0.181 0.649-02.58
|
||||
**Table I: Cap rate correlations** **Cap Rate Correlation With:*** **BBB Corp** **10-Year Bond Yield S&P Dividend** **Treasury (10-15 yr) Yield** Multifamily 0.187 0.771 0.068 Industrial-0.221 0.748-0.307 CBD Office-0.449 0.694-0.458 Retail-0.181 0.649-02.58
|
||||
|
||||
* Based on 25 years of data for the 10-yrT & S&P DivYld; and 14 years for BBB.
|
||||
**Figure 1:** NCREIF cap rates vs. 10-yearTreasury
|
||||
@@ -20,9 +20,7 @@ R E V I E W 8 5
|
||||
|
||||
**Figure 2:** Capratespreadsover10-yearTreasury
|
||||
|
||||
**Basis Points** -200
|
||||
|
||||
-400
|
||||
**Basis Points -200** -400
|
||||
|
||||
-600
|
||||
|
||||
@@ -34,7 +32,7 @@ R E V I E W 8 5
|
||||
|
||||
1982 1986 1990 1994 1998 2002 2006
|
||||
|
||||
**Table II:** Correlationsofspreadsbypropertytype **Correlation of Cap Rate Spreads Over Treasury** **Multifamily Industrial CBD Office**
|
||||
**Table II: Correlationsofspreadsbypropertytype** **Correlation of Cap Rate Spreads Over Treasury** **Multifamily Industrial CBD Office**
|
||||
|
||||
||Multifamily|Industrial|CBD Office|
|
||||
|---|---|---|---|
|
||||
|
||||
+16
-16
@@ -1,8 +1,8 @@
|
||||
(e) [Reserved]. For further guidance, see §1.1563-3T(e)(1). Par. 50. Section 1.1563-3T is added to read as follows:
|
||||
<u>§1.1563-3T Rules for determining stock ownership (temporary)</u>.
|
||||
§1.1563-3T Rules for determining stock ownership (temporary).
|
||||
|
||||
(a) through (d)(2)(iii) [Reserved]. For further guidance, see §1.1563-3(a)
|
||||
through (d)(2)(iii). (iv) <u>Statement</u>. If the application of paragraph (d)(2)(ii) or (iii) of §1.1563-3 does not result in a corporation being treated as a component member of only one controlled group of corporations on a December 31, then such corporation will be treated as a component member of only one such group on such date. Such corporation may elect the group in which it is to be included by including on or with its income tax return a statement entitled, “STATEMENT TO ELECT CONTROLLED GROUP PURSUANT TO §1.1563-3T(d)(2)(iv).” The statement must include--
|
||||
through (d)(2)(iii). (iv) Statement. If the application of paragraph (d)(2)(ii) or (iii) of §1.1563-3 does not result in a corporation being treated as a component member of only one controlled group of corporations on a December 31, then such corporation will be treated as a component member of only one such group on such date. Such corporation may elect the group in which it is to be included by including on or with its income tax return a statement entitled, “STATEMENT TO ELECT CONTROLLED GROUP PURSUANT TO §1.1563-3T(d)(2)(iv).” The statement must include--
|
||||
|
||||
(A) A description of each of the controlled groups in which the corporation
|
||||
could be included. The description must include the name and employer identification number of each component member of each such group and the stock ownership of the component members of each such group; and
|
||||
@@ -10,7 +10,7 @@ could be included. The description must include the name and employer identifica
|
||||
(B) The following representation: [INSERT NAME AND EMPLOYER
|
||||
IDENTIFICATION NUMBER OF CORPORATION] ELECTS TO BE TREATED AS A COMPONENT MEMBER OF THE [INSERT DESIGNATION OF GROUP].
|
||||
|
||||
(v) <u>Election</u>-- (A) <u>Election filed</u>. An election filed under paragraph (d)(2)(iv) of
|
||||
(v) Election-- (A) Election filed. An election filed under paragraph (d)(2)(iv) of
|
||||
this section is irrevocable and effective until paragraph (d)(2)(ii) or (iii) of §1.1563-3 applies or until a change in the stock ownership of the corporation results in
|
||||
|
||||
|termination of membership in the controlled group in which such corporation has||
|
||||
@@ -30,47 +30,47 @@ Federal income tax return (including any amended return filed on or before the d
|
||||
|
||||
2006.
|
||||
(2) Expiration date. The applicability of this section will expire on May 26,
|
||||
2009. Par. 51. Section 1.6012-2 is amended by revising paragraph (c) and adding paragraph (k) to read as follows: <u>§1.6012-2 Corporations required to make returns of income</u>.
|
||||
2009. Par. 51. Section 1.6012-2 is amended by revising paragraph (c) and adding paragraph (k) to read as follows: §1.6012-2 Corporations required to make returns of income.
|
||||
* * * * *
|
||||
(c) [Reserved]. For further guidance, see §1.6012-2T(c).
|
||||
* * * * *
|
||||
(k) [Reserved]. For further guidance, see §1.6012-2T(k)(1).
|
||||
|
||||
Par. 52. Section 1.6012-2T is added to read as follows: <u>§1.6012-2T Corporations required to make returns of income (temporary)</u>.
|
||||
Par. 52. Section 1.6012-2T is added to read as follows: §1.6012-2T Corporations required to make returns of income (temporary).
|
||||
|
||||
(a) through (b) [Reserved]. For further guidance, see §1.6012-2(a) through
|
||||
(b).
|
||||
(c) Insurance companies-- (1) Domestic life insurance companies-- (i) In
|
||||
<u>general</u>. A life insurance company subject to tax under section 801 shall make a return on Form 1120L. Except as provided in paragraph (c)(4) of this section, such company shall file with its return--
|
||||
general. A life insurance company subject to tax under section 801 shall make a return on Form 1120L. Except as provided in paragraph (c)(4) of this section, such company shall file with its return--
|
||||
|
||||
(A) A copy of its annual statement which shows the reserves used by the
|
||||
company in computing the taxable income reported on its return; and
|
||||
|
||||
(B) A copy of Schedule A (real estate) and of Schedule D (bonds and stocks),
|
||||
or any successor thereto, of such annual statement. (ii) <u>Mutual savings banks</u>. Mutual savings banks conducting life insurance business and meeting the requirements of section 594 are subject to partial tax computed on Form 1120 and partial tax computed on Form 1120L. The Form 1120L is attached as a schedule to Form 1120, together with the annual statement and schedules required to be filed with Form 1120L.
|
||||
or any successor thereto, of such annual statement. (ii) Mutual savings banks. Mutual savings banks conducting life insurance business and meeting the requirements of section 594 are subject to partial tax computed on Form 1120 and partial tax computed on Form 1120L. The Form 1120L is attached as a schedule to Form 1120, together with the annual statement and schedules required to be filed with Form 1120L.
|
||||
|
||||
(2) <u>Domestic nonlife insurance companies</u>. Every domestic insurance
|
||||
(2) Domestic nonlife insurance companies. Every domestic insurance
|
||||
company other than a life insurance company shall make a return on Form 1120PC. This includes organizations described in section 501(m)(1) that provide commercial- type insurance and organizations described in section 833. Except as provided in paragraph (c)(4) of this section, such company shall file with its return a copy of its
|
||||
|
||||
annual statement (or a pro forma annual statement), including the underwriting and investment exhibit for the year covered by such return.
|
||||
|
||||
(3) <u>Foreign insurance companies</u>. The provisions of paragraphs (c)(1) and
|
||||
(3) Foreign insurance companies. The provisions of paragraphs (c)(1) and
|
||||
(c)(2) of this section concerning the returns and statements of insurance companies subject to tax under section 801 or section 831 also apply to foreign insurance companies subject to tax under those sections, except that the copy of the annual statement required to be submitted with the return shall, in the case of a foreign insurance company that is not required to file an annual statement, be a copy of the pro forma annual statement relating to the United States business of such company.
|
||||
(4) <u>Exception for insurance companies filing their Federal income tax returns</u>
|
||||
<u>electronically</u>. If an insurance company described in paragraph (c)(1), (c)(2), or
|
||||
(4) Exception for insurance companies filing their Federal income tax returns
|
||||
electronically. If an insurance company described in paragraph (c)(1), (c)(2), or
|
||||
|
||||
(c)(3) of this section files its Federal income tax return electronically, it should not include on or with such return its annual statement (or pro forma annual statement), or any portion thereof. Such statement must be available at all times for inspection by authorized Internal Revenue Service officers or employees and retained for so long as such statements may be material in the administration of any internal revenue law. See §1.6001-1(e).
|
||||
(5) <u>Definition</u>. For purposes of this section, the term <u>annual statement</u> means
|
||||
(5) Definition. For purposes of this section, the term annual statement means
|
||||
the annual statement, the form of which is approved by the National Association of Insurance Commissioners (NAIC), which is filed by an insurance company for the year with the insurance departments of States, Territories, and the District of
|
||||
|
||||
Columbia. The term annual statement also includes a pro forma annual statement if the insurance company is not required to file the NAIC annual statement.
|
||||
|
||||
(d) through (j) [Reserved]. For further guidance, see §1.6012-2(d) through (j).
|
||||
(k) <u>Effective date</u>-- (1) <u>Applicability date</u>. This section applies to any original
|
||||
(k) Effective date-- (1) Applicability date. This section applies to any original
|
||||
Federal income tax return (including any amended return filed on or before the due date (including extensions) of such original return) timely filed on or after May 30,
|
||||
|
||||
2006.
|
||||
(2) <u>Expiration date</u>. The applicability of this section will expire on May 26,
|
||||
(2) Expiration date. The applicability of this section will expire on May 26,
|
||||
2009.
|
||||
|
||||
|||Par. 53. For each entry in the “Location” column of the following table,|
|
||||
@@ -165,7 +165,7 @@ section and paragraph
|
||||
PART 602--OMB CONTROL NUMBERS UNDER THE PAPERWORK REDUCTION ACT Par. 54. The authority citation for part 602 continues to read as follows: Authority: 26 U.S.C. 7805. Par. 55. In §602.101, paragraph (b) is amended to read as follows:
|
||||
|
||||
1. The following entries to the table are removed:
|
||||
<u>§602.101 OMB Control numbers</u>.
|
||||
§602.101 OMB Control numbers.
|
||||
|
||||
* * * * *
|
||||
(b) * * *
|
||||
@@ -180,7 +180,7 @@ CFR part or section where Current OMB identified or described control No.
|
||||
1.1081-11………………………………………………………………. 1545-2019
|
||||
* * * * * **______________________________________________________________**
|
||||
2. The following entries are added in numerical order to the table:
|
||||
<u>§602.101 OMB Control numbers</u>.
|
||||
§602.101 OMB Control numbers.
|
||||
|
||||
* * * * *
|
||||
(b) * * *
|
||||
|
||||
@@ -26,7 +26,7 @@ A.P., NIST Standard Reference in cubic meters per kilogram Database 23, NIST the
|
||||
|
||||
##### Physical Properties
|
||||
|
||||
|Chemical Formula|CCl₂F₂|
|
||||
|Chemical Formula|CCl2F2|
|
||||
|---|---|
|
||||
|Molecular mass|120.91|
|
||||
|Boiling Point At one atmosphere|-29.75°C|
|
||||
@@ -45,7 +45,7 @@ l
|
||||
|
||||
|Temp|Pressure||Volume|||Density||Enthalpy|||Entropy|Temp|
|
||||
|---|---|---|---|---|---|---|---|---|---|---|---|---|
|
||||
|°C|[kPa]|[m³ Liquid v f|/kg]|Vapour v g|Liquid d f|[kg/m³] Vapour d g|Liquid H f|[kJ/kg] Latent H fg|Vapour H g|Liquid S f|[kJ/K-kg] Vapour S g|°C|
|
||||
|°C|[kPa]|[m3 Liquid v f|/kg]|Vapour v g|Liquid d f|[kg/m3] Vapour d g|Liquid H f|[kJ/kg] Latent H fg|Vapour H g|Liquid S f|[kJ/K-kg] Vapour S g|°C|
|
||||
|
||||
|-100|1.2|0.0006|10.0000|1679.0|0.100|113.3|192.8|306.1|0.6077|1.7210|-100|
|
||||
|---|---|---|---|---|---|---|---|---|---|---|---|
|
||||
|
||||
Reference in New Issue
Block a user