Compare commits
32
Commits
@@ -4,6 +4,9 @@ on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths: ['Cargo.toml']
|
||||
# Manual fallback: retry a publish that failed after the version was
|
||||
# already merged (a plain re-push won't register as a version change).
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -14,6 +17,10 @@ env:
|
||||
jobs:
|
||||
check-version:
|
||||
name: Check version change
|
||||
# Guard manual dispatches: crates.io trusted publishing matches
|
||||
# repo+workflow+environment but NOT branch, so without this a
|
||||
# workflow_dispatch from any branch could publish unmerged code.
|
||||
if: github.ref == 'refs/heads/main'
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
changed: ${{ steps.check.outputs.changed }}
|
||||
@@ -28,17 +35,26 @@ jobs:
|
||||
id: check
|
||||
run: |
|
||||
NEW_VERSION=$(python3 -c 'import pathlib, tomllib; print(tomllib.loads(pathlib.Path("Cargo.toml").read_text())["package"]["version"])')
|
||||
OLD_VERSION=$(git show HEAD~1:Cargo.toml | python3 -c 'import sys, tomllib; print(tomllib.loads(sys.stdin.read())["package"]["version"])')
|
||||
echo "old=$OLD_VERSION new=$NEW_VERSION"
|
||||
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
|
||||
|
||||
if [ "$NEW_VERSION" = "$OLD_VERSION" ]; then
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
|
||||
# Manual dispatch publishes the current version regardless of the
|
||||
# previous commit; the crates.io check below still prevents
|
||||
# double-publishing an already-released version.
|
||||
echo "manual dispatch: publishing v$NEW_VERSION"
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
OLD_VERSION=$(git show HEAD~1:Cargo.toml | python3 -c 'import sys, tomllib; print(tomllib.loads(sys.stdin.read())["package"]["version"])')
|
||||
echo "old=$OLD_VERSION new=$NEW_VERSION"
|
||||
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
if [ "$NEW_VERSION" = "$OLD_VERSION" ]; then
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
echo "published=false" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
HTTP_STATUS=$(curl --silent --show-error --output /tmp/crate-version.json --write-out "%{http_code}" \
|
||||
-H "User-Agent: firecrawl/pdf-inspector publish workflow (https://github.com/firecrawl/pdf-inspector)" \
|
||||
|
||||
@@ -4,6 +4,9 @@ on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths: ['napi/package.json']
|
||||
# Manual fallback: retry a publish that failed partway (per-package
|
||||
# already-published checks make re-runs idempotent).
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -12,6 +15,10 @@ permissions:
|
||||
jobs:
|
||||
check-version:
|
||||
name: Check version change
|
||||
# Guard manual dispatches: npm trusted publishing matches
|
||||
# repo+workflow+environment but NOT branch, so without this a
|
||||
# workflow_dispatch from any branch could publish unmerged code.
|
||||
if: github.ref == 'refs/heads/main'
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
changed: ${{ steps.check.outputs.changed }}
|
||||
@@ -25,11 +32,21 @@ jobs:
|
||||
id: check
|
||||
run: |
|
||||
NEW_VERSION=$(node -p "require('./napi/package.json').version")
|
||||
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
|
||||
|
||||
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
|
||||
# Manual dispatch rebuilds and publishes the current version; the
|
||||
# per-package already-published checks in the publish job skip
|
||||
# anything that made it out in a previous partial run.
|
||||
echo "manual dispatch: publishing v$NEW_VERSION"
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
OLD_VERSION=$(git show HEAD~1:napi/package.json | node -p "JSON.parse(require('fs').readFileSync('/dev/stdin','utf8')).version")
|
||||
echo "old=$OLD_VERSION new=$NEW_VERSION"
|
||||
if [ "$NEW_VERSION" != "$OLD_VERSION" ]; then
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
@@ -115,14 +132,81 @@ jobs:
|
||||
with:
|
||||
path: napi/artifacts
|
||||
|
||||
- name: Collect binaries and publish
|
||||
- name: Publish platform packages
|
||||
working-directory: napi
|
||||
run: |
|
||||
cp artifacts/bindings-*/*.node .
|
||||
VERSION="${{ needs.check-version.outputs.version }}"
|
||||
|
||||
for node_file in artifacts/bindings-*/pdf-inspector.*.node; do
|
||||
base=$(basename "$node_file")
|
||||
suffix=${base#pdf-inspector.}
|
||||
suffix=${suffix%.node}
|
||||
pkg="@firecrawl/pdf-inspector-$suffix"
|
||||
|
||||
if npm view "$pkg@$VERSION" version >/dev/null 2>&1; then
|
||||
echo "$pkg@$VERSION already published — skipping"
|
||||
continue
|
||||
fi
|
||||
|
||||
dir="npm-dist/$suffix"
|
||||
mkdir -p "$dir"
|
||||
cp "$node_file" "$dir/"
|
||||
node -e '
|
||||
const [suffix, version] = process.argv.slice(1)
|
||||
const meta = {
|
||||
"linux-x64-gnu": { os: ["linux"], cpu: ["x64"], libc: ["glibc"] },
|
||||
"darwin-arm64": { os: ["darwin"], cpu: ["arm64"] },
|
||||
"win32-x64-msvc": { os: ["win32"], cpu: ["x64"] },
|
||||
}[suffix]
|
||||
if (!meta) {
|
||||
console.error(`unknown platform suffix: ${suffix} — add it to the meta map`)
|
||||
process.exit(1)
|
||||
}
|
||||
const pkg = {
|
||||
name: `@firecrawl/pdf-inspector-${suffix}`,
|
||||
version,
|
||||
description: `Prebuilt ${suffix} binary for @firecrawl/pdf-inspector`,
|
||||
main: `pdf-inspector.${suffix}.node`,
|
||||
files: [`pdf-inspector.${suffix}.node`],
|
||||
license: "MIT",
|
||||
engines: { node: ">= 10" },
|
||||
repository: { type: "git", url: "https://github.com/firecrawl/pdf-inspector" },
|
||||
publishConfig: { access: "public" },
|
||||
...meta,
|
||||
}
|
||||
require("fs").writeFileSync(`npm-dist/${suffix}/package.json`, JSON.stringify(pkg, null, 2) + "\n")
|
||||
' "$suffix" "$VERSION"
|
||||
|
||||
echo "=== $pkg@$VERSION ==="
|
||||
ls -la "$dir"
|
||||
(cd "$dir" && npm publish --provenance --access public)
|
||||
done
|
||||
|
||||
- name: Publish main package
|
||||
working-directory: napi
|
||||
run: |
|
||||
VERSION="${{ needs.check-version.outputs.version }}"
|
||||
|
||||
if npm view "@firecrawl/pdf-inspector@$VERSION" version >/dev/null 2>&1; then
|
||||
echo "@firecrawl/pdf-inspector@$VERSION already published — skipping"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
cp artifacts/js-bindings/index.js .
|
||||
cp artifacts/js-bindings/index.d.ts .
|
||||
|
||||
echo "=== Package contents ==="
|
||||
ls -la *.node index.js index.d.ts
|
||||
# Stamp optionalDependencies to this exact version so the platform
|
||||
# pins can never drift from the main package version.
|
||||
node -e '
|
||||
const fs = require("fs")
|
||||
const pkg = JSON.parse(fs.readFileSync("package.json", "utf8"))
|
||||
for (const dep of Object.keys(pkg.optionalDependencies ?? {})) {
|
||||
pkg.optionalDependencies[dep] = pkg.version
|
||||
}
|
||||
fs.writeFileSync("package.json", JSON.stringify(pkg, null, 2) + "\n")
|
||||
'
|
||||
|
||||
echo "=== Main package contents ==="
|
||||
npm pack --dry-run
|
||||
|
||||
npm publish --provenance --access public
|
||||
|
||||
+14
-1
@@ -1,12 +1,25 @@
|
||||
[package]
|
||||
name = "pdf-inspector"
|
||||
version = "0.1.4"
|
||||
version = "0.1.6"
|
||||
edition = "2021"
|
||||
autobins = false
|
||||
authors = ["Firecrawl Team"]
|
||||
description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection"
|
||||
license = "MIT"
|
||||
repository = "https://github.com/firecrawl/pdf-inspector"
|
||||
readme = "docs/rust-api.md"
|
||||
# Explicit allowlist: crates.io caps uploads at 10 MiB and tests/fixtures
|
||||
# alone exceeds that. external/bcmaps ships in the crate — tounicode.rs
|
||||
# loads it at runtime relative to CARGO_MANIFEST_DIR.
|
||||
include = [
|
||||
"src/**",
|
||||
"external/bcmaps/**",
|
||||
"docs/rust-api.md",
|
||||
"LICENSE",
|
||||
# maturin derives the sdist file list from this allowlist; the stub must
|
||||
# ship so wheels built from the sdist keep their type hints.
|
||||
"pdf_inspector.pyi",
|
||||
]
|
||||
|
||||
[lib]
|
||||
name = "pdf_inspector"
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
[](https://crates.io/crates/pdf-inspector)
|
||||
[](https://www.npmjs.com/package/@firecrawl/pdf-inspector)
|
||||
[](https://pypi.org/project/pdf-inspector/)
|
||||
[](LICENSE)
|
||||
|
||||
Fast Rust library for PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Includes bindings for [Python](docs/python.md) and [Node.js](napi/README.md).
|
||||
@@ -26,7 +27,7 @@ Evaluated on the [opendataloader-bench](https://github.com/opendataloader-projec
|
||||
|
||||
| Engine | Overall | Reading Order (NID) | Tables (TEDS) | Headings (MHS) | Speed (200 docs) |
|
||||
|---|---|---|---|---|---|
|
||||
| pdf-inspector | 0.83 | 0.88 | 0.66 | 0.74 | 4s |
|
||||
| pdf-inspector | 0.83 | 0.89 | 0.66 | 0.74 | 4s |
|
||||
| opendataloader | 0.84 | 0.91 | 0.49 | 0.74 | 11s |
|
||||
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
|
||||
| markitdown | 0.58 | 0.88 | 0.00 | 0.00 | 8s |
|
||||
@@ -118,6 +119,9 @@ pdf2md document.pdf --items-json
|
||||
# Raw markdown only (no headers)
|
||||
pdf2md document.pdf --raw
|
||||
|
||||
# Token-efficient output (collapses long dot leaders and similar source padding)
|
||||
pdf2md document.pdf --compact
|
||||
|
||||
# Insert page break markers (<!-- Page N -->)
|
||||
pdf2md document.pdf --pages
|
||||
|
||||
|
||||
+73
-10
@@ -1,9 +1,37 @@
|
||||
# Python API
|
||||
# pdf-inspector
|
||||
|
||||
Python bindings via [PyO3](https://pyo3.rs). Requires Rust toolchain for building from source.
|
||||
Fast PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Python bindings via [PyO3](https://pyo3.rs) for the [pdf-inspector](https://github.com/firecrawl/pdf-inspector) Rust library.
|
||||
|
||||
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
|
||||
|
||||
## Features
|
||||
|
||||
- **Smart classification** — `text_based` / `scanned` / `image_based` / `mixed` in ~10–50ms, with a confidence score and per-page OCR routing.
|
||||
- **Markdown conversion** — headings, lists, code blocks, bold/italic, URL linking, and dual-mode table detection (PDF drawing ops + text-alignment heuristics).
|
||||
- **Layout-aware extraction** — multi-column reading order, position and font info per text item, RTL support.
|
||||
- **Robust text decoding** — CID/Type0 fonts via ToUnicode CMaps, plus automatic flagging of broken encodings so callers can fall back to OCR.
|
||||
- **Lightweight** — native Rust core, no ML models, no external services; ships type stubs.
|
||||
|
||||
## Benchmark
|
||||
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), direct-extraction engines only — no OCR, no ML. Scores 0–1, higher is better:
|
||||
|
||||
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|
||||
|---|---|---|---|---|---|
|
||||
| **pdf-inspector** | 0.83 | 0.88 | **0.66** | 0.74 | **4s** |
|
||||
| opendataloader | 0.84 | 0.91 | 0.49 | 0.74 | 11s |
|
||||
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
|
||||
|
||||
OCR/ML engines (docling, marker, mineru) score 0.83–0.88 overall but take 2–180 minutes on the same corpus. Full numbers in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
pip install pdf-inspector
|
||||
```
|
||||
|
||||
Prebuilt wheels cover CPython ≥3.8 on Linux (x86_64, aarch64), macOS (Intel, Apple Silicon), and Windows (x64). Other platforms build from source, which requires a Rust toolchain. For local development in a repo checkout:
|
||||
|
||||
```bash
|
||||
pip install maturin
|
||||
maturin develop --release
|
||||
@@ -73,16 +101,51 @@ result = pdf_inspector.extract_pages_markdown("document.pdf", pages=[0, 2])
|
||||
|
||||
## Types
|
||||
|
||||
**`PdfResult` fields:** `pdf_type`, `markdown`, `page_count`, `processing_time_ms`, `pages_needing_ocr`, `title`, `confidence`, `is_complex_layout`, `pages_with_tables`, `pages_with_columns`, `has_encoding_issues`
|
||||
Type stubs (`pdf_inspector.pyi`) ship with the package. Result types at a glance:
|
||||
|
||||
**`PdfClassification` fields:** `pdf_type`, `page_count`, `pages_needing_ocr` (0-indexed), `confidence`
|
||||
```python
|
||||
class PdfResult: # process_pdf / detect_pdf
|
||||
pdf_type: str # "text_based" | "scanned" | "image_based" | "mixed"
|
||||
markdown: str | None # extracted Markdown (None for detect_pdf)
|
||||
page_count: int
|
||||
processing_time_ms: int
|
||||
pages_needing_ocr: list[int]
|
||||
title: str | None
|
||||
confidence: float # 0.0 - 1.0
|
||||
is_complex_layout: bool
|
||||
pages_with_tables: list[int]
|
||||
pages_with_columns: list[int]
|
||||
has_encoding_issues: bool # broken font encodings — consider OCR fallback
|
||||
|
||||
**`TextItem` fields:** `text`, `x`, `y`, `width`, `height`, `font`, `font_size`, `page`, `is_bold`, `is_italic`, `item_type`
|
||||
class PdfClassification: # classify_pdf
|
||||
pdf_type: str
|
||||
page_count: int
|
||||
pages_needing_ocr: list[int] # 0-indexed
|
||||
confidence: float
|
||||
|
||||
**`RegionText` fields:** `text`, `needs_ocr`
|
||||
class TextItem: # extract_text_with_positions
|
||||
text: str
|
||||
x: float
|
||||
y: float
|
||||
width: float
|
||||
height: float
|
||||
font: str
|
||||
font_size: float
|
||||
page: int
|
||||
is_bold: bool
|
||||
is_italic: bool
|
||||
is_underline: bool
|
||||
is_strikeout: bool
|
||||
item_type: str
|
||||
|
||||
**`PageRegionTexts` fields:** `page` (0-indexed), `regions` (list of RegionText)
|
||||
class PageRegionTexts: # extract_text_in_regions
|
||||
page: int # 0-indexed
|
||||
regions: list[RegionText] # RegionText: text: str, needs_ocr: bool
|
||||
|
||||
**`PageMarkdown` fields:** `page` (0-indexed), `markdown`, `needs_ocr`
|
||||
|
||||
**`PagesExtractionResult` fields:** `pages` (list of PageMarkdown), `pages_with_tables` (1-indexed), `pages_with_columns` (1-indexed), `pages_needing_ocr` (1-indexed), `is_complex`
|
||||
class PagesExtractionResult: # extract_pages_markdown
|
||||
pages: list[PageMarkdown] # PageMarkdown: page (0-indexed), markdown, needs_ocr
|
||||
pages_with_tables: list[int] # 1-indexed
|
||||
pages_with_columns: list[int] # 1-indexed
|
||||
pages_needing_ocr: list[int] # 1-indexed
|
||||
is_complex: bool # any page has tables or multi-column layout
|
||||
```
|
||||
|
||||
+38
-2
@@ -1,12 +1,48 @@
|
||||
# Rust API
|
||||
# pdf-inspector
|
||||
|
||||
Add to your `Cargo.toml`:
|
||||
Fast PDF classification and text extraction. Detects whether a PDF is text-based or scanned, extracts text with position awareness, and converts to clean Markdown — all without OCR. Pure Rust, no ML models, no external services; the only PDF dependency is [lopdf](https://crates.io/crates/lopdf). Also available for [Python](https://pypi.org/project/pdf-inspector/) and [Node.js](https://www.npmjs.com/package/@firecrawl/pdf-inspector).
|
||||
|
||||
Built by [Firecrawl](https://firecrawl.dev) to handle text-based PDFs locally in under 200ms, skipping expensive OCR services for the ~54% of PDFs that don't need them.
|
||||
|
||||
## Features
|
||||
|
||||
- **Smart classification** — TextBased / Scanned / ImageBased / Mixed in ~10–50ms, with a confidence score and per-page OCR routing.
|
||||
- **Markdown conversion** — headings, lists, code blocks, bold/italic, URL linking, and dual-mode table detection (PDF drawing ops + text-alignment heuristics).
|
||||
- **Layout-aware extraction** — multi-column reading order, position and font info per text item, RTL support.
|
||||
- **Robust text decoding** — CID/Type0 fonts via ToUnicode CMaps, plus automatic flagging of broken encodings so callers can fall back to OCR.
|
||||
- **Lightweight** — pure Rust, no ML models, no external services; single PDF dependency ([lopdf](https://crates.io/crates/lopdf)).
|
||||
|
||||
## Benchmark
|
||||
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), direct-extraction engines only — no OCR, no ML. Scores 0–1, higher is better:
|
||||
|
||||
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|
||||
|---|---|---|---|---|---|
|
||||
| **pdf-inspector** | 0.83 | 0.88 | **0.66** | 0.74 | **4s** |
|
||||
| opendataloader | 0.84 | 0.91 | 0.49 | 0.74 | 11s |
|
||||
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
|
||||
|
||||
OCR/ML engines (docling, marker, mineru) score 0.83–0.88 overall but take 2–180 minutes on the same corpus. Full numbers in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
cargo add pdf-inspector
|
||||
```
|
||||
|
||||
For the latest unreleased changes, use the git dependency instead:
|
||||
|
||||
```toml
|
||||
[dependencies]
|
||||
pdf-inspector = { git = "https://github.com/firecrawl/pdf-inspector" }
|
||||
```
|
||||
|
||||
The crate also ships CLI binaries — `pdf2md` (PDF → Markdown, with `--json`, `--pages`, `--select-pages`, and the opt-in token-saving `--compact` profile) and `detect-pdf` (classification, with `--analyze --json`):
|
||||
|
||||
```bash
|
||||
cargo install pdf-inspector
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
Detect and extract in one call:
|
||||
|
||||
Generated
+1
-1
@@ -830,7 +830,7 @@ checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe"
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector"
|
||||
version = "0.1.4"
|
||||
version = "0.1.6"
|
||||
dependencies = [
|
||||
"env_logger",
|
||||
"log",
|
||||
|
||||
+28
-5
@@ -4,6 +4,26 @@ Fast PDF classification and region-based text extraction for Node.js/Bun. Native
|
||||
|
||||
Built by [Firecrawl](https://firecrawl.dev) for hybrid OCR pipelines — extract text from PDF structure where possible, fall back to OCR only when needed.
|
||||
|
||||
## Features
|
||||
|
||||
- **Smart classification** — text-based / scanned / image-based / mixed in ~10–50ms, with a confidence score and per-page OCR routing.
|
||||
- **Region-based extraction** — pull text from bounding boxes with per-region quality checks (`needsOcr`).
|
||||
- **Layout-aware** — multi-column reading order, position and font info per text item, RTL support.
|
||||
- **Robust text decoding** — CID/Type0 fonts via ToUnicode CMaps, plus automatic flagging of broken encodings so callers can fall back to OCR.
|
||||
- **Lightweight** — native Rust core via napi-rs, no ML models, no external services; ~5–6 MB platform binary, TypeScript definitions included.
|
||||
|
||||
## Benchmark
|
||||
|
||||
[opendataloader-bench](https://github.com/opendataloader-project/opendataloader-bench) corpus (200 PDFs), direct-extraction engines only — no OCR, no ML. Scores 0–1, higher is better:
|
||||
|
||||
| Engine | Overall | Reading order | Tables (TEDS) | Headings | Speed |
|
||||
|---|---|---|---|---|---|
|
||||
| **pdf-inspector** | 0.83 | 0.88 | **0.66** | 0.74 | **4s** |
|
||||
| opendataloader | 0.84 | 0.91 | 0.49 | 0.74 | 11s |
|
||||
| pymupdf4llm | 0.73 | 0.89 | 0.40 | 0.41 | 18s |
|
||||
|
||||
OCR/ML engines (docling, marker, mineru) score 0.83–0.88 overall but take 2–180 minutes on the same corpus. Full numbers in the [repo README](https://github.com/firecrawl/pdf-inspector#benchmark).
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
@@ -12,7 +32,7 @@ npm install @firecrawl/pdf-inspector
|
||||
bun add @firecrawl/pdf-inspector
|
||||
```
|
||||
|
||||
Prebuilt binaries included for **linux-x64** and **macOS ARM64**. No Rust toolchain needed.
|
||||
Prebuilt binaries for **Linux x64**, **macOS ARM64**, and **Windows x64** — npm installs only the one matching your platform. No Rust toolchain needed.
|
||||
|
||||
## API
|
||||
|
||||
@@ -90,10 +110,13 @@ interface RegionText {
|
||||
|
||||
## Platforms
|
||||
|
||||
| Platform | Architecture | Supported |
|
||||
|----------|-------------|-----------|
|
||||
| Linux | x64 | Yes |
|
||||
| macOS | ARM64 | Yes |
|
||||
Prebuilt binaries ship as platform-specific packages installed automatically via `optionalDependencies`:
|
||||
|
||||
| Platform | Architecture | Package |
|
||||
|----------|-------------|---------|
|
||||
| Linux | x64 (glibc) | `@firecrawl/pdf-inspector-linux-x64-gnu` |
|
||||
| macOS | ARM64 | `@firecrawl/pdf-inspector-darwin-arm64` |
|
||||
| Windows | x64 | `@firecrawl/pdf-inspector-win32-x64-msvc` |
|
||||
|
||||
## License
|
||||
|
||||
|
||||
@@ -7,6 +7,11 @@
|
||||
"devDependencies": {
|
||||
"@napi-rs/cli": "^3.4.1",
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.11.0",
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.11.0",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.11.0",
|
||||
},
|
||||
},
|
||||
},
|
||||
"packages": {
|
||||
|
||||
+7
-6
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "@firecrawl/pdf-inspector",
|
||||
"version": "1.10.3",
|
||||
"version": "1.11.1",
|
||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||
"main": "index.js",
|
||||
"types": "index.d.ts",
|
||||
@@ -22,7 +22,6 @@
|
||||
"files": [
|
||||
"index.js",
|
||||
"index.d.ts",
|
||||
"*.node",
|
||||
"bin/",
|
||||
"README.md"
|
||||
],
|
||||
@@ -40,10 +39,7 @@
|
||||
"x86_64-unknown-linux-gnu",
|
||||
"aarch64-apple-darwin",
|
||||
"x86_64-pc-windows-msvc"
|
||||
],
|
||||
"package": {
|
||||
"name": "@firecrawl/pdf-inspector-js"
|
||||
}
|
||||
]
|
||||
},
|
||||
"scripts": {
|
||||
"build": "napi build --platform --release",
|
||||
@@ -51,5 +47,10 @@
|
||||
},
|
||||
"devDependencies": {
|
||||
"@napi-rs/cli": "^3.4.1"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@firecrawl/pdf-inspector-linux-x64-gnu": "1.11.1",
|
||||
"@firecrawl/pdf-inspector-darwin-arm64": "1.11.1",
|
||||
"@firecrawl/pdf-inspector-win32-x64-msvc": "1.11.1"
|
||||
}
|
||||
}
|
||||
|
||||
+7
-1
@@ -6,8 +6,9 @@ build-backend = "maturin"
|
||||
name = "pdf-inspector"
|
||||
# Bump this to publish to PyPI — CI publishes automatically when the version
|
||||
# changes on main (same flow as napi/package.json for npm).
|
||||
version = "0.2.1"
|
||||
version = "0.2.5"
|
||||
description = "Fast PDF inspection, classification, and text extraction with smart scanned vs text-based detection"
|
||||
readme = "docs/python.md"
|
||||
license = { text = "MIT" }
|
||||
requires-python = ">=3.8"
|
||||
classifiers = [
|
||||
@@ -19,5 +20,10 @@ classifiers = [
|
||||
"Topic :: Text Processing",
|
||||
]
|
||||
|
||||
[project.urls]
|
||||
Homepage = "https://github.com/firecrawl/pdf-inspector"
|
||||
Repository = "https://github.com/firecrawl/pdf-inspector"
|
||||
Documentation = "https://github.com/firecrawl/pdf-inspector/blob/main/docs/python.md"
|
||||
|
||||
[tool.maturin]
|
||||
features = ["python"]
|
||||
|
||||
+1
-1
@@ -310,7 +310,7 @@
|
||||
<tr><th>Engine</th><th>Overall</th><th>Reading order</th><th>Tables</th><th>Headings</th><th>200 docs</th></tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
<tr class="us"><td>pdf-inspector</td><td>0.83</td><td>0.88</td><td>0.66</td><td>0.74</td><td>4s</td></tr>
|
||||
<tr class="us"><td>pdf-inspector</td><td>0.83</td><td>0.89</td><td>0.66</td><td>0.74</td><td>4s</td></tr>
|
||||
<tr><td>opendataloader</td><td>0.84</td><td>0.91</td><td>0.49</td><td>0.74</td><td>11s</td></tr>
|
||||
<tr><td>pymupdf4llm</td><td>0.73</td><td>0.89</td><td>0.40</td><td>0.41</td><td>18s</td></tr>
|
||||
<tr><td>markitdown</td><td>0.58</td><td>0.88</td><td>0.00</td><td>0.00</td><td>8s</td></tr>
|
||||
|
||||
@@ -206,6 +206,9 @@ fn main() {
|
||||
eprintln!(" --json Output result as JSON");
|
||||
eprintln!(" --items-json Output positioned TextItem JSON");
|
||||
eprintln!(" --raw Output only markdown (no headers)");
|
||||
eprintln!(
|
||||
" --compact Collapse token-heavy source formatting such as dot leaders"
|
||||
);
|
||||
eprintln!(" --pages Insert page break markers (<!-- Page N -->)");
|
||||
eprintln!(" --select-pages N Only process specified pages (e.g. 1,3,5-10)");
|
||||
eprintln!(" --password PW Password for an encrypted PDF");
|
||||
@@ -218,6 +221,7 @@ fn main() {
|
||||
let json_output = args.iter().any(|a| a == "--json");
|
||||
let items_json_output = args.iter().any(|a| a == "--items-json");
|
||||
let raw_output = args.iter().any(|a| a == "--raw");
|
||||
let compact_output = args.iter().any(|a| a == "--compact");
|
||||
let page_numbers = args.iter().any(|a| a == "--pages");
|
||||
let detect_only = args.iter().any(|a| a == "--detect-only");
|
||||
let analyze = args.iter().any(|a| a == "--analyze");
|
||||
@@ -276,6 +280,9 @@ fn main() {
|
||||
};
|
||||
|
||||
let mut options = PdfOptions::new().mode(process_mode);
|
||||
if compact_output {
|
||||
options.markdown.profile = pdf_inspector::MarkdownProfile::Compact;
|
||||
}
|
||||
options.markdown.include_page_numbers = page_numbers;
|
||||
if let Some(pages) = page_filter {
|
||||
options.page_filter = Some(pages);
|
||||
|
||||
@@ -1187,10 +1187,37 @@ pub(crate) fn extract_text_from_operand(
|
||||
})();
|
||||
result.map(|text| {
|
||||
let text = clean_symbol_pua(text);
|
||||
let text = remap_texcm_math_symbols(text, base_font_name);
|
||||
normalize_cp1252_controls(text, use_cp1252_fallback)
|
||||
})
|
||||
}
|
||||
|
||||
/// Fix a known producer bug in "TeXCMMathsSymbols" subset fonts (IntechOpen
|
||||
/// and sibling academic pipelines): the Computer Modern symbol glyphs are
|
||||
/// misnamed after Latin lookalikes (equal → /onequarter, plus → /thorn, …)
|
||||
/// and the generated ToUnicode faithfully propagates the wrong names. The
|
||||
/// remap applies only to text decoded from that font, keyed on the glyphs'
|
||||
/// observed misnames.
|
||||
fn remap_texcm_math_symbols(text: String, base_font_name: Option<&str>) -> String {
|
||||
let is_texcm = base_font_name.is_some_and(|n| {
|
||||
let n = n.rsplit_once('+').map_or(n, |(_, s)| s);
|
||||
n.eq_ignore_ascii_case("TeXCMMathsSymbols")
|
||||
});
|
||||
if !is_texcm {
|
||||
return text;
|
||||
}
|
||||
text.chars()
|
||||
.map(|c| match c {
|
||||
'¼' => '=',
|
||||
'½' => '-',
|
||||
'þ' => '+',
|
||||
'ð' => '(',
|
||||
'Þ' => ')',
|
||||
_ => c,
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn decode_single_byte_fallback(bytes: &[u8], use_cp1252_fallback: bool) -> String {
|
||||
bytes
|
||||
.iter()
|
||||
@@ -1409,6 +1436,21 @@ fn score_text(text: &str) -> i32 {
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
|
||||
#[test]
|
||||
fn texcm_math_symbols_remap() {
|
||||
assert_eq!(
|
||||
super::remap_texcm_math_symbols("S ¼ kB þ 1".into(), Some("EEKVNO+TeXCMMathsSymbols")),
|
||||
"S = kB + 1"
|
||||
);
|
||||
// Other fonts keep their genuine fractions/thorns.
|
||||
assert_eq!(
|
||||
super::remap_texcm_math_symbols("¼ cup þorn".into(), Some("Times-Roman")),
|
||||
"¼ cup þorn"
|
||||
);
|
||||
assert_eq!(super::remap_texcm_math_symbols("¼".into(), None), "¼");
|
||||
}
|
||||
|
||||
use super::*;
|
||||
use lopdf::dictionary;
|
||||
|
||||
|
||||
+151
-5
@@ -674,9 +674,24 @@ fn validate_and_build_columns(
|
||||
page: u32,
|
||||
center_assign: bool,
|
||||
) -> Vec<ColumnRegion> {
|
||||
// Compute Y range of the page
|
||||
let y_min = page_items.iter().map(|i| i.y).fold(f32::INFINITY, f32::min);
|
||||
let y_max = page_items
|
||||
// Compute the Y range from column-eligible items only — the same items
|
||||
// the histogram counted. Spanning items (full-width captions, titles)
|
||||
// are excluded from the projection, so letting them stretch the page's
|
||||
// vertical extent here would sink the overlap ratio for column regions
|
||||
// that legitimately occupy only part of the page (e.g. two-column text
|
||||
// below a figure).
|
||||
let x_span = page_items
|
||||
.iter()
|
||||
.map(|i| i.x + effective_width(i))
|
||||
.fold(f32::NEG_INFINITY, f32::max)
|
||||
- page_items.iter().map(|i| i.x).fold(f32::INFINITY, f32::min);
|
||||
let narrow: Vec<&&TextItem> = page_items
|
||||
.iter()
|
||||
.filter(|i| effective_width(i) <= x_span * 0.6)
|
||||
.collect();
|
||||
let span_items: &[&&TextItem] = if narrow.is_empty() { &[] } else { &narrow };
|
||||
let y_min = span_items.iter().map(|i| i.y).fold(f32::INFINITY, f32::min);
|
||||
let y_max = span_items
|
||||
.iter()
|
||||
.map(|i| i.y)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
@@ -722,6 +737,10 @@ fn validate_and_build_columns(
|
||||
(right_items.len(), left_items.len())
|
||||
};
|
||||
if larger < min_items || smaller < 3 {
|
||||
debug!(
|
||||
" valley rejected: counts smaller={} larger={}",
|
||||
smaller, larger
|
||||
);
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -735,6 +754,7 @@ fn validate_and_build_columns(
|
||||
&right_items
|
||||
};
|
||||
if is_list_marker_column(smaller_items) {
|
||||
debug!(" valley rejected: list-marker column");
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -759,6 +779,13 @@ fn validate_and_build_columns(
|
||||
let overlap = (overlap_max - overlap_min).max(0.0);
|
||||
|
||||
if overlap / y_range < min_vertical_span {
|
||||
debug!(
|
||||
" valley rejected: overlap {:.0}/{:.0} = {:.2} < {:.2}",
|
||||
overlap,
|
||||
y_range,
|
||||
overlap / y_range,
|
||||
min_vertical_span
|
||||
);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
@@ -1133,6 +1160,25 @@ pub(crate) fn group_into_lines_with_thresholds(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
table_pages: &HashSet<u32>,
|
||||
) -> Vec<TextLine> {
|
||||
group_into_lines_with_thresholds_and_charts(
|
||||
items,
|
||||
page_thresholds,
|
||||
table_pages,
|
||||
&HashMap::new(),
|
||||
)
|
||||
}
|
||||
|
||||
/// Like `group_into_lines_with_thresholds`, but items inside chart regions
|
||||
/// are excluded from column detection: chart text scattered across the page
|
||||
/// fills the gutter in the projection histogram, so two-column pages read as
|
||||
/// one column and same-baseline items from both columns fuse into one line
|
||||
/// (headings absorbed into the neighboring column's body text).
|
||||
pub(crate) fn group_into_lines_with_thresholds_and_charts(
|
||||
items: Vec<TextItem>,
|
||||
page_thresholds: &HashMap<u32, f32>,
|
||||
table_pages: &HashSet<u32>,
|
||||
chart_regions: &HashMap<u32, Vec<(f32, f32, f32, f32)>>,
|
||||
) -> Vec<TextLine> {
|
||||
if items.is_empty() {
|
||||
return Vec::new();
|
||||
@@ -1159,8 +1205,35 @@ pub(crate) fn group_into_lines_with_thresholds(
|
||||
// Non-Canva pages use the default 0.10 threshold.
|
||||
let adaptive_threshold = page_thresholds.get(&page).copied().unwrap_or(0.10);
|
||||
|
||||
// Detect columns for this page
|
||||
let columns = detect_columns(&page_items, page, table_pages.contains(&page));
|
||||
// Detect columns for this page, blind to chart text.
|
||||
debug!(
|
||||
"page {}: grouping chart-aware={} regions={:?}",
|
||||
page,
|
||||
chart_regions.contains_key(&page),
|
||||
chart_regions.get(&page).map(|v| v
|
||||
.iter()
|
||||
.map(|&(a, b, c, d)| (a as i32, b as i32, c as i32, d as i32))
|
||||
.collect::<Vec<_>>())
|
||||
);
|
||||
let columns = match chart_regions.get(&page).filter(|r| !r.is_empty()) {
|
||||
Some(regions) => {
|
||||
let col_input: Vec<TextItem> = page_items
|
||||
.iter()
|
||||
.filter(|it| {
|
||||
let cx = it.x + it.width / 2.0;
|
||||
// Tight bounds: this only blinds the histogram to
|
||||
// chart-internal text; rows adjacent to the chart
|
||||
// belong to the column layout.
|
||||
!regions.iter().any(|&(x0, y0, x1, y1)| {
|
||||
cx >= x0 - 2.0 && cx <= x1 + 2.0 && it.y >= y0 - 2.0 && it.y <= y1 + 2.0
|
||||
})
|
||||
})
|
||||
.cloned()
|
||||
.collect();
|
||||
detect_columns(&col_input, page, table_pages.contains(&page))
|
||||
}
|
||||
None => detect_columns(&page_items, page, table_pages.contains(&page)),
|
||||
};
|
||||
|
||||
if columns.len() <= 1 {
|
||||
// Single column - use simple sorting
|
||||
@@ -1446,6 +1519,56 @@ fn group_single_column(items: Vec<TextItem>, adaptive_threshold: f32) -> Vec<Tex
|
||||
}
|
||||
}
|
||||
}
|
||||
// Same baseline, but separated by a wide void, with the incoming
|
||||
// run starting alphabetic: the neighboring column's body text
|
||||
// sharing a y with this line, in gutters too narrow for column
|
||||
// detection. Both sides must be multi-word prose — TOC page
|
||||
// numbers, dot leaders, and outline-numbered table cells (which
|
||||
// start with digits) stay joined.
|
||||
if let Some(last_item) = last_line.items.last() {
|
||||
let gap = item.x - (last_item.x + last_item.width);
|
||||
if gap > (item.font_size.max(last_item.font_size) * 3.0).max(30.0)
|
||||
&& item
|
||||
.text
|
||||
.trim()
|
||||
.chars()
|
||||
.next()
|
||||
.is_some_and(|c| c.is_alphabetic())
|
||||
{
|
||||
// The incoming run must be substantial prose; the line
|
||||
// side may be short (a wrapped heading's last words).
|
||||
let incoming_wordy = {
|
||||
let t = item.text.trim();
|
||||
t.split_whitespace().count() >= 3
|
||||
&& t.chars().filter(|c| c.is_alphabetic()).count() >= 10
|
||||
};
|
||||
let line_text = last_line
|
||||
.items
|
||||
.iter()
|
||||
.map(|i| i.text.trim())
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
let line_wordy = line_text.split_whitespace().count() >= 2
|
||||
&& line_text.chars().filter(|c| c.is_alphabetic()).count() >= 8;
|
||||
// Lowercase starts are mid-sentence continuations and
|
||||
// split on prose signals alone. Uppercase starts also
|
||||
// need a bold-style mismatch between the runs — a bold
|
||||
// heading beside regular body text — otherwise same-style
|
||||
// label rows (feature tiles, legends) would shatter.
|
||||
let starts_lower = item
|
||||
.text
|
||||
.trim()
|
||||
.chars()
|
||||
.next()
|
||||
.is_some_and(|c| c.is_lowercase());
|
||||
// The whole line must be bold (a heading), not merely
|
||||
// its last run — mixed bold-label/value rows stay joined.
|
||||
let style_mismatch = last_line.items.iter().all(|i| i.is_bold) && !item.is_bold;
|
||||
if line_wordy && incoming_wordy && (starts_lower || style_mismatch) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
true
|
||||
});
|
||||
|
||||
@@ -1519,6 +1642,29 @@ mod tests {
|
||||
items
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn same_baseline_wide_gap_lowercase_continuation_splits() {
|
||||
// Heading in the left column, mid-sentence body text from the right
|
||||
// column at the same y, separated by a wide void: two lines.
|
||||
let items = vec![
|
||||
make_item(1, 94.0, 242.0, "6.2. Expectations for Re-Hiring Staff"),
|
||||
make_item(1, 380.0, 242.0, "they had no plans to re-hire and more"),
|
||||
];
|
||||
let lines = group_single_column(items, 0.10);
|
||||
assert_eq!(lines.len(), 2, "independent column runs must not fuse");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn same_baseline_wide_gap_table_label_stays_joined() {
|
||||
// Outline-numbered cell content to the right: table-ish, keep joined.
|
||||
let items = vec![
|
||||
make_item(1, 94.0, 242.0, "2. Embracing complexity in"),
|
||||
make_item(1, 380.0, 242.0, "2.1 Systems thinking and practice"),
|
||||
];
|
||||
let lines = group_single_column(items, 0.10);
|
||||
assert_eq!(lines.len(), 1, "numbered table cells stay on one line");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn three_zone_layout_detected() {
|
||||
// Left months (x=15..330), right months (x=345..660), sidebar (x=675..800)
|
||||
|
||||
+136
-9
@@ -28,6 +28,7 @@ pub(crate) use fonts::FontStyleCache;
|
||||
pub(crate) use layout::detect_columns;
|
||||
pub use layout::group_into_lines;
|
||||
pub(crate) use layout::group_into_lines_with_thresholds;
|
||||
pub(crate) use layout::group_into_lines_with_thresholds_and_charts;
|
||||
pub(crate) use layout::is_newspaper_layout;
|
||||
pub(crate) use layout::ColumnRegion;
|
||||
|
||||
@@ -176,14 +177,89 @@ fn extract_positioned_text_impl(
|
||||
continue;
|
||||
}
|
||||
}
|
||||
let ((mut items, rects, lines), has_gid_fonts, _coords_rotated) = extract_page_text_items(
|
||||
doc,
|
||||
page_id,
|
||||
*page_num,
|
||||
font_cmaps,
|
||||
include_invisible,
|
||||
&mut style_cache,
|
||||
)?;
|
||||
let ((mut items, mut rects, mut lines), has_gid_fonts, coords_rotated) =
|
||||
extract_page_text_items(
|
||||
doc,
|
||||
page_id,
|
||||
*page_num,
|
||||
font_cmaps,
|
||||
include_invisible,
|
||||
&mut style_cache,
|
||||
)?;
|
||||
// Clip to the visible page box: single-page extracts and imposed
|
||||
// spreads keep neighboring pages' content in the stream, positioned
|
||||
// outside the CropBox. Extracting it interleaves invisible text into
|
||||
// the page and poisons font statistics. Rotated pages are left alone
|
||||
// — their item coordinates are already transformed out of box space.
|
||||
let mut clipped_box: Option<(f32, f32, f32, f32)> = None;
|
||||
if !coords_rotated {
|
||||
if let Some((bx0, by0, bx1, by1)) = get_page_box(doc, page_id) {
|
||||
const TOL: f32 = 6.0;
|
||||
let outside = |it: &TextItem| {
|
||||
let cx = it.x + it.width / 2.0;
|
||||
!(cx >= bx0 - TOL && cx <= bx1 + TOL && it.y >= by0 - TOL && it.y <= by1 + TOL)
|
||||
};
|
||||
// Only clip when the off-page material reads as coherent text
|
||||
// (neighboring-page paragraphs). Curved/rotated display text
|
||||
// leaves short glyph fragments with artifact coordinates
|
||||
// outside the box, and those must stay.
|
||||
let off: Vec<&TextItem> = items.iter().filter(|it| outside(it)).collect();
|
||||
// Judge by character mass: paragraphs are dominated by long
|
||||
// word runs even when interleaved with short math fragments,
|
||||
// while glyph-confetti is short items through and through.
|
||||
let total_chars: usize = off.iter().map(|it| it.text.trim().chars().count()).sum();
|
||||
let wordy_chars: usize = off
|
||||
.iter()
|
||||
.map(|it| it.text.trim().chars().count())
|
||||
.filter(|&n| n >= 4)
|
||||
.sum();
|
||||
// Genuine neighboring-page content is cleanly separated from
|
||||
// on-page text. When an off-page item continues an on-page
|
||||
// line (same baseline, near-adjacent x), the coordinates are
|
||||
// artifacts of transforms we mis-model — don't clip those.
|
||||
let straddles = off.iter().any(|o| {
|
||||
items.iter().any(|i| {
|
||||
!outside(i)
|
||||
&& (i.y - o.y).abs() <= 2.0
|
||||
&& (o.x - (i.x + i.width)).abs() <= 10.0
|
||||
})
|
||||
});
|
||||
let coherent =
|
||||
off.len() >= 10 && wordy_chars * 2 >= total_chars.max(1) && !straddles;
|
||||
if bx1 - bx0 >= 72.0 && by1 - by0 >= 72.0 && coherent {
|
||||
let before = items.len();
|
||||
items.retain(|it| !outside(it));
|
||||
if items.len() < before {
|
||||
debug!(
|
||||
"page {}: clipped {} items outside page box ({:.0},{:.0})-({:.0},{:.0})",
|
||||
page_num,
|
||||
before - items.len(),
|
||||
bx0,
|
||||
by0,
|
||||
bx1,
|
||||
by1
|
||||
);
|
||||
// Only prune off-page geometry when off-page text
|
||||
// existed — same neighboring-page content.
|
||||
let overlaps = |x: f32, y: f32, w: f32, h: f32| {
|
||||
let (x0, x1) = if w < 0.0 { (x + w, x) } else { (x, x + w) };
|
||||
let (y0, y1) = if h < 0.0 { (y + h, y) } else { (y, y + h) };
|
||||
x0 < bx1 + TOL && x1 > bx0 - TOL && y0 < by1 + TOL && y1 > by0 - TOL
|
||||
};
|
||||
rects.retain(|r| overlaps(r.x, r.y, r.width, r.height));
|
||||
clipped_box = Some((bx0, by0, bx1, by1));
|
||||
lines.retain(|l| {
|
||||
overlaps(
|
||||
l.x1.min(l.x2),
|
||||
l.y1.min(l.y2),
|
||||
(l.x2 - l.x1).abs(),
|
||||
(l.y2 - l.y1).abs(),
|
||||
)
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if has_gid_fonts {
|
||||
gid_encoded_pages.insert(*page_num);
|
||||
}
|
||||
@@ -223,7 +299,18 @@ fn extract_positioned_text_impl(
|
||||
all_lines.extend(lines);
|
||||
|
||||
// Extract hyperlinks from page annotations
|
||||
let links = extract_page_links(doc, page_id, *page_num);
|
||||
let mut links = extract_page_links(doc, page_id, *page_num);
|
||||
// Annotations from the neighboring page are off-box too.
|
||||
if let Some((bx0, by0, bx1, by1)) = clipped_box {
|
||||
links.retain(|it| {
|
||||
let cx = it.x + it.width / 2.0;
|
||||
// Center-y, not it.y: link items carry an annotation rect,
|
||||
// so y is a box edge — unlike text items, where y is a
|
||||
// baseline and testing it directly is the natural semantics.
|
||||
let cy = it.y + it.height / 2.0;
|
||||
cx >= bx0 - 6.0 && cx <= bx1 + 6.0 && cy >= by0 - 6.0 && cy <= by1 + 6.0
|
||||
});
|
||||
}
|
||||
all_items.extend(links);
|
||||
}
|
||||
|
||||
@@ -967,6 +1054,46 @@ pub(crate) fn get_number(obj: &Object) -> Option<f32> {
|
||||
}
|
||||
}
|
||||
|
||||
/// Visible page box: CropBox if present, else MediaBox, walking page-tree
|
||||
/// inheritance (both attributes are inheritable). Returns normalized
|
||||
/// (x0, y0, x1, y1) in PDF space.
|
||||
fn get_page_box(doc: &Document, page_id: ObjectId) -> Option<(f32, f32, f32, f32)> {
|
||||
fn find_box(doc: &Document, page_id: ObjectId, key: &[u8]) -> Option<Vec<f32>> {
|
||||
let mut id = page_id;
|
||||
for _ in 0..32 {
|
||||
let dict = doc.get_dictionary(id).ok()?;
|
||||
if let Ok(obj) = dict.get(key) {
|
||||
let arr = match obj {
|
||||
Object::Array(a) => Some(a.clone()),
|
||||
Object::Reference(r) => match doc.get_object(*r) {
|
||||
Ok(Object::Array(a)) => Some(a.clone()),
|
||||
_ => None,
|
||||
},
|
||||
_ => None,
|
||||
};
|
||||
if let Some(arr) = arr {
|
||||
let vals: Vec<f32> = arr.iter().filter_map(get_number).collect();
|
||||
if vals.len() >= 4 {
|
||||
return Some(vals);
|
||||
}
|
||||
}
|
||||
}
|
||||
match dict.get(b"Parent") {
|
||||
Ok(Object::Reference(p)) => id = *p,
|
||||
_ => return None,
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
let v = find_box(doc, page_id, b"CropBox").or_else(|| find_box(doc, page_id, b"MediaBox"))?;
|
||||
Some((
|
||||
v[0].min(v[2]),
|
||||
v[1].min(v[3]),
|
||||
v[0].max(v[2]),
|
||||
v[1].max(v[3]),
|
||||
))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
+5
-10
@@ -54,6 +54,7 @@ pub use extractor::{
|
||||
};
|
||||
pub use markdown::{
|
||||
to_markdown, to_markdown_from_items, to_markdown_from_items_with_rects, MarkdownOptions,
|
||||
MarkdownProfile,
|
||||
};
|
||||
pub use process_mode::ProcessMode;
|
||||
pub use types::{LayoutComplexity, PdfLine, PdfRect, TextItem};
|
||||
@@ -1173,7 +1174,7 @@ pub fn detect_vector_grid_in_region_mem(
|
||||
}
|
||||
|
||||
let line_tables =
|
||||
tables::detect_tables_from_lines(&items_in_region, &lines_in_region, page_1idx);
|
||||
tables::detect_vector_grid_tables_from_lines(&items_in_region, &lines_in_region, page_1idx);
|
||||
for table in line_tables {
|
||||
if let Some(result) = vector_grid_result_from_table(
|
||||
&table,
|
||||
@@ -1526,16 +1527,10 @@ mod vector_grid_tests {
|
||||
/// strip while body rows are drawn with `m`/`l` operators, so the rect
|
||||
/// cluster has only 2 Y-edges and `try_build_grid` rejects.
|
||||
///
|
||||
/// IGNORED: lifting this shape required the exact-duplicate early-dedup
|
||||
/// (PR #76 first iteration), which had broad collateral damage on
|
||||
/// SEC 10-K TOCs and similar docs that draw rule-rects above + below
|
||||
/// section dividers (production diff: 0001104659-25-093871 lost its
|
||||
/// TOC structure, perf-graph data table, and qualifications matrix).
|
||||
/// Re-enable once a more surgical lift exists in `try_build_grid` or
|
||||
/// `snap_edges` that handles cell-border + inner-fill + text-bg rect
|
||||
/// triplets without page-wide dedup.
|
||||
/// The page repeats a full-page background many times. Those fills must be
|
||||
/// removed from clustering before chart/table evidence is evaluated, or
|
||||
/// they swamp the real cell rectangles and make this table look chart-like.
|
||||
#[test]
|
||||
#[ignore]
|
||||
fn greencomp_competence_two_cols() {
|
||||
let tables = detect_rect_tables_in_fixture("tests/fixtures/greencomp_competence.pdf");
|
||||
assert!(
|
||||
|
||||
@@ -374,6 +374,15 @@ pub(crate) fn compute_heading_tiers(lines: &[TextLine], base_size: f32) -> Vec<f
|
||||
for line in lines {
|
||||
if let Some(first) = line.items.first() {
|
||||
if first.font_size / base_size >= 1.2 {
|
||||
// Digit-only lines (page numbers, issue numbers) must not
|
||||
// define heading tiers: a large bold folio claims tier 0 and
|
||||
// blocks the bold-size fallback for the document's real
|
||||
// same-size headings.
|
||||
let text = line.text();
|
||||
let t = text.trim();
|
||||
if !t.is_empty() && t.chars().all(|c| !c.is_alphabetic()) {
|
||||
continue;
|
||||
}
|
||||
heading_sizes.push(first.font_size);
|
||||
}
|
||||
}
|
||||
@@ -391,11 +400,45 @@ pub(crate) fn compute_heading_tiers(lines: &[TextLine], base_size: f32) -> Vec<f
|
||||
}
|
||||
}
|
||||
|
||||
// Books often set section headings barely above body size (e.g. 11pt
|
||||
// bold over 10pt text). When nothing clears the 1.2x ratio gate, fall
|
||||
// back to bold lines modestly larger than body so those documents still
|
||||
// get an H1 instead of every bold heading defaulting to H2.
|
||||
if tiers.is_empty() {
|
||||
let mut bold_sizes: Vec<f32> = lines
|
||||
.iter()
|
||||
.filter(|line| {
|
||||
let text = line.text();
|
||||
let t = text.trim();
|
||||
!t.is_empty() && t.chars().any(|c| c.is_alphabetic())
|
||||
})
|
||||
.filter_map(|line| line.items.first())
|
||||
.filter(|it| it.is_bold && it.font_size / base_size >= 1.05)
|
||||
.map(|it| it.font_size)
|
||||
.collect();
|
||||
bold_sizes.sort_by(|a, b| b.total_cmp(a));
|
||||
for size in bold_sizes {
|
||||
if !tiers.iter().any(|&t| (t - size).abs() < 0.5) {
|
||||
tiers.push(size);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Cap at 4 tiers
|
||||
tiers.truncate(4);
|
||||
tiers
|
||||
}
|
||||
|
||||
/// Boldness of a line judged by character mass, so a heading with an
|
||||
/// unbold section-number prefix ("4. " + bold title) still counts as bold.
|
||||
pub(crate) fn line_is_mostly_bold(line: &TextLine) -> bool {
|
||||
let (bold, total) = line.items.iter().fold((0usize, 0usize), |(b, t), it| {
|
||||
let n = it.text.trim().chars().count();
|
||||
(b + if it.is_bold { n } else { 0 }, t + n)
|
||||
});
|
||||
total > 0 && bold * 2 >= total
|
||||
}
|
||||
|
||||
/// Detect header level from font size using document-specific heading tiers.
|
||||
/// When tiers are available, maps tier 0→H1, tier 1→H2, etc.
|
||||
/// Falls back to ratio-based thresholds when no tiers exist.
|
||||
@@ -403,9 +446,21 @@ pub(crate) fn detect_header_level(
|
||||
font_size: f32,
|
||||
base_size: f32,
|
||||
heading_tiers: &[f32],
|
||||
is_bold: bool,
|
||||
) -> Option<usize> {
|
||||
let ratio = font_size / base_size;
|
||||
|
||||
// Tier matches are trusted below the 1.2x gate (down to 1.05x) only for
|
||||
// bold lines: sub-gate tiers come from the bold fallback, and honoring
|
||||
// them for non-bold text at the same size would promote captions.
|
||||
if (1.05..1.2).contains(&ratio) && is_bold && !heading_tiers.is_empty() {
|
||||
for (i, &tier_size) in heading_tiers.iter().enumerate() {
|
||||
if (font_size - tier_size).abs() < 0.5 {
|
||||
return Some(i + 1); // tier 0 → H1, tier 1 → H2, etc.
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if ratio < 1.2 {
|
||||
return None; // Regular text
|
||||
}
|
||||
@@ -442,6 +497,75 @@ pub(crate) fn detect_header_level(
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn line_of(text: &str, font_size: f32, bold: bool, y: f32) -> crate::types::TextLine {
|
||||
let item = crate::types::TextItem {
|
||||
text: text.into(),
|
||||
x: 72.0,
|
||||
y,
|
||||
width: text.len() as f32 * font_size * 0.5,
|
||||
height: font_size,
|
||||
font: "Test".into(),
|
||||
font_size,
|
||||
page: 1,
|
||||
is_bold: bold,
|
||||
is_italic: false,
|
||||
is_underline: false,
|
||||
is_strikeout: false,
|
||||
item_type: crate::types::ItemType::Text,
|
||||
mcid: None,
|
||||
};
|
||||
crate::types::TextLine {
|
||||
items: vec![item],
|
||||
y,
|
||||
page: 1,
|
||||
adaptive_threshold: 0.10,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn digit_only_lines_do_not_define_tiers() {
|
||||
// A 14pt bold page number must not claim tier 0 — that both demotes
|
||||
// every real heading a level and blocks the bold-size fallback.
|
||||
let lines = vec![
|
||||
line_of("76", 14.0, true, 760.0),
|
||||
line_of("Replace", 11.0, true, 700.0),
|
||||
line_of("body text at eleven points", 11.0, false, 680.0),
|
||||
];
|
||||
let tiers = compute_heading_tiers(&lines, 11.0);
|
||||
assert!(tiers.is_empty(), "page number claimed a tier: {tiers:?}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bold_fallback_tiers_when_nothing_clears_ratio_gate() {
|
||||
// 10pt body, 11pt bold section headings (book-style): no size clears
|
||||
// 1.2x, so bold sizes modestly above body form the tiers.
|
||||
let lines = vec![
|
||||
line_of("4. Entropy", 11.0, true, 700.0),
|
||||
line_of("body text about entropy", 10.0, false, 680.0),
|
||||
line_of("5. The dynamics", 11.0, true, 500.0),
|
||||
];
|
||||
let tiers = compute_heading_tiers(&lines, 10.0);
|
||||
assert_eq!(tiers, vec![11.0]);
|
||||
assert_eq!(detect_header_level(11.0, 10.0, &tiers, true), Some(1));
|
||||
// Non-bold text at the fallback size must not become a heading.
|
||||
assert_eq!(detect_header_level(11.0, 10.0, &tiers, false), None);
|
||||
// Non-tier body text stays regular.
|
||||
assert_eq!(detect_header_level(10.0, 10.0, &tiers, true), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bold_fallback_skipped_when_real_tiers_exist() {
|
||||
let lines = vec![
|
||||
line_of("Chapter One", 18.0, false, 700.0),
|
||||
line_of("bold label", 11.0, true, 600.0),
|
||||
line_of("body", 10.0, false, 580.0),
|
||||
];
|
||||
let tiers = compute_heading_tiers(&lines, 10.0);
|
||||
assert_eq!(tiers, vec![18.0]);
|
||||
// The 11pt bold label does not match any tier and stays non-heading.
|
||||
assert_eq!(detect_header_level(11.0, 10.0, &tiers, true), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn toc_entry_with_single_dot_group() {
|
||||
assert!(is_toc_entry_line("Measurement Lab worksheet ... 3"));
|
||||
|
||||
+519
-97
@@ -1,5 +1,6 @@
|
||||
//! Core line-to-markdown conversion loop with table/image interleaving.
|
||||
|
||||
use std::cmp::Ordering;
|
||||
use std::collections::{HashMap, HashSet};
|
||||
|
||||
use crate::structure_tree::StructRole;
|
||||
@@ -13,9 +14,149 @@ use super::analysis::{
|
||||
use super::classify::{
|
||||
format_list_item, is_caption_line, is_list_item, is_monospace_font, starts_with_bullet_marker,
|
||||
};
|
||||
use super::heading::classify_heading_sequences;
|
||||
use super::postprocess::clean_markdown;
|
||||
use super::preprocess::{merge_drop_caps, merge_heading_lines};
|
||||
use super::MarkdownOptions;
|
||||
use super::{item_is_in_chart_region, MarkdownOptions, CHART_SEPARATOR_PAD};
|
||||
|
||||
/// Logical stream geometry for a page where one full-width chart separates
|
||||
/// two prose columns. Positioned non-text blocks use this same ordering so a
|
||||
/// right-column table or image cannot jump ahead of left-column prose.
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub(super) struct ChartProseOrder {
|
||||
split_x: f32,
|
||||
chart_region: (f32, f32, f32, f32),
|
||||
}
|
||||
|
||||
impl ChartProseOrder {
|
||||
pub(super) fn new(split_x: f32, chart_region: (f32, f32, f32, f32)) -> Self {
|
||||
Self {
|
||||
split_x,
|
||||
chart_region,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Markdown block with its physical position and optional logical chart-page
|
||||
/// stream. Tables and images share this representation because both are
|
||||
/// removed before text-line grouping and reinserted during conversion.
|
||||
#[derive(Debug, Clone)]
|
||||
pub(super) struct PositionedMarkdown {
|
||||
y: f32,
|
||||
x: f32,
|
||||
markdown: String,
|
||||
chart_order: Option<ChartProseOrder>,
|
||||
}
|
||||
|
||||
impl PositionedMarkdown {
|
||||
pub(super) fn new(
|
||||
y: f32,
|
||||
x: f32,
|
||||
markdown: String,
|
||||
chart_order: Option<ChartProseOrder>,
|
||||
) -> Self {
|
||||
Self {
|
||||
y,
|
||||
x,
|
||||
markdown,
|
||||
chart_order,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn chart_stream_position(
|
||||
y: f32,
|
||||
x: f32,
|
||||
claimed_by_chart: bool,
|
||||
order: ChartProseOrder,
|
||||
) -> (u8, u8) {
|
||||
let (_, y0, _, y1) = order.chart_region;
|
||||
let low = y0.min(y1) - CHART_SEPARATOR_PAD;
|
||||
let high = y0.max(y1) + CHART_SEPARATOR_PAD;
|
||||
let in_chart_zone = claimed_by_chart || (y >= low && y <= high);
|
||||
let zone = if in_chart_zone {
|
||||
1
|
||||
} else if y > high {
|
||||
0
|
||||
} else {
|
||||
2
|
||||
};
|
||||
let column = if in_chart_zone || x < order.split_x {
|
||||
0
|
||||
} else {
|
||||
1
|
||||
};
|
||||
(zone, column)
|
||||
}
|
||||
|
||||
fn positioned_block_precedes_line(block: &PositionedMarkdown, line: &TextLine) -> bool {
|
||||
let Some(order) = block.chart_order else {
|
||||
return block.y > line.y;
|
||||
};
|
||||
let line_x = line.items.first().map(|item| item.x).unwrap_or(0.0);
|
||||
let line_claimed_by_chart = line
|
||||
.items
|
||||
.iter()
|
||||
.any(|item| item_is_in_chart_region(item, &[order.chart_region]));
|
||||
let block_position = chart_stream_position(block.y, block.x, false, order);
|
||||
let line_position = chart_stream_position(line.y, line_x, line_claimed_by_chart, order);
|
||||
block_position < line_position || (block_position == line_position && block.y > line.y)
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)]
|
||||
enum PositionedBlockKind {
|
||||
Table,
|
||||
Image,
|
||||
}
|
||||
|
||||
type PositionedBlockRef<'a> = (PositionedBlockKind, usize, &'a PositionedMarkdown);
|
||||
|
||||
fn compare_positioned_blocks(
|
||||
(a_kind, a_idx, a): &PositionedBlockRef<'_>,
|
||||
(b_kind, b_idx, b): &PositionedBlockRef<'_>,
|
||||
) -> Ordering {
|
||||
if let (Some(a_order), Some(b_order)) = (a.chart_order, b.chart_order) {
|
||||
let a_position = chart_stream_position(a.y, a.x, false, a_order);
|
||||
let b_position = chart_stream_position(b.y, b.x, false, b_order);
|
||||
return a_position
|
||||
.cmp(&b_position)
|
||||
.then_with(|| b.y.total_cmp(&a.y))
|
||||
.then_with(|| a.x.total_cmp(&b.x))
|
||||
.then_with(|| a_kind.cmp(b_kind))
|
||||
.then_with(|| a_idx.cmp(b_idx));
|
||||
}
|
||||
|
||||
// Preserve the legacy ordering for ordinary pages: tables in detection
|
||||
// order, followed by images in input order. Chart pages give every block
|
||||
// a chart order and use the logical stream comparison above.
|
||||
a_kind.cmp(b_kind).then_with(|| a_idx.cmp(b_idx))
|
||||
}
|
||||
|
||||
fn positioned_blocks_for_page<'a>(
|
||||
page: u32,
|
||||
page_tables: &'a HashMap<u32, Vec<PositionedMarkdown>>,
|
||||
page_images: &'a HashMap<u32, Vec<PositionedMarkdown>>,
|
||||
) -> Vec<PositionedBlockRef<'a>> {
|
||||
let mut blocks = Vec::new();
|
||||
if let Some(tables) = page_tables.get(&page) {
|
||||
blocks.extend(
|
||||
tables
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(idx, table)| (PositionedBlockKind::Table, idx, table)),
|
||||
);
|
||||
}
|
||||
if let Some(images) = page_images.get(&page) {
|
||||
blocks.extend(
|
||||
images
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(idx, image)| (PositionedBlockKind::Image, idx, image)),
|
||||
);
|
||||
}
|
||||
blocks.sort_by(compare_positioned_blocks);
|
||||
blocks
|
||||
}
|
||||
|
||||
/// Pre-scan struct heading tags to find levels that are overused — i.e., tagged on
|
||||
/// so many lines that they clearly represent body text, not real headings.
|
||||
@@ -160,6 +301,112 @@ fn find_isolated_lines(lines: &[TextLine], base_size: f32, para_threshold: f32)
|
||||
/// wrapped visual line as "standalone" once the first line is misclassified,
|
||||
/// producing a stack of `##` headings. Multi-line body-size bold runs with a
|
||||
/// paragraph-sized word count should stay paragraph text.
|
||||
/// Merge 2-3 consecutive all-bold body-size lines into one line when the
|
||||
/// group is isolated (paragraph break before and after) and short enough to
|
||||
/// be a heading. Longer/wordier bold runs are wrapped bold paragraphs and
|
||||
/// are left for `find_wrapped_bold_paragraph_lines` to suppress.
|
||||
/// "9.5. ", "12.3.1. " — section-numbered heading prefix followed by a word.
|
||||
fn starts_with_section_number(t: &str) -> bool {
|
||||
let t = t.trim_start();
|
||||
let mut rest = t;
|
||||
let mut groups = 0;
|
||||
loop {
|
||||
let digits = rest.chars().take_while(|c| c.is_ascii_digit()).count();
|
||||
if digits == 0 || digits > 3 {
|
||||
break;
|
||||
}
|
||||
groups += 1;
|
||||
rest = &rest[digits..];
|
||||
if let Some(r) = rest.strip_prefix('.') {
|
||||
rest = r;
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
// Two components minimum ("9.5. "): a single "1. " is an ordered list
|
||||
// item, and this prefix bypasses the isolation checks entirely.
|
||||
groups >= 2
|
||||
&& rest.starts_with(char::is_whitespace)
|
||||
&& rest.trim_start().starts_with(|c: char| c.is_alphabetic())
|
||||
}
|
||||
|
||||
fn merge_wrapped_bold_heading_groups(
|
||||
lines: Vec<TextLine>,
|
||||
base_size: f32,
|
||||
para_threshold: f32,
|
||||
) -> Vec<TextLine> {
|
||||
let mut out: Vec<TextLine> = Vec::with_capacity(lines.len());
|
||||
let mut i = 0usize;
|
||||
while i < lines.len() {
|
||||
if !is_body_size_all_bold_line(&lines[i], base_size) {
|
||||
out.push(lines[i].clone());
|
||||
i += 1;
|
||||
continue;
|
||||
}
|
||||
let start = i;
|
||||
let mut end = i;
|
||||
let mut word_count = lines[i].text().split_whitespace().count();
|
||||
while end + 1 < lines.len()
|
||||
&& is_body_size_all_bold_line(&lines[end + 1], base_size)
|
||||
&& is_wrapped_same_style_line(&lines[end], &lines[end + 1], para_threshold)
|
||||
{
|
||||
end += 1;
|
||||
word_count += lines[end].text().split_whitespace().count();
|
||||
}
|
||||
let line_count = end - start + 1;
|
||||
// Column-local isolation: on interleaved multi-column pages the
|
||||
// vector neighbors may be the other column's lines, so judge the
|
||||
// break by x-overlapping lines only.
|
||||
let gx0 = lines[start..=end]
|
||||
.iter()
|
||||
.flat_map(|l| l.items.iter().map(|i| i.x))
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let gx1 = lines[start..=end]
|
||||
.iter()
|
||||
.flat_map(|l| l.items.iter().map(|i| i.x + i.width))
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
let overlaps_x = |l: &TextLine| {
|
||||
let lx0 = l.items.iter().map(|i| i.x).fold(f32::INFINITY, f32::min);
|
||||
let lx1 = l
|
||||
.items
|
||||
.iter()
|
||||
.map(|i| i.x + i.width)
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
lx0 < gx1 && lx1 > gx0
|
||||
};
|
||||
let page = lines[start].page;
|
||||
let break_before = !lines.iter().any(|l| {
|
||||
l.page == page
|
||||
&& l.y > lines[start].y
|
||||
&& l.y - lines[start].y <= para_threshold
|
||||
&& overlaps_x(l)
|
||||
});
|
||||
let break_after = !lines.iter().any(|l| {
|
||||
l.page == page
|
||||
&& l.y < lines[end].y
|
||||
&& lines[end].y - l.y <= para_threshold
|
||||
&& overlaps_x(l)
|
||||
});
|
||||
let numbered = starts_with_section_number(&lines[start].text());
|
||||
if (2..=3).contains(&line_count)
|
||||
&& word_count <= 15
|
||||
&& ((break_before && break_after) || numbered)
|
||||
{
|
||||
let mut merged = lines[start].clone();
|
||||
for l in &lines[start + 1..=end] {
|
||||
merged.items.extend(l.items.iter().cloned());
|
||||
}
|
||||
out.push(merged);
|
||||
} else {
|
||||
for l in &lines[start..=end] {
|
||||
out.push(l.clone());
|
||||
}
|
||||
}
|
||||
i = end + 1;
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
fn find_wrapped_bold_paragraph_lines(
|
||||
lines: &[TextLine],
|
||||
base_size: f32,
|
||||
@@ -277,7 +524,7 @@ fn struct_role_heading_level(role: &StructRole) -> Option<usize> {
|
||||
/// We strip their header+separator rows and append their data rows to the first page's
|
||||
/// table, then remove them from later pages.
|
||||
pub(super) fn merge_continuation_tables(
|
||||
page_tables: &mut std::collections::HashMap<u32, Vec<(f32, String)>>,
|
||||
page_tables: &mut std::collections::HashMap<u32, Vec<PositionedMarkdown>>,
|
||||
table_only_pages: &HashSet<u32>,
|
||||
) {
|
||||
let mut sorted_pages: Vec<u32> = page_tables.keys().copied().collect();
|
||||
@@ -305,7 +552,7 @@ pub(super) fn merge_continuation_tables(
|
||||
continue;
|
||||
}
|
||||
|
||||
let first_col_count = count_table_columns(&first_tables[0].1);
|
||||
let first_col_count = count_table_columns(&first_tables[0].markdown);
|
||||
if first_col_count == 0 {
|
||||
i += 1;
|
||||
continue;
|
||||
@@ -336,7 +583,7 @@ pub(super) fn merge_continuation_tables(
|
||||
_ => break,
|
||||
};
|
||||
|
||||
let next_col_count = count_table_columns(&next_tables[0].1);
|
||||
let next_col_count = count_table_columns(&next_tables[0].markdown);
|
||||
if next_col_count != first_col_count {
|
||||
break;
|
||||
}
|
||||
@@ -350,7 +597,7 @@ pub(super) fn merge_continuation_tables(
|
||||
let mut extra_rows = String::new();
|
||||
for &cont_page in &continuation_pages {
|
||||
if let Some(tables) = page_tables.get(&cont_page) {
|
||||
let table_md = &tables[0].1;
|
||||
let table_md = &tables[0].markdown;
|
||||
// Skip header row (line 1) and separator row (line 2), keep the rest
|
||||
for (line_idx, line) in table_md.lines().enumerate() {
|
||||
if line_idx >= 2 {
|
||||
@@ -363,7 +610,7 @@ pub(super) fn merge_continuation_tables(
|
||||
|
||||
// Append continuation rows to the first page's table
|
||||
if let Some(tables) = page_tables.get_mut(&first_page) {
|
||||
tables[0].1.push_str(&extra_rows);
|
||||
tables[0].markdown.push_str(&extra_rows);
|
||||
}
|
||||
|
||||
// Remove continuation pages from the map
|
||||
@@ -395,37 +642,35 @@ fn count_table_columns(table_md: &str) -> usize {
|
||||
/// Flush any remaining tables and images for a given page
|
||||
fn flush_page_tables_and_images(
|
||||
page: u32,
|
||||
page_tables: &std::collections::HashMap<u32, Vec<(f32, String)>>,
|
||||
page_images: &std::collections::HashMap<u32, Vec<(f32, String)>>,
|
||||
page_blocks: &HashMap<u32, Vec<PositionedBlockRef<'_>>>,
|
||||
inserted_tables: &mut HashSet<(u32, usize)>,
|
||||
inserted_images: &mut HashSet<(u32, usize)>,
|
||||
output: &mut String,
|
||||
in_paragraph: &mut bool,
|
||||
) {
|
||||
if let Some(tables) = page_tables.get(&page) {
|
||||
for (idx, (_, table_md)) in tables.iter().enumerate() {
|
||||
if !inserted_tables.contains(&(page, idx)) {
|
||||
if *in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
*in_paragraph = false;
|
||||
}
|
||||
output.push('\n');
|
||||
output.push_str(table_md);
|
||||
output.push('\n');
|
||||
let Some(blocks) = page_blocks.get(&page) else {
|
||||
return;
|
||||
};
|
||||
for &(kind, idx, block) in blocks {
|
||||
let already_inserted = match kind {
|
||||
PositionedBlockKind::Table => inserted_tables.contains(&(page, idx)),
|
||||
PositionedBlockKind::Image => inserted_images.contains(&(page, idx)),
|
||||
};
|
||||
if already_inserted {
|
||||
continue;
|
||||
}
|
||||
if *in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
*in_paragraph = false;
|
||||
}
|
||||
output.push('\n');
|
||||
output.push_str(&block.markdown);
|
||||
output.push('\n');
|
||||
match kind {
|
||||
PositionedBlockKind::Table => {
|
||||
inserted_tables.insert((page, idx));
|
||||
}
|
||||
}
|
||||
}
|
||||
if let Some(images) = page_images.get(&page) {
|
||||
for (idx, (_, image_md)) in images.iter().enumerate() {
|
||||
if !inserted_images.contains(&(page, idx)) {
|
||||
if *in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
*in_paragraph = false;
|
||||
}
|
||||
output.push('\n');
|
||||
output.push_str(image_md);
|
||||
output.push('\n');
|
||||
PositionedBlockKind::Image => {
|
||||
inserted_images.insert((page, idx));
|
||||
}
|
||||
}
|
||||
@@ -436,8 +681,9 @@ fn flush_page_tables_and_images(
|
||||
pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
lines: Vec<TextLine>,
|
||||
options: MarkdownOptions,
|
||||
page_tables: std::collections::HashMap<u32, Vec<(f32, String)>>,
|
||||
page_images: std::collections::HashMap<u32, Vec<(f32, String)>>,
|
||||
page_tables: std::collections::HashMap<u32, Vec<PositionedMarkdown>>,
|
||||
page_images: std::collections::HashMap<u32, Vec<PositionedMarkdown>>,
|
||||
page_chart_regions: &std::collections::HashMap<u32, Vec<(f32, f32, f32, f32)>>,
|
||||
band_split_pages: &HashSet<u32>,
|
||||
struct_roles: Option<
|
||||
&std::collections::HashMap<u32, std::collections::HashMap<i64, StructRole>>,
|
||||
@@ -468,6 +714,17 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
// threshold and cause every line to be treated as a paragraph break.
|
||||
let para_threshold = compute_paragraph_threshold(&lines, base_size);
|
||||
|
||||
// Merge wrapped bold headings: a 2-3 line group of consecutive all-bold
|
||||
// body-size lines that is isolated as a group (paragraph break before
|
||||
// and after) is one heading that wrapped. Left split, the internal line
|
||||
// gap breaks each line's isolation and neither classifies as a heading —
|
||||
// the whole group then merges into the following body paragraph.
|
||||
let lines = if std::env::var("PI_NO_MERGE").is_ok() {
|
||||
lines
|
||||
} else {
|
||||
merge_wrapped_bold_heading_groups(lines, base_size, para_threshold)
|
||||
};
|
||||
|
||||
// Pre-scan: identify isolated lines (paragraph break before AND after).
|
||||
// These are heading candidates even without bold/large font — common in
|
||||
// academic papers where section titles like "Acknowledgements" sit alone
|
||||
@@ -477,6 +734,33 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
let wrapped_bold_paragraph_lines =
|
||||
find_wrapped_bold_paragraph_lines(&lines, base_size, para_threshold);
|
||||
|
||||
let mut sequence_excluded_lines = wrapped_bold_paragraph_lines.clone();
|
||||
for (line_idx, line) in lines.iter().enumerate() {
|
||||
if page_chart_regions.get(&line.page).is_some_and(|regions| {
|
||||
line.items
|
||||
.iter()
|
||||
.any(|item| item_is_in_chart_region(item, regions))
|
||||
}) {
|
||||
sequence_excluded_lines.insert(line_idx);
|
||||
}
|
||||
}
|
||||
if let Some(roles) = struct_roles {
|
||||
for (line_idx, line) in lines.iter().enumerate() {
|
||||
if resolve_line_struct_role(line, roles)
|
||||
.is_some_and(|role| role.is_non_heading_content())
|
||||
{
|
||||
sequence_excluded_lines.insert(line_idx);
|
||||
}
|
||||
}
|
||||
}
|
||||
let sequence_heading_levels = classify_heading_sequences(
|
||||
&lines,
|
||||
base_size,
|
||||
&heading_tiers,
|
||||
&isolated_lines,
|
||||
&sequence_excluded_lines,
|
||||
);
|
||||
|
||||
// Detect struct heading levels that are overused (body text mistagged as headings)
|
||||
let overused_heading_levels = detect_overused_struct_heading_levels(&lines, struct_roles);
|
||||
|
||||
@@ -502,6 +786,18 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
.collect();
|
||||
all_content_pages.sort();
|
||||
all_content_pages.dedup();
|
||||
// Build the unified table/image order once per page. This is only a
|
||||
// meaningful sort on chart/prose pages; ordinary pages retain their
|
||||
// legacy table-then-image order without repeating work for every line.
|
||||
let page_blocks: HashMap<u32, Vec<PositionedBlockRef<'_>>> = all_content_pages
|
||||
.iter()
|
||||
.map(|&page| {
|
||||
(
|
||||
page,
|
||||
positioned_blocks_for_page(page, &page_tables, &page_images),
|
||||
)
|
||||
})
|
||||
.collect();
|
||||
|
||||
for (line_idx, line) in lines.iter().enumerate() {
|
||||
// Page break
|
||||
@@ -514,8 +810,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
}
|
||||
flush_page_tables_and_images(
|
||||
current_page,
|
||||
&page_tables,
|
||||
&page_images,
|
||||
&page_blocks,
|
||||
&mut inserted_tables,
|
||||
&mut inserted_images,
|
||||
&mut output,
|
||||
@@ -539,8 +834,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
}
|
||||
flush_page_tables_and_images(
|
||||
p,
|
||||
&page_tables,
|
||||
&page_images,
|
||||
&page_blocks,
|
||||
&mut inserted_tables,
|
||||
&mut inserted_images,
|
||||
&mut output,
|
||||
@@ -563,38 +857,32 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
}
|
||||
}
|
||||
|
||||
// Check if we should insert a table before this line
|
||||
if let Some(tables) = page_tables.get(¤t_page) {
|
||||
for (idx, (table_y, table_md)) in tables.iter().enumerate() {
|
||||
// Insert table when we pass its Y position
|
||||
if *table_y > line.y && !inserted_tables.contains(&(current_page, idx)) {
|
||||
// Insert tables and images through one ordered stream. Chart/prose
|
||||
// pages sort by zone, column, and physical Y; ordinary pages retain
|
||||
// the legacy table-then-image input order.
|
||||
if let Some(blocks) = page_blocks.get(¤t_page) {
|
||||
for &(kind, idx, block) in blocks {
|
||||
let already_inserted = match kind {
|
||||
PositionedBlockKind::Table => inserted_tables.contains(&(current_page, idx)),
|
||||
PositionedBlockKind::Image => inserted_images.contains(&(current_page, idx)),
|
||||
};
|
||||
if positioned_block_precedes_line(block, line) && !already_inserted {
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
in_paragraph = false;
|
||||
paragraph_in_wrapped_bold_run = false;
|
||||
}
|
||||
output.push('\n');
|
||||
output.push_str(table_md);
|
||||
output.push_str(&block.markdown);
|
||||
output.push('\n');
|
||||
inserted_tables.insert((current_page, idx));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Check if we should insert an image before this line
|
||||
if let Some(images) = page_images.get(¤t_page) {
|
||||
for (idx, (image_y, image_md)) in images.iter().enumerate() {
|
||||
// Insert image when we pass its Y position
|
||||
if *image_y > line.y && !inserted_images.contains(&(current_page, idx)) {
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
in_paragraph = false;
|
||||
paragraph_in_wrapped_bold_run = false;
|
||||
match kind {
|
||||
PositionedBlockKind::Table => {
|
||||
inserted_tables.insert((current_page, idx));
|
||||
}
|
||||
PositionedBlockKind::Image => {
|
||||
inserted_images.insert((current_page, idx));
|
||||
}
|
||||
}
|
||||
output.push('\n');
|
||||
output.push_str(image_md);
|
||||
output.push('\n');
|
||||
inserted_images.insert((current_page, idx));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -721,7 +1009,13 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
&& toc_suppress_page != Some(line.page)
|
||||
{
|
||||
let line_font_size = line.items.first().map(|i| i.font_size).unwrap_or(base_size);
|
||||
detect_header_level(line_font_size, base_size, &heading_tiers).or_else(|| {
|
||||
detect_header_level(
|
||||
line_font_size,
|
||||
base_size,
|
||||
&heading_tiers,
|
||||
crate::markdown::analysis::line_is_mostly_bold(line),
|
||||
)
|
||||
.or_else(|| {
|
||||
// Rarity-based heading detection (inspired by opendataloader).
|
||||
// Heading probability scoring with lookahead context.
|
||||
// Score = rarity * 0.5 + bold * 0.3 + standalone * 0.2
|
||||
@@ -754,16 +1048,23 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
// paragraph continuity and minor font-size variation
|
||||
// inflates rarity scores.
|
||||
let has_strong_signal = all_bold || isolated || (rarity >= 0.97 && word_count <= 8);
|
||||
// Single-word headings ("IMPLEMENTATION", "CONTENTS") are common;
|
||||
// accept them only with the strongest signal combination.
|
||||
let enough_words =
|
||||
word_count >= 2 || (all_bold && isolated && plain_trimmed.len() >= 4);
|
||||
if score >= 0.5 && standalone && enough_words && has_strong_signal {
|
||||
// Single-word headings ("IMPLEMENTATION", "CONTENTS",
|
||||
// "Replace") are common. All-bold single words qualify when
|
||||
// standalone (paragraph break before / page top) — headings
|
||||
// hug their section's first paragraph, so requiring a break
|
||||
// after as well missed most of them. Mixed bold lead-ins
|
||||
// ("Note: ...") are excluded by all_bold.
|
||||
let enough_words = word_count >= 2 || (all_bold && plain_trimmed.len() >= 4);
|
||||
let numbered_bold = all_bold && starts_with_section_number(plain_trimmed);
|
||||
if numbered_bold
|
||||
|| (score >= 0.5 && standalone && enough_words && has_strong_signal)
|
||||
{
|
||||
Some(bold_heading_level(&heading_tiers))
|
||||
} else {
|
||||
None
|
||||
}
|
||||
})
|
||||
.or_else(|| sequence_heading_levels.get(&line_idx).copied())
|
||||
} else {
|
||||
None
|
||||
};
|
||||
@@ -915,8 +1216,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
// (handles table-only pages after the last text line, and trailing image-only pages)
|
||||
flush_page_tables_and_images(
|
||||
current_page,
|
||||
&page_tables,
|
||||
&page_images,
|
||||
&page_blocks,
|
||||
&mut inserted_tables,
|
||||
&mut inserted_images,
|
||||
&mut output,
|
||||
@@ -928,8 +1228,7 @@ pub(super) fn to_markdown_from_lines_with_tables_and_images(
|
||||
}
|
||||
flush_page_tables_and_images(
|
||||
p,
|
||||
&page_tables,
|
||||
&page_images,
|
||||
&page_blocks,
|
||||
&mut inserted_tables,
|
||||
&mut inserted_images,
|
||||
&mut output,
|
||||
@@ -973,6 +1272,13 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
let isolated_lines = find_isolated_lines(&lines, base_size, para_threshold);
|
||||
let wrapped_bold_paragraph_lines =
|
||||
find_wrapped_bold_paragraph_lines(&lines, base_size, para_threshold);
|
||||
let sequence_heading_levels = classify_heading_sequences(
|
||||
&lines,
|
||||
base_size,
|
||||
&heading_tiers,
|
||||
&isolated_lines,
|
||||
&wrapped_bold_paragraph_lines,
|
||||
);
|
||||
|
||||
let mut output = String::new();
|
||||
let mut current_page = 0u32;
|
||||
@@ -1067,33 +1373,39 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
&& !(options.detect_code && line.items.iter().any(|i| is_monospace_font(&i.font)))
|
||||
{
|
||||
let line_font_size = line.items.first().map(|i| i.font_size).unwrap_or(base_size);
|
||||
if let Some(header_level) =
|
||||
detect_header_level(line_font_size, base_size, &heading_tiers).or_else(|| {
|
||||
if line_font_size < base_size * 0.95 {
|
||||
return None;
|
||||
}
|
||||
let word_count = plain_trimmed.split_whitespace().count();
|
||||
if !(1..=15).contains(&word_count) {
|
||||
return None;
|
||||
}
|
||||
if wrapped_bold_paragraph_lines.contains(&line_idx) {
|
||||
return None;
|
||||
}
|
||||
let rarity = font_size_rarity(line_font_size, &font_stats);
|
||||
let all_bold = !line.items.is_empty() && line.items.iter().all(|i| i.is_bold);
|
||||
let standalone = !in_paragraph;
|
||||
let isolated = isolated_lines.contains(&line_idx);
|
||||
let score = rarity * 0.5
|
||||
+ if all_bold { 0.3 } else { 0.0 }
|
||||
+ if standalone { 0.2 } else { 0.0 }
|
||||
+ if isolated { 0.3 } else { 0.0 };
|
||||
let enough_words =
|
||||
word_count >= 2 || (all_bold && isolated && plain_trimmed.len() >= 4);
|
||||
if score >= 0.5 && standalone && enough_words {
|
||||
return Some(bold_heading_level(&heading_tiers));
|
||||
}
|
||||
None
|
||||
})
|
||||
if let Some(header_level) = detect_header_level(
|
||||
line_font_size,
|
||||
base_size,
|
||||
&heading_tiers,
|
||||
crate::markdown::analysis::line_is_mostly_bold(line),
|
||||
)
|
||||
.or_else(|| {
|
||||
if line_font_size < base_size * 0.95 {
|
||||
return None;
|
||||
}
|
||||
let word_count = plain_trimmed.split_whitespace().count();
|
||||
if !(1..=15).contains(&word_count) {
|
||||
return None;
|
||||
}
|
||||
if wrapped_bold_paragraph_lines.contains(&line_idx) {
|
||||
return None;
|
||||
}
|
||||
let rarity = font_size_rarity(line_font_size, &font_stats);
|
||||
let all_bold = !line.items.is_empty() && line.items.iter().all(|i| i.is_bold);
|
||||
let standalone = !in_paragraph;
|
||||
let isolated = isolated_lines.contains(&line_idx);
|
||||
let score = rarity * 0.5
|
||||
+ if all_bold { 0.3 } else { 0.0 }
|
||||
+ if standalone { 0.2 } else { 0.0 }
|
||||
+ if isolated { 0.3 } else { 0.0 };
|
||||
let enough_words =
|
||||
word_count >= 2 || (all_bold && isolated && plain_trimmed.len() >= 4);
|
||||
if score >= 0.5 && standalone && enough_words {
|
||||
return Some(bold_heading_level(&heading_tiers));
|
||||
}
|
||||
None
|
||||
})
|
||||
.or_else(|| sequence_heading_levels.get(&line_idx).copied())
|
||||
{
|
||||
if in_paragraph {
|
||||
output.push_str("\n\n");
|
||||
@@ -1204,6 +1516,20 @@ pub fn to_markdown_from_lines(lines: Vec<TextLine>, options: MarkdownOptions) ->
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
|
||||
#[test]
|
||||
fn section_number_prefix_detection() {
|
||||
assert!(starts_with_section_number(
|
||||
"9.5. Adapting to the New Normal"
|
||||
));
|
||||
assert!(starts_with_section_number("12.3.1. Deep subsection"));
|
||||
assert!(starts_with_section_number("2.1 Systems thinking"));
|
||||
assert!(!starts_with_section_number("1. First item in a list"));
|
||||
assert!(!starts_with_section_number("24% in October 2020."));
|
||||
assert!(!starts_with_section_number("2020 was a hard year"));
|
||||
assert!(!starts_with_section_number("Introduction"));
|
||||
}
|
||||
|
||||
use super::*;
|
||||
use crate::structure_tree::StructRole;
|
||||
use crate::types::TextItem;
|
||||
@@ -1245,6 +1571,91 @@ mod tests {
|
||||
make_line(vec![item])
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn chart_page_blocks_follow_zone_and_column_stream() {
|
||||
let line = |text: &str, x: f32, y: f32| {
|
||||
let mut item = make_item(text, 1, None);
|
||||
item.x = x;
|
||||
item.y = y;
|
||||
make_line(vec![item])
|
||||
};
|
||||
// Logical newspaper order for the prose zone: all left-column lines,
|
||||
// then all right-column lines, even though their physical Y values
|
||||
// jump back upward at the column switch.
|
||||
let lines = vec![
|
||||
line("Left column upper prose.", 90.0, 700.0),
|
||||
line("Left column lower prose.", 90.0, 500.0),
|
||||
line("Right column upper prose.", 340.0, 700.0),
|
||||
line("Right column lower prose.", 340.0, 500.0),
|
||||
];
|
||||
let order = ChartProseOrder::new(280.0, (100.0, 300.0, 500.0, 400.0));
|
||||
let mut tables = HashMap::new();
|
||||
tables.insert(
|
||||
1,
|
||||
vec![
|
||||
// Detection order is deliberately right before left.
|
||||
PositionedMarkdown::new(
|
||||
600.0,
|
||||
340.0,
|
||||
"| Right metric | Value |\n|---|---|\n| A | 1 |\n".into(),
|
||||
Some(order),
|
||||
),
|
||||
PositionedMarkdown::new(
|
||||
550.0,
|
||||
90.0,
|
||||
"| Left metric | Value |\n|---|---|\n| B | 2 |\n".into(),
|
||||
Some(order),
|
||||
),
|
||||
],
|
||||
);
|
||||
let mut images = HashMap::new();
|
||||
images.insert(
|
||||
1,
|
||||
vec\n".into(),
|
||||
Some(order),
|
||||
),
|
||||
PositionedMarkdown::new(
|
||||
575.0,
|
||||
340.0,
|
||||
"\n".into(),
|
||||
Some(order),
|
||||
),
|
||||
],
|
||||
);
|
||||
|
||||
let md = to_markdown_from_lines_with_tables_and_images(
|
||||
lines,
|
||||
MarkdownOptions::default(),
|
||||
tables,
|
||||
images,
|
||||
&HashMap::new(),
|
||||
&HashSet::from([1]),
|
||||
None,
|
||||
);
|
||||
let positions = [
|
||||
"Left column upper prose.",
|
||||
"",
|
||||
"| Left metric | Value |",
|
||||
"Left column lower prose.",
|
||||
"Right column upper prose.",
|
||||
"| Right metric | Value |",
|
||||
"",
|
||||
"Right column lower prose.",
|
||||
]
|
||||
.map(|needle| {
|
||||
md.find(needle)
|
||||
.unwrap_or_else(|| panic!("missing {needle:?} in {md}"))
|
||||
});
|
||||
assert!(
|
||||
positions.windows(2).all(|pair| pair[0] < pair[1]),
|
||||
"blocks must follow the logical chart-page stream: {md}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn isolated_lines_kept_on_sparse_pages() {
|
||||
// A ToC page with a lone title and one entry far below: the density
|
||||
@@ -1297,6 +1708,7 @@ mod tests {
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
Some(&roles),
|
||||
);
|
||||
@@ -1325,6 +1737,7 @@ mod tests {
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
Some(&roles),
|
||||
);
|
||||
@@ -1366,6 +1779,7 @@ mod tests {
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
Some(&roles),
|
||||
);
|
||||
@@ -1402,6 +1816,7 @@ mod tests {
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
Some(&roles),
|
||||
);
|
||||
@@ -1437,6 +1852,7 @@ mod tests {
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
Some(&roles),
|
||||
);
|
||||
@@ -1466,6 +1882,7 @@ mod tests {
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
None,
|
||||
);
|
||||
@@ -1507,6 +1924,7 @@ mod tests {
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
Some(&roles),
|
||||
);
|
||||
@@ -1555,6 +1973,7 @@ mod tests {
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
None,
|
||||
);
|
||||
@@ -1656,6 +2075,7 @@ mod tests {
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
None,
|
||||
);
|
||||
@@ -1703,6 +2123,7 @@ mod tests {
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
Some(&roles),
|
||||
);
|
||||
@@ -1842,6 +2263,7 @@ mod tests {
|
||||
MarkdownOptions::default(),
|
||||
HashMap::new(),
|
||||
HashMap::new(),
|
||||
&HashMap::new(),
|
||||
&std::collections::HashSet::new(),
|
||||
None,
|
||||
);
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
+1214
-85
File diff suppressed because it is too large
Load Diff
@@ -2,12 +2,15 @@
|
||||
|
||||
use regex::Regex;
|
||||
|
||||
use super::MarkdownOptions;
|
||||
use super::{MarkdownOptions, MarkdownProfile};
|
||||
|
||||
/// Clean up markdown output with post-processing
|
||||
pub(crate) fn clean_markdown(mut text: String, options: &MarkdownOptions) -> String {
|
||||
// Collapse dot leaders (e.g. TOC entries: "Introduction...............................1")
|
||||
text = collapse_dot_leaders(&text);
|
||||
if options.profile == MarkdownProfile::Compact {
|
||||
// Dot-leader collapse saves tokens but changes source text, so it is
|
||||
// reserved for the explicit compact profile.
|
||||
text = collapse_dot_leaders(&text);
|
||||
}
|
||||
|
||||
// Fix hyphenation first (before other processing)
|
||||
if options.fix_hyphenation {
|
||||
@@ -355,6 +358,23 @@ fn format_urls(text: &str) -> String {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn fidelity_profile_preserves_dot_leaders() {
|
||||
let input = "Introduction............................1".to_string();
|
||||
let result = clean_markdown(input.clone(), &MarkdownOptions::default());
|
||||
assert_eq!(result, format!("{input}\n"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn compact_profile_collapses_dot_leaders() {
|
||||
let input = "Introduction............................1".to_string();
|
||||
let options = MarkdownOptions {
|
||||
profile: MarkdownProfile::Compact,
|
||||
..MarkdownOptions::default()
|
||||
};
|
||||
assert_eq!(clean_markdown(input, &options), "Introduction ... 1\n");
|
||||
}
|
||||
|
||||
// --- collapse_dot_leaders ---
|
||||
|
||||
#[test]
|
||||
|
||||
@@ -42,7 +42,12 @@ fn effective_heading_level(
|
||||
|
||||
// Fall back to font-size heuristic
|
||||
let font = line.items.first().map(|i| i.font_size).unwrap_or(base_size);
|
||||
detect_header_level(font, base_size, heading_tiers)
|
||||
detect_header_level(
|
||||
font,
|
||||
base_size,
|
||||
heading_tiers,
|
||||
crate::markdown::analysis::line_is_mostly_bold(line),
|
||||
)
|
||||
}
|
||||
|
||||
/// Merge consecutive heading lines at the same level into a single line.
|
||||
|
||||
@@ -914,7 +914,126 @@ fn looks_like_number(s: &str) -> bool {
|
||||
///
|
||||
/// Used by format.rs to render TOCs as flat lists instead of markdown tables.
|
||||
pub fn is_table_of_contents(cells: &[Vec<String>]) -> bool {
|
||||
is_dot_leader_toc(cells) || is_tabular_toc(cells)
|
||||
is_dot_leader_toc(cells) || is_tabular_toc(cells) || is_page_number_toc(cells)
|
||||
}
|
||||
|
||||
/// Parse a page-number-like token: a short arabic integer (≤4 digits) or a
|
||||
/// canonical roman numeral (front-matter pages: i, ii, …, xxxviii). Roman
|
||||
/// parsing is shared with the formatter via `super::canonical_roman_value` so
|
||||
/// the two stay in sync.
|
||||
fn page_number_value(token: &str) -> Option<u32> {
|
||||
let t = token.trim();
|
||||
if t.is_empty() {
|
||||
return None;
|
||||
}
|
||||
if t.chars().all(|c| c.is_ascii_digit()) && t.len() <= 4 {
|
||||
return t.parse().ok();
|
||||
}
|
||||
super::canonical_roman_value(t)
|
||||
}
|
||||
|
||||
/// Page-number-column TOC: title-based contents with no dot leaders and no
|
||||
/// section numbers (e.g. "About the Publisher vii", "Experiment #1 … 3").
|
||||
/// The signature is a text-title first column and a last column that is almost
|
||||
/// entirely page numbers whose values are *mostly non-decreasing* — the
|
||||
/// monotonic run is what separates a real TOC from an incidental 2-column
|
||||
/// numeric data table.
|
||||
pub(super) fn is_page_number_toc(cells: &[Vec<String>]) -> bool {
|
||||
let num_cols = cells.first().map(|r| r.len()).unwrap_or(0);
|
||||
// A page-number TOC is a narrow list (title + page, optionally a leader
|
||||
// column). Wider grids are data tables, not contents.
|
||||
if !(2..=3).contains(&num_cols) || cells.len() < 5 {
|
||||
return false;
|
||||
}
|
||||
let last = num_cols - 1;
|
||||
|
||||
// No header row: a TOC's first row is already an entry, so its last cell is
|
||||
// a page number. A data table's first row is a column header (non-numeric,
|
||||
// or an empty units cell like "Category | ") — the tell that separates
|
||||
// "Mineral | CEC" tables from real contents. Check the actual first row,
|
||||
// not the first non-empty one, so a blank header cell still rejects.
|
||||
let first_last = cells[0].get(last).map(|s| s.trim()).unwrap_or("");
|
||||
if page_number_value(first_last).is_none() {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Last column: page numbers on ≥70% of filled rows; collect their values.
|
||||
let mut filled = 0u32;
|
||||
let mut page_vals: Vec<u32> = Vec::new();
|
||||
for row in cells {
|
||||
let cell = row.get(last).map(|s| s.trim()).unwrap_or("");
|
||||
if cell.is_empty() {
|
||||
continue;
|
||||
}
|
||||
filled += 1;
|
||||
if let Some(v) = page_number_value(cell) {
|
||||
page_vals.push(v);
|
||||
}
|
||||
}
|
||||
if filled < 4 || (page_vals.len() as f32) < 0.7 * filled as f32 {
|
||||
return false;
|
||||
}
|
||||
|
||||
// First column: mostly text titles (has alphabetic content). This rejects
|
||||
// numeric-vs-numeric grids.
|
||||
let text_first = cells
|
||||
.iter()
|
||||
.filter(|row| {
|
||||
row.first()
|
||||
.is_some_and(|c| c.chars().any(|ch| ch.is_alphabetic()))
|
||||
})
|
||||
.count();
|
||||
if (text_first as f32) < 0.6 * cells.len() as f32 {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Page numbers mostly ascend (allow front-matter→body resets and noise).
|
||||
if page_vals.len() < 2 {
|
||||
return false;
|
||||
}
|
||||
let non_decreasing = page_vals.windows(2).filter(|w| w[1] >= w[0]).count();
|
||||
if (non_decreasing as f32) < 0.7 * (page_vals.len() - 1) as f32 {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Stronger TOC signal. Real page numbers SPAN the document — entries skip
|
||||
// (3, 6, 13, 24, …) so their range exceeds the entry count. A rank / ID /
|
||||
// ordinal column is instead a *perfectly dense* consecutive run (1,2,3,… or
|
||||
// 100,101,102,…). Accept anything with page gaps; for a dense run — which a
|
||||
// one-page-per-entry TOC can also produce — fall back to a title signal:
|
||||
// real contents entries are multi-word headings, rank labels are short.
|
||||
let min = *page_vals.iter().min().unwrap();
|
||||
let max = *page_vals.iter().max().unwrap();
|
||||
let span = max.saturating_sub(min);
|
||||
if span > page_vals.len() as u32 {
|
||||
return true;
|
||||
}
|
||||
let dense_consecutive = (span as usize) + 1 == page_vals.len() && {
|
||||
let mut sorted = page_vals.clone();
|
||||
sorted.sort_unstable();
|
||||
sorted.dedup();
|
||||
sorted.len() == page_vals.len()
|
||||
};
|
||||
if !dense_consecutive {
|
||||
// Narrow range but with a gap or repeat — still contents-like.
|
||||
return true;
|
||||
}
|
||||
// Dense counter: only a TOC if the titles read like headings, not the
|
||||
// short single-word labels typical of rank/leaderboard/ID tables.
|
||||
let (total_words, titled_rows) = cells
|
||||
.iter()
|
||||
.filter_map(|row| row.first())
|
||||
.filter(|c| c.chars().any(|ch| ch.is_alphabetic()))
|
||||
.fold((0usize, 0usize), |(w, n), c| {
|
||||
(
|
||||
w + c
|
||||
.split_whitespace()
|
||||
.filter(|t| t.chars().any(|ch| ch.is_alphabetic()))
|
||||
.count(),
|
||||
n + 1,
|
||||
)
|
||||
});
|
||||
titled_rows > 0 && (total_words as f32) / titled_rows as f32 >= 1.8
|
||||
}
|
||||
|
||||
/// Dot-leader TOC: any "Chapter 1 ........ 42" style with explicit leader
|
||||
@@ -1886,4 +2005,166 @@ mod tests {
|
||||
assert!(!starts_with_section_number(""));
|
||||
assert!(!starts_with_section_number("Hello world"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_value_rejects_roman_lookalike_words() {
|
||||
// Ordinary words made only of {i,v,x,l,c} are not page numbers.
|
||||
assert!(page_number_value("civil").is_none());
|
||||
assert!(page_number_value("mix").is_none());
|
||||
assert!(page_number_value("ill").is_none());
|
||||
assert!(page_number_value("lil").is_none());
|
||||
// Canonical roman numerals still parse.
|
||||
assert_eq!(page_number_value("vii"), Some(7));
|
||||
assert_eq!(page_number_value("ix"), Some(9));
|
||||
assert_eq!(page_number_value("xii"), Some(12));
|
||||
assert_eq!(page_number_value("42"), Some(42));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_matches_consecutive_pages_with_titles() {
|
||||
// A short chapter-per-page contents: pages are a dense 1..n run, but
|
||||
// the multi-word titles mark it as a real TOC (recovered by the title
|
||||
// signal rather than rejected for lacking page gaps).
|
||||
let cells: Vec<Vec<String>> = vec![
|
||||
vec!["Introduction to the Study".into(), "1".into()],
|
||||
vec!["Materials and Methods".into(), "2".into()],
|
||||
vec!["Results and Discussion".into(), "3".into()],
|
||||
vec!["Summary of Findings".into(), "4".into()],
|
||||
vec!["References and Notes".into(), "5".into()],
|
||||
];
|
||||
assert!(is_page_number_toc(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_rejects_dense_ordinal_column() {
|
||||
// Headerless title | rank table: values are a consecutive 1..n
|
||||
// sequence (monotonic, no header, text first column) but their range
|
||||
// ~= the row count, so it is data, not a table of contents.
|
||||
let cells: Vec<Vec<String>> = vec![
|
||||
vec!["Alice".into(), "1".into()],
|
||||
vec!["Bob".into(), "2".into()],
|
||||
vec!["Carol".into(), "3".into()],
|
||||
vec!["Dave".into(), "4".into()],
|
||||
vec!["Erin".into(), "5".into()],
|
||||
vec!["Frank".into(), "6".into()],
|
||||
];
|
||||
assert!(!is_page_number_toc(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_rejects_blank_header_cell() {
|
||||
// First row is a header whose last cell is blank ("Category | ");
|
||||
// must not be flattened even though later rows look TOC-like.
|
||||
let cells = vec![
|
||||
vec!["Category".into(), "".into()],
|
||||
vec!["Alpha".into(), "3".into()],
|
||||
vec!["Beta".into(), "9".into()],
|
||||
vec!["Gamma".into(), "14".into()],
|
||||
vec!["Delta".into(), "20".into()],
|
||||
];
|
||||
assert!(!is_page_number_toc(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_matches_title_based_contents() {
|
||||
// Title-left, page-number-right, no dot leaders, no section numbers.
|
||||
let cells = vec![
|
||||
vec!["About the Publisher".into(), "vii".into()],
|
||||
vec!["About This Project".into(), "ix".into()],
|
||||
vec!["Acknowledgments".into(), "xi".into()],
|
||||
vec!["Experiment #1: Hydrostatic Pressure".into(), "3".into()],
|
||||
vec!["Experiment #2: Bernoulli's Theorem".into(), "13".into()],
|
||||
vec![
|
||||
"Experiment #3: Energy Loss in Pipe Fittings".into(),
|
||||
"24".into(),
|
||||
],
|
||||
];
|
||||
assert!(is_page_number_toc(&cells));
|
||||
assert!(is_table_of_contents(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_rejects_numeric_data_table() {
|
||||
// Real 2-col data table: numeric first column, non-monotonic values.
|
||||
let cells = vec![
|
||||
vec!["101".into(), "45".into()],
|
||||
vec!["102".into(), "12".into()],
|
||||
vec!["103".into(), "88".into()],
|
||||
vec!["104".into(), "7".into()],
|
||||
vec!["105".into(), "63".into()],
|
||||
];
|
||||
assert!(!is_page_number_toc(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_rejects_non_monotonic_pages() {
|
||||
// Text labels but the "page" column jumps around — a small data table,
|
||||
// not a contents listing. 5 rows so the row-count guard passes and the
|
||||
// monotonicity check is what does the rejecting.
|
||||
let cells: Vec<Vec<String>> = vec![
|
||||
vec!["Apples".into(), "42".into()],
|
||||
vec!["Oranges".into(), "7".into()],
|
||||
vec!["Pears".into(), "91".into()],
|
||||
vec!["Plums".into(), "3".into()],
|
||||
vec!["Grapes".into(), "60".into()],
|
||||
];
|
||||
// Sanity: this input clears the row-count and header guards, so a
|
||||
// failure here is genuinely the monotonicity check.
|
||||
assert!(cells.len() >= 5 && page_number_value(cells[0][1].trim()).is_some());
|
||||
assert!(!is_page_number_toc(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_rejects_header_row_data_table() {
|
||||
// Real 2-col data table with a header row ("Mineral | CEC") and
|
||||
// ascending values that mimic page numbers — the header tells us it
|
||||
// is data, not contents.
|
||||
let cells = vec![
|
||||
vec![
|
||||
"Mineral or colloid type".into(),
|
||||
"CEC of pure colloid".into(),
|
||||
],
|
||||
vec!["kaolinite".into(), "10".into()],
|
||||
vec!["illite".into(), "30".into()],
|
||||
vec!["montmorillonite".into(), "100".into()],
|
||||
vec!["vermiculite".into(), "150".into()],
|
||||
];
|
||||
assert!(!is_page_number_toc(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_rejects_wide_data_grid() {
|
||||
// A 4-column regional data table must not be read as a TOC even with a
|
||||
// text first column and integer last column.
|
||||
let cells = vec![
|
||||
vec![
|
||||
"REGIONS".into(),
|
||||
"2007".into(),
|
||||
"2010".into(),
|
||||
"2016".into(),
|
||||
],
|
||||
vec![
|
||||
"National Capital Region".into(),
|
||||
"9".into(),
|
||||
"8".into(),
|
||||
"5".into(),
|
||||
],
|
||||
vec!["Cordillera".into(), "1".into(), "2".into(), "1".into()],
|
||||
vec!["Ilocos Region".into(), "1".into(), "5".into(), "4".into()],
|
||||
vec!["Cagayan Valley".into(), "1".into(), "3".into(), "5".into()],
|
||||
];
|
||||
assert!(!is_page_number_toc(&cells));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn page_number_toc_needs_page_number_last_column() {
|
||||
// Last column is prose, not page numbers.
|
||||
let cells = vec![
|
||||
vec!["Section A".into(), "see appendix".into()],
|
||||
vec!["Section B".into(), "see notes".into()],
|
||||
vec!["Section C".into(), "later".into()],
|
||||
vec!["Section D".into(), "TBD".into()],
|
||||
];
|
||||
assert!(!is_page_number_toc(&cells));
|
||||
}
|
||||
}
|
||||
|
||||
+2111
-21
File diff suppressed because it is too large
Load Diff
+762
-8
@@ -8,6 +8,9 @@ use crate::types::{PdfRect, TextItem};
|
||||
|
||||
use super::Table;
|
||||
|
||||
const DOMINANT_PAGE_BACKGROUND_MIN_REPETITIONS: usize = 8;
|
||||
const COMPETING_TABLE_MIN_ROWS: usize = 8;
|
||||
|
||||
/// Disjoint-set (union-find) with component sizes for clustering indices.
|
||||
struct UnionFind {
|
||||
parent: Vec<usize>,
|
||||
@@ -226,6 +229,77 @@ pub struct RectHintRegion {
|
||||
/// Also returns hint regions: bounding boxes of cell-sized rects from clusters
|
||||
/// that failed full grid validation. These can be used to scope heuristic
|
||||
/// detection and prevent unrelated items from being merged into tables.
|
||||
/// Bounding boxes of chart-bar clusters on the page. Text inside these
|
||||
/// regions (axis labels, data values, legends) belongs to a figure and must
|
||||
/// not be gridded into a table by any detection strategy.
|
||||
pub fn detect_chart_regions(
|
||||
items: &[TextItem],
|
||||
rects: &[PdfRect],
|
||||
page: u32,
|
||||
) -> Vec<(f32, f32, f32, f32)> {
|
||||
// Match detect_tables_from_rects: image placeholders are not text and
|
||||
// would defeat the bar-content check.
|
||||
let items_owned: Vec<TextItem> = items
|
||||
.iter()
|
||||
.filter(|i| crate::extractor::is_text_layout_item(i))
|
||||
.cloned()
|
||||
.collect();
|
||||
let items = items_owned.as_slice();
|
||||
let page_rects: Vec<(f32, f32, f32, f32)> = rects
|
||||
.iter()
|
||||
.filter(|r| r.page == page)
|
||||
.map(|r| {
|
||||
let (x, w) = if r.width < 0.0 {
|
||||
(r.x + r.width, -r.width)
|
||||
} else {
|
||||
(r.x, r.width)
|
||||
};
|
||||
let (y, h) = if r.height < 0.0 {
|
||||
(r.y + r.height, -r.height)
|
||||
} else {
|
||||
(r.y, r.height)
|
||||
};
|
||||
(x, y, w, h)
|
||||
})
|
||||
// Origin-anchored page backgrounds/clipping paths are never chart
|
||||
// geometry, and letting one bridge into a bar cluster would inflate
|
||||
// the region to the whole page.
|
||||
.filter(|&(x, y, w, h)| w >= 5.0 && h >= 5.0 && !(x < 5.0 && y < 5.0))
|
||||
.collect();
|
||||
if page_rects.len() < 6 {
|
||||
return Vec::new();
|
||||
}
|
||||
let mut regions = Vec::new();
|
||||
for cluster in &cluster_rects(&page_rects, 3.0, 6) {
|
||||
let group: Vec<(f32, f32, f32, f32)> = cluster.iter().map(|&i| page_rects[i]).collect();
|
||||
if is_chart_bar_cluster(items, &group, page) {
|
||||
let bbox = group.iter().fold(
|
||||
(
|
||||
f32::INFINITY,
|
||||
f32::INFINITY,
|
||||
f32::NEG_INFINITY,
|
||||
f32::NEG_INFINITY,
|
||||
),
|
||||
|(x0, y0, x1, y1), &(x, y, w, h)| {
|
||||
(x0.min(x), y0.min(y), x1.max(x + w), y1.max(y + h))
|
||||
},
|
||||
);
|
||||
regions.push(bbox);
|
||||
}
|
||||
}
|
||||
regions
|
||||
}
|
||||
|
||||
fn detect_direct_rect_table(
|
||||
items: &[TextItem],
|
||||
rects: &[(f32, f32, f32, f32)],
|
||||
page: u32,
|
||||
) -> Option<Table> {
|
||||
detect_table_from_rect_group(items, rects, page)
|
||||
.or_else(|| detect_row_stripe_table(items, rects, page))
|
||||
.or_else(|| detect_stacked_box_table(items, rects, page))
|
||||
}
|
||||
|
||||
pub fn detect_tables_from_rects(
|
||||
items: &[TextItem],
|
||||
rects: &[PdfRect],
|
||||
@@ -379,12 +453,77 @@ pub fn detect_tables_from_rects(
|
||||
.collect();
|
||||
|
||||
debug!("page {}: {} clusters with >= 6 rects", page, clusters.len());
|
||||
for cluster_indices in &clusters {
|
||||
let mut merge_excluded_cluster_ids: Vec<usize> = Vec::new();
|
||||
for (cluster_id, cluster_indices) in clusters.iter().enumerate() {
|
||||
let group_rects: Vec<(f32, f32, f32, f32)> =
|
||||
cluster_indices.iter().map(|&i| page_rects[i]).collect();
|
||||
if let Some(table) = detect_table_from_rect_group(items, &group_rects, page) {
|
||||
tables.push(table);
|
||||
} else if let Some(table) = detect_row_stripe_table(items, &group_rects, page) {
|
||||
// Chart bars are neither table cells nor a hint region — gridding
|
||||
// a chart's axis labels scrambles the page. Skip the cluster
|
||||
// entirely so it can't reach any detector, the merged fallback,
|
||||
// or the hint fallback.
|
||||
if is_chart_bar_cluster(items, &group_rects, page) {
|
||||
// Repeated page fills can dominate the geometry and make a
|
||||
// real shaded-cell table look like a chart. Remove those fills,
|
||||
// re-cluster the remaining geometry, and evaluate valid table
|
||||
// candidates as a competing hypothesis before the chart
|
||||
// rejection wins.
|
||||
let normalized = without_dominant_page_backgrounds(&group_rects);
|
||||
let normalized_table = (normalized.len() < group_rects.len())
|
||||
.then(|| {
|
||||
cluster_rects(&normalized, 3.0, 6)
|
||||
.iter()
|
||||
.filter_map(|indices| {
|
||||
let candidate: Vec<(f32, f32, f32, f32)> =
|
||||
indices.iter().map(|&i| normalized[i]).collect();
|
||||
if is_chart_bar_cluster(items, &candidate, page) {
|
||||
None
|
||||
} else {
|
||||
detect_table_from_rect_group(items, &candidate, page)
|
||||
.or_else(|| {
|
||||
detect_row_stripe_table_from_cell_rects(
|
||||
items, &candidate, page,
|
||||
)
|
||||
})
|
||||
// Small chart panels can still form
|
||||
// plausible grids from their labels.
|
||||
// Require sustained row evidence; the
|
||||
// motivating table has 17 rows.
|
||||
.filter(|table| {
|
||||
table.rows.len() >= COMPETING_TABLE_MIN_ROWS
|
||||
})
|
||||
}
|
||||
})
|
||||
.max_by_key(|table| table.rows.len() * table.columns.len())
|
||||
})
|
||||
.flatten();
|
||||
if let Some(table) = normalized_table {
|
||||
debug!(
|
||||
"page {}: chart-like cluster normalized from {} to {} rects; accepted {}x{} table hypothesis",
|
||||
page,
|
||||
group_rects.len(),
|
||||
normalized.len(),
|
||||
table.rows.len(),
|
||||
table.columns.len()
|
||||
);
|
||||
// The accepted hypothesis is based on normalized
|
||||
// geometry. Keep the original chart-like cluster out of
|
||||
// the merged fallback: reintroducing its repeated page
|
||||
// fills can manufacture a wider candidate that replaces
|
||||
// this valid narrow table below.
|
||||
merge_excluded_cluster_ids.push(cluster_id);
|
||||
tables.push(table);
|
||||
continue;
|
||||
} else {
|
||||
debug!(
|
||||
"page {}: skipping chart-bar cluster ({} rects)",
|
||||
page,
|
||||
group_rects.len()
|
||||
);
|
||||
merge_excluded_cluster_ids.push(cluster_id);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
if let Some(table) = detect_direct_rect_table(items, &group_rects, page) {
|
||||
tables.push(table);
|
||||
} else if let Some((left, right)) = split_wide_cluster(&group_rects, 15.0, 6) {
|
||||
// Cluster was too wide — retry each half independently
|
||||
@@ -419,12 +558,19 @@ pub fn detect_tables_from_rects(
|
||||
// text-based column detection.
|
||||
let only_narrow = !tables.is_empty() && tables.iter().all(|t| t.columns.len() <= 3);
|
||||
if tables.is_empty() || only_narrow {
|
||||
let total_clustered: usize = clusters.iter().map(|c| c.len()).sum();
|
||||
if clusters.len() >= 3 && total_clustered >= 50 {
|
||||
// Chart clusters stay out of the merge as well.
|
||||
let table_clusters: Vec<&Vec<usize>> = clusters
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(id, _)| !merge_excluded_cluster_ids.contains(id))
|
||||
.map(|(_, c)| c)
|
||||
.collect();
|
||||
let total_clustered: usize = table_clusters.iter().map(|c| c.len()).sum();
|
||||
if table_clusters.len() >= 3 && total_clustered >= 50 {
|
||||
debug!(
|
||||
"page {}: trying merged-cluster fallback ({} clusters, {} rects{})",
|
||||
page,
|
||||
clusters.len(),
|
||||
table_clusters.len(),
|
||||
total_clustered,
|
||||
if only_narrow {
|
||||
", replacing narrow tables"
|
||||
@@ -432,7 +578,7 @@ pub fn detect_tables_from_rects(
|
||||
""
|
||||
}
|
||||
);
|
||||
let all_cluster_rects: Vec<(f32, f32, f32, f32)> = clusters
|
||||
let all_cluster_rects: Vec<(f32, f32, f32, f32)> = table_clusters
|
||||
.iter()
|
||||
.flat_map(|idxs| idxs.iter().map(|&i| page_rects[i]))
|
||||
.collect();
|
||||
@@ -491,6 +637,14 @@ pub fn detect_tables_from_rects(
|
||||
}
|
||||
}
|
||||
|
||||
// NOTE: 3-5 box stacks never reach detect_stacked_box_table — the main
|
||||
// loop requires >=6-rect clusters (and a >=6-rect page). This is a
|
||||
// deliberate precision gate: routing smaller clusters through the
|
||||
// detector was tried and regressed four pdf-evals documents (striped
|
||||
// bullet lists, wrapped regulation text, stats-table columns) while
|
||||
// improving nothing — with so few boxes the anti-prose guards have too
|
||||
// little signal to discriminate. See stacked_box_three_rows_below_
|
||||
// cluster_minimum for the pinned behavior.
|
||||
if tables.is_empty() {
|
||||
// When no tables detected but clusters exist, generate XY hint regions
|
||||
// from cluster bounding boxes to scope heuristic table detection.
|
||||
@@ -631,6 +785,240 @@ pub fn detect_tables_from_rects(
|
||||
/// overlap or are close (gap < 50pt). This handles calendar-style layouts where a
|
||||
/// month zone's decorative rects split into 2-3 adjacent clusters with small X gaps.
|
||||
/// Runs iteratively until no more merges occur.
|
||||
/// Detect a single-column table drawn as a vertical stack of boxes, each
|
||||
/// holding one short line of text (framework/step lists on slide-style
|
||||
/// pages). The normal grid path rejects these — one column means only two
|
||||
/// x-edges — so the rows would otherwise flow into surrounding prose as a
|
||||
/// run-on paragraph.
|
||||
fn detect_stacked_box_table(
|
||||
items: &[TextItem],
|
||||
group_rects: &[(f32, f32, f32, f32)],
|
||||
page: u32,
|
||||
) -> Option<Table> {
|
||||
// Candidate row boxes: single-text-line height, substantial width.
|
||||
let cands: Vec<(f32, f32, f32, f32)> = group_rects
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|&(_, _, w, h)| w >= 100.0 && (8.0..=80.0).contains(&h))
|
||||
.collect();
|
||||
// The row boxes form the largest family of same-width, x-aligned rects
|
||||
// (backgrounds and decor have their own geometry and stay out).
|
||||
let mut boxes: Vec<(f32, f32, f32, f32)> = Vec::new();
|
||||
for &anchor in &cands {
|
||||
let family: Vec<(f32, f32, f32, f32)> = cands
|
||||
.iter()
|
||||
.copied()
|
||||
.filter(|&(x, _, w, h)| {
|
||||
(x - anchor.0).abs() <= 12.0
|
||||
&& (w - anchor.2).abs() <= anchor.2 * 0.15
|
||||
&& (h - anchor.3).abs() <= anchor.3 * 0.3
|
||||
})
|
||||
.collect();
|
||||
if family.len() > boxes.len() {
|
||||
boxes = family;
|
||||
}
|
||||
}
|
||||
if boxes.len() < 3 {
|
||||
return None;
|
||||
}
|
||||
// Boxes flanked at the same y-level — by other rects or by text outside
|
||||
// the family's x-range — are one column of a wider structure. Leave
|
||||
// those to the grid/cell-rect paths instead of collapsing to one column.
|
||||
let flanked = boxes
|
||||
.iter()
|
||||
.filter(|&&(bx, by, bw, bh)| {
|
||||
let rect_sibling = group_rects.iter().any(|&(ox, oy, ow, oh)| {
|
||||
let y_overlap = (by + bh).min(oy + oh) - by.max(oy);
|
||||
oh >= 8.0
|
||||
&& y_overlap > bh * 0.5
|
||||
&& (ox + ow <= bx + 2.0 || ox >= bx + bw - 2.0)
|
||||
&& ow >= 30.0
|
||||
});
|
||||
let text_sibling = items.iter().any(|it| {
|
||||
let cx = it.x + it.width / 2.0;
|
||||
it.page == page
|
||||
&& it.y >= by - 2.0
|
||||
&& it.y <= by + bh + 2.0
|
||||
&& (cx < bx - 5.0 || cx > bx + bw + 5.0)
|
||||
&& it.width >= 10.0
|
||||
});
|
||||
rect_sibling || text_sibling
|
||||
})
|
||||
.count();
|
||||
if flanked * 3 >= boxes.len() {
|
||||
debug!(
|
||||
" stacked-box rejected: {}/{} boxes flanked by rects or text",
|
||||
flanked,
|
||||
boxes.len()
|
||||
);
|
||||
return None;
|
||||
}
|
||||
boxes.sort_by(|a, b| b.1.total_cmp(&a.1)); // top to bottom (descending y)
|
||||
|
||||
// Merge duplicates (border + fill pairs draw the same box twice), then
|
||||
// require a clean vertical stack: no overlaps beyond a small tolerance.
|
||||
boxes.dedup_by(|a, b| (a.1 - b.1).abs() <= 3.0 && (a.3 - b.3).abs() <= 6.0);
|
||||
if boxes.len() < 3 {
|
||||
return None;
|
||||
}
|
||||
for w in boxes.windows(2) {
|
||||
let (upper, lower) = (w[0], w[1]);
|
||||
let upper_bottom = upper.1;
|
||||
let lower_top = lower.1 + lower.3;
|
||||
if lower_top > upper_bottom + 4.0 {
|
||||
return None; // vertical overlap — not a stack
|
||||
}
|
||||
if upper_bottom - lower_top > upper.3.max(lower.3) {
|
||||
return None; // gap larger than a row — unrelated boxes
|
||||
}
|
||||
}
|
||||
|
||||
// Assign items to boxes; every box needs text and cells must stay short
|
||||
// (prose paragraphs inside stacked frames are page decor, not a table).
|
||||
let mut cells: Vec<Vec<String>> = Vec::with_capacity(boxes.len());
|
||||
let mut item_indices: Vec<usize> = Vec::new();
|
||||
let mut multi_run_boxes = 0usize;
|
||||
for &(bx, by, bw, bh) in &boxes {
|
||||
let mut in_box: Vec<(usize, &TextItem)> = items
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(_, it)| {
|
||||
it.page == page
|
||||
&& it.y >= by - 2.0
|
||||
&& it.y <= by + bh + 2.0
|
||||
&& it.x + it.width / 2.0 >= bx
|
||||
&& it.x + it.width / 2.0 <= bx + bw
|
||||
})
|
||||
.collect();
|
||||
if in_box.is_empty() {
|
||||
return None;
|
||||
}
|
||||
in_box.sort_by(|a, b| {
|
||||
b.1.y
|
||||
.partial_cmp(&a.1.y)
|
||||
.unwrap_or(std::cmp::Ordering::Equal)
|
||||
.then_with(|| {
|
||||
a.1.x
|
||||
.partial_cmp(&b.1.x)
|
||||
.unwrap_or(std::cmp::Ordering::Equal)
|
||||
})
|
||||
});
|
||||
// Count horizontally separated text runs inside the box. A single
|
||||
// list row flows as one run; two-plus runs across most boxes means
|
||||
// multi-column content (striped prose or a real grid) that must not
|
||||
// collapse into a one-column table. Same-baseline only: boxed
|
||||
// display/diagram rows legitimately scatter segments at mixed
|
||||
// baselines, and those must stay one row.
|
||||
let mut runs = 1usize;
|
||||
for pair in in_box.windows(2) {
|
||||
let (prev, item) = (pair[0].1, pair[1].1);
|
||||
if (prev.y - item.y).abs() <= 2.0 && item.x - (prev.x + prev.width) > 15.0 {
|
||||
runs += 1;
|
||||
}
|
||||
}
|
||||
if runs >= 2 {
|
||||
multi_run_boxes += 1;
|
||||
}
|
||||
let text = in_box
|
||||
.iter()
|
||||
.map(|(_, it)| it.text.trim())
|
||||
.filter(|t| !t.is_empty())
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
if text.is_empty() || text.chars().count() > 120 {
|
||||
return None;
|
||||
}
|
||||
item_indices.extend(in_box.iter().map(|(i, _)| *i));
|
||||
cells.push(vec![text]);
|
||||
}
|
||||
if multi_run_boxes * 2 >= boxes.len() {
|
||||
debug!(
|
||||
" stacked-box rejected: {}/{} boxes hold multiple text runs",
|
||||
multi_run_boxes,
|
||||
boxes.len()
|
||||
);
|
||||
return None;
|
||||
}
|
||||
|
||||
// Reject prose behind per-line stripe rects: sentence fragments flowing
|
||||
// across rows read as long, function-word-dense cells, while genuine
|
||||
// list-table rows are short labels/titles.
|
||||
const PROSE_WORDS: &[&str] = &[
|
||||
"a", "an", "the", "of", "to", "is", "was", "are", "were", "be", "been", "in", "on", "at",
|
||||
"with", "for", "by", "as", "and", "or", "but", "this", "that", "these", "those", "from",
|
||||
"into", "has", "have", "had", "not", "it", "its", "their", "such", "shall", "which",
|
||||
];
|
||||
let total_chars: usize = cells.iter().map(|r| r[0].chars().count()).sum();
|
||||
let mean_chars = total_chars / cells.len().max(1);
|
||||
let prose_cells = cells
|
||||
.iter()
|
||||
.filter(|r| {
|
||||
r[0].to_ascii_lowercase()
|
||||
.split(|c: char| !c.is_ascii_alphabetic() && c != '\'')
|
||||
.any(|w| PROSE_WORDS.contains(&w))
|
||||
})
|
||||
.count();
|
||||
if mean_chars > 60 && prose_cells * 5 >= cells.len() * 2 {
|
||||
debug!(
|
||||
" stacked-box rejected: prose rows (mean {} chars, prose words {}/{})",
|
||||
mean_chars,
|
||||
prose_cells,
|
||||
cells.len()
|
||||
);
|
||||
return None;
|
||||
}
|
||||
// Sentences wrapping across stripe rects: a row ending with a comma, or
|
||||
// a row without terminal punctuation followed by a row starting
|
||||
// lowercase, is mid-sentence flow — not list rows. Genuine label/title
|
||||
// rows produce none of these, so even a small share is disqualifying.
|
||||
let continuations = cells
|
||||
.windows(2)
|
||||
.filter(|pair| {
|
||||
let prev = pair[0][0].trim_end();
|
||||
let next = pair[1][0].trim_start();
|
||||
let prev_open = !prev.ends_with(['.', ':', ';', '!', '?', ')', '"', '%']);
|
||||
let next_lower = next.chars().next().is_some_and(|c| c.is_lowercase());
|
||||
prev.ends_with(',') || (prev_open && next_lower)
|
||||
})
|
||||
.count();
|
||||
if cells.len() >= 2 && (continuations >= 2 || continuations * 4 >= cells.len() - 1) {
|
||||
debug!(
|
||||
" stacked-box rejected: {}/{} row pairs continue a sentence",
|
||||
continuations,
|
||||
cells.len() - 1
|
||||
);
|
||||
return None;
|
||||
}
|
||||
// Numbered/lettered list items behind decorative stripes stay lists:
|
||||
// "1) content..." / "(ii) content..." / "a. content...".
|
||||
let list_marker = |t: &str| {
|
||||
let t = t.trim_start().strip_prefix('(').unwrap_or(t.trim_start());
|
||||
let marker_len = t.chars().take_while(|c| c.is_ascii_alphanumeric()).count();
|
||||
(1..=3).contains(&marker_len)
|
||||
&& t.chars()
|
||||
.nth(marker_len)
|
||||
.is_some_and(|c| c == ')' || c == '.')
|
||||
};
|
||||
let list_rows = cells.iter().filter(|r| list_marker(&r[0])).count();
|
||||
if list_rows * 2 >= cells.len() {
|
||||
debug!(
|
||||
" stacked-box rejected: {}/{} rows are numbered list items",
|
||||
list_rows,
|
||||
cells.len()
|
||||
);
|
||||
return None;
|
||||
}
|
||||
|
||||
debug!(
|
||||
"page {}: stacked-box table: {} single-column rows",
|
||||
page,
|
||||
cells.len()
|
||||
);
|
||||
let columns = vec![boxes[0].0 + boxes[0].2 / 2.0];
|
||||
let rows: Vec<f32> = boxes.iter().map(|b| b.1 + b.3 / 2.0).collect();
|
||||
Some(Table::new(columns, rows, cells, item_indices))
|
||||
}
|
||||
|
||||
fn merge_overlapping_hints(mut hints: Vec<RectHintRegion>) -> Vec<RectHintRegion> {
|
||||
if hints.len() <= 1 {
|
||||
return hints;
|
||||
@@ -1578,11 +1966,153 @@ fn row_stripe_is_sparse_prose_outline(cells: &[Vec<String>]) -> bool {
|
||||
long_dense_cells * 2 >= dense_count
|
||||
}
|
||||
|
||||
/// Remove repeated page-scale fills from a chart-like cluster so the actual
|
||||
/// cell/bar geometry can be evaluated independently. A small number of
|
||||
/// coincident origin frames may be meaningful table structure, so repetition
|
||||
/// only becomes normalization evidence when it dominates the cluster.
|
||||
fn without_dominant_page_backgrounds(rects: &[(f32, f32, f32, f32)]) -> Vec<(f32, f32, f32, f32)> {
|
||||
let x_max = rects
|
||||
.iter()
|
||||
.map(|&(x, _, width, _)| x + width)
|
||||
.fold(0.0_f32, f32::max);
|
||||
let y_max = rects
|
||||
.iter()
|
||||
.map(|&(_, y, _, height)| y + height)
|
||||
.fold(0.0_f32, f32::max);
|
||||
let is_page_scale = |&(x, y, width, height): &(f32, f32, f32, f32)| {
|
||||
x < 5.0 && y < 5.0 && width >= x_max * 0.9 && height >= y_max * 0.9
|
||||
};
|
||||
|
||||
if rects.iter().filter(|rect| is_page_scale(rect)).count()
|
||||
< DOMINANT_PAGE_BACKGROUND_MIN_REPETITIONS
|
||||
{
|
||||
return rects.to_vec();
|
||||
}
|
||||
|
||||
rects
|
||||
.iter()
|
||||
.filter(|rect| !is_page_scale(rect))
|
||||
.copied()
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Detect a table from cell-background rects that failed grid detection.
|
||||
///
|
||||
/// Uses rect Y-edges for row boundaries and text X-position clustering for
|
||||
/// columns. Handles tables with cell backgrounds that don't form a clean
|
||||
/// X-edge grid (variable column widths, decorative fills).
|
||||
/// Chart-bar signature: ≥3 rects sharing an aligned bottom edge (the axis),
|
||||
/// with similar widths (bars) but strongly varying heights (data-driven),
|
||||
/// holding at most a single numeric data label each. Bar charts drawn as
|
||||
/// filled rects otherwise read as cell rects and grid their axis labels
|
||||
/// into a phantom table. The mirrored check catches horizontal bar charts.
|
||||
fn is_chart_bar_cluster(
|
||||
items: &[TextItem],
|
||||
group_rects: &[(f32, f32, f32, f32)],
|
||||
page: u32,
|
||||
) -> bool {
|
||||
let numeric_or_empty = |(rx, ry, rw, rh): (f32, f32, f32, f32)| {
|
||||
let inside: Vec<&TextItem> = items
|
||||
.iter()
|
||||
.filter(|it| {
|
||||
let cx = it.x + it.width / 2.0;
|
||||
it.page == page && cx >= rx && cx <= rx + rw && it.y >= ry && it.y <= ry + rh
|
||||
})
|
||||
.collect();
|
||||
// Any number of numeric data labels is chart-like; a single run of
|
||||
// word text inside means a table cell.
|
||||
inside.iter().all(|it| {
|
||||
let t = it.text.trim();
|
||||
let data = t
|
||||
.chars()
|
||||
.filter(|c| c.is_ascii_digit() || ",.%-".contains(*c))
|
||||
.count();
|
||||
t.is_empty() || data * 2 >= t.chars().count()
|
||||
})
|
||||
};
|
||||
|
||||
// Bars: the dominant equal-width family, arranged in >=2 spaced columns
|
||||
// (inter-column gap >= half a bar width — table cell rects touch), with
|
||||
// data-driven height variation (checkbox/cell grids are uniform).
|
||||
// Mirrored predicate catches horizontal bar charts.
|
||||
let bar_family = |pos: fn(&(f32, f32, f32, f32)) -> f32,
|
||||
breadth: fn(&(f32, f32, f32, f32)) -> f32,
|
||||
length: fn(&(f32, f32, f32, f32)) -> f32,
|
||||
along: fn(&(f32, f32, f32, f32)) -> f32| {
|
||||
group_rects.iter().any(|anchor| {
|
||||
let bw = breadth(anchor);
|
||||
if bw <= 0.0 {
|
||||
return false;
|
||||
}
|
||||
let family: Vec<&(f32, f32, f32, f32)> = group_rects
|
||||
.iter()
|
||||
.filter(|r| {
|
||||
(breadth(r) - bw).abs() <= (bw * 0.1).max(2.0)
|
||||
&& length(r) > 0.0
|
||||
&& length(r) < bw * 20.0
|
||||
})
|
||||
.collect();
|
||||
if family.len() < 4 {
|
||||
return false;
|
||||
}
|
||||
// Distinct positions along the axis (bar columns).
|
||||
let mut positions: Vec<f32> = Vec::new();
|
||||
for r in &family {
|
||||
let p = pos(r);
|
||||
if !positions.iter().any(|&q| (q - p).abs() <= 2.0) {
|
||||
positions.push(p);
|
||||
}
|
||||
}
|
||||
if positions.len() < 2 {
|
||||
return false;
|
||||
}
|
||||
positions.sort_by(|a, b| a.total_cmp(b));
|
||||
let min_gap = positions
|
||||
.windows(2)
|
||||
.map(|w| w[1] - w[0] - bw)
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
if min_gap < bw * 0.5 {
|
||||
return false;
|
||||
}
|
||||
// Data-driven variation along the bar direction.
|
||||
let len_min = family
|
||||
.iter()
|
||||
.map(|r| length(r))
|
||||
.fold(f32::INFINITY, f32::min);
|
||||
let len_max = family
|
||||
.iter()
|
||||
.map(|r| length(r))
|
||||
.fold(f32::NEG_INFINITY, f32::max);
|
||||
if len_max < len_min * 1.3 {
|
||||
return false;
|
||||
}
|
||||
// Grid rows disguise as bars: a table's cell rects have same-y,
|
||||
// same-height partners in other columns (uniform row heights).
|
||||
// Chart segments start where the previous datum ended, so their
|
||||
// extents rarely pair up across positions.
|
||||
let matched = family
|
||||
.iter()
|
||||
.filter(|r| {
|
||||
family.iter().any(|s| {
|
||||
(pos(s) - pos(r)).abs() > 2.0
|
||||
&& (along(s) - along(r)).abs() <= 3.0
|
||||
&& (length(s) - length(r)).abs() <= 3.0
|
||||
})
|
||||
})
|
||||
.count();
|
||||
if matched * 5 >= family.len() * 3 {
|
||||
return false;
|
||||
}
|
||||
family.iter().filter(|r| numeric_or_empty(***r)).count() * 3 >= family.len() * 2
|
||||
})
|
||||
};
|
||||
|
||||
// vertical bars: position/breadth = x/width, length = height, along = y
|
||||
bar_family(|r| r.0, |r| r.2, |r| r.3, |r| r.1)
|
||||
// horizontal bars: position/breadth = y/height, length = width, along = x
|
||||
|| bar_family(|r| r.1, |r| r.3, |r| r.2, |r| r.0)
|
||||
}
|
||||
|
||||
fn detect_row_stripe_table_from_cell_rects(
|
||||
items: &[TextItem],
|
||||
group_rects: &[(f32, f32, f32, f32)],
|
||||
@@ -2458,6 +2988,187 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
// --- is_chart_bar_cluster / detect_chart_regions ---
|
||||
|
||||
/// Stacked bar chart: frame + 3 columns of equal-width segments with
|
||||
/// data-driven heights, holding numeric labels.
|
||||
fn chart_rects() -> Vec<PdfRect> {
|
||||
let mut rects = vec![PdfRect {
|
||||
x: 126.0,
|
||||
y: 548.0,
|
||||
width: 396.0,
|
||||
height: 216.0,
|
||||
page: 1,
|
||||
}];
|
||||
let bars = [
|
||||
(208.0, 618.0, 59.0),
|
||||
(208.0, 661.0, 39.0),
|
||||
(208.0, 696.0, 37.0),
|
||||
(313.0, 618.0, 67.0),
|
||||
(313.0, 670.0, 49.0),
|
||||
(313.0, 691.0, 42.0),
|
||||
(419.0, 618.0, 73.0),
|
||||
(419.0, 684.0, 37.0),
|
||||
(419.0, 708.0, 25.0),
|
||||
];
|
||||
for (x, y, h) in bars {
|
||||
rects.push(PdfRect {
|
||||
x,
|
||||
y,
|
||||
width: 46.0,
|
||||
height: h,
|
||||
page: 1,
|
||||
});
|
||||
}
|
||||
rects
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn chart_bars_produce_region_not_table() {
|
||||
let items: Vec<TextItem> = [
|
||||
("38", 228.0, 638.0),
|
||||
("30", 228.0, 676.0),
|
||||
("46", 333.0, 643.0),
|
||||
("17", 333.0, 679.0),
|
||||
("57", 438.0, 650.0),
|
||||
("20", 438.0, 694.0),
|
||||
]
|
||||
.iter()
|
||||
.map(|&(t, x, y)| make_item(t, x, y, 9.0))
|
||||
.collect();
|
||||
let rects = chart_rects();
|
||||
let regions = detect_chart_regions(&items, &rects, 1);
|
||||
assert_eq!(regions.len(), 1, "expected one chart region");
|
||||
let (tables, hints) = detect_tables_from_rects(&items, &rects, 1);
|
||||
assert!(tables.is_empty(), "chart bars must not become a table");
|
||||
assert!(hints.is_empty(), "chart bars must not become a hint region");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dominant_page_backgrounds_are_normalized_only_after_repetition() {
|
||||
let page_fill = (0.0, 0.0, 600.0, 800.0);
|
||||
let cell = (100.0, 500.0, 120.0, 20.0);
|
||||
|
||||
let mut dominant = vec![page_fill; DOMINANT_PAGE_BACKGROUND_MIN_REPETITIONS];
|
||||
dominant.push(cell);
|
||||
assert_eq!(without_dominant_page_backgrounds(&dominant), vec![cell]);
|
||||
|
||||
let mut incidental = vec![page_fill; DOMINANT_PAGE_BACKGROUND_MIN_REPETITIONS - 1];
|
||||
incidental.push(cell);
|
||||
assert_eq!(without_dominant_page_backgrounds(&incidental), incidental);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn uniform_cell_grid_is_not_a_chart() {
|
||||
// Touching, uniform-height cell rects (a real table) must not match:
|
||||
// no inter-column gap and no bar-length variation.
|
||||
let mut rects = Vec::new();
|
||||
for row in 0..4 {
|
||||
for col in 0..3 {
|
||||
rects.push(PdfRect {
|
||||
x: 100.0 + col as f32 * 80.0,
|
||||
y: 600.0 - row as f32 * 20.0,
|
||||
width: 80.0,
|
||||
height: 20.0,
|
||||
page: 1,
|
||||
});
|
||||
}
|
||||
}
|
||||
let items: Vec<TextItem> = (0..4)
|
||||
.flat_map(|r| {
|
||||
(0..3).map(move |c| (100.0 + c as f32 * 80.0 + 10.0, 605.0 - r as f32 * 20.0))
|
||||
})
|
||||
.map(|(x, y)| make_item("42", x, y, 9.0))
|
||||
.collect();
|
||||
assert!(detect_chart_regions(&items, &rects, 1).is_empty());
|
||||
}
|
||||
|
||||
// --- detect_stacked_box_table ---
|
||||
|
||||
/// N stacked boxes at x=100, w=300, h=22, top-to-bottom from y=600.
|
||||
fn stacked_boxes(n: usize) -> Vec<(f32, f32, f32, f32)> {
|
||||
(0..n)
|
||||
.map(|i| (100.0, 600.0 - i as f32 * 22.0, 300.0, 22.0))
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stacked_box_list_becomes_single_column_table() {
|
||||
let rects = stacked_boxes(5);
|
||||
let items: Vec<TextItem> = (0..5)
|
||||
.map(|i| make_item("#1: Recycling Basics", 120.0, 605.0 - i as f32 * 22.0, 10.0))
|
||||
.collect();
|
||||
let table = detect_stacked_box_table(&items, &rects, 1).expect("stacked-box table");
|
||||
assert_eq!(table.cells.len(), 5);
|
||||
assert_eq!(table.cells[0].len(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stacked_box_rejects_wrapped_sentences() {
|
||||
// Line stripes behind flowing prose: rows continue mid-sentence.
|
||||
let rects = stacked_boxes(4);
|
||||
let texts = [
|
||||
"the provisions of this section apply to",
|
||||
"companies subject to tax under those",
|
||||
"sections, except that the copy of the",
|
||||
"annual statement must be retained.",
|
||||
];
|
||||
let items: Vec<TextItem> = texts
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, t)| make_item(t, 120.0, 605.0 - i as f32 * 22.0, 10.0))
|
||||
.collect();
|
||||
assert!(detect_stacked_box_table(&items, &rects, 1).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stacked_box_rejects_flanking_text() {
|
||||
// A ruled label column with plain-text data columns beside it is one
|
||||
// column of a wider table, not a single-column list.
|
||||
let rects = stacked_boxes(4);
|
||||
let mut items = Vec::new();
|
||||
for i in 0..4 {
|
||||
let y = 605.0 - i as f32 * 22.0;
|
||||
items.push(make_item("Section 1.382", 120.0, y, 10.0));
|
||||
items.push(make_item("removed text", 450.0, y, 10.0)); // beside the box
|
||||
}
|
||||
assert!(detect_stacked_box_table(&items, &rects, 1).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stacked_box_rejects_two_column_content() {
|
||||
// Boxes holding two separated runs are striped multi-column content.
|
||||
let rects = stacked_boxes(4);
|
||||
let mut items = Vec::new();
|
||||
for i in 0..4 {
|
||||
let y = 605.0 - i as f32 * 22.0;
|
||||
let mut left = make_item("left words", 110.0, y, 10.0);
|
||||
left.width = 60.0;
|
||||
let mut right = make_item("right words", 250.0, y, 10.0);
|
||||
right.width = 60.0;
|
||||
items.push(left);
|
||||
items.push(right);
|
||||
}
|
||||
assert!(detect_stacked_box_table(&items, &rects, 1).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stacked_box_rejects_mixed_height_stripes() {
|
||||
// Mixed 13/27pt stripes (redline markup) — height uniformity splits
|
||||
// the family and the gap check rejects the remainder.
|
||||
let mut rects = Vec::new();
|
||||
let mut y = 600.0;
|
||||
for i in 0..8 {
|
||||
let h = if i % 3 == 0 { 27.0 } else { 13.5 };
|
||||
y -= h;
|
||||
rects.push((100.0, y, 300.0, h));
|
||||
}
|
||||
let items: Vec<TextItem> = (0..8)
|
||||
.map(|i| make_item("PART 602 OMB CONTROL", 120.0, 590.0 - i as f32 * 18.0, 10.0))
|
||||
.collect();
|
||||
assert!(detect_stacked_box_table(&items, &rects, 1).is_none());
|
||||
}
|
||||
|
||||
// --- has_dominant_prose_cell ---
|
||||
|
||||
fn cells_of(rows: &[&[&str]]) -> Vec<Vec<String>> {
|
||||
@@ -3480,6 +4191,49 @@ mod tests {
|
||||
assert!((merged[0].x_right - 340.0).abs() < 0.01);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stacked_box_three_rows_below_cluster_minimum() {
|
||||
// Pins a deliberate precision gate: a 3-box stack stays below the
|
||||
// main loop's 6-rect cluster minimum and is NOT detected end-to-end.
|
||||
// Routing smaller clusters through detect_stacked_box_table was
|
||||
// tried and regressed four pdf-evals documents (striped bullet
|
||||
// lists, wrapped regulation text, stats-table columns) with no
|
||||
// corpus gains — too few boxes for the anti-prose guards to work.
|
||||
// If this ever becomes worth revisiting, the guards need stronger
|
||||
// signals first; flipping this assertion is the entry point.
|
||||
let mut rects: Vec<PdfRect> = (0..3)
|
||||
.map(|i| PdfRect {
|
||||
x: 100.0,
|
||||
y: 600.0 - i as f32 * 22.0,
|
||||
width: 300.0,
|
||||
height: 22.0,
|
||||
page: 1,
|
||||
})
|
||||
.collect();
|
||||
// Unrelated scattered rects push the page past the 6-rect page gate
|
||||
// so the run reaches clustering, while the 3-box stack itself stays
|
||||
// below the 6-rect cluster minimum.
|
||||
for i in 0..4 {
|
||||
rects.push(PdfRect {
|
||||
x: 100.0 + i as f32 * 120.0,
|
||||
y: 100.0,
|
||||
width: 40.0,
|
||||
height: 15.0,
|
||||
page: 1,
|
||||
});
|
||||
}
|
||||
let items: Vec<TextItem> = ["Step One: Plan", "Step Two: Build", "Step Three: Ship"]
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, t)| make_item(t, 120.0, 605.0 - i as f32 * 22.0, 10.0))
|
||||
.collect();
|
||||
let (tables, _) = detect_tables_from_rects(&items, &rects, 1);
|
||||
assert!(
|
||||
tables.is_empty(),
|
||||
"3-box stacks are intentionally below the detection floor"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn failed_cluster_generates_hint_with_items() {
|
||||
// A cluster of rects forming an outer border (2 x-edges after snapping)
|
||||
|
||||
+62
-2
@@ -123,6 +123,7 @@ fn format_toc_as_list(cells: &[Vec<String>], footnotes: &[String]) -> String {
|
||||
|
||||
/// True when the cell looks like a page number. Accepts:
|
||||
/// - plain digit tokens: "42", "86 86"
|
||||
/// - canonical roman numerals (front-matter pages): "vii", "ix", "xii"
|
||||
/// - dashed section-page IDs: "5-21", "A-1", "B--3", "TC-2" (common in
|
||||
/// technical manuals)
|
||||
fn is_page_number_cell(cell: &str) -> bool {
|
||||
@@ -138,6 +139,9 @@ fn is_page_number_cell(cell: &str) -> bool {
|
||||
if all_digits {
|
||||
return t.len() <= 4;
|
||||
}
|
||||
if super::canonical_roman_value(t).is_some() {
|
||||
return true;
|
||||
}
|
||||
// Section-page form: uppercase letters, digits, dashes; at least
|
||||
// one digit present.
|
||||
t.chars()
|
||||
@@ -184,6 +188,21 @@ fn starts_with_numbered_label(cell: &str) -> bool {
|
||||
.is_some_and(|c| matches!(c, '.' | ')' | '-' | ':'))
|
||||
}
|
||||
|
||||
fn starts_with_hierarchical_numbered_label(cell: &str) -> bool {
|
||||
let token = cell
|
||||
.split_whitespace()
|
||||
.next()
|
||||
.unwrap_or("")
|
||||
.trim_end_matches(['.', ')', ':', '-']);
|
||||
let levels: Vec<&str> = token.split('.').collect();
|
||||
(2..=4).contains(&levels.len())
|
||||
&& levels.iter().all(|level| {
|
||||
!level.is_empty()
|
||||
&& level.len() <= 3
|
||||
&& level.chars().all(|character| character.is_ascii_digit())
|
||||
})
|
||||
}
|
||||
|
||||
fn alpha_word_count(cell: &str) -> usize {
|
||||
cell.split_whitespace()
|
||||
.filter(|word| word.chars().any(|c| c.is_alphabetic()))
|
||||
@@ -337,11 +356,12 @@ fn clean_table_cells(cells: &[Vec<String>]) -> (Vec<Vec<String>>, Vec<String>) {
|
||||
// mid-sentence/lowercase ("continued text here", "with 3.5%...") or
|
||||
// carry lowercase fragments in the later cells, so keep those mergeable.
|
||||
let looks_like_hierarchical_subrow = first_cell.is_empty()
|
||||
&& row.len() >= 3
|
||||
&& first_non_empty_col == Some(1)
|
||||
&& looks_like_compact_entry_label(first_non_empty_cell)
|
||||
&& ((non_first_cells.len() >= 2 && title_like_later_cells > 0)
|
||||
&& ((row.len() == 2 && starts_with_hierarchical_numbered_label(first_non_empty_cell))
|
||||
|| (row.len() >= 3 && non_first_cells.len() >= 2 && title_like_later_cells > 0)
|
||||
|| (non_first_cells.len() == 1
|
||||
&& row.len() >= 3
|
||||
&& prev_first_cell_empty
|
||||
&& alpha_word_count(first_non_empty_cell) >= 2));
|
||||
let looks_like_new_first_column_entry = !first_cell.is_empty()
|
||||
@@ -684,6 +704,46 @@ mod tests {
|
||||
assert_eq!(cleaned[4][1], "Model training");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_clean_table_cells_two_column_numbered_subrows_not_merged() {
|
||||
let cells = vec![
|
||||
vec!["Area".into(), "Competence".into()],
|
||||
vec![
|
||||
"1. Embodying sustainability values".into(),
|
||||
"1.1 Valuing sustainability".into(),
|
||||
],
|
||||
vec!["".into(), "1.2 Supporting fairness".into()],
|
||||
vec!["".into(), "1.3 Promoting nature".into()],
|
||||
vec![
|
||||
"2. Embracing complexity".into(),
|
||||
"2.1 Systems thinking".into(),
|
||||
],
|
||||
vec!["".into(), "2.2 Critical thinking".into()],
|
||||
];
|
||||
let (cleaned, _) = clean_table_cells(&cells);
|
||||
|
||||
assert_eq!(cleaned.len(), 6);
|
||||
assert_eq!(cleaned[2], vec!["", "1.2 Supporting fairness"]);
|
||||
assert_eq!(cleaned[3], vec!["", "1.3 Promoting nature"]);
|
||||
assert_eq!(cleaned[5], vec!["", "2.2 Critical thinking"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_clean_table_cells_two_column_numbered_continuation_merges() {
|
||||
let cells = vec![
|
||||
vec!["Area".into(), "Requirement".into()],
|
||||
vec!["Safety".into(), "The program includes".into()],
|
||||
vec!["".into(), "1. First requirement for every operator".into()],
|
||||
];
|
||||
let (cleaned, _) = clean_table_cells(&cells);
|
||||
|
||||
assert_eq!(cleaned.len(), 2);
|
||||
assert_eq!(
|
||||
cleaned[1][1],
|
||||
"The program includes 1. First requirement for every operator"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_clean_table_cells_partial_hierarchical_subrow_not_merged() {
|
||||
let cells = vec![
|
||||
|
||||
+57
-2
@@ -4,7 +4,7 @@
|
||||
|
||||
mod detect_heuristic;
|
||||
mod detect_lines;
|
||||
mod detect_rects;
|
||||
pub(crate) mod detect_rects;
|
||||
mod detect_struct;
|
||||
mod financial;
|
||||
mod format;
|
||||
@@ -14,8 +14,9 @@ pub mod structured;
|
||||
pub use detect_heuristic::detect_tables;
|
||||
pub(crate) use detect_heuristic::is_table_of_contents;
|
||||
pub use detect_lines::detect_tables_from_lines;
|
||||
pub(crate) use detect_lines::detect_vector_grid_tables_from_lines;
|
||||
pub(crate) use detect_rects::cluster_rects;
|
||||
pub use detect_rects::{detect_tables_from_rects, RectHintRegion};
|
||||
pub use detect_rects::{detect_chart_regions, detect_tables_from_rects, RectHintRegion};
|
||||
pub use detect_struct::detect_tables_from_struct_tree;
|
||||
pub use format::table_to_markdown;
|
||||
pub use structured::{cells_to_markdown, StructuredCell};
|
||||
@@ -177,6 +178,60 @@ pub(crate) fn try_build_rect_guided_table(
|
||||
))
|
||||
}
|
||||
|
||||
/// Canonical lowercase roman numeral for `n` (the i/v/x/l/c range).
|
||||
pub(super) fn to_roman_lower(mut n: u32) -> String {
|
||||
const TABLE: [(u32, &str); 9] = [
|
||||
(100, "c"),
|
||||
(90, "xc"),
|
||||
(50, "l"),
|
||||
(40, "xl"),
|
||||
(10, "x"),
|
||||
(9, "ix"),
|
||||
(5, "v"),
|
||||
(4, "iv"),
|
||||
(1, "i"),
|
||||
];
|
||||
let mut out = String::new();
|
||||
for (val, sym) in TABLE {
|
||||
while n >= val {
|
||||
out.push_str(sym);
|
||||
n -= val;
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// Parse a *canonical* roman numeral (i/v/x/l/c range, ≤8 chars) to its value.
|
||||
/// Returns `None` for non-canonical strings, so ordinary words made of those
|
||||
/// letters — "civil", "mix", "ill" — are not mistaken for numbers. Shared by
|
||||
/// the TOC detector and the TOC formatter so the two stay in sync.
|
||||
pub(super) fn canonical_roman_value(token: &str) -> Option<u32> {
|
||||
let lower = token.trim().to_ascii_lowercase();
|
||||
if lower.is_empty() || lower.len() > 8 || !lower.chars().all(|c| "ivxlc".contains(c)) {
|
||||
return None;
|
||||
}
|
||||
let mut total = 0i32;
|
||||
let mut prev = 0i32;
|
||||
for c in lower.chars().rev() {
|
||||
let v = match c {
|
||||
'i' => 1,
|
||||
'v' => 5,
|
||||
'x' => 10,
|
||||
'l' => 50,
|
||||
'c' => 100,
|
||||
_ => return None,
|
||||
};
|
||||
if v < prev {
|
||||
total -= v;
|
||||
} else {
|
||||
total += v;
|
||||
prev = v;
|
||||
}
|
||||
}
|
||||
let value = u32::try_from(total).ok().filter(|&n| n > 0)?;
|
||||
(to_roman_lower(value) == lower).then_some(value)
|
||||
}
|
||||
|
||||
/// Split a TextItem whose text contains multiple whitespace-separated tokens
|
||||
/// (like "10 11 12 ... 31") into individual TextItems, each assigned to the
|
||||
/// nearest column boundary.
|
||||
|
||||
@@ -337,6 +337,7 @@ fn test_group_into_lines_sorting_by_x() {
|
||||
#[test]
|
||||
fn test_markdown_options_default() {
|
||||
let opts = MarkdownOptions::default();
|
||||
assert_eq!(opts.profile, pdf_inspector::MarkdownProfile::Fidelity);
|
||||
assert!(opts.detect_headers);
|
||||
assert!(opts.detect_lists);
|
||||
assert!(opts.detect_code);
|
||||
@@ -346,6 +347,7 @@ fn test_markdown_options_default() {
|
||||
#[test]
|
||||
fn test_markdown_options_custom() {
|
||||
let opts = MarkdownOptions {
|
||||
profile: pdf_inspector::MarkdownProfile::Compact,
|
||||
detect_headers: false,
|
||||
detect_lists: true,
|
||||
detect_code: false,
|
||||
@@ -361,6 +363,7 @@ fn test_markdown_options_custom() {
|
||||
..Default::default()
|
||||
};
|
||||
assert!(!opts.detect_headers);
|
||||
assert_eq!(opts.profile, pdf_inspector::MarkdownProfile::Compact);
|
||||
assert!(opts.detect_lists);
|
||||
assert!(!opts.detect_code);
|
||||
assert_eq!(opts.base_font_size, Some(14.0));
|
||||
|
||||
@@ -22,7 +22,9 @@ Name and address of employee
|
||||
|
||||
**Publication 1244 (Rev. 7-96)** Cat. No. 44472W
|
||||
|
||||
**Instructions** You must keep sufficient proof to show the amount of your tip income for the year. A daily record of your tip income is considered sufficient proof. Keep a daily record for each workday showing the amount of cash and credit card tips received directly from customers or other employees. Also keep a record of the amount of tips, if any, you paid to other employees through tip sharing, tip pooling or other arrangements, and the names of employees to whom you paid tips. Show the date that each entry is made. This date should be on or near the date you received the tip income. You may use **Form 4070A**, Employee’s Daily Record of Tips, or any other daily record to record your tips. **Reporting Tips to Your Employer.—**If you receive tips that total $20 or more for any month while working for one employer, you must report the tips to your employer. Tips include cash left by customers, tips customers add to credit card charges, and tips you receive from other employees. You must report your tips for any one month by the 10th day of the next month. If the 10th day falls on a Saturday, Sunday, or legal holiday, you may give the report to your employer on the next business day that is not a Saturday, Sunday, or legal holiday. You must report tips that total $20 or more every month regardless of your total wages and tips for the year. You may use **Form 4070**, Employee’s Report of Tips to Employer, to report your tips to your employer. See the instructions on the back of Form 4070. You must include all tips, including tips not reported to your employer, as wages on your income tax return. You may use the last page of this publication to total your tips for the year. Your employer must withhold income, social security, and Medicare (or railroad retirement) taxes on tips you report. Your employer usually deducts the withholding due on tips from your regular wages.
|
||||
### Instructions
|
||||
|
||||
You must keep sufficient proof to show the amount of your tip income for the year. A daily record of your tip income is considered sufficient proof. Keep a daily record for each workday showing the amount of cash and credit card tips received directly from customers or other employees. Also keep a record of the amount of tips, if any, you paid to other employees through tip sharing, tip pooling or other arrangements, and the names of employees to whom you paid tips. Show the date that each entry is made. This date should be on or near the date you received the tip income. You may use **Form 4070A**, Employee’s Daily Record of Tips, or any other daily record to record your tips. **Reporting Tips to Your Employer.—**If you receive tips that total $20 or more for any month while working for one employer, you must report the tips to your employer. Tips include cash left by customers, tips customers add to credit card charges, and tips you receive from other employees. You must report your tips for any one month by the 10th day of the next month. If the 10th day falls on a Saturday, Sunday, or legal holiday, you may give the report to your employer on the next business day that is not a Saturday, Sunday, or legal holiday. You must report tips that total $20 or more every month regardless of your total wages and tips for the year. You may use **Form 4070**, Employee’s Report of Tips to Employer, to report your tips to your employer. See the instructions on the back of Form 4070. You must include all tips, including tips not reported to your employer, as wages on your income tax return. You may use the last page of this publication to total your tips for the year. Your employer must withhold income, social security, and Medicare (or railroad retirement) taxes on tips you report. Your employer usually deducts the withholding due on tips from your regular wages.
|
||||
|
||||
*(continued on inside of back cover)*
|
||||
|
||||
|
||||
@@ -6,20 +6,22 @@
|
||||
|
||||
#### Thermodynamic Properties
|
||||
|
||||
**of**
|
||||
|
||||
®
|
||||
**of** ®
|
||||
|
||||
# Freon 12
|
||||
|
||||
**(R-12)** **Technical Information** **Technical Information**
|
||||
##### (R-12)
|
||||
|
||||
##### Technical Information Technical Information
|
||||
|
||||
**®** **Thermodynamic Properties of Freon 12 Refrigerant** **(R-12)** **SI Units**
|
||||
|
||||
Tables of the thermodynamic **Units** properties of R-12 have been developed and are presented here. P = Pressure in kPa. Absolute This information is based on values calculated using the NIST REFPROP T = Temperature in Celcius Database (McLinden, M.O., Klein,
|
||||
|
||||
S.A., Lemmon, E.W., and Peskin, Vf = Fluid (liquid) specific volume
|
||||
A.P., NIST Standard Reference in cubic meters per kilogram Database 23, NIST thermodynamic and transport properties of Vg = Vapour (gas) specific volume refrigerants and refrigerant in cubic meters per kilogram mixtures – REFPROP version 6.01, Standard Reference Data Program, df and dg = Fluid and Vapour National Institute of Standards and (respectively) densities in Technology, 1998). kilograms per cubic meter
|
||||
A.P., NIST Standard Reference in cubic meters per kilogram Database 23, NIST thermodynamic and transport properties of Vg = Vapour (gas) specific volume refrigerants and refrigerant in cubic meters per kilogram mixtures – REFPROP version 6.01, Standard Reference Data Program, df and dg = Fluid and Vapour National Institute of Standards and (respectively) densities in Technology, 1998).
|
||||
kilograms per cubic meter
|
||||
|
||||
##### H = Enthalpy (kJ/kg)
|
||||
|
||||
##### S = Entropy (kJ/kg.K)
|
||||
|
||||
Reference in New Issue
Block a user