add logging
This commit is contained in:
@@ -47,15 +47,6 @@ src/
|
|||||||
bin/
|
bin/
|
||||||
pdf2md.rs — CLI: PDF → Markdown
|
pdf2md.rs — CLI: PDF → Markdown
|
||||||
detect_pdf.rs — CLI: detect PDF type
|
detect_pdf.rs — CLI: detect PDF type
|
||||||
debug_spaces.rs — Debug: dump text items with x/y/width per page
|
|
||||||
dump_ops.rs — Debug: dump raw PDF content stream operators
|
|
||||||
debug_ygaps.rs — Debug: Y-gap analysis between lines
|
|
||||||
debug_fonts.rs — Debug: font information
|
|
||||||
debug_ligatures.rs — Debug: ligature expansion
|
|
||||||
debug_order.rs — Debug: reading order
|
|
||||||
debug_pages.rs — Debug: page-level info
|
|
||||||
detection_report.rs — Batch detection report on PDF directory
|
|
||||||
profile_stages.rs — Performance profiling of pipeline stages
|
|
||||||
```
|
```
|
||||||
|
|
||||||
## Data Flow
|
## Data Flow
|
||||||
@@ -120,6 +111,23 @@ cargo fmt --check # Format check
|
|||||||
cargo run --release --bin pdf2md -- <file.pdf> # Smoke test
|
cargo run --release --bin pdf2md -- <file.pdf> # Smoke test
|
||||||
```
|
```
|
||||||
|
|
||||||
|
## Debugging with RUST_LOG
|
||||||
|
|
||||||
|
All debug output uses structured logging via the `log` crate. Set `RUST_LOG` to control output:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
RUST_LOG=pdf_inspector::extractor::content_stream=trace # raw PDF operators
|
||||||
|
RUST_LOG=pdf_inspector::extractor::fonts=debug # font metadata + encodings
|
||||||
|
RUST_LOG=pdf_inspector::tounicode=debug # CMap parsing
|
||||||
|
RUST_LOG=pdf_inspector::extractor=debug # text items per page
|
||||||
|
RUST_LOG=pdf_inspector::extractor::layout=debug # columns, reading order
|
||||||
|
RUST_LOG=pdf_inspector::markdown::analysis=debug # Y-gaps, paragraph threshold
|
||||||
|
RUST_LOG=pdf_inspector::tables=debug # table detection
|
||||||
|
RUST_LOG=pdf_inspector=debug # everything
|
||||||
|
```
|
||||||
|
|
||||||
|
Example: `RUST_LOG=pdf_inspector::extractor::fonts=debug cargo run --bin pdf2md -- file.pdf > /dev/null`
|
||||||
|
|
||||||
## Common Tasks
|
## Common Tasks
|
||||||
|
|
||||||
| Task | Where to Edit |
|
| Task | Where to Edit |
|
||||||
|
|||||||
+1
-19
@@ -23,6 +23,7 @@ rayon = "1.10"
|
|||||||
|
|
||||||
# Logging
|
# Logging
|
||||||
log = "0.4"
|
log = "0.4"
|
||||||
|
env_logger = "0.11"
|
||||||
|
|
||||||
# Text processing
|
# Text processing
|
||||||
regex = "1.10"
|
regex = "1.10"
|
||||||
@@ -43,22 +44,3 @@ name = "detect-pdf"
|
|||||||
path = "src/bin/detect_pdf.rs"
|
path = "src/bin/detect_pdf.rs"
|
||||||
|
|
||||||
|
|
||||||
[[bin]]
|
|
||||||
name = "debug_ygaps"
|
|
||||||
path = "src/bin/debug_ygaps.rs"
|
|
||||||
|
|
||||||
[[bin]]
|
|
||||||
name = "debug_pages"
|
|
||||||
path = "src/bin/debug_pages.rs"
|
|
||||||
|
|
||||||
[[bin]]
|
|
||||||
name = "debug_fonts"
|
|
||||||
path = "src/bin/debug_fonts.rs"
|
|
||||||
|
|
||||||
[[bin]]
|
|
||||||
name = "dump_ops"
|
|
||||||
path = "src/bin/dump_ops.rs"
|
|
||||||
|
|
||||||
[[bin]]
|
|
||||||
name = "debug_spaces"
|
|
||||||
path = "src/bin/debug_spaces.rs"
|
|
||||||
|
|||||||
@@ -150,7 +150,7 @@ src/
|
|||||||
extractor/ — Text extraction pipeline
|
extractor/ — Text extraction pipeline
|
||||||
tables/ — Table detection and formatting
|
tables/ — Table detection and formatting
|
||||||
markdown/ — Markdown conversion and structure detection
|
markdown/ — Markdown conversion and structure detection
|
||||||
bin/ — CLI tools and debug utilities
|
bin/ — CLI tools (pdf2md, detect_pdf)
|
||||||
```
|
```
|
||||||
|
|
||||||
## How classification works
|
## How classification works
|
||||||
@@ -226,14 +226,34 @@ The converter handles:
|
|||||||
| Drop caps | Large initial letters merged with following text |
|
| Drop caps | Large initial letters merged with following text |
|
||||||
| Dot leaders | TOC-style dots collapsed to " ... " |
|
| Dot leaders | TOC-style dots collapsed to " ... " |
|
||||||
|
|
||||||
## Debug tools
|
## Debugging with RUST_LOG
|
||||||
|
|
||||||
|
Structured logging via `RUST_LOG` replaces the former debug binaries. Set the environment variable to control which sections emit debug output on stderr:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
cargo run --bin debug_spaces -- file.pdf # Text items with x/y/width per page
|
# Raw PDF content stream operators (replaces dump_ops)
|
||||||
cargo run --bin dump_ops -- file.pdf # Raw PDF content stream operators
|
RUST_LOG=pdf_inspector::extractor::content_stream=trace cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||||
cargo run --bin debug_ygaps -- file.pdf # Y-gap analysis between lines
|
|
||||||
cargo run --bin debug_fonts -- file.pdf # Font information
|
# Font metadata, encodings, ligatures (replaces debug_fonts / debug_ligatures)
|
||||||
cargo run --bin debug_order -- file.pdf # Reading order visualization
|
RUST_LOG=pdf_inspector::extractor::fonts=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||||
|
|
||||||
|
# ToUnicode CMap parsing
|
||||||
|
RUST_LOG=pdf_inspector::tounicode=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||||
|
|
||||||
|
# Text items per page with x/y/width (replaces debug_spaces / debug_pages)
|
||||||
|
RUST_LOG=pdf_inspector::extractor=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||||
|
|
||||||
|
# Column detection and reading order (replaces debug_order)
|
||||||
|
RUST_LOG=pdf_inspector::extractor::layout=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||||
|
|
||||||
|
# Y-gap analysis and paragraph thresholds (replaces debug_ygaps)
|
||||||
|
RUST_LOG=pdf_inspector::markdown::analysis=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||||
|
|
||||||
|
# Table detection
|
||||||
|
RUST_LOG=pdf_inspector::tables=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||||
|
|
||||||
|
# Everything
|
||||||
|
RUST_LOG=pdf_inspector=debug cargo run --bin pdf2md -- file.pdf > /dev/null
|
||||||
```
|
```
|
||||||
|
|
||||||
## Use case: smart PDF routing
|
## Use case: smart PDF routing
|
||||||
|
|||||||
@@ -6,6 +6,7 @@ use std::process;
|
|||||||
use std::time::Instant;
|
use std::time::Instant;
|
||||||
|
|
||||||
fn main() {
|
fn main() {
|
||||||
|
env_logger::init();
|
||||||
let args: Vec<String> = env::args().collect();
|
let args: Vec<String> = env::args().collect();
|
||||||
|
|
||||||
if args.len() < 2 {
|
if args.len() < 2 {
|
||||||
|
|||||||
@@ -43,6 +43,7 @@ fn parse_page_spec(spec: &str) -> Result<HashSet<u32>, String> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fn main() {
|
fn main() {
|
||||||
|
env_logger::init();
|
||||||
let args: Vec<String> = env::args().collect();
|
let args: Vec<String> = env::args().collect();
|
||||||
|
|
||||||
if args.len() < 2 {
|
if args.len() < 2 {
|
||||||
|
|||||||
@@ -9,6 +9,7 @@ use crate::text_utils::{
|
|||||||
use crate::tounicode::FontCMaps;
|
use crate::tounicode::FontCMaps;
|
||||||
use crate::types::{ItemType, PdfRect, TextItem};
|
use crate::types::{ItemType, PdfRect, TextItem};
|
||||||
use crate::PdfError;
|
use crate::PdfError;
|
||||||
|
use log::trace;
|
||||||
use lopdf::{Document, Encoding, Object, ObjectId};
|
use lopdf::{Document, Encoding, Object, ObjectId};
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
|
|
||||||
@@ -100,6 +101,7 @@ pub(crate) fn extract_page_text_items(
|
|||||||
let mut actual_text_start_tm: Option<[f32; 6]> = None; // text matrix at BDC entry
|
let mut actual_text_start_tm: Option<[f32; 6]> = None; // text matrix at BDC entry
|
||||||
|
|
||||||
for op in &content.operations {
|
for op in &content.operations {
|
||||||
|
trace!("{} {:?}", op.operator, op.operands);
|
||||||
match op.operator.as_str() {
|
match op.operator.as_str() {
|
||||||
"q" => {
|
"q" => {
|
||||||
// Save graphics state
|
// Save graphics state
|
||||||
|
|||||||
@@ -3,6 +3,7 @@
|
|||||||
use crate::glyph_names::glyph_to_char;
|
use crate::glyph_names::glyph_to_char;
|
||||||
use crate::tounicode::FontCMaps;
|
use crate::tounicode::FontCMaps;
|
||||||
use crate::types::{FontEncodingMap, FontWidthInfo, PageFontEncodings, PageFontWidths};
|
use crate::types::{FontEncodingMap, FontWidthInfo, PageFontEncodings, PageFontWidths};
|
||||||
|
use log::debug;
|
||||||
use lopdf::{Document, Encoding, Object};
|
use lopdf::{Document, Encoding, Object};
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
|
|
||||||
@@ -42,6 +43,37 @@ pub(crate) fn build_font_widths(
|
|||||||
|
|
||||||
for (font_name, font_dict) in fonts {
|
for (font_name, font_dict) in fonts {
|
||||||
let resource_name = String::from_utf8_lossy(font_name).to_string();
|
let resource_name = String::from_utf8_lossy(font_name).to_string();
|
||||||
|
|
||||||
|
let subtype = font_dict
|
||||||
|
.get(b"Subtype")
|
||||||
|
.ok()
|
||||||
|
.and_then(|o| o.as_name().ok())
|
||||||
|
.map(|n| String::from_utf8_lossy(n).to_string())
|
||||||
|
.unwrap_or_default();
|
||||||
|
let base_font = font_dict
|
||||||
|
.get(b"BaseFont")
|
||||||
|
.ok()
|
||||||
|
.and_then(|o| o.as_name().ok())
|
||||||
|
.map(|n| String::from_utf8_lossy(n).to_string())
|
||||||
|
.unwrap_or_default();
|
||||||
|
let has_tounicode = font_dict.get(b"ToUnicode").is_ok();
|
||||||
|
let has_descendants = font_dict.get(b"DescendantFonts").is_ok();
|
||||||
|
let encoding_str = font_dict
|
||||||
|
.get(b"Encoding")
|
||||||
|
.ok()
|
||||||
|
.map(|o| match o {
|
||||||
|
Object::Name(n) => String::from_utf8_lossy(n).to_string(),
|
||||||
|
Object::Reference(_) => "ref(dict)".to_string(),
|
||||||
|
Object::Dictionary(_) => "dict".to_string(),
|
||||||
|
_ => format!("{:?}", o),
|
||||||
|
})
|
||||||
|
.unwrap_or_else(|| "none".to_string());
|
||||||
|
|
||||||
|
debug!(
|
||||||
|
"font {:<10} sub={:<12} base={:<45} toUni={:<6} enc={:<20} cid={}",
|
||||||
|
resource_name, subtype, base_font, has_tounicode, encoding_str, has_descendants
|
||||||
|
);
|
||||||
|
|
||||||
if let Some(info) = parse_font_widths(doc, font_dict) {
|
if let Some(info) = parse_font_widths(doc, font_dict) {
|
||||||
widths.insert(resource_name, info);
|
widths.insert(resource_name, info);
|
||||||
}
|
}
|
||||||
@@ -452,6 +484,7 @@ pub(crate) fn parse_encoding_dictionary(
|
|||||||
|
|
||||||
let mut encoding_map = FontEncodingMap::new();
|
let mut encoding_map = FontEncodingMap::new();
|
||||||
let mut current_code: u8 = 0;
|
let mut current_code: u8 = 0;
|
||||||
|
let mut ligature_count = 0u32;
|
||||||
|
|
||||||
for item in diff_array {
|
for item in diff_array {
|
||||||
match item {
|
match item {
|
||||||
@@ -462,6 +495,17 @@ pub(crate) fn parse_encoding_dictionary(
|
|||||||
Object::Name(name) => {
|
Object::Name(name) => {
|
||||||
// Map current code to glyph name -> Unicode
|
// Map current code to glyph name -> Unicode
|
||||||
let glyph_name = String::from_utf8_lossy(&name).to_string();
|
let glyph_name = String::from_utf8_lossy(&name).to_string();
|
||||||
|
if glyph_name == "fi"
|
||||||
|
|| glyph_name == "fl"
|
||||||
|
|| glyph_name == "ffi"
|
||||||
|
|| glyph_name == "ffl"
|
||||||
|
{
|
||||||
|
debug!(
|
||||||
|
" Differences: code=0x{:02X} glyph={:?} (ligature)",
|
||||||
|
current_code, glyph_name
|
||||||
|
);
|
||||||
|
ligature_count += 1;
|
||||||
|
}
|
||||||
if let Some(ch) = glyph_to_char(&glyph_name) {
|
if let Some(ch) = glyph_to_char(&glyph_name) {
|
||||||
encoding_map.insert(current_code, ch);
|
encoding_map.insert(current_code, ch);
|
||||||
}
|
}
|
||||||
@@ -471,6 +515,14 @@ pub(crate) fn parse_encoding_dictionary(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if ligature_count > 0 {
|
||||||
|
debug!(
|
||||||
|
" Differences: {} total entries, {} ligatures",
|
||||||
|
encoding_map.len(),
|
||||||
|
ligature_count
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
if encoding_map.is_empty() {
|
if encoding_map.is_empty() {
|
||||||
None
|
None
|
||||||
} else {
|
} else {
|
||||||
|
|||||||
+54
-1
@@ -2,6 +2,7 @@
|
|||||||
|
|
||||||
use crate::text_utils::{effective_width, sort_line_items};
|
use crate::text_utils::{effective_width, sort_line_items};
|
||||||
use crate::types::{TextItem, TextLine};
|
use crate::types::{TextItem, TextLine};
|
||||||
|
use log::debug;
|
||||||
|
|
||||||
/// Represents a column region on a page
|
/// Represents a column region on a page
|
||||||
#[derive(Debug, Clone)]
|
#[derive(Debug, Clone)]
|
||||||
@@ -170,6 +171,16 @@ pub(crate) fn detect_columns(items: &[TextItem], page: u32) -> Vec<ColumnRegion>
|
|||||||
return vec![ColumnRegion { x_min, x_max }];
|
return vec![ColumnRegion { x_min, x_max }];
|
||||||
}
|
}
|
||||||
|
|
||||||
|
debug!(
|
||||||
|
"page {}: {} columns detected (boundaries: {:?})",
|
||||||
|
page,
|
||||||
|
valid_valleys.len() + 1,
|
||||||
|
valid_valleys
|
||||||
|
.iter()
|
||||||
|
.map(|(s, e)| x_min + ((*s + *e) as f32 / 2.0) * BIN_WIDTH)
|
||||||
|
.collect::<Vec<_>>()
|
||||||
|
);
|
||||||
|
|
||||||
// Limit to at most 3 gutters (4 columns) — keep the widest if more found
|
// Limit to at most 3 gutters (4 columns) — keep the widest if more found
|
||||||
if valid_valleys.len() > 3 {
|
if valid_valleys.len() > 3 {
|
||||||
valid_valleys.sort_by(|a, b| {
|
valid_valleys.sort_by(|a, b| {
|
||||||
@@ -401,6 +412,39 @@ pub fn group_into_lines(items: Vec<TextItem>) -> Vec<TextLine> {
|
|||||||
col_buckets[best_col].push(item.clone());
|
col_buckets[best_col].push(item.clone());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
debug!(
|
||||||
|
"page {}: {} columns, {} spanning items",
|
||||||
|
page,
|
||||||
|
columns.len(),
|
||||||
|
spanning_items.len()
|
||||||
|
);
|
||||||
|
for (ci, col) in columns.iter().enumerate() {
|
||||||
|
debug!(
|
||||||
|
" col {}: x=[{:.0}..{:.0}] {} items",
|
||||||
|
ci,
|
||||||
|
col.x_min,
|
||||||
|
col.x_max,
|
||||||
|
col_buckets[ci].len()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
if log::log_enabled!(log::Level::Trace) {
|
||||||
|
for (ci, bucket) in col_buckets.iter().enumerate() {
|
||||||
|
for item in bucket {
|
||||||
|
log::trace!(
|
||||||
|
" col {} <- x={:7.1} y={:7.1} {:?}",
|
||||||
|
ci,
|
||||||
|
item.x,
|
||||||
|
item.y,
|
||||||
|
if item.text.len() > 60 {
|
||||||
|
&item.text[..60]
|
||||||
|
} else {
|
||||||
|
&item.text
|
||||||
|
}
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
let mut per_column_lines: Vec<Vec<TextLine>> = Vec::new();
|
let mut per_column_lines: Vec<Vec<TextLine>> = Vec::new();
|
||||||
for col_items in col_buckets {
|
for col_items in col_buckets {
|
||||||
let lines = group_single_column(col_items);
|
let lines = group_single_column(col_items);
|
||||||
@@ -410,7 +454,14 @@ pub fn group_into_lines(items: Vec<TextItem>) -> Vec<TextLine> {
|
|||||||
// Process spanning items as their own group
|
// Process spanning items as their own group
|
||||||
let spanning_lines = group_single_column(spanning_items);
|
let spanning_lines = group_single_column(spanning_items);
|
||||||
|
|
||||||
if is_newspaper_layout(&per_column_lines) {
|
let is_newspaper = is_newspaper_layout(&per_column_lines);
|
||||||
|
debug!(
|
||||||
|
"page {}: layout={}",
|
||||||
|
page,
|
||||||
|
if is_newspaper { "newspaper" } else { "tabular" }
|
||||||
|
);
|
||||||
|
|
||||||
|
if is_newspaper {
|
||||||
// Newspaper: columns are independent text flows.
|
// Newspaper: columns are independent text flows.
|
||||||
// 1. Split each column into its densest cluster (core) and stragglers
|
// 1. Split each column into its densest cluster (core) and stragglers
|
||||||
// 2. Use core columns to determine the above/below threshold
|
// 2. Use core columns to determine the above/below threshold
|
||||||
@@ -631,5 +682,7 @@ fn group_single_column(items: Vec<TextItem>) -> Vec<TextLine> {
|
|||||||
sort_line_items(&mut line.items);
|
sort_line_items(&mut line.items);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
debug!("group_single_column: {} lines", lines.len());
|
||||||
|
|
||||||
lines
|
lines
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -12,6 +12,7 @@ use crate::text_utils::is_rtl_text;
|
|||||||
use crate::tounicode::FontCMaps;
|
use crate::tounicode::FontCMaps;
|
||||||
use crate::types::{PdfRect, TextItem};
|
use crate::types::{PdfRect, TextItem};
|
||||||
use crate::PdfError;
|
use crate::PdfError;
|
||||||
|
use log::debug;
|
||||||
use lopdf::{Document, Object, ObjectId};
|
use lopdf::{Document, Object, ObjectId};
|
||||||
use std::collections::{HashMap, HashSet};
|
use std::collections::{HashMap, HashSet};
|
||||||
use std::path::Path;
|
use std::path::Path;
|
||||||
@@ -145,6 +146,30 @@ fn extract_positioned_text_from_doc(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
let (items, rects) = extract_page_text_items(doc, page_id, *page_num, font_cmaps)?;
|
let (items, rects) = extract_page_text_items(doc, page_id, *page_num, font_cmaps)?;
|
||||||
|
debug!(
|
||||||
|
"page {}: {} text items, {} rects",
|
||||||
|
page_num,
|
||||||
|
items.len(),
|
||||||
|
rects.len()
|
||||||
|
);
|
||||||
|
if log::log_enabled!(log::Level::Trace) {
|
||||||
|
for item in &items {
|
||||||
|
log::trace!(
|
||||||
|
" p={} x={:7.1} y={:7.1} w={:7.1} fs={:5.1} font={:6} {:?}",
|
||||||
|
page_num,
|
||||||
|
item.x,
|
||||||
|
item.y,
|
||||||
|
item.width,
|
||||||
|
item.font_size,
|
||||||
|
item.font,
|
||||||
|
if item.text.len() > 80 {
|
||||||
|
&item.text[..80]
|
||||||
|
} else {
|
||||||
|
&item.text
|
||||||
|
}
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
all_items.extend(items);
|
all_items.extend(items);
|
||||||
all_rects.extend(rects);
|
all_rects.extend(rects);
|
||||||
|
|
||||||
|
|||||||
@@ -3,6 +3,7 @@
|
|||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
|
|
||||||
use crate::types::{TextItem, TextLine};
|
use crate::types::{TextItem, TextLine};
|
||||||
|
use log::debug;
|
||||||
|
|
||||||
/// Font statistics for a document
|
/// Font statistics for a document
|
||||||
pub(crate) struct FontStats {
|
pub(crate) struct FontStats {
|
||||||
@@ -118,13 +119,85 @@ pub(crate) fn compute_paragraph_threshold(lines: &[TextLine], base_size: f32) ->
|
|||||||
|
|
||||||
let median = gaps[gaps.len() / 2];
|
let median = gaps[gaps.len() / 2];
|
||||||
|
|
||||||
// The paragraph threshold should be larger than the typical line spacing.
|
let threshold = (median * 1.3).max(base_size * 1.5);
|
||||||
// Use 1.3x the median gap. This means:
|
|
||||||
// - Single-spaced (median ~14pt for 12pt font): threshold = 18.2pt
|
debug!(
|
||||||
// - Double-spaced (median ~28pt for 12pt font): threshold = 36.4pt
|
"paragraph_threshold: base_size={:.1} median_gap={:.1} threshold={:.1} ({} gaps sampled)",
|
||||||
// Also ensure it's at least base_size * 1.5 to avoid false paragraph breaks
|
base_size,
|
||||||
// in tightly-spaced documents.
|
median,
|
||||||
(median * 1.3).max(base_size * 1.5)
|
threshold,
|
||||||
|
gaps.len()
|
||||||
|
);
|
||||||
|
|
||||||
|
if log::log_enabled!(log::Level::Debug) {
|
||||||
|
// Gap histogram
|
||||||
|
let buckets: &[f32] = &[0.0, 0.5, 1.0, 1.2, 1.5, 1.8, 2.0, 2.5, 3.0, 5.0, 10.0];
|
||||||
|
for i in 0..buckets.len() - 1 {
|
||||||
|
let count = gaps
|
||||||
|
.iter()
|
||||||
|
.filter(|&&g| {
|
||||||
|
let r = g / base_size;
|
||||||
|
r >= buckets[i] && r < buckets[i + 1]
|
||||||
|
})
|
||||||
|
.count();
|
||||||
|
if count > 0 {
|
||||||
|
debug!(
|
||||||
|
" gap_ratio {:.1}-{:.1}: {}",
|
||||||
|
buckets[i],
|
||||||
|
buckets[i + 1],
|
||||||
|
count
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let over = gaps.iter().filter(|&&g| g / base_size >= 10.0).count();
|
||||||
|
if over > 0 {
|
||||||
|
debug!(" gap_ratio 10.0+: {}", over);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Per-line detail: Y position, gap, ratio, bold, text preview, paragraph marker
|
||||||
|
if log::log_enabled!(log::Level::Trace) {
|
||||||
|
let mut prev: Option<(u32, f32)> = None;
|
||||||
|
for line in lines {
|
||||||
|
let font_size = line.items.first().map(|i| i.font_size).unwrap_or(0.0);
|
||||||
|
let is_bold = line.items.first().map(|i| i.is_bold).unwrap_or(false);
|
||||||
|
let text = line.text();
|
||||||
|
let display: String = text.chars().take(80).collect();
|
||||||
|
|
||||||
|
let (gap_str, ratio_str, marker) = if let Some((pp, py)) = prev {
|
||||||
|
if pp == line.page {
|
||||||
|
let gap = py - line.y;
|
||||||
|
let ratio = gap / base_size;
|
||||||
|
let is_para = gap > threshold;
|
||||||
|
(
|
||||||
|
format!("{:8.1}", gap),
|
||||||
|
format!("{:8.2}", ratio),
|
||||||
|
if is_para { " <<PARA>>" } else { "" },
|
||||||
|
)
|
||||||
|
} else {
|
||||||
|
(" ---".to_string(), " ---".to_string(), "")
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
(" ---".to_string(), " ---".to_string(), "")
|
||||||
|
};
|
||||||
|
|
||||||
|
log::trace!(
|
||||||
|
" p={} y={:8.1} gap={} ratio={} fs={:5.1} {} {}{}",
|
||||||
|
line.page,
|
||||||
|
line.y,
|
||||||
|
gap_str,
|
||||||
|
ratio_str,
|
||||||
|
font_size,
|
||||||
|
if is_bold { "B" } else { " " },
|
||||||
|
display,
|
||||||
|
marker
|
||||||
|
);
|
||||||
|
|
||||||
|
prev = Some((line.page, line.y));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
threshold
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Discover distinct heading font-size tiers in the document.
|
/// Discover distinct heading font-size tiers in the document.
|
||||||
|
|||||||
@@ -2,6 +2,7 @@
|
|||||||
|
|
||||||
use crate::text_utils::is_rtl_text;
|
use crate::text_utils::is_rtl_text;
|
||||||
use crate::types::TextItem;
|
use crate::types::TextItem;
|
||||||
|
use log::debug;
|
||||||
|
|
||||||
use super::financial::try_split_financial_item;
|
use super::financial::try_split_financial_item;
|
||||||
use super::grid::{
|
use super::grid::{
|
||||||
@@ -578,6 +579,13 @@ fn detect_table_in_region(items: &[(usize, &TextItem)], mode: TableDetectionMode
|
|||||||
return None;
|
return None;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
debug!(
|
||||||
|
"table detected: {} rows x {} cols, {} items",
|
||||||
|
rows.len(),
|
||||||
|
columns.len(),
|
||||||
|
item_indices.len()
|
||||||
|
);
|
||||||
|
|
||||||
Some(Table {
|
Some(Table {
|
||||||
columns,
|
columns,
|
||||||
rows,
|
rows,
|
||||||
|
|||||||
@@ -3,6 +3,7 @@
|
|||||||
//! This module parses ToUnicode CMaps to convert CID-encoded text to Unicode.
|
//! This module parses ToUnicode CMaps to convert CID-encoded text to Unicode.
|
||||||
|
|
||||||
use flate2::read::ZlibDecoder;
|
use flate2::read::ZlibDecoder;
|
||||||
|
use log::debug;
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
use std::io::Read;
|
use std::io::Read;
|
||||||
|
|
||||||
@@ -599,6 +600,18 @@ impl FontCMaps {
|
|||||||
// Copy the by_obj map
|
// Copy the by_obj map
|
||||||
let by_obj_num = cmaps_by_obj;
|
let by_obj_num = cmaps_by_obj;
|
||||||
|
|
||||||
|
for (name, cmap) in &by_name {
|
||||||
|
if !name.contains('_') || name.ends_with(|c: char| c.is_ascii_digit()) {
|
||||||
|
debug!(
|
||||||
|
"CMap font={:30} code_byte_length={} char_map={} ranges={}",
|
||||||
|
name,
|
||||||
|
cmap.code_byte_length,
|
||||||
|
cmap.char_map.len(),
|
||||||
|
cmap.ranges.len()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
FontCMaps {
|
FontCMaps {
|
||||||
by_name,
|
by_name,
|
||||||
by_obj_num,
|
by_obj_num,
|
||||||
|
|||||||
Reference in New Issue
Block a user