Compare commits
10
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7a5ad20072 | ||
|
|
35cf88eae5 | ||
|
|
44b58a1f6b | ||
|
|
c409e3b8ca | ||
|
|
704ca588f5 | ||
|
|
6f4a523a06 | ||
|
|
2f23f07f6e | ||
|
|
0a9c120a6b | ||
|
|
7c8b09be67 | ||
|
|
35445c3208 |
@@ -2,14 +2,41 @@ name: Publish npm package
|
||||
|
||||
on:
|
||||
push:
|
||||
tags: ['v*']
|
||||
branches: [main]
|
||||
paths: ['napi/package.json']
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
id-token: write
|
||||
|
||||
jobs:
|
||||
check-version:
|
||||
name: Check version change
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
changed: ${{ steps.check.outputs.changed }}
|
||||
version: ${{ steps.check.outputs.version }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 2
|
||||
|
||||
- name: Check if version changed
|
||||
id: check
|
||||
run: |
|
||||
NEW_VERSION=$(node -p "require('./napi/package.json').version")
|
||||
OLD_VERSION=$(git show HEAD~1:napi/package.json | node -p "JSON.parse(require('fs').readFileSync('/dev/stdin','utf8')).version")
|
||||
echo "old=$OLD_VERSION new=$NEW_VERSION"
|
||||
if [ "$NEW_VERSION" != "$OLD_VERSION" ]; then
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
echo "version=$NEW_VERSION" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
build:
|
||||
needs: check-version
|
||||
if: needs.check-version.outputs.changed == 'true'
|
||||
name: Build ${{ matrix.target }}
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
@@ -68,7 +95,7 @@ jobs:
|
||||
|
||||
publish:
|
||||
name: Publish to npm
|
||||
needs: build
|
||||
needs: [check-version, build]
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
@@ -55,12 +55,12 @@ print(result.markdown) # Markdown string or None
|
||||
### Node.js
|
||||
|
||||
```bash
|
||||
npm install @firecrawl/pdf-inspector-js
|
||||
npm install firecrawl-pdf-inspector
|
||||
```
|
||||
|
||||
```javascript
|
||||
import { readFileSync } from 'fs';
|
||||
import { processPdf, classifyPdf } from '@firecrawl/pdf-inspector-js';
|
||||
import { processPdf, classifyPdf } from 'firecrawl-pdf-inspector';
|
||||
|
||||
const result = processPdf(readFileSync('document.pdf'));
|
||||
console.log(result.pdfType); // "TextBased", "Scanned", "ImageBased", "Mixed"
|
||||
|
||||
Executable
+131
@@ -0,0 +1,131 @@
|
||||
#!/usr/bin/env node
|
||||
|
||||
import { readFileSync, writeFileSync } from "fs";
|
||||
import { createRequire } from "module";
|
||||
|
||||
const require = createRequire(import.meta.url);
|
||||
const { version } = require("../package.json");
|
||||
|
||||
const HELP = `pdf-inspector v${version} — Fast PDF text extraction to Markdown
|
||||
|
||||
Usage:
|
||||
pdf-inspector <file> Extract markdown (default)
|
||||
pdf-inspector detect <file> Classify PDF type
|
||||
|
||||
Options:
|
||||
--json Output as JSON
|
||||
--pages <pages> Comma-separated page numbers (e.g. 1,3,5)
|
||||
-o, --output <file> Write output to file instead of stdout
|
||||
-h, --help Show this help
|
||||
-v, --version Show version
|
||||
|
||||
Examples:
|
||||
pdf-inspector document.pdf
|
||||
pdf-inspector document.pdf --json
|
||||
pdf-inspector document.pdf --pages 1,2,3
|
||||
pdf-inspector detect document.pdf --json
|
||||
cat document.pdf | pdf-inspector -`;
|
||||
|
||||
function die(msg) {
|
||||
process.stderr.write(`error: ${msg}\n`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
function parseArgs(argv) {
|
||||
const opts = { json: false, pages: null, output: null, file: null, command: "extract" };
|
||||
let i = 0;
|
||||
|
||||
// Check for subcommand
|
||||
if (argv[0] === "detect") {
|
||||
opts.command = "detect";
|
||||
i = 1;
|
||||
}
|
||||
|
||||
while (i < argv.length) {
|
||||
const arg = argv[i];
|
||||
if (arg === "-h" || arg === "--help") {
|
||||
process.stdout.write(HELP + "\n");
|
||||
process.exit(0);
|
||||
} else if (arg === "-v" || arg === "--version") {
|
||||
process.stdout.write(`${version}\n`);
|
||||
process.exit(0);
|
||||
} else if (arg === "--json") {
|
||||
opts.json = true;
|
||||
} else if (arg === "--pages") {
|
||||
i++;
|
||||
if (!argv[i]) die("--pages requires a value (e.g. 1,3,5)");
|
||||
opts.pages = argv[i].split(",").map((p) => {
|
||||
const n = parseInt(p.trim(), 10);
|
||||
if (Number.isNaN(n) || n < 1) die(`invalid page number: ${p}`);
|
||||
return n;
|
||||
});
|
||||
} else if (arg === "-o" || arg === "--output") {
|
||||
i++;
|
||||
if (!argv[i]) die("-o requires a filename");
|
||||
opts.output = argv[i];
|
||||
} else if (arg === "-" || !arg.startsWith("-")) {
|
||||
if (opts.file) die(`unexpected argument: ${arg}`);
|
||||
opts.file = arg;
|
||||
} else {
|
||||
die(`unknown option: ${arg}`);
|
||||
}
|
||||
i++;
|
||||
}
|
||||
|
||||
return opts;
|
||||
}
|
||||
|
||||
function readInput(file) {
|
||||
if (file === "-") {
|
||||
return readFileSync(0); // stdin fd
|
||||
}
|
||||
try {
|
||||
return readFileSync(file);
|
||||
} catch (err) {
|
||||
if (err.code === "ENOENT") die(`file not found: ${file}`);
|
||||
die(err.message);
|
||||
}
|
||||
}
|
||||
|
||||
function output(text, outputPath) {
|
||||
if (outputPath) {
|
||||
writeFileSync(outputPath, text);
|
||||
} else {
|
||||
process.stdout.write(text);
|
||||
}
|
||||
}
|
||||
|
||||
// ---- main ----
|
||||
|
||||
const opts = parseArgs(process.argv.slice(2));
|
||||
|
||||
if (!opts.file) {
|
||||
// Check if stdin is piped
|
||||
if (process.stdin.isTTY !== false) {
|
||||
process.stderr.write(HELP + "\n");
|
||||
process.exit(1);
|
||||
}
|
||||
opts.file = "-";
|
||||
}
|
||||
|
||||
const { processPdf, classifyPdf } = await import("../index.js");
|
||||
const buffer = readInput(opts.file);
|
||||
|
||||
if (opts.command === "detect") {
|
||||
const result = classifyPdf(buffer);
|
||||
if (opts.json) {
|
||||
output(JSON.stringify(result, null, 2) + "\n", opts.output);
|
||||
} else {
|
||||
const ocr = result.pagesNeedingOcr.length > 0
|
||||
? `, ${result.pagesNeedingOcr.length} pages need OCR`
|
||||
: "";
|
||||
output(`${result.pdfType} (${result.pageCount} pages, confidence: ${result.confidence.toFixed(2)}${ocr})\n`, opts.output);
|
||||
}
|
||||
} else {
|
||||
const result = processPdf(buffer, opts.pages ?? undefined);
|
||||
if (opts.json) {
|
||||
output(JSON.stringify(result, null, 2) + "\n", opts.output);
|
||||
} else {
|
||||
output((result.markdown ?? "") + "\n", opts.output);
|
||||
}
|
||||
}
|
||||
+5
-1
@@ -1,9 +1,12 @@
|
||||
{
|
||||
"name": "firecrawl-pdf-inspector",
|
||||
"version": "0.7.0",
|
||||
"version": "1.0.0",
|
||||
"description": "Fast PDF classification and text extraction. Detect text-based vs scanned PDFs, extract text by region with quality checks. Native Rust performance via napi-rs.",
|
||||
"main": "index.js",
|
||||
"types": "index.d.ts",
|
||||
"bin": {
|
||||
"pdf-inspector": "bin/pdf-inspector.mjs"
|
||||
},
|
||||
"license": "MIT",
|
||||
"keywords": [
|
||||
"pdf",
|
||||
@@ -20,6 +23,7 @@
|
||||
"index.js",
|
||||
"index.d.ts",
|
||||
"*.node",
|
||||
"bin/",
|
||||
"README.md"
|
||||
],
|
||||
"repository": {
|
||||
|
||||
+1954
-79
File diff suppressed because it is too large
Load Diff
@@ -216,6 +216,7 @@ pub(crate) fn extract_page_text_items(
|
||||
let mut marked_content_stack: Vec<MarkedContentEntry> = Vec::new();
|
||||
let mut suppress_glyph_extraction = false;
|
||||
let mut actual_text_start_tm: Option<[f32; 6]> = None; // text matrix at BDC entry
|
||||
let mut actual_text_glyph_tm: Option<[f32; 6]> = None; // text matrix at first glyph inside BDC
|
||||
/// Get the innermost MCID from the marked content stack.
|
||||
fn current_mcid(stack: &[MarkedContentEntry]) -> Option<i64> {
|
||||
stack.iter().rev().find_map(|e| e.mcid)
|
||||
@@ -349,8 +350,15 @@ pub(crate) fn extract_page_text_items(
|
||||
)
|
||||
})
|
||||
});
|
||||
// ActualText: suppress glyph extraction, just advance text matrix
|
||||
// ActualText: suppress glyph extraction, just advance text matrix.
|
||||
// Capture the FIRST glyph's text matrix as the rendering position
|
||||
// for the ActualText item. Td ops between BDC and the first Tj
|
||||
// may have moved the position to the correct line — the BDC-entry
|
||||
// position (actual_text_start_tm) can be on the previous line.
|
||||
if suppress_glyph_extraction {
|
||||
if actual_text_glyph_tm.is_none() {
|
||||
actual_text_glyph_tm = Some(text_matrix);
|
||||
}
|
||||
if let Some(w_ts) = w_ts_opt {
|
||||
text_matrix[4] += w_ts * text_matrix[0];
|
||||
text_matrix[5] += w_ts * text_matrix[1];
|
||||
@@ -425,6 +433,10 @@ pub(crate) fn extract_page_text_items(
|
||||
let font_info = font_widths.get(¤t_font);
|
||||
let is_invisible = (text_rendering_mode == 3 && !include_invisible)
|
||||
|| suppress_glyph_extraction;
|
||||
// Capture first-glyph position for ActualText
|
||||
if suppress_glyph_extraction && actual_text_glyph_tm.is_none() {
|
||||
actual_text_glyph_tm = Some(text_matrix);
|
||||
}
|
||||
|
||||
// Compute space threshold based on font metrics when available
|
||||
let space_threshold = if let Some(font_info) = font_info {
|
||||
@@ -700,6 +712,7 @@ pub(crate) fn extract_page_text_items(
|
||||
if actual_text.is_some() {
|
||||
suppress_glyph_extraction = true;
|
||||
actual_text_start_tm = Some(text_matrix);
|
||||
actual_text_glyph_tm = None; // reset — will be captured at first Tj/TJ
|
||||
}
|
||||
marked_content_stack.push(MarkedContentEntry { actual_text, mcid });
|
||||
}
|
||||
@@ -707,8 +720,13 @@ pub(crate) fn extract_page_text_items(
|
||||
// End Marked Content — emit ActualText item with correct width
|
||||
if let Some(entry) = marked_content_stack.pop() {
|
||||
if let Some(at) = entry.actual_text {
|
||||
// Compute width from text matrix advancement during BDC..EMC
|
||||
if let Some(start_tm) = actual_text_start_tm.take() {
|
||||
// Use the first-glyph position (if available) instead of the
|
||||
// BDC-entry position. Td operators between BDC and the first
|
||||
// Tj may have moved the text position to the correct line —
|
||||
// the BDC-entry position can be on the previous line.
|
||||
let glyph_tm = actual_text_glyph_tm.take();
|
||||
let entry_tm = actual_text_start_tm.take();
|
||||
if let Some(start_tm) = glyph_tm.or(entry_tm) {
|
||||
let combined = multiply_matrices(&start_tm, &ctm);
|
||||
if combined[0].abs() >= combined[1].abs() {
|
||||
rotation_votes.horizontal += 1;
|
||||
|
||||
@@ -1,3 +1,9 @@
|
||||
// Rust 1.95 introduced collapsible_match for `if` inside match arms.
|
||||
// The content-stream parsers use this pattern extensively (match on operator
|
||||
// name, then check `in_text_block && !op.operands.is_empty()`). Collapsing
|
||||
// these into match guards would hurt readability. Allow crate-wide.
|
||||
#![allow(clippy::collapsible_match)]
|
||||
|
||||
//! Smart PDF detection and text extraction using lopdf
|
||||
//!
|
||||
//! # Quick start
|
||||
|
||||
@@ -1144,6 +1144,33 @@ pub(crate) fn find_first_table_row(
|
||||
continue;
|
||||
}
|
||||
|
||||
// Skip rows that have duplicate non-empty cells. These are spanning
|
||||
// super-headers (e.g., "First Degree | First Degree | Higher Degree")
|
||||
// that sit above the real column header row. Using them as the markdown
|
||||
// header produces duplicate column names that downstream validation
|
||||
// rejects. Only skip if a subsequent row looks like a better header
|
||||
// (denser fill or has data).
|
||||
if filled_count >= 2 && !has_data {
|
||||
let mut text_counts: std::collections::HashMap<&str, usize> =
|
||||
std::collections::HashMap::new();
|
||||
for cell in &filled_cells {
|
||||
*text_counts.entry(cell.trim()).or_insert(0) += 1;
|
||||
}
|
||||
let has_duplicates = text_counts.values().any(|&count| count >= 2);
|
||||
if has_duplicates {
|
||||
// Check if a later row is a better header candidate
|
||||
let has_better_below = cells.iter().skip(row_idx + 1).take(3).any(|r| {
|
||||
let next_filled = r.iter().filter(|c| !c.trim().is_empty()).count();
|
||||
let next_fill = next_filled as f32 / total_cols as f32;
|
||||
let next_numeric = r.iter().filter(|c| looks_like_number(c.trim())).count();
|
||||
next_fill >= 0.4 || next_numeric >= 2
|
||||
});
|
||||
if has_better_below {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Data rows are definitely table content
|
||||
if has_data {
|
||||
first_table_row = row_idx;
|
||||
|
||||
+132
-12
@@ -82,33 +82,42 @@ pub(crate) fn find_column_boundaries(
|
||||
}
|
||||
}
|
||||
|
||||
let mut columns = Vec::new();
|
||||
let mut cluster_items: Vec<f32> = vec![x_positions[0]];
|
||||
// Track cluster membership: for each cluster, store the list of x positions
|
||||
let mut cluster_xs: Vec<Vec<f32>> = vec![vec![x_positions[0]]];
|
||||
|
||||
for &x in &x_positions[1..] {
|
||||
let last_cluster = cluster_xs.last().unwrap();
|
||||
// For dense columns (gap-histogram triggered), use edge-based clustering:
|
||||
// compare with the last item to avoid center-drift that merges adjacent
|
||||
// narrow columns. For normal tables, use center-based (original behavior).
|
||||
let reference = if use_edge_clustering {
|
||||
*cluster_items.last().unwrap()
|
||||
*last_cluster.last().unwrap()
|
||||
} else {
|
||||
cluster_items.iter().sum::<f32>() / cluster_items.len() as f32
|
||||
last_cluster.iter().sum::<f32>() / last_cluster.len() as f32
|
||||
};
|
||||
|
||||
if x - reference > cluster_threshold {
|
||||
let cluster_center = cluster_items.iter().sum::<f32>() / cluster_items.len() as f32;
|
||||
columns.push(cluster_center);
|
||||
cluster_items = vec![x];
|
||||
cluster_xs.push(vec![x]);
|
||||
} else {
|
||||
cluster_items.push(x);
|
||||
cluster_xs.last_mut().unwrap().push(x);
|
||||
}
|
||||
}
|
||||
|
||||
// Don't forget last cluster
|
||||
if !cluster_items.is_empty() {
|
||||
columns.push(cluster_items.iter().sum::<f32>() / cluster_items.len() as f32);
|
||||
// Numeric column merge pass: when a sparse cluster (few items, typically
|
||||
// header text) is adjacent to a dense numeric cluster and within 1.5×
|
||||
// threshold, merge them. This fixes tables where multi-line wrapped
|
||||
// headers have slightly different X positions than the data columns,
|
||||
// causing the header and data to split into separate clusters.
|
||||
let columns_before_merge = cluster_xs.len();
|
||||
if columns_before_merge >= 3 {
|
||||
cluster_xs = merge_numeric_adjacent_clusters(cluster_xs, items, cluster_threshold);
|
||||
}
|
||||
|
||||
let columns: Vec<f32> = cluster_xs
|
||||
.iter()
|
||||
.map(|xs| xs.iter().sum::<f32>() / xs.len() as f32)
|
||||
.collect();
|
||||
|
||||
// Filter columns - each should have multiple items
|
||||
let min_items_per_col = (items.len() / columns.len().max(1) / 4).max(2);
|
||||
let columns: Vec<f32> = columns
|
||||
@@ -123,8 +132,9 @@ pub(crate) fn find_column_boundaries(
|
||||
.collect();
|
||||
|
||||
log::debug!(
|
||||
" find_column_boundaries: {} columns before filter, threshold={:.1}, {} items",
|
||||
" find_column_boundaries: {} columns (merged from {}), threshold={:.1}, {} items",
|
||||
columns.len(),
|
||||
columns_before_merge,
|
||||
cluster_threshold,
|
||||
items.len()
|
||||
);
|
||||
@@ -148,6 +158,116 @@ pub(crate) fn find_column_boundaries(
|
||||
columns
|
||||
}
|
||||
|
||||
/// Check if a text string looks like a number (digits, decimals, sign, comma).
|
||||
fn is_numeric_text(s: &str) -> bool {
|
||||
let s = s.trim();
|
||||
if s.is_empty() {
|
||||
return false;
|
||||
}
|
||||
// Match patterns like: 8.23, -1.05, 9.99, 7.12, 100, 3,456.78, +5%, ---
|
||||
// But NOT: BIO, Department, Core Courses
|
||||
s.chars()
|
||||
.all(|c| c.is_ascii_digit() || c == '.' || c == ',' || c == '-' || c == '+' || c == '%')
|
||||
&& s.chars().any(|c| c.is_ascii_digit())
|
||||
}
|
||||
|
||||
/// Merge adjacent X-position clusters when one is a sparse header cluster
|
||||
/// and the other is a dense numeric data cluster. This prevents multi-line
|
||||
/// wrapped headers from splitting a logical column into two clusters.
|
||||
fn merge_numeric_adjacent_clusters(
|
||||
mut clusters: Vec<Vec<f32>>,
|
||||
items: &[(usize, &TextItem)],
|
||||
threshold: f32,
|
||||
) -> Vec<Vec<f32>> {
|
||||
// For each cluster, compute: center, item count, numeric fraction
|
||||
struct ClusterInfo {
|
||||
center: f32,
|
||||
count: usize,
|
||||
numeric_frac: f32,
|
||||
}
|
||||
|
||||
let compute_info = |xs: &[f32]| -> ClusterInfo {
|
||||
let center = xs.iter().sum::<f32>() / xs.len() as f32;
|
||||
// Count items and numeric fraction for items near this cluster center
|
||||
let mut total = 0;
|
||||
let mut numeric = 0;
|
||||
for (_, item) in items {
|
||||
if (item.x - center).abs() < threshold {
|
||||
total += 1;
|
||||
if is_numeric_text(&item.text) {
|
||||
numeric += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
ClusterInfo {
|
||||
center,
|
||||
count: total,
|
||||
numeric_frac: if total > 0 {
|
||||
numeric as f32 / total as f32
|
||||
} else {
|
||||
0.0
|
||||
},
|
||||
}
|
||||
};
|
||||
|
||||
// Merge distance: allow merging clusters that are slightly beyond the
|
||||
// original threshold. Use 1.5× threshold to catch header-vs-data splits.
|
||||
let merge_dist = threshold * 1.5;
|
||||
|
||||
// Iterate and merge adjacent pairs. Use a simple left-to-right scan.
|
||||
let mut merged = true;
|
||||
while merged {
|
||||
merged = false;
|
||||
let mut i = 0;
|
||||
while i + 1 < clusters.len() {
|
||||
let info_a = compute_info(&clusters[i]);
|
||||
let info_b = compute_info(&clusters[i + 1]);
|
||||
let dist = (info_b.center - info_a.center).abs();
|
||||
|
||||
if dist > merge_dist {
|
||||
i += 1;
|
||||
continue;
|
||||
}
|
||||
|
||||
// Determine if one cluster is sparse (header) and the other
|
||||
// is dense and numeric (data). A cluster is "sparse" if it has
|
||||
// significantly fewer items than the other.
|
||||
let (sparse, dense) = if info_a.count < info_b.count {
|
||||
(&info_a, &info_b)
|
||||
} else {
|
||||
(&info_b, &info_a)
|
||||
};
|
||||
|
||||
// Merge if the dense cluster is predominantly numeric (>50%)
|
||||
// and the sparse cluster has at most 1/3 the items of the dense one.
|
||||
let should_merge =
|
||||
dense.numeric_frac > 0.50 && sparse.count <= dense.count / 2 && sparse.count <= 5;
|
||||
|
||||
if should_merge {
|
||||
log::debug!(
|
||||
" merging column clusters: center {:.1} ({} items, {:.0}% numeric) + {:.1} ({} items, {:.0}% numeric), dist={:.1}",
|
||||
info_a.center,
|
||||
info_a.count,
|
||||
info_a.numeric_frac * 100.0,
|
||||
info_b.center,
|
||||
info_b.count,
|
||||
info_b.numeric_frac * 100.0,
|
||||
dist,
|
||||
);
|
||||
// Merge cluster i+1 into cluster i
|
||||
let next = clusters.remove(i + 1);
|
||||
clusters[i].extend(next);
|
||||
merged = true;
|
||||
// Don't increment i — check if the merged cluster can merge further
|
||||
} else {
|
||||
i += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
clusters
|
||||
}
|
||||
|
||||
/// Find row boundaries by clustering Y positions
|
||||
pub(crate) fn find_row_boundaries(items: &[(usize, &TextItem)]) -> Vec<f32> {
|
||||
let mut y_positions: Vec<f32> = items.iter().map(|(_, i)| i.y).collect();
|
||||
|
||||
BIN
Binary file not shown.
@@ -1438,6 +1438,42 @@ fn test_extract_tables_in_regions_nonexistent_page() {
|
||||
assert!(region.text.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_bits_pilani_page4_table_detection() {
|
||||
// Page 4 (0-indexed 3) has a table with multi-line wrapped headers and
|
||||
// numeric data columns. The heuristic detector previously failed because:
|
||||
// 1. Header items at different X positions than data created extra column
|
||||
// clusters (6 cols instead of 4)
|
||||
// 2. Spanning super-header row ("First Degree | First Degree") produced
|
||||
// duplicate header cells that looks_like_partial_table_ex rejected
|
||||
let buf = std::fs::read("tests/fixtures/bits_pilani_feedback.pdf").unwrap();
|
||||
let results =
|
||||
extract_tables_in_regions_mem(&buf, &[(3, vec![[0.0, 0.0, 612.0, 792.0]])]).unwrap();
|
||||
assert_eq!(results.len(), 1);
|
||||
let region = &results[0].regions[0];
|
||||
assert!(
|
||||
!region.needs_ocr,
|
||||
"Page 4 table should be detected, got needs_ocr=true"
|
||||
);
|
||||
assert!(
|
||||
region.text.contains("BIO"),
|
||||
"Should contain department name BIO"
|
||||
);
|
||||
assert!(region.text.contains("8.23"), "Should contain numeric data");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_bits_pilani_page8_table_detection() {
|
||||
// Page 8 (0-indexed 7) has a numbered-row table that already worked.
|
||||
// Verify it still works after changes.
|
||||
let buf = std::fs::read("tests/fixtures/bits_pilani_feedback.pdf").unwrap();
|
||||
let results =
|
||||
extract_tables_in_regions_mem(&buf, &[(7, vec![[0.0, 0.0, 612.0, 792.0]])]).unwrap();
|
||||
assert_eq!(results.len(), 1);
|
||||
let region = &results[0].regions[0];
|
||||
assert!(!region.needs_ocr, "Page 8 table should still be detected");
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
// extract_pages_markdown_mem tests
|
||||
// =========================================================================
|
||||
|
||||
Reference in New Issue
Block a user