extract-pdf 0.1.260 → 0.1.262

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,180 @@
1
+ /**
2
+ * @file server.js
3
+ * @description Entry point: wires middleware, route handlers, OpenAPI docs, and starts the server.
4
+ */
5
+ import { serve } from "@hono/node-server";
6
+ import { OpenAPIHono } from "@hono/zod-openapi";
7
+ import { swaggerUI } from "@hono/swagger-ui";
8
+ import { cors } from "hono/cors";
9
+ import { logger } from "hono/logger";
10
+
11
+ import { initializeModel, isModelLoaded, generateFromImage, load_image } from "./model.js";
12
+ import { convertImageRoute, convertImageBase64Route, healthRoute } from "./routes.js";
13
+
14
+ const app = new OpenAPIHono();
15
+
16
+ app.use("*", cors());
17
+ app.use("*", logger());
18
+
19
+ /** Timestamp used to calculate uptime for the health endpoint. */
20
+ const startTime = Date.now();
21
+
22
+ // ── Health ────────────────────────────────────────────────────────────────────
23
+
24
+ app.openapi(healthRoute, (c) =>
25
+ c.json({
26
+ status: "healthy",
27
+ modelLoaded: isModelLoaded(),
28
+ uptime: Date.now() - startTime,
29
+ version: "1.0.0",
30
+ })
31
+ );
32
+
33
+ // ── Convert (URL) ─────────────────────────────────────────────────────────────
34
+
35
+ app.openapi(convertImageRoute, async (c) => {
36
+ const startProcessing = Date.now();
37
+
38
+ try {
39
+ const { imageUrl, prompt, maxTokens, streaming } = c.req.valid("json");
40
+
41
+ let image;
42
+ try {
43
+ image = await load_image(imageUrl);
44
+ } catch {
45
+ return c.json({ success: false, error: "Failed to load image from URL", code: "IMAGE_LOAD_ERROR" }, 400);
46
+ }
47
+
48
+ const { generatedText } = await generateFromImage({ image, prompt, maxTokens, streaming });
49
+
50
+ return c.json({
51
+ success: true,
52
+ result: generatedText,
53
+ metadata: { processingTime: Date.now() - startProcessing },
54
+ });
55
+ } catch (error) {
56
+ console.error("Conversion error:", error);
57
+ return c.json({ success: false, error: error.message || "Internal processing error", code: "PROCESSING_ERROR" }, 500);
58
+ }
59
+ });
60
+
61
+ // ── Convert (base64) ──────────────────────────────────────────────────────────
62
+
63
+ app.openapi(convertImageBase64Route, async (c) => {
64
+ const startProcessing = Date.now();
65
+
66
+ try {
67
+ const { imageBase64, mimeType, prompt, maxTokens, streaming } = c.req.valid("json");
68
+
69
+ let image;
70
+ try {
71
+ image = await load_image(`data:${mimeType};base64,${imageBase64}`);
72
+ } catch {
73
+ return c.json({ success: false, error: "Failed to decode base64 image", code: "IMAGE_DECODE_ERROR" }, 400);
74
+ }
75
+
76
+ const { generatedText, generated_ids } = await generateFromImage({ image, prompt, maxTokens, streaming });
77
+
78
+ return c.json({
79
+ success: true,
80
+ result: generatedText,
81
+ metadata: {
82
+ processingTime: Date.now() - startProcessing,
83
+ tokenCount: generated_ids?.dims?.[1],
84
+ },
85
+ });
86
+ } catch (error) {
87
+ console.error("Conversion error:", error);
88
+ return c.json({ success: false, error: error.message || "Internal processing error", code: "PROCESSING_ERROR" }, 500);
89
+ }
90
+ });
91
+
92
+ // ── Streaming SSE endpoint ────────────────────────────────────────────────────
93
+
94
+ /**
95
+ * POST /api/v1/convert-stream
96
+ * Streams generated tokens via Server-Sent Events.
97
+ * Not registered with OpenAPI because SSE responses fall outside the JSON schema.
98
+ */
99
+ app.post("/api/v1/convert-stream", async (c) => {
100
+ const { imageUrl, prompt = "Convert this page to docling.", maxTokens = 4096 } = await c.req.json();
101
+
102
+ c.header("Content-Type", "text/event-stream");
103
+ c.header("Cache-Control", "no-cache");
104
+ c.header("Connection", "keep-alive");
105
+
106
+ const encoder = new TextEncoder();
107
+ const stream = new ReadableStream({
108
+ async start(controller) {
109
+ const send = (payload) =>
110
+ controller.enqueue(encoder.encode(`data: ${JSON.stringify(payload)}\n\n`));
111
+ try {
112
+ const { model, processor } = await initializeModel();
113
+ const image = await load_image(imageUrl);
114
+
115
+ const messages = [{
116
+ role: "user",
117
+ content: [{ type: "image" }, { type: "text", text: prompt }],
118
+ }];
119
+ const text = processor.apply_chat_template(messages, { add_generation_prompt: true });
120
+ const inputs = await processor(text, [image], { do_image_splitting: true });
121
+
122
+ let fullText = "";
123
+ const { TextStreamer } = await import("@huggingface/transformers");
124
+ const streamer = new TextStreamer(processor.tokenizer, {
125
+ skip_prompt: true,
126
+ skip_special_tokens: false,
127
+ on_finalized_text: (chunk) => {
128
+ fullText += chunk;
129
+ send({ text: chunk, done: false });
130
+ },
131
+ });
132
+
133
+ await model.generate({ ...inputs, max_new_tokens: maxTokens, streamer });
134
+ send({ text: "", done: true, fullText });
135
+ } catch (error) {
136
+ send({ error: error.message, done: true });
137
+ }
138
+ controller.close();
139
+ },
140
+ });
141
+
142
+ return new Response(stream);
143
+ });
144
+
145
+ // ── Docs & error handling ─────────────────────────────────────────────────────
146
+
147
+ app.doc("/openapi.json", {
148
+ openapi: "3.0.0",
149
+ info: {
150
+ version: "1.0.0",
151
+ title: "Granite Docling API",
152
+ description: "API for converting images to docling format using the Granite Docling model",
153
+ },
154
+ servers: [{ url: "http://localhost:3000", description: "Development server" }],
155
+ });
156
+
157
+ app.get("/docs", swaggerUI({ url: "/openapi.json" }));
158
+ app.get("/", (c) => c.redirect("/docs"));
159
+
160
+ app.onError((err, c) => {
161
+ console.error(`${err}`);
162
+ return c.json({ success: false, error: err.message || "Internal server error", code: "INTERNAL_ERROR" }, 500);
163
+ });
164
+
165
+ app.notFound((c) =>
166
+ c.json({ success: false, error: "Endpoint not found", code: "NOT_FOUND" }, 404)
167
+ );
168
+
169
+ // ── Start ─────────────────────────────────────────────────────────────────────
170
+
171
+ // Workers runtime uses the default export; Node.js uses serve().
172
+ export default app;
173
+
174
+ if (typeof process !== "undefined" && process.env.NODE_ENV !== "worker") {
175
+ initializeModel().catch(console.error);
176
+ const port = process.env.PORT || 3000;
177
+ console.log(`Server is running on port ${port}`);
178
+ console.log(`OpenAPI documentation available at http://localhost:${port}/docs`);
179
+ serve({ fetch: app.fetch, port });
180
+ }
@@ -0,0 +1,18 @@
1
+ {
2
+ "$schema": "node_modules/wrangler/config-schema.json",
3
+ "name": "pdf-to-html-docling",
4
+ "main": "server.js",
5
+ "compatibility_date": "2025-05-01",
6
+ // Required for Node.js built-ins (crypto, stream, path, etc.) used by
7
+ // @huggingface/transformers and @hono/node-server.
8
+ "compatibility_flags": ["nodejs_compat_v2"],
9
+ "dev": {
10
+ "port": 3000,
11
+ "local_protocol": "http"
12
+ },
13
+ // Workers free tier: 128 MB. Model inference is heavy — bump this when
14
+ // deploying to a paid account or Workers AI.
15
+ "limits": {
16
+ "cpu_ms": 30000
17
+ }
18
+ }
@@ -0,0 +1,271 @@
1
+ /**
2
+ * @fileoverview OCR of PDF pages with the Granite Docling vision model —
3
+ * either in-process (via the optional `@huggingface/transformers` dependency
4
+ * and the ONNX build of `ibm-granite/granite-docling-258M`) or by POSTing
5
+ * page images to a docling-compatible HTTP processor (the Hono service in
6
+ * this package's `server/` folder, or any other deployment of it).
7
+ *
8
+ * Everything heavy is imported lazily: requiring this module costs nothing
9
+ * until a page is actually OCR'd.
10
+ */
11
+
12
+ /** Hugging Face id of the ONNX Granite Docling build used for local OCR. */
13
+ const MODEL_ID = "onnx-community/granite-docling-258M-ONNX";
14
+
15
+ /** Default instruction sent to the model per page. */
16
+ const DEFAULT_PROMPT = "Convert this page to docling.";
17
+
18
+ /** Options accepted by the docling OCR helpers. */
19
+ export interface DoclingOcrOptions {
20
+ /**
21
+ * Base URL of a remote docling-compatible processor
22
+ * (e.g. `"http://localhost:3000"`). When set, page images are POSTed to
23
+ * `{processorUrl}/api/v1/convert-base64` instead of running the model
24
+ * in-process.
25
+ */
26
+ processorUrl?: string;
27
+ /** Instruction for the model. default="Convert this page to docling." */
28
+ prompt?: string;
29
+ /** Max tokens generated per page. default=4096 */
30
+ maxTokens?: number;
31
+ /** Rasterization scale for PDF pages (1 = 72 DPI). default=2 */
32
+ scale?: number;
33
+ }
34
+
35
+ // ── Local model (optional @huggingface/transformers) ─────────────────────────
36
+
37
+ let localModelPromise: Promise<{ model: any; processor: any; lib: any }> | null =
38
+ null;
39
+
40
+ /** Loads the Granite Docling ONNX model once and caches it. */
41
+ async function loadLocalModel() {
42
+ if (!localModelPromise) {
43
+ localModelPromise = (async () => {
44
+ const lib = await import("@huggingface/transformers");
45
+ const processor = await lib.AutoProcessor.from_pretrained(MODEL_ID);
46
+ const model = await lib.AutoModelForVision2Seq.from_pretrained(MODEL_ID, {
47
+ dtype: "fp32",
48
+ });
49
+ return { model, processor, lib };
50
+ })();
51
+ }
52
+ return localModelPromise;
53
+ }
54
+
55
+ /**
56
+ * OCRs one image with Granite Docling and returns the raw doctags output.
57
+ * Uses the remote processor when `processorUrl` is set, otherwise the local
58
+ * ONNX model.
59
+ *
60
+ * @param imageBase64 - Base64-encoded PNG of the page/image (no data: prefix)
61
+ * @category Extract
62
+ */
63
+ export async function ocrImageWithDocling(
64
+ imageBase64: string,
65
+ options: DoclingOcrOptions = {},
66
+ ): Promise<string> {
67
+ const {
68
+ processorUrl,
69
+ prompt = DEFAULT_PROMPT,
70
+ maxTokens = 4096,
71
+ } = options;
72
+
73
+ if (processorUrl) {
74
+ const response = await fetch(
75
+ `${processorUrl.replace(/\/$/, "")}/api/v1/convert-base64`,
76
+ {
77
+ method: "POST",
78
+ headers: { "Content-Type": "application/json" },
79
+ body: JSON.stringify({
80
+ imageBase64,
81
+ mimeType: "image/png",
82
+ prompt,
83
+ maxTokens,
84
+ }),
85
+ },
86
+ );
87
+ const data: any = await response.json();
88
+ if (!response.ok || !data?.success)
89
+ throw new Error(data?.error || `Processor error ${response.status}`);
90
+ return data.result;
91
+ }
92
+
93
+ const { model, processor, lib } = await loadLocalModel();
94
+ const image = await lib.load_image(`data:image/png;base64,${imageBase64}`);
95
+ const messages = [
96
+ {
97
+ role: "user",
98
+ content: [{ type: "image" }, { type: "text", text: prompt }],
99
+ },
100
+ ];
101
+ const text = processor.apply_chat_template(messages, {
102
+ add_generation_prompt: true,
103
+ });
104
+ const inputs = await processor(text, [image], { do_image_splitting: true });
105
+ const generated_ids = await model.generate({
106
+ ...inputs,
107
+ max_new_tokens: maxTokens,
108
+ });
109
+ const [decoded] = processor.batch_decode(
110
+ generated_ids.slice(null, [inputs.input_ids.dims.at(-1), null]),
111
+ { skip_special_tokens: true },
112
+ );
113
+ return decoded;
114
+ }
115
+
116
+ // ── Page rasterization ───────────────────────────────────────────────────────
117
+
118
+ /** Creates a 2D canvas in whatever environment is available. */
119
+ async function createCanvas(width: number, height: number): Promise<any> {
120
+ if (typeof document !== "undefined") {
121
+ const canvas = document.createElement("canvas");
122
+ canvas.width = width;
123
+ canvas.height = height;
124
+ return canvas;
125
+ }
126
+ try {
127
+ const napi: any = await import("@napi-rs/canvas" as any);
128
+ return napi.createCanvas(width, height);
129
+ } catch {
130
+ /* optional dependency not installed */
131
+ }
132
+ if (typeof OffscreenCanvas !== "undefined")
133
+ return new OffscreenCanvas(width, height);
134
+ throw new Error(
135
+ "No canvas available to rasterize PDF pages — install @napi-rs/canvas in Node.js, or run in a browser/Worker with OffscreenCanvas",
136
+ );
137
+ }
138
+
139
+ /** Encodes a canvas (DOM, Offscreen, or napi) as base64 PNG. */
140
+ async function canvasToPngBase64(canvas: any): Promise<string> {
141
+ if (typeof canvas.toBuffer === "function")
142
+ return canvas.toBuffer("image/png").toString("base64");
143
+ if (typeof canvas.convertToBlob === "function") {
144
+ const blob = await canvas.convertToBlob({ type: "image/png" });
145
+ const bytes = new Uint8Array(await blob.arrayBuffer());
146
+ let binary = "";
147
+ for (const byte of bytes) binary += String.fromCharCode(byte);
148
+ return btoa(binary);
149
+ }
150
+ return canvas.toDataURL("image/png").split(",")[1];
151
+ }
152
+
153
+ /**
154
+ * Rasterizes one PDF.js page object to a base64 PNG.
155
+ * @param page - A page from `pdfDocument.getPage(n)`
156
+ * @category Extract
157
+ */
158
+ export async function renderPdfPageToPngBase64(
159
+ page: any,
160
+ scale = 2,
161
+ ): Promise<string> {
162
+ const viewport = page.getViewport({ scale });
163
+ const canvas = await createCanvas(
164
+ Math.ceil(viewport.width),
165
+ Math.ceil(viewport.height),
166
+ );
167
+ const canvasContext = canvas.getContext("2d");
168
+ await page.render({ canvasContext, viewport }).promise;
169
+ return canvasToPngBase64(canvas);
170
+ }
171
+
172
+ /**
173
+ * OCRs a set of pages from a loaded PDF.js document with Granite Docling.
174
+ *
175
+ * @param pdfDocument - Document from PDF.js `getDocument(...).promise`
176
+ * @param pageNumbers - 1-based page numbers to OCR
177
+ * @returns Map of page number → HTML (converted from the model's doctags);
178
+ * pages whose OCR failed are absent from the map.
179
+ * @category Extract
180
+ */
181
+ export async function ocrPdfPagesWithDocling(
182
+ pdfDocument: any,
183
+ pageNumbers: number[],
184
+ options: DoclingOcrOptions = {},
185
+ ): Promise<Map<number, string>> {
186
+ const results = new Map<number, string>();
187
+ for (const pageNumber of pageNumbers) {
188
+ try {
189
+ const page = await pdfDocument.getPage(pageNumber);
190
+ const imageBase64 = await renderPdfPageToPngBase64(page, options.scale);
191
+ const doctags = await ocrImageWithDocling(imageBase64, options);
192
+ results.set(pageNumber, doctagsToHtml(doctags));
193
+ } catch (error) {
194
+ console.error(`Docling OCR failed for page ${pageNumber}:`, error);
195
+ }
196
+ }
197
+ return results;
198
+ }
199
+
200
+ // ── Doctags → HTML ───────────────────────────────────────────────────────────
201
+
202
+ /** Converts an OTSL table body (`<fcel>a<fcel>b<nl>...`) to an HTML table. */
203
+ function otslToHtmlTable(body: string): string {
204
+ const rows = body
205
+ .split("<nl>")
206
+ .map((row) => row.trim())
207
+ .filter(Boolean);
208
+ const html = rows
209
+ .map((row) => {
210
+ const cells = [...row.matchAll(/<(fcel|ched|rhed|ecel|lcel|ucel|xcel)>([^<]*)/g)];
211
+ if (!cells.length) return "";
212
+ const cellsHtml = cells
213
+ .map(([, kind, content]) => {
214
+ const tag = kind === "ched" || kind === "rhed" ? "th" : "td";
215
+ return `<${tag}>${content.trim()}</${tag}>`;
216
+ })
217
+ .join("");
218
+ return `<tr>${cellsHtml}</tr>`;
219
+ })
220
+ .filter(Boolean)
221
+ .join("");
222
+ return `<table>${html}</table>`;
223
+ }
224
+
225
+ /**
226
+ * Converts Granite Docling's doctags output to plain HTML: strips location
227
+ * tokens, maps doctags elements (section headers, text, lists, code,
228
+ * formulas, captions, OTSL tables) to their HTML equivalents, and drops page
229
+ * furniture (running headers/footers). Best-effort — unknown tags are
230
+ * removed, their text content kept.
231
+ * @category Extract
232
+ */
233
+ export function doctagsToHtml(doctags: string): string {
234
+ let html = (doctags || "")
235
+ // location / special tokens
236
+ .replace(/<\/?doctag>/g, "")
237
+ .replace(/<loc_\d+>/g, "")
238
+ .replace(/<page_break>/g, "")
239
+ .replace(/<end_of_utterance>/g, "")
240
+ // page furniture carries no content value
241
+ .replace(/<page_(?:header|footer)>[\s\S]*?<\/page_(?:header|footer)>/g, "");
242
+
243
+ // OTSL tables
244
+ html = html.replace(/<otsl>([\s\S]*?)<\/otsl>/g, (_, body) =>
245
+ otslToHtmlTable(body),
246
+ );
247
+
248
+ const tagMap: Array<[RegExp, string, string]> = [
249
+ [/<title>([\s\S]*?)<\/title>/g, "<h1>", "</h1>"],
250
+ [/<section_header_level_1>([\s\S]*?)<\/section_header_level_1>/g, "<h2>", "</h2>"],
251
+ [/<section_header_level_2>([\s\S]*?)<\/section_header_level_2>/g, "<h3>", "</h3>"],
252
+ [/<section_header_level_3>([\s\S]*?)<\/section_header_level_3>/g, "<h4>", "</h4>"],
253
+ [/<section_header_level_[4-9]>([\s\S]*?)<\/section_header_level_[4-9]>/g, "<h5>", "</h5>"],
254
+ [/<(?:text|paragraph)>([\s\S]*?)<\/(?:text|paragraph)>/g, "<p>", "</p>"],
255
+ [/<caption>([\s\S]*?)<\/caption>/g, "<figcaption>", "</figcaption>"],
256
+ [/<(?:picture|chart)>([\s\S]*?)<\/(?:picture|chart)>/g, "<figure>", "</figure>"],
257
+ [/<code>([\s\S]*?)<\/code>/g, "<pre><code>", "</code></pre>"],
258
+ [/<formula>([\s\S]*?)<\/formula>/g, '<code class="formula">', "</code>"],
259
+ [/<footnote>([\s\S]*?)<\/footnote>/g, '<p class="footnote">', "</p>"],
260
+ [/<list_item>([\s\S]*?)<\/list_item>/g, "<li>", "</li>"],
261
+ [/<unordered_list>([\s\S]*?)<\/unordered_list>/g, "<ul>", "</ul>"],
262
+ [/<ordered_list>([\s\S]*?)<\/ordered_list>/g, "<ol>", "</ol>"],
263
+ ];
264
+ for (const [regex, open, close] of tagMap)
265
+ html = html.replace(regex, (_, content) => `${open}${content.trim()}${close}`);
266
+
267
+ // Drop any leftover doctags-style tokens, keeping their inner text.
268
+ html = html.replace(/<\/?(?:[a-z][a-z0-9]*_[a-z0-9_]+|otsl|smiles)>/g, "");
269
+
270
+ return html.trim();
271
+ }
@@ -0,0 +1,141 @@
1
+ /**
2
+ * @fileoverview Dependency-free, regex-based scan of per-page PDF text that
3
+ * flags which pages contain infographics, figures, charts, or tables — the
4
+ * pages worth sending through a heavy OCR model (e.g. Granite Docling) instead
5
+ * of the fast text-layer pipeline. Complements `detectPdfNeedsOcr` (which uses
6
+ * LiteParse's native complexity analysis and is Node.js only): this scan runs
7
+ * anywhere on plain strings, so it also powers the `"hybrid"` processor mode
8
+ * of {@link convertPDFToHTML}.
9
+ */
10
+
11
+ /**
12
+ * Captions and labels that indicate a figure/table/graphic on the page,
13
+ * e.g. "Figure 3", "Fig. 2:", "Table IV", "Chart 1", "Infographic 2".
14
+ */
15
+ const CAPTION_REGEX =
16
+ /\b(?:fig(?:ure)?s?|table|chart|graph|diagram|infographic|exhibit|illustration|plate|scheme)\s*\.?\s*(?:\d+(?:\.\d+)*|[ivxlcdm]+)\b/gi;
17
+
18
+ /**
19
+ * A line that reads like a table row: 3+ cells separated by tabs or runs of
20
+ * 2+ spaces (column-aligned text layers keep those gaps).
21
+ */
22
+ const TABLE_ROW_REGEX = /^\s*\S[^\t\n]{0,60}?(?:\t+| {2,})\S[^\t\n]{0,60}?(?:\t+| {2,})\S/;
23
+
24
+ /**
25
+ * A run of 3+ numeric cells (optionally with %, $, commas) in a row — data
26
+ * tables and chart axis labels produce these even without column gaps.
27
+ */
28
+ const NUMERIC_ROW_REGEX = /(?:[-+]?[\d][\d.,]*\s*[%$€£]?\s+){3,}[-+]?[\d]/;
29
+
30
+ /** Characters that show up when a text layer is garbled/unmapped glyphs. */
31
+ const GARBLED_REGEX = /[\uFFFD\u0000-\u0008\u000B\u000C\u000E-\u001F]/g;
32
+
33
+ /** Per-page result of {@link scanPagesForOCR}. */
34
+ export interface PageOcrScan {
35
+ /** 1-based page number. */
36
+ page: number;
37
+ /** True when the page should be routed through OCR. */
38
+ needsOcr: boolean;
39
+ /**
40
+ * Why the page was flagged: `"no-text"`, `"sparse-text"`, `"garbled-text"`,
41
+ * `"figure-caption"`, `"table-caption"`, `"table-rows"`, `"numeric-grid"`.
42
+ * Empty when `needsOcr` is false.
43
+ */
44
+ reasons: string[];
45
+ /** Caption strings matched on the page (e.g. `"Figure 3"`, `"Table 2"`). */
46
+ captions: string[];
47
+ }
48
+
49
+ /** Result of {@link scanPagesForOCR}. */
50
+ export interface OcrScanResult {
51
+ /** True when any page was flagged. */
52
+ needsOcr: boolean;
53
+ /** 1-based page numbers of every flagged page. */
54
+ pagesNeedingOcr: number[];
55
+ /** Per-page detail, one entry per page in input order. */
56
+ pages: PageOcrScan[];
57
+ }
58
+
59
+ /** Options for {@link scanPagesForOCR}. */
60
+ export interface ScanPagesForOCROptions {
61
+ /**
62
+ * Pages with fewer non-whitespace characters than this are flagged as
63
+ * `"sparse-text"` (likely a scanned image or a full-page graphic).
64
+ * default=200
65
+ */
66
+ sparseTextThreshold?: number;
67
+ /** Minimum table-looking lines before flagging `"table-rows"`. default=3 */
68
+ minTableRows?: number;
69
+ /** Minimum numeric-run lines before flagging `"numeric-grid"`. default=2 */
70
+ minNumericRows?: number;
71
+ }
72
+
73
+ /**
74
+ * Scans per-page extracted text with regex heuristics and reports which pages
75
+ * contain infographics/figures/tables (or have no usable text layer) and
76
+ * therefore need to be OCR'd for full fidelity.
77
+ *
78
+ * @param pageTexts - Extracted plain text of each page, in page order
79
+ * @param options - Threshold tuning, see {@link ScanPagesForOCROptions}
80
+ * @category Extract
81
+ */
82
+ export function scanPagesForOCR(
83
+ pageTexts: string[],
84
+ options: ScanPagesForOCROptions = {},
85
+ ): OcrScanResult {
86
+ const {
87
+ sparseTextThreshold = 200,
88
+ minTableRows = 3,
89
+ minNumericRows = 2,
90
+ } = options;
91
+
92
+ const pages: PageOcrScan[] = pageTexts.map((text, index) => {
93
+ const reasons: string[] = [];
94
+ const compact = (text || "").replace(/\s+/g, " ").trim();
95
+
96
+ // Text-layer health: nothing/near-nothing extractable means the page is
97
+ // an image (scan or full-page infographic) as far as pdfjs is concerned.
98
+ if (compact.length === 0) reasons.push("no-text");
99
+ else if (compact.length < sparseTextThreshold) reasons.push("sparse-text");
100
+
101
+ const garbled = (text || "").match(GARBLED_REGEX)?.length || 0;
102
+ if (compact.length > 0 && garbled / compact.length > 0.05)
103
+ reasons.push("garbled-text");
104
+
105
+ // Captions referencing figures/charts vs tables. Only line-leading
106
+ // matches count as flags — a real caption starts its own line, while
107
+ // prose like "see Figure 1" does not mean the figure is on this page.
108
+ const captions = [...(text || "").matchAll(CAPTION_REGEX)].map((m) =>
109
+ m[0].replace(/\s+/g, " ").trim(),
110
+ );
111
+ const leadingCaptions = [
112
+ ...(text || "").matchAll(new RegExp(`^\\s*${CAPTION_REGEX.source}`, "gim")),
113
+ ].map((m) => m[0].trim());
114
+ if (leadingCaptions.some((c) => /^table/i.test(c)))
115
+ reasons.push("table-caption");
116
+ if (leadingCaptions.some((c) => !/^table/i.test(c)))
117
+ reasons.push("figure-caption");
118
+
119
+ // Column-aligned or numeric rows.
120
+ const lines = (text || "").split(/\r?\n/);
121
+ const tableRows = lines.filter((line) => TABLE_ROW_REGEX.test(line)).length;
122
+ const numericRows = lines.filter((line) =>
123
+ NUMERIC_ROW_REGEX.test(line),
124
+ ).length;
125
+ if (tableRows >= minTableRows) reasons.push("table-rows");
126
+ if (numericRows >= minNumericRows) reasons.push("numeric-grid");
127
+
128
+ return {
129
+ page: index + 1,
130
+ needsOcr: reasons.length > 0,
131
+ reasons,
132
+ captions,
133
+ };
134
+ });
135
+
136
+ const pagesNeedingOcr = pages
137
+ .filter((page) => page.needsOcr)
138
+ .map((page) => page.page);
139
+
140
+ return { needsOcr: pagesNeedingOcr.length > 0, pagesNeedingOcr, pages };
141
+ }