extract-pdf 0.1.260 → 0.1.261
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +88 -1
- package/dist/docling-ocr.d.ts +60 -0
- package/dist/ocr-page-scan.d.ts +56 -0
- package/dist/pdf-to-html.cjs.js +1 -1
- package/dist/pdf-to-html.d.ts +42 -2
- package/dist/pdf-to-html.es.js +1 -1
- package/dist/utils/load-pdfjs.d.ts +19 -0
- package/package.json +18 -6
- package/server/model.js +96 -0
- package/server/routes.js +91 -0
- package/server/schemas.js +91 -0
- package/server/server.js +180 -0
- package/server/wrangler.jsonc +18 -0
- package/src/docling-ocr.ts +271 -0
- package/src/ocr-page-scan.ts +141 -0
- package/src/pdf-to-html.ts +121 -9
- package/src/utils/load-pdfjs.ts +32 -0
package/server/server.js
ADDED
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @file server.js
|
|
3
|
+
* @description Entry point: wires middleware, route handlers, OpenAPI docs, and starts the server.
|
|
4
|
+
*/
|
|
5
|
+
import { serve } from "@hono/node-server";
|
|
6
|
+
import { OpenAPIHono } from "@hono/zod-openapi";
|
|
7
|
+
import { swaggerUI } from "@hono/swagger-ui";
|
|
8
|
+
import { cors } from "hono/cors";
|
|
9
|
+
import { logger } from "hono/logger";
|
|
10
|
+
|
|
11
|
+
import { initializeModel, isModelLoaded, generateFromImage, load_image } from "./model.js";
|
|
12
|
+
import { convertImageRoute, convertImageBase64Route, healthRoute } from "./routes.js";
|
|
13
|
+
|
|
14
|
+
const app = new OpenAPIHono();
|
|
15
|
+
|
|
16
|
+
app.use("*", cors());
|
|
17
|
+
app.use("*", logger());
|
|
18
|
+
|
|
19
|
+
/** Timestamp used to calculate uptime for the health endpoint. */
|
|
20
|
+
const startTime = Date.now();
|
|
21
|
+
|
|
22
|
+
// ── Health ────────────────────────────────────────────────────────────────────
|
|
23
|
+
|
|
24
|
+
app.openapi(healthRoute, (c) =>
|
|
25
|
+
c.json({
|
|
26
|
+
status: "healthy",
|
|
27
|
+
modelLoaded: isModelLoaded(),
|
|
28
|
+
uptime: Date.now() - startTime,
|
|
29
|
+
version: "1.0.0",
|
|
30
|
+
})
|
|
31
|
+
);
|
|
32
|
+
|
|
33
|
+
// ── Convert (URL) ─────────────────────────────────────────────────────────────
|
|
34
|
+
|
|
35
|
+
app.openapi(convertImageRoute, async (c) => {
|
|
36
|
+
const startProcessing = Date.now();
|
|
37
|
+
|
|
38
|
+
try {
|
|
39
|
+
const { imageUrl, prompt, maxTokens, streaming } = c.req.valid("json");
|
|
40
|
+
|
|
41
|
+
let image;
|
|
42
|
+
try {
|
|
43
|
+
image = await load_image(imageUrl);
|
|
44
|
+
} catch {
|
|
45
|
+
return c.json({ success: false, error: "Failed to load image from URL", code: "IMAGE_LOAD_ERROR" }, 400);
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
const { generatedText } = await generateFromImage({ image, prompt, maxTokens, streaming });
|
|
49
|
+
|
|
50
|
+
return c.json({
|
|
51
|
+
success: true,
|
|
52
|
+
result: generatedText,
|
|
53
|
+
metadata: { processingTime: Date.now() - startProcessing },
|
|
54
|
+
});
|
|
55
|
+
} catch (error) {
|
|
56
|
+
console.error("Conversion error:", error);
|
|
57
|
+
return c.json({ success: false, error: error.message || "Internal processing error", code: "PROCESSING_ERROR" }, 500);
|
|
58
|
+
}
|
|
59
|
+
});
|
|
60
|
+
|
|
61
|
+
// ── Convert (base64) ──────────────────────────────────────────────────────────
|
|
62
|
+
|
|
63
|
+
app.openapi(convertImageBase64Route, async (c) => {
|
|
64
|
+
const startProcessing = Date.now();
|
|
65
|
+
|
|
66
|
+
try {
|
|
67
|
+
const { imageBase64, mimeType, prompt, maxTokens, streaming } = c.req.valid("json");
|
|
68
|
+
|
|
69
|
+
let image;
|
|
70
|
+
try {
|
|
71
|
+
image = await load_image(`data:${mimeType};base64,${imageBase64}`);
|
|
72
|
+
} catch {
|
|
73
|
+
return c.json({ success: false, error: "Failed to decode base64 image", code: "IMAGE_DECODE_ERROR" }, 400);
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
const { generatedText, generated_ids } = await generateFromImage({ image, prompt, maxTokens, streaming });
|
|
77
|
+
|
|
78
|
+
return c.json({
|
|
79
|
+
success: true,
|
|
80
|
+
result: generatedText,
|
|
81
|
+
metadata: {
|
|
82
|
+
processingTime: Date.now() - startProcessing,
|
|
83
|
+
tokenCount: generated_ids?.dims?.[1],
|
|
84
|
+
},
|
|
85
|
+
});
|
|
86
|
+
} catch (error) {
|
|
87
|
+
console.error("Conversion error:", error);
|
|
88
|
+
return c.json({ success: false, error: error.message || "Internal processing error", code: "PROCESSING_ERROR" }, 500);
|
|
89
|
+
}
|
|
90
|
+
});
|
|
91
|
+
|
|
92
|
+
// ── Streaming SSE endpoint ────────────────────────────────────────────────────
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* POST /api/v1/convert-stream
|
|
96
|
+
* Streams generated tokens via Server-Sent Events.
|
|
97
|
+
* Not registered with OpenAPI because SSE responses fall outside the JSON schema.
|
|
98
|
+
*/
|
|
99
|
+
app.post("/api/v1/convert-stream", async (c) => {
|
|
100
|
+
const { imageUrl, prompt = "Convert this page to docling.", maxTokens = 4096 } = await c.req.json();
|
|
101
|
+
|
|
102
|
+
c.header("Content-Type", "text/event-stream");
|
|
103
|
+
c.header("Cache-Control", "no-cache");
|
|
104
|
+
c.header("Connection", "keep-alive");
|
|
105
|
+
|
|
106
|
+
const encoder = new TextEncoder();
|
|
107
|
+
const stream = new ReadableStream({
|
|
108
|
+
async start(controller) {
|
|
109
|
+
const send = (payload) =>
|
|
110
|
+
controller.enqueue(encoder.encode(`data: ${JSON.stringify(payload)}\n\n`));
|
|
111
|
+
try {
|
|
112
|
+
const { model, processor } = await initializeModel();
|
|
113
|
+
const image = await load_image(imageUrl);
|
|
114
|
+
|
|
115
|
+
const messages = [{
|
|
116
|
+
role: "user",
|
|
117
|
+
content: [{ type: "image" }, { type: "text", text: prompt }],
|
|
118
|
+
}];
|
|
119
|
+
const text = processor.apply_chat_template(messages, { add_generation_prompt: true });
|
|
120
|
+
const inputs = await processor(text, [image], { do_image_splitting: true });
|
|
121
|
+
|
|
122
|
+
let fullText = "";
|
|
123
|
+
const { TextStreamer } = await import("@huggingface/transformers");
|
|
124
|
+
const streamer = new TextStreamer(processor.tokenizer, {
|
|
125
|
+
skip_prompt: true,
|
|
126
|
+
skip_special_tokens: false,
|
|
127
|
+
on_finalized_text: (chunk) => {
|
|
128
|
+
fullText += chunk;
|
|
129
|
+
send({ text: chunk, done: false });
|
|
130
|
+
},
|
|
131
|
+
});
|
|
132
|
+
|
|
133
|
+
await model.generate({ ...inputs, max_new_tokens: maxTokens, streamer });
|
|
134
|
+
send({ text: "", done: true, fullText });
|
|
135
|
+
} catch (error) {
|
|
136
|
+
send({ error: error.message, done: true });
|
|
137
|
+
}
|
|
138
|
+
controller.close();
|
|
139
|
+
},
|
|
140
|
+
});
|
|
141
|
+
|
|
142
|
+
return new Response(stream);
|
|
143
|
+
});
|
|
144
|
+
|
|
145
|
+
// ── Docs & error handling ─────────────────────────────────────────────────────
|
|
146
|
+
|
|
147
|
+
app.doc("/openapi.json", {
|
|
148
|
+
openapi: "3.0.0",
|
|
149
|
+
info: {
|
|
150
|
+
version: "1.0.0",
|
|
151
|
+
title: "Granite Docling API",
|
|
152
|
+
description: "API for converting images to docling format using the Granite Docling model",
|
|
153
|
+
},
|
|
154
|
+
servers: [{ url: "http://localhost:3000", description: "Development server" }],
|
|
155
|
+
});
|
|
156
|
+
|
|
157
|
+
app.get("/docs", swaggerUI({ url: "/openapi.json" }));
|
|
158
|
+
app.get("/", (c) => c.redirect("/docs"));
|
|
159
|
+
|
|
160
|
+
app.onError((err, c) => {
|
|
161
|
+
console.error(`${err}`);
|
|
162
|
+
return c.json({ success: false, error: err.message || "Internal server error", code: "INTERNAL_ERROR" }, 500);
|
|
163
|
+
});
|
|
164
|
+
|
|
165
|
+
app.notFound((c) =>
|
|
166
|
+
c.json({ success: false, error: "Endpoint not found", code: "NOT_FOUND" }, 404)
|
|
167
|
+
);
|
|
168
|
+
|
|
169
|
+
// ── Start ─────────────────────────────────────────────────────────────────────
|
|
170
|
+
|
|
171
|
+
// Workers runtime uses the default export; Node.js uses serve().
|
|
172
|
+
export default app;
|
|
173
|
+
|
|
174
|
+
if (typeof process !== "undefined" && process.env.NODE_ENV !== "worker") {
|
|
175
|
+
initializeModel().catch(console.error);
|
|
176
|
+
const port = process.env.PORT || 3000;
|
|
177
|
+
console.log(`Server is running on port ${port}`);
|
|
178
|
+
console.log(`OpenAPI documentation available at http://localhost:${port}/docs`);
|
|
179
|
+
serve({ fetch: app.fetch, port });
|
|
180
|
+
}
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "node_modules/wrangler/config-schema.json",
|
|
3
|
+
"name": "pdf-to-html-docling",
|
|
4
|
+
"main": "server.js",
|
|
5
|
+
"compatibility_date": "2025-05-01",
|
|
6
|
+
// Required for Node.js built-ins (crypto, stream, path, etc.) used by
|
|
7
|
+
// @huggingface/transformers and @hono/node-server.
|
|
8
|
+
"compatibility_flags": ["nodejs_compat_v2"],
|
|
9
|
+
"dev": {
|
|
10
|
+
"port": 3000,
|
|
11
|
+
"local_protocol": "http"
|
|
12
|
+
},
|
|
13
|
+
// Workers free tier: 128 MB. Model inference is heavy — bump this when
|
|
14
|
+
// deploying to a paid account or Workers AI.
|
|
15
|
+
"limits": {
|
|
16
|
+
"cpu_ms": 30000
|
|
17
|
+
}
|
|
18
|
+
}
|
|
@@ -0,0 +1,271 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview OCR of PDF pages with the Granite Docling vision model —
|
|
3
|
+
* either in-process (via the optional `@huggingface/transformers` dependency
|
|
4
|
+
* and the ONNX build of `ibm-granite/granite-docling-258M`) or by POSTing
|
|
5
|
+
* page images to a docling-compatible HTTP processor (the Hono service in
|
|
6
|
+
* this package's `server/` folder, or any other deployment of it).
|
|
7
|
+
*
|
|
8
|
+
* Everything heavy is imported lazily: requiring this module costs nothing
|
|
9
|
+
* until a page is actually OCR'd.
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
/** Hugging Face id of the ONNX Granite Docling build used for local OCR. */
|
|
13
|
+
const MODEL_ID = "onnx-community/granite-docling-258M-ONNX";
|
|
14
|
+
|
|
15
|
+
/** Default instruction sent to the model per page. */
|
|
16
|
+
const DEFAULT_PROMPT = "Convert this page to docling.";
|
|
17
|
+
|
|
18
|
+
/** Options accepted by the docling OCR helpers. */
|
|
19
|
+
export interface DoclingOcrOptions {
|
|
20
|
+
/**
|
|
21
|
+
* Base URL of a remote docling-compatible processor
|
|
22
|
+
* (e.g. `"http://localhost:3000"`). When set, page images are POSTed to
|
|
23
|
+
* `{processorUrl}/api/v1/convert-base64` instead of running the model
|
|
24
|
+
* in-process.
|
|
25
|
+
*/
|
|
26
|
+
processorUrl?: string;
|
|
27
|
+
/** Instruction for the model. default="Convert this page to docling." */
|
|
28
|
+
prompt?: string;
|
|
29
|
+
/** Max tokens generated per page. default=4096 */
|
|
30
|
+
maxTokens?: number;
|
|
31
|
+
/** Rasterization scale for PDF pages (1 = 72 DPI). default=2 */
|
|
32
|
+
scale?: number;
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
// ── Local model (optional @huggingface/transformers) ─────────────────────────
|
|
36
|
+
|
|
37
|
+
let localModelPromise: Promise<{ model: any; processor: any; lib: any }> | null =
|
|
38
|
+
null;
|
|
39
|
+
|
|
40
|
+
/** Loads the Granite Docling ONNX model once and caches it. */
|
|
41
|
+
async function loadLocalModel() {
|
|
42
|
+
if (!localModelPromise) {
|
|
43
|
+
localModelPromise = (async () => {
|
|
44
|
+
const lib = await import("@huggingface/transformers");
|
|
45
|
+
const processor = await lib.AutoProcessor.from_pretrained(MODEL_ID);
|
|
46
|
+
const model = await lib.AutoModelForVision2Seq.from_pretrained(MODEL_ID, {
|
|
47
|
+
dtype: "fp32",
|
|
48
|
+
});
|
|
49
|
+
return { model, processor, lib };
|
|
50
|
+
})();
|
|
51
|
+
}
|
|
52
|
+
return localModelPromise;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* OCRs one image with Granite Docling and returns the raw doctags output.
|
|
57
|
+
* Uses the remote processor when `processorUrl` is set, otherwise the local
|
|
58
|
+
* ONNX model.
|
|
59
|
+
*
|
|
60
|
+
* @param imageBase64 - Base64-encoded PNG of the page/image (no data: prefix)
|
|
61
|
+
* @category Extract
|
|
62
|
+
*/
|
|
63
|
+
export async function ocrImageWithDocling(
|
|
64
|
+
imageBase64: string,
|
|
65
|
+
options: DoclingOcrOptions = {},
|
|
66
|
+
): Promise<string> {
|
|
67
|
+
const {
|
|
68
|
+
processorUrl,
|
|
69
|
+
prompt = DEFAULT_PROMPT,
|
|
70
|
+
maxTokens = 4096,
|
|
71
|
+
} = options;
|
|
72
|
+
|
|
73
|
+
if (processorUrl) {
|
|
74
|
+
const response = await fetch(
|
|
75
|
+
`${processorUrl.replace(/\/$/, "")}/api/v1/convert-base64`,
|
|
76
|
+
{
|
|
77
|
+
method: "POST",
|
|
78
|
+
headers: { "Content-Type": "application/json" },
|
|
79
|
+
body: JSON.stringify({
|
|
80
|
+
imageBase64,
|
|
81
|
+
mimeType: "image/png",
|
|
82
|
+
prompt,
|
|
83
|
+
maxTokens,
|
|
84
|
+
}),
|
|
85
|
+
},
|
|
86
|
+
);
|
|
87
|
+
const data: any = await response.json();
|
|
88
|
+
if (!response.ok || !data?.success)
|
|
89
|
+
throw new Error(data?.error || `Processor error ${response.status}`);
|
|
90
|
+
return data.result;
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
const { model, processor, lib } = await loadLocalModel();
|
|
94
|
+
const image = await lib.load_image(`data:image/png;base64,${imageBase64}`);
|
|
95
|
+
const messages = [
|
|
96
|
+
{
|
|
97
|
+
role: "user",
|
|
98
|
+
content: [{ type: "image" }, { type: "text", text: prompt }],
|
|
99
|
+
},
|
|
100
|
+
];
|
|
101
|
+
const text = processor.apply_chat_template(messages, {
|
|
102
|
+
add_generation_prompt: true,
|
|
103
|
+
});
|
|
104
|
+
const inputs = await processor(text, [image], { do_image_splitting: true });
|
|
105
|
+
const generated_ids = await model.generate({
|
|
106
|
+
...inputs,
|
|
107
|
+
max_new_tokens: maxTokens,
|
|
108
|
+
});
|
|
109
|
+
const [decoded] = processor.batch_decode(
|
|
110
|
+
generated_ids.slice(null, [inputs.input_ids.dims.at(-1), null]),
|
|
111
|
+
{ skip_special_tokens: true },
|
|
112
|
+
);
|
|
113
|
+
return decoded;
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
// ── Page rasterization ───────────────────────────────────────────────────────
|
|
117
|
+
|
|
118
|
+
/** Creates a 2D canvas in whatever environment is available. */
|
|
119
|
+
async function createCanvas(width: number, height: number): Promise<any> {
|
|
120
|
+
if (typeof document !== "undefined") {
|
|
121
|
+
const canvas = document.createElement("canvas");
|
|
122
|
+
canvas.width = width;
|
|
123
|
+
canvas.height = height;
|
|
124
|
+
return canvas;
|
|
125
|
+
}
|
|
126
|
+
try {
|
|
127
|
+
const napi: any = await import("@napi-rs/canvas" as any);
|
|
128
|
+
return napi.createCanvas(width, height);
|
|
129
|
+
} catch {
|
|
130
|
+
/* optional dependency not installed */
|
|
131
|
+
}
|
|
132
|
+
if (typeof OffscreenCanvas !== "undefined")
|
|
133
|
+
return new OffscreenCanvas(width, height);
|
|
134
|
+
throw new Error(
|
|
135
|
+
"No canvas available to rasterize PDF pages — install @napi-rs/canvas in Node.js, or run in a browser/Worker with OffscreenCanvas",
|
|
136
|
+
);
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
/** Encodes a canvas (DOM, Offscreen, or napi) as base64 PNG. */
|
|
140
|
+
async function canvasToPngBase64(canvas: any): Promise<string> {
|
|
141
|
+
if (typeof canvas.toBuffer === "function")
|
|
142
|
+
return canvas.toBuffer("image/png").toString("base64");
|
|
143
|
+
if (typeof canvas.convertToBlob === "function") {
|
|
144
|
+
const blob = await canvas.convertToBlob({ type: "image/png" });
|
|
145
|
+
const bytes = new Uint8Array(await blob.arrayBuffer());
|
|
146
|
+
let binary = "";
|
|
147
|
+
for (const byte of bytes) binary += String.fromCharCode(byte);
|
|
148
|
+
return btoa(binary);
|
|
149
|
+
}
|
|
150
|
+
return canvas.toDataURL("image/png").split(",")[1];
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
/**
|
|
154
|
+
* Rasterizes one PDF.js page object to a base64 PNG.
|
|
155
|
+
* @param page - A page from `pdfDocument.getPage(n)`
|
|
156
|
+
* @category Extract
|
|
157
|
+
*/
|
|
158
|
+
export async function renderPdfPageToPngBase64(
|
|
159
|
+
page: any,
|
|
160
|
+
scale = 2,
|
|
161
|
+
): Promise<string> {
|
|
162
|
+
const viewport = page.getViewport({ scale });
|
|
163
|
+
const canvas = await createCanvas(
|
|
164
|
+
Math.ceil(viewport.width),
|
|
165
|
+
Math.ceil(viewport.height),
|
|
166
|
+
);
|
|
167
|
+
const canvasContext = canvas.getContext("2d");
|
|
168
|
+
await page.render({ canvasContext, viewport }).promise;
|
|
169
|
+
return canvasToPngBase64(canvas);
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
/**
|
|
173
|
+
* OCRs a set of pages from a loaded PDF.js document with Granite Docling.
|
|
174
|
+
*
|
|
175
|
+
* @param pdfDocument - Document from PDF.js `getDocument(...).promise`
|
|
176
|
+
* @param pageNumbers - 1-based page numbers to OCR
|
|
177
|
+
* @returns Map of page number → HTML (converted from the model's doctags);
|
|
178
|
+
* pages whose OCR failed are absent from the map.
|
|
179
|
+
* @category Extract
|
|
180
|
+
*/
|
|
181
|
+
export async function ocrPdfPagesWithDocling(
|
|
182
|
+
pdfDocument: any,
|
|
183
|
+
pageNumbers: number[],
|
|
184
|
+
options: DoclingOcrOptions = {},
|
|
185
|
+
): Promise<Map<number, string>> {
|
|
186
|
+
const results = new Map<number, string>();
|
|
187
|
+
for (const pageNumber of pageNumbers) {
|
|
188
|
+
try {
|
|
189
|
+
const page = await pdfDocument.getPage(pageNumber);
|
|
190
|
+
const imageBase64 = await renderPdfPageToPngBase64(page, options.scale);
|
|
191
|
+
const doctags = await ocrImageWithDocling(imageBase64, options);
|
|
192
|
+
results.set(pageNumber, doctagsToHtml(doctags));
|
|
193
|
+
} catch (error) {
|
|
194
|
+
console.error(`Docling OCR failed for page ${pageNumber}:`, error);
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
return results;
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
// ── Doctags → HTML ───────────────────────────────────────────────────────────
|
|
201
|
+
|
|
202
|
+
/** Converts an OTSL table body (`<fcel>a<fcel>b<nl>...`) to an HTML table. */
|
|
203
|
+
function otslToHtmlTable(body: string): string {
|
|
204
|
+
const rows = body
|
|
205
|
+
.split("<nl>")
|
|
206
|
+
.map((row) => row.trim())
|
|
207
|
+
.filter(Boolean);
|
|
208
|
+
const html = rows
|
|
209
|
+
.map((row) => {
|
|
210
|
+
const cells = [...row.matchAll(/<(fcel|ched|rhed|ecel|lcel|ucel|xcel)>([^<]*)/g)];
|
|
211
|
+
if (!cells.length) return "";
|
|
212
|
+
const cellsHtml = cells
|
|
213
|
+
.map(([, kind, content]) => {
|
|
214
|
+
const tag = kind === "ched" || kind === "rhed" ? "th" : "td";
|
|
215
|
+
return `<${tag}>${content.trim()}</${tag}>`;
|
|
216
|
+
})
|
|
217
|
+
.join("");
|
|
218
|
+
return `<tr>${cellsHtml}</tr>`;
|
|
219
|
+
})
|
|
220
|
+
.filter(Boolean)
|
|
221
|
+
.join("");
|
|
222
|
+
return `<table>${html}</table>`;
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
/**
|
|
226
|
+
* Converts Granite Docling's doctags output to plain HTML: strips location
|
|
227
|
+
* tokens, maps doctags elements (section headers, text, lists, code,
|
|
228
|
+
* formulas, captions, OTSL tables) to their HTML equivalents, and drops page
|
|
229
|
+
* furniture (running headers/footers). Best-effort — unknown tags are
|
|
230
|
+
* removed, their text content kept.
|
|
231
|
+
* @category Extract
|
|
232
|
+
*/
|
|
233
|
+
export function doctagsToHtml(doctags: string): string {
|
|
234
|
+
let html = (doctags || "")
|
|
235
|
+
// location / special tokens
|
|
236
|
+
.replace(/<\/?doctag>/g, "")
|
|
237
|
+
.replace(/<loc_\d+>/g, "")
|
|
238
|
+
.replace(/<page_break>/g, "")
|
|
239
|
+
.replace(/<end_of_utterance>/g, "")
|
|
240
|
+
// page furniture carries no content value
|
|
241
|
+
.replace(/<page_(?:header|footer)>[\s\S]*?<\/page_(?:header|footer)>/g, "");
|
|
242
|
+
|
|
243
|
+
// OTSL tables
|
|
244
|
+
html = html.replace(/<otsl>([\s\S]*?)<\/otsl>/g, (_, body) =>
|
|
245
|
+
otslToHtmlTable(body),
|
|
246
|
+
);
|
|
247
|
+
|
|
248
|
+
const tagMap: Array<[RegExp, string, string]> = [
|
|
249
|
+
[/<title>([\s\S]*?)<\/title>/g, "<h1>", "</h1>"],
|
|
250
|
+
[/<section_header_level_1>([\s\S]*?)<\/section_header_level_1>/g, "<h2>", "</h2>"],
|
|
251
|
+
[/<section_header_level_2>([\s\S]*?)<\/section_header_level_2>/g, "<h3>", "</h3>"],
|
|
252
|
+
[/<section_header_level_3>([\s\S]*?)<\/section_header_level_3>/g, "<h4>", "</h4>"],
|
|
253
|
+
[/<section_header_level_[4-9]>([\s\S]*?)<\/section_header_level_[4-9]>/g, "<h5>", "</h5>"],
|
|
254
|
+
[/<(?:text|paragraph)>([\s\S]*?)<\/(?:text|paragraph)>/g, "<p>", "</p>"],
|
|
255
|
+
[/<caption>([\s\S]*?)<\/caption>/g, "<figcaption>", "</figcaption>"],
|
|
256
|
+
[/<(?:picture|chart)>([\s\S]*?)<\/(?:picture|chart)>/g, "<figure>", "</figure>"],
|
|
257
|
+
[/<code>([\s\S]*?)<\/code>/g, "<pre><code>", "</code></pre>"],
|
|
258
|
+
[/<formula>([\s\S]*?)<\/formula>/g, '<code class="formula">', "</code>"],
|
|
259
|
+
[/<footnote>([\s\S]*?)<\/footnote>/g, '<p class="footnote">', "</p>"],
|
|
260
|
+
[/<list_item>([\s\S]*?)<\/list_item>/g, "<li>", "</li>"],
|
|
261
|
+
[/<unordered_list>([\s\S]*?)<\/unordered_list>/g, "<ul>", "</ul>"],
|
|
262
|
+
[/<ordered_list>([\s\S]*?)<\/ordered_list>/g, "<ol>", "</ol>"],
|
|
263
|
+
];
|
|
264
|
+
for (const [regex, open, close] of tagMap)
|
|
265
|
+
html = html.replace(regex, (_, content) => `${open}${content.trim()}${close}`);
|
|
266
|
+
|
|
267
|
+
// Drop any leftover doctags-style tokens, keeping their inner text.
|
|
268
|
+
html = html.replace(/<\/?(?:[a-z][a-z0-9]*_[a-z0-9_]+|otsl|smiles)>/g, "");
|
|
269
|
+
|
|
270
|
+
return html.trim();
|
|
271
|
+
}
|
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Dependency-free, regex-based scan of per-page PDF text that
|
|
3
|
+
* flags which pages contain infographics, figures, charts, or tables — the
|
|
4
|
+
* pages worth sending through a heavy OCR model (e.g. Granite Docling) instead
|
|
5
|
+
* of the fast text-layer pipeline. Complements `detectPdfNeedsOcr` (which uses
|
|
6
|
+
* LiteParse's native complexity analysis and is Node.js only): this scan runs
|
|
7
|
+
* anywhere on plain strings, so it also powers the `"hybrid"` processor mode
|
|
8
|
+
* of {@link convertPDFToHTML}.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* Captions and labels that indicate a figure/table/graphic on the page,
|
|
13
|
+
* e.g. "Figure 3", "Fig. 2:", "Table IV", "Chart 1", "Infographic 2".
|
|
14
|
+
*/
|
|
15
|
+
const CAPTION_REGEX =
|
|
16
|
+
/\b(?:fig(?:ure)?s?|table|chart|graph|diagram|infographic|exhibit|illustration|plate|scheme)\s*\.?\s*(?:\d+(?:\.\d+)*|[ivxlcdm]+)\b/gi;
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* A line that reads like a table row: 3+ cells separated by tabs or runs of
|
|
20
|
+
* 2+ spaces (column-aligned text layers keep those gaps).
|
|
21
|
+
*/
|
|
22
|
+
const TABLE_ROW_REGEX = /^\s*\S[^\t\n]{0,60}?(?:\t+| {2,})\S[^\t\n]{0,60}?(?:\t+| {2,})\S/;
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* A run of 3+ numeric cells (optionally with %, $, commas) in a row — data
|
|
26
|
+
* tables and chart axis labels produce these even without column gaps.
|
|
27
|
+
*/
|
|
28
|
+
const NUMERIC_ROW_REGEX = /(?:[-+]?[\d][\d.,]*\s*[%$€£]?\s+){3,}[-+]?[\d]/;
|
|
29
|
+
|
|
30
|
+
/** Characters that show up when a text layer is garbled/unmapped glyphs. */
|
|
31
|
+
const GARBLED_REGEX = /[\uFFFD\u0000-\u0008\u000B\u000C\u000E-\u001F]/g;
|
|
32
|
+
|
|
33
|
+
/** Per-page result of {@link scanPagesForOCR}. */
|
|
34
|
+
export interface PageOcrScan {
|
|
35
|
+
/** 1-based page number. */
|
|
36
|
+
page: number;
|
|
37
|
+
/** True when the page should be routed through OCR. */
|
|
38
|
+
needsOcr: boolean;
|
|
39
|
+
/**
|
|
40
|
+
* Why the page was flagged: `"no-text"`, `"sparse-text"`, `"garbled-text"`,
|
|
41
|
+
* `"figure-caption"`, `"table-caption"`, `"table-rows"`, `"numeric-grid"`.
|
|
42
|
+
* Empty when `needsOcr` is false.
|
|
43
|
+
*/
|
|
44
|
+
reasons: string[];
|
|
45
|
+
/** Caption strings matched on the page (e.g. `"Figure 3"`, `"Table 2"`). */
|
|
46
|
+
captions: string[];
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
/** Result of {@link scanPagesForOCR}. */
|
|
50
|
+
export interface OcrScanResult {
|
|
51
|
+
/** True when any page was flagged. */
|
|
52
|
+
needsOcr: boolean;
|
|
53
|
+
/** 1-based page numbers of every flagged page. */
|
|
54
|
+
pagesNeedingOcr: number[];
|
|
55
|
+
/** Per-page detail, one entry per page in input order. */
|
|
56
|
+
pages: PageOcrScan[];
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/** Options for {@link scanPagesForOCR}. */
|
|
60
|
+
export interface ScanPagesForOCROptions {
|
|
61
|
+
/**
|
|
62
|
+
* Pages with fewer non-whitespace characters than this are flagged as
|
|
63
|
+
* `"sparse-text"` (likely a scanned image or a full-page graphic).
|
|
64
|
+
* default=200
|
|
65
|
+
*/
|
|
66
|
+
sparseTextThreshold?: number;
|
|
67
|
+
/** Minimum table-looking lines before flagging `"table-rows"`. default=3 */
|
|
68
|
+
minTableRows?: number;
|
|
69
|
+
/** Minimum numeric-run lines before flagging `"numeric-grid"`. default=2 */
|
|
70
|
+
minNumericRows?: number;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/**
|
|
74
|
+
* Scans per-page extracted text with regex heuristics and reports which pages
|
|
75
|
+
* contain infographics/figures/tables (or have no usable text layer) and
|
|
76
|
+
* therefore need to be OCR'd for full fidelity.
|
|
77
|
+
*
|
|
78
|
+
* @param pageTexts - Extracted plain text of each page, in page order
|
|
79
|
+
* @param options - Threshold tuning, see {@link ScanPagesForOCROptions}
|
|
80
|
+
* @category Extract
|
|
81
|
+
*/
|
|
82
|
+
export function scanPagesForOCR(
|
|
83
|
+
pageTexts: string[],
|
|
84
|
+
options: ScanPagesForOCROptions = {},
|
|
85
|
+
): OcrScanResult {
|
|
86
|
+
const {
|
|
87
|
+
sparseTextThreshold = 200,
|
|
88
|
+
minTableRows = 3,
|
|
89
|
+
minNumericRows = 2,
|
|
90
|
+
} = options;
|
|
91
|
+
|
|
92
|
+
const pages: PageOcrScan[] = pageTexts.map((text, index) => {
|
|
93
|
+
const reasons: string[] = [];
|
|
94
|
+
const compact = (text || "").replace(/\s+/g, " ").trim();
|
|
95
|
+
|
|
96
|
+
// Text-layer health: nothing/near-nothing extractable means the page is
|
|
97
|
+
// an image (scan or full-page infographic) as far as pdfjs is concerned.
|
|
98
|
+
if (compact.length === 0) reasons.push("no-text");
|
|
99
|
+
else if (compact.length < sparseTextThreshold) reasons.push("sparse-text");
|
|
100
|
+
|
|
101
|
+
const garbled = (text || "").match(GARBLED_REGEX)?.length || 0;
|
|
102
|
+
if (compact.length > 0 && garbled / compact.length > 0.05)
|
|
103
|
+
reasons.push("garbled-text");
|
|
104
|
+
|
|
105
|
+
// Captions referencing figures/charts vs tables. Only line-leading
|
|
106
|
+
// matches count as flags — a real caption starts its own line, while
|
|
107
|
+
// prose like "see Figure 1" does not mean the figure is on this page.
|
|
108
|
+
const captions = [...(text || "").matchAll(CAPTION_REGEX)].map((m) =>
|
|
109
|
+
m[0].replace(/\s+/g, " ").trim(),
|
|
110
|
+
);
|
|
111
|
+
const leadingCaptions = [
|
|
112
|
+
...(text || "").matchAll(new RegExp(`^\\s*${CAPTION_REGEX.source}`, "gim")),
|
|
113
|
+
].map((m) => m[0].trim());
|
|
114
|
+
if (leadingCaptions.some((c) => /^table/i.test(c)))
|
|
115
|
+
reasons.push("table-caption");
|
|
116
|
+
if (leadingCaptions.some((c) => !/^table/i.test(c)))
|
|
117
|
+
reasons.push("figure-caption");
|
|
118
|
+
|
|
119
|
+
// Column-aligned or numeric rows.
|
|
120
|
+
const lines = (text || "").split(/\r?\n/);
|
|
121
|
+
const tableRows = lines.filter((line) => TABLE_ROW_REGEX.test(line)).length;
|
|
122
|
+
const numericRows = lines.filter((line) =>
|
|
123
|
+
NUMERIC_ROW_REGEX.test(line),
|
|
124
|
+
).length;
|
|
125
|
+
if (tableRows >= minTableRows) reasons.push("table-rows");
|
|
126
|
+
if (numericRows >= minNumericRows) reasons.push("numeric-grid");
|
|
127
|
+
|
|
128
|
+
return {
|
|
129
|
+
page: index + 1,
|
|
130
|
+
needsOcr: reasons.length > 0,
|
|
131
|
+
reasons,
|
|
132
|
+
captions,
|
|
133
|
+
};
|
|
134
|
+
});
|
|
135
|
+
|
|
136
|
+
const pagesNeedingOcr = pages
|
|
137
|
+
.filter((page) => page.needsOcr)
|
|
138
|
+
.map((page) => page.page);
|
|
139
|
+
|
|
140
|
+
return { needsOcr: pagesNeedingOcr.length > 0, pagesNeedingOcr, pages };
|
|
141
|
+
}
|