extract-pdf 0.1.260 → 0.1.262
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +88 -1
- package/dist/docling-ocr.d.ts +60 -0
- package/dist/ocr-page-scan.d.ts +56 -0
- package/dist/pdf-to-html.cjs.js +1 -1
- package/dist/pdf-to-html.d.ts +42 -2
- package/dist/pdf-to-html.es.js +1 -1
- package/dist/utils/load-pdfjs.d.ts +19 -0
- package/package.json +18 -6
- package/server/model.js +96 -0
- package/server/routes.js +91 -0
- package/server/schemas.js +91 -0
- package/server/server.js +180 -0
- package/server/wrangler.jsonc +18 -0
- package/src/docling-ocr.ts +271 -0
- package/src/ocr-page-scan.ts +141 -0
- package/src/pdf-to-html.ts +121 -9
- package/src/utils/load-pdfjs.ts +32 -0
package/src/pdf-to-html.ts
CHANGED
|
@@ -33,6 +33,34 @@ import {
|
|
|
33
33
|
convertPDFToHTMLWithLiteParseWasm,
|
|
34
34
|
type LiteParseWasmHTMLOptions,
|
|
35
35
|
} from "./liteparse-wasm-to-html";
|
|
36
|
+
import { loadPdfJs } from "./utils/load-pdfjs";
|
|
37
|
+
import {
|
|
38
|
+
scanPagesForOCR,
|
|
39
|
+
type OcrScanResult,
|
|
40
|
+
type ScanPagesForOCROptions,
|
|
41
|
+
} from "./ocr-page-scan";
|
|
42
|
+
import {
|
|
43
|
+
ocrPdfPagesWithDocling,
|
|
44
|
+
type DoclingOcrOptions,
|
|
45
|
+
} from "./docling-ocr";
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* Where the OCR-capable Docling processing happens for
|
|
49
|
+
* {@link convertPDFToHTML}.
|
|
50
|
+
* - `"frontend"` (default) — all pages parsed by the pure-JS text-layer
|
|
51
|
+
* pipeline; no OCR, no model, works everywhere.
|
|
52
|
+
* - `"hybrid"` — every page goes through the JS pipeline, then a regex scan
|
|
53
|
+
* ({@link scanPagesForOCR}) flags pages containing infographics/figures/
|
|
54
|
+
* tables (or with no usable text layer) and only those pages are re-done
|
|
55
|
+
* with the Granite Docling OCR model.
|
|
56
|
+
* - `"docling"` — every page is rasterized and OCR'd with Granite Docling
|
|
57
|
+
* (in-process via the optional `@huggingface/transformers` dependency, or
|
|
58
|
+
* remotely when `processorUrl` is set).
|
|
59
|
+
* - any `http(s)://` URL — like `"docling"`, but all pages are sent to that
|
|
60
|
+
* docling-compatible processor API (this package's `server/`, or another
|
|
61
|
+
* deployment).
|
|
62
|
+
*/
|
|
63
|
+
export type ProcessorMode = "frontend" | "hybrid" | "docling" | (string & {});
|
|
36
64
|
|
|
37
65
|
/**
|
|
38
66
|
* Which parsing engine {@link convertPDFToHTML} runs.
|
|
@@ -65,6 +93,16 @@ export type ParseMethod = "ts-block-algorithm" | "liteparse" | "liteparse-wasm";
|
|
|
65
93
|
* @param {boolean} options.removePageHeaders default=true - Removes repeated headers found on each page
|
|
66
94
|
* @param {ParseMethod} options.method default="ts-block-algorithm" - Parsing engine to use;
|
|
67
95
|
* `"liteparse"` delegates to LiteParse (Node.js only, see {@link ParseMethod})
|
|
96
|
+
* @param {ProcessorMode} options.processor default="frontend" - Where OCR happens:
|
|
97
|
+
* `"frontend"` (all JS, no OCR), `"hybrid"` (regex-scan pages for
|
|
98
|
+
* infographics/tables and OCR only those), `"docling"` (OCR every page), or
|
|
99
|
+
* the URL of a docling-compatible processor API (see {@link ProcessorMode})
|
|
100
|
+
* @param {string} options.processorUrl - Remote docling-compatible API base URL
|
|
101
|
+
* used by `"hybrid"`/`"docling"` instead of the in-process model
|
|
102
|
+
* @param {Object} options.ocrScanOptions - Threshold tuning for the hybrid
|
|
103
|
+
* page scan, see {@link ScanPagesForOCROptions}
|
|
104
|
+
* @param {Object} options.doclingOptions - Prompt/maxTokens/scale for the OCR
|
|
105
|
+
* model, see {@link DoclingOcrOptions}
|
|
68
106
|
* @returns {string|Object} HTML formatted text
|
|
69
107
|
* @category Extract
|
|
70
108
|
* @author [vtempest (2025)](https://github.com/vtempest),
|
|
@@ -77,6 +115,10 @@ export async function convertPDFToHTML(
|
|
|
77
115
|
addPageNumbers?: boolean;
|
|
78
116
|
addCitation?: boolean;
|
|
79
117
|
method?: ParseMethod;
|
|
118
|
+
processor?: ProcessorMode;
|
|
119
|
+
processorUrl?: string;
|
|
120
|
+
ocrScanOptions?: ScanPagesForOCROptions;
|
|
121
|
+
doclingOptions?: Omit<DoclingOcrOptions, "processorUrl">;
|
|
80
122
|
} & Pick<LiteParseHTMLOptions, "liteParseOptions"> = {},
|
|
81
123
|
) {
|
|
82
124
|
if (options.method === "liteparse") {
|
|
@@ -95,6 +137,16 @@ export async function convertPDFToHTML(
|
|
|
95
137
|
// try {
|
|
96
138
|
var { addPageNumbers = false, addCitation = true } = options;
|
|
97
139
|
|
|
140
|
+
// Resolve where OCR happens: "frontend" | "hybrid" | "docling" | a
|
|
141
|
+
// processor URL (which means "docling" against that remote API).
|
|
142
|
+
const processor = options.processor ?? "frontend";
|
|
143
|
+
const processorIsUrl = /^https?:\/\//i.test(processor);
|
|
144
|
+
const processorMode = processorIsUrl ? "docling" : processor;
|
|
145
|
+
const doclingOptions: DoclingOcrOptions = {
|
|
146
|
+
...options.doclingOptions,
|
|
147
|
+
processorUrl: processorIsUrl ? processor : options.processorUrl,
|
|
148
|
+
};
|
|
149
|
+
|
|
98
150
|
// pass in databuffer or download all pdf data
|
|
99
151
|
// and convert to array buffer
|
|
100
152
|
var buffer =
|
|
@@ -107,9 +159,7 @@ export async function convertPDFToHTML(
|
|
|
107
159
|
|
|
108
160
|
let pdfDocument;
|
|
109
161
|
try {
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
const { getDocument } = await resolvePDFJS();
|
|
162
|
+
const { getDocument } = await loadPdfJs();
|
|
113
163
|
pdfDocument = await getDocument({
|
|
114
164
|
data: new Uint8Array(buffer),
|
|
115
165
|
useSystemFonts: true,
|
|
@@ -178,6 +228,19 @@ export async function convertPDFToHTML(
|
|
|
178
228
|
pages[page.pageNumber - 1].items = textItems;
|
|
179
229
|
}
|
|
180
230
|
|
|
231
|
+
// Raw per-page text (items grouped into lines by y) captured before the
|
|
232
|
+
// transforms mutate the pages — input for the OCR page scan.
|
|
233
|
+
const pageTexts = pages.map((page) =>
|
|
234
|
+
Object.values(
|
|
235
|
+
(page.items as any[]).reduce((lines: any, item: any) => {
|
|
236
|
+
(lines[item.y] = lines[item.y] || []).push(item.text);
|
|
237
|
+
return lines;
|
|
238
|
+
}, {}),
|
|
239
|
+
)
|
|
240
|
+
.map((line: any) => line.join(" "))
|
|
241
|
+
.join("\n"),
|
|
242
|
+
);
|
|
243
|
+
|
|
181
244
|
var parseResult = new ParseResult({ pages });
|
|
182
245
|
|
|
183
246
|
let lastTransformation: (typeof transformations)[number] | undefined,
|
|
@@ -206,14 +269,35 @@ export async function convertPDFToHTML(
|
|
|
206
269
|
lastTransformation = transformation;
|
|
207
270
|
});
|
|
208
271
|
|
|
209
|
-
var
|
|
210
|
-
|
|
211
|
-
acc +
|
|
272
|
+
var pageHtmls = parseResult.pages.map(
|
|
273
|
+
(page, pageNumber) =>
|
|
212
274
|
`<p id="page-${pageNumber + 1}">${
|
|
213
275
|
addPageNumbers ? ` [${pageNumber + 1}] ` : ""
|
|
214
|
-
}${page.items.join('</p><p id="page-' + pageNumber + '">')}</p
|
|
276
|
+
}${page.items.join('</p><p id="page-' + pageNumber + '">')}</p>`,
|
|
277
|
+
);
|
|
278
|
+
|
|
279
|
+
// Regex scan flagging pages with infographics/figures/tables (or no usable
|
|
280
|
+
// text layer) — the pages worth OCR'ing in hybrid mode.
|
|
281
|
+
const ocrScan = scanPagesForOCR(pageTexts, options.ocrScanOptions);
|
|
282
|
+
|
|
283
|
+
if (processorMode === "docling" || processorMode === "hybrid") {
|
|
284
|
+
const targetPages =
|
|
285
|
+
processorMode === "docling"
|
|
286
|
+
? pageHtmls.map((_, index) => index + 1)
|
|
287
|
+
: ocrScan.pagesNeedingOcr;
|
|
288
|
+
const ocrResults = await ocrPdfPagesWithDocling(
|
|
289
|
+
pdfDocument,
|
|
290
|
+
targetPages,
|
|
291
|
+
doclingOptions,
|
|
215
292
|
);
|
|
216
|
-
|
|
293
|
+
// Pages whose OCR failed keep their frontend-parsed HTML.
|
|
294
|
+
for (const [pageNumber, ocrHtml] of ocrResults)
|
|
295
|
+
pageHtmls[pageNumber - 1] = `<section class="ocr-page" id="page-${pageNumber}">${
|
|
296
|
+
addPageNumbers ? ` [${pageNumber}] ` : ""
|
|
297
|
+
}${ocrHtml}</section>`;
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
var html = pageHtmls.join("");
|
|
217
301
|
|
|
218
302
|
if (addCitation) {
|
|
219
303
|
// Get metadata
|
|
@@ -234,7 +318,21 @@ export async function convertPDFToHTML(
|
|
|
234
318
|
title = html.slice(0, 400).match(/<h[0-9]>(.*?)<\/h[0-9]>/)?.[1] || title;
|
|
235
319
|
}
|
|
236
320
|
|
|
237
|
-
return {
|
|
321
|
+
return {
|
|
322
|
+
author,
|
|
323
|
+
title,
|
|
324
|
+
html,
|
|
325
|
+
format: "pdf",
|
|
326
|
+
processor: processorMode,
|
|
327
|
+
ocrScan,
|
|
328
|
+
} as {
|
|
329
|
+
author?: string;
|
|
330
|
+
title?: string;
|
|
331
|
+
html: string;
|
|
332
|
+
format: string;
|
|
333
|
+
processor: string;
|
|
334
|
+
ocrScan: OcrScanResult;
|
|
335
|
+
};
|
|
238
336
|
}
|
|
239
337
|
|
|
240
338
|
export { convertPDFToHTMLWithLiteParse };
|
|
@@ -246,4 +344,18 @@ export type {
|
|
|
246
344
|
DetectPdfNeedsOcrOptions,
|
|
247
345
|
PdfOcrAssessment,
|
|
248
346
|
} from "./detect-needs-ocr";
|
|
347
|
+
export { scanPagesForOCR } from "./ocr-page-scan";
|
|
348
|
+
export type {
|
|
349
|
+
OcrScanResult,
|
|
350
|
+
PageOcrScan,
|
|
351
|
+
ScanPagesForOCROptions,
|
|
352
|
+
} from "./ocr-page-scan";
|
|
353
|
+
export {
|
|
354
|
+
doctagsToHtml,
|
|
355
|
+
ocrImageWithDocling,
|
|
356
|
+
ocrPdfPagesWithDocling,
|
|
357
|
+
renderPdfPageToPngBase64,
|
|
358
|
+
} from "./docling-ocr";
|
|
359
|
+
export type { DoclingOcrOptions } from "./docling-ocr";
|
|
360
|
+
export { loadPdfJs } from "./utils/load-pdfjs";
|
|
249
361
|
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Lazy loader for the slim serverless PDF.js build.
|
|
3
|
+
*
|
|
4
|
+
* The package deliberately does NOT depend on `pdfjs-dist`: PDF.js is pulled
|
|
5
|
+
* at runtime from jsDelivr's ESM build of
|
|
6
|
+
* [pdfjs-serverless](https://github.com/johannschopplich/pdfjs-serverless) —
|
|
7
|
+
* a zero-dependency, single-file (~1.6 MB minified) redistribution of Mozilla
|
|
8
|
+
* PDF.js that works in Workers/edge runtimes, Node.js, and browsers. Pinned
|
|
9
|
+
* to the major version so compatible patch/minor releases update
|
|
10
|
+
* automatically. This keeps `extract-pdf`'s default export slim: nothing
|
|
11
|
+
* PDF.js-related is bundled or installed until a document is actually parsed.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
const PDFJS_SERVERLESS_CDN_URL =
|
|
15
|
+
"https://cdn.jsdelivr.net/npm/pdfjs-serverless@1/+esm";
|
|
16
|
+
|
|
17
|
+
let pdfjsPromise: Promise<any> | null = null;
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* Resolves the PDF.js API (`getDocument`, ...) from the pdfjs-serverless CDN
|
|
21
|
+
* build, caching the module after the first call. Node.js and Bun cannot
|
|
22
|
+
* import remote URLs, so when the CDN import throws we fall back to the
|
|
23
|
+
* locally-installed `pdfjs-serverless` optional dependency.
|
|
24
|
+
*/
|
|
25
|
+
export async function loadPdfJs(): Promise<any> {
|
|
26
|
+
if (!pdfjsPromise) {
|
|
27
|
+
pdfjsPromise = import(/* @vite-ignore */ PDFJS_SERVERLESS_CDN_URL as any)
|
|
28
|
+
.catch(() => import("pdfjs-serverless" as any))
|
|
29
|
+
.then(({ resolvePDFJS }) => resolvePDFJS());
|
|
30
|
+
}
|
|
31
|
+
return pdfjsPromise;
|
|
32
|
+
}
|