extract-pdf 0.1.259 → 0.1.261

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -33,6 +33,34 @@ import {
33
33
  convertPDFToHTMLWithLiteParseWasm,
34
34
  type LiteParseWasmHTMLOptions,
35
35
  } from "./liteparse-wasm-to-html";
36
+ import { loadPdfJs } from "./utils/load-pdfjs";
37
+ import {
38
+ scanPagesForOCR,
39
+ type OcrScanResult,
40
+ type ScanPagesForOCROptions,
41
+ } from "./ocr-page-scan";
42
+ import {
43
+ ocrPdfPagesWithDocling,
44
+ type DoclingOcrOptions,
45
+ } from "./docling-ocr";
46
+
47
+ /**
48
+ * Where the OCR-capable Docling processing happens for
49
+ * {@link convertPDFToHTML}.
50
+ * - `"frontend"` (default) — all pages parsed by the pure-JS text-layer
51
+ * pipeline; no OCR, no model, works everywhere.
52
+ * - `"hybrid"` — every page goes through the JS pipeline, then a regex scan
53
+ * ({@link scanPagesForOCR}) flags pages containing infographics/figures/
54
+ * tables (or with no usable text layer) and only those pages are re-done
55
+ * with the Granite Docling OCR model.
56
+ * - `"docling"` — every page is rasterized and OCR'd with Granite Docling
57
+ * (in-process via the optional `@huggingface/transformers` dependency, or
58
+ * remotely when `processorUrl` is set).
59
+ * - any `http(s)://` URL — like `"docling"`, but all pages are sent to that
60
+ * docling-compatible processor API (this package's `server/`, or another
61
+ * deployment).
62
+ */
63
+ export type ProcessorMode = "frontend" | "hybrid" | "docling" | (string & {});
36
64
 
37
65
  /**
38
66
  * Which parsing engine {@link convertPDFToHTML} runs.
@@ -65,6 +93,16 @@ export type ParseMethod = "ts-block-algorithm" | "liteparse" | "liteparse-wasm";
65
93
  * @param {boolean} options.removePageHeaders default=true - Removes repeated headers found on each page
66
94
  * @param {ParseMethod} options.method default="ts-block-algorithm" - Parsing engine to use;
67
95
  * `"liteparse"` delegates to LiteParse (Node.js only, see {@link ParseMethod})
96
+ * @param {ProcessorMode} options.processor default="frontend" - Where OCR happens:
97
+ * `"frontend"` (all JS, no OCR), `"hybrid"` (regex-scan pages for
98
+ * infographics/tables and OCR only those), `"docling"` (OCR every page), or
99
+ * the URL of a docling-compatible processor API (see {@link ProcessorMode})
100
+ * @param {string} options.processorUrl - Remote docling-compatible API base URL
101
+ * used by `"hybrid"`/`"docling"` instead of the in-process model
102
+ * @param {Object} options.ocrScanOptions - Threshold tuning for the hybrid
103
+ * page scan, see {@link ScanPagesForOCROptions}
104
+ * @param {Object} options.doclingOptions - Prompt/maxTokens/scale for the OCR
105
+ * model, see {@link DoclingOcrOptions}
68
106
  * @returns {string|Object} HTML formatted text
69
107
  * @category Extract
70
108
  * @author [vtempest (2025)](https://github.com/vtempest),
@@ -77,6 +115,10 @@ export async function convertPDFToHTML(
77
115
  addPageNumbers?: boolean;
78
116
  addCitation?: boolean;
79
117
  method?: ParseMethod;
118
+ processor?: ProcessorMode;
119
+ processorUrl?: string;
120
+ ocrScanOptions?: ScanPagesForOCROptions;
121
+ doclingOptions?: Omit<DoclingOcrOptions, "processorUrl">;
80
122
  } & Pick<LiteParseHTMLOptions, "liteParseOptions"> = {},
81
123
  ) {
82
124
  if (options.method === "liteparse") {
@@ -95,6 +137,16 @@ export async function convertPDFToHTML(
95
137
  // try {
96
138
  var { addPageNumbers = false, addCitation = true } = options;
97
139
 
140
+ // Resolve where OCR happens: "frontend" | "hybrid" | "docling" | a
141
+ // processor URL (which means "docling" against that remote API).
142
+ const processor = options.processor ?? "frontend";
143
+ const processorIsUrl = /^https?:\/\//i.test(processor);
144
+ const processorMode = processorIsUrl ? "docling" : processor;
145
+ const doclingOptions: DoclingOcrOptions = {
146
+ ...options.doclingOptions,
147
+ processorUrl: processorIsUrl ? processor : options.processorUrl,
148
+ };
149
+
98
150
  // pass in databuffer or download all pdf data
99
151
  // and convert to array buffer
100
152
  var buffer =
@@ -107,9 +159,7 @@ export async function convertPDFToHTML(
107
159
 
108
160
  let pdfDocument;
109
161
  try {
110
- let { resolvePDFJS } = await import("https://cdn.jsdelivr.net/npm/pdfjs-serverless@1.1.0/+esm" as any);
111
-
112
- const { getDocument } = await resolvePDFJS();
162
+ const { getDocument } = await loadPdfJs();
113
163
  pdfDocument = await getDocument({
114
164
  data: new Uint8Array(buffer),
115
165
  useSystemFonts: true,
@@ -178,6 +228,19 @@ export async function convertPDFToHTML(
178
228
  pages[page.pageNumber - 1].items = textItems;
179
229
  }
180
230
 
231
+ // Raw per-page text (items grouped into lines by y) captured before the
232
+ // transforms mutate the pages — input for the OCR page scan.
233
+ const pageTexts = pages.map((page) =>
234
+ Object.values(
235
+ (page.items as any[]).reduce((lines: any, item: any) => {
236
+ (lines[item.y] = lines[item.y] || []).push(item.text);
237
+ return lines;
238
+ }, {}),
239
+ )
240
+ .map((line: any) => line.join(" "))
241
+ .join("\n"),
242
+ );
243
+
181
244
  var parseResult = new ParseResult({ pages });
182
245
 
183
246
  let lastTransformation: (typeof transformations)[number] | undefined,
@@ -206,14 +269,35 @@ export async function convertPDFToHTML(
206
269
  lastTransformation = transformation;
207
270
  });
208
271
 
209
- var html = parseResult.pages.reduce((acc, page, pageNumber) => {
210
- return (
211
- acc +
272
+ var pageHtmls = parseResult.pages.map(
273
+ (page, pageNumber) =>
212
274
  `<p id="page-${pageNumber + 1}">${
213
275
  addPageNumbers ? ` [${pageNumber + 1}] ` : ""
214
- }${page.items.join('</p><p id="page-' + pageNumber + '">')}</p>`
276
+ }${page.items.join('</p><p id="page-' + pageNumber + '">')}</p>`,
277
+ );
278
+
279
+ // Regex scan flagging pages with infographics/figures/tables (or no usable
280
+ // text layer) — the pages worth OCR'ing in hybrid mode.
281
+ const ocrScan = scanPagesForOCR(pageTexts, options.ocrScanOptions);
282
+
283
+ if (processorMode === "docling" || processorMode === "hybrid") {
284
+ const targetPages =
285
+ processorMode === "docling"
286
+ ? pageHtmls.map((_, index) => index + 1)
287
+ : ocrScan.pagesNeedingOcr;
288
+ const ocrResults = await ocrPdfPagesWithDocling(
289
+ pdfDocument,
290
+ targetPages,
291
+ doclingOptions,
215
292
  );
216
- }, "");
293
+ // Pages whose OCR failed keep their frontend-parsed HTML.
294
+ for (const [pageNumber, ocrHtml] of ocrResults)
295
+ pageHtmls[pageNumber - 1] = `<section class="ocr-page" id="page-${pageNumber}">${
296
+ addPageNumbers ? ` [${pageNumber}] ` : ""
297
+ }${ocrHtml}</section>`;
298
+ }
299
+
300
+ var html = pageHtmls.join("");
217
301
 
218
302
  if (addCitation) {
219
303
  // Get metadata
@@ -234,7 +318,21 @@ export async function convertPDFToHTML(
234
318
  title = html.slice(0, 400).match(/<h[0-9]>(.*?)<\/h[0-9]>/)?.[1] || title;
235
319
  }
236
320
 
237
- return { author, title, html, format: "pdf" };
321
+ return {
322
+ author,
323
+ title,
324
+ html,
325
+ format: "pdf",
326
+ processor: processorMode,
327
+ ocrScan,
328
+ } as {
329
+ author?: string;
330
+ title?: string;
331
+ html: string;
332
+ format: string;
333
+ processor: string;
334
+ ocrScan: OcrScanResult;
335
+ };
238
336
  }
239
337
 
240
338
  export { convertPDFToHTMLWithLiteParse };
@@ -246,4 +344,18 @@ export type {
246
344
  DetectPdfNeedsOcrOptions,
247
345
  PdfOcrAssessment,
248
346
  } from "./detect-needs-ocr";
347
+ export { scanPagesForOCR } from "./ocr-page-scan";
348
+ export type {
349
+ OcrScanResult,
350
+ PageOcrScan,
351
+ ScanPagesForOCROptions,
352
+ } from "./ocr-page-scan";
353
+ export {
354
+ doctagsToHtml,
355
+ ocrImageWithDocling,
356
+ ocrPdfPagesWithDocling,
357
+ renderPdfPageToPngBase64,
358
+ } from "./docling-ocr";
359
+ export type { DoclingOcrOptions } from "./docling-ocr";
360
+ export { loadPdfJs } from "./utils/load-pdfjs";
249
361
 
@@ -0,0 +1,32 @@
1
+ /**
2
+ * @fileoverview Lazy loader for the slim serverless PDF.js build.
3
+ *
4
+ * The package deliberately does NOT depend on `pdfjs-dist`: PDF.js is pulled
5
+ * at runtime from jsDelivr's ESM build of
6
+ * [pdfjs-serverless](https://github.com/johannschopplich/pdfjs-serverless) —
7
+ * a zero-dependency, single-file (~1.6 MB minified) redistribution of Mozilla
8
+ * PDF.js that works in Workers/edge runtimes, Node.js, and browsers. Pinned
9
+ * to the major version so compatible patch/minor releases update
10
+ * automatically. This keeps `extract-pdf`'s default export slim: nothing
11
+ * PDF.js-related is bundled or installed until a document is actually parsed.
12
+ */
13
+
14
+ const PDFJS_SERVERLESS_CDN_URL =
15
+ "https://cdn.jsdelivr.net/npm/pdfjs-serverless@1/+esm";
16
+
17
+ let pdfjsPromise: Promise<any> | null = null;
18
+
19
+ /**
20
+ * Resolves the PDF.js API (`getDocument`, ...) from the pdfjs-serverless CDN
21
+ * build, caching the module after the first call. Node.js and Bun cannot
22
+ * import remote URLs, so when the CDN import throws we fall back to the
23
+ * locally-installed `pdfjs-serverless` optional dependency.
24
+ */
25
+ export async function loadPdfJs(): Promise<any> {
26
+ if (!pdfjsPromise) {
27
+ pdfjsPromise = import(/* @vite-ignore */ PDFJS_SERVERLESS_CDN_URL as any)
28
+ .catch(() => import("pdfjs-serverless" as any))
29
+ .then(({ resolvePDFJS }) => resolvePDFJS());
30
+ }
31
+ return pdfjsPromise;
32
+ }