tablefacts 0.2.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,330 @@
1
+ // Reads a PDF menu. A digital PDF carries its text, so a page is transcribed
2
+ // from that; a scanned page has no text and is rendered to a picture for the
3
+ // vision model. The same render lets a product's printed photo be screenshotted:
4
+ // pdfjs records where every image lands on the page while rendering
5
+ // (`recordImages`), and that rectangle is cropped out of the page render.
6
+ //
7
+ // pdfjs-dist is optional and loads through pdfjs.mjs. See raw/README in
8
+ // src/menu/README.md for how the pages flow through the importer.
9
+ import { readFile } from "node:fs/promises";
10
+ import { fileURLToPath } from "node:url";
11
+ import { TablefactsError } from "../../lib/errors.mjs";
12
+ import { resolveIn } from "../../lib/project.mjs";
13
+ import { get } from "./source.mjs";
14
+ import { loadPdfjs, renderError } from "./pdfjs.mjs";
15
+
16
+ // A menu PDF is small; this keeps a stray file from filling memory. The vision
17
+ // providers cap a picture at 5 MB, which a rendered page respects by itself.
18
+ const MAX_BYTES = 25 * 1024 * 1024;
19
+ // A PDF can declare a huge MediaBox; rendering it at `scale` would allocate an
20
+ // enormous canvas, so the scale is reduced to stay under this many pixels.
21
+ const MAX_RENDER_PIXELS = 30_000_000;
22
+ const PDF_MAGIC = "%PDF-";
23
+
24
+ const hasPdfMagic = (bytes) => bytes.length >= 5 && String.fromCharCode(...bytes.subarray(0, 5)) === PDF_MAGIC;
25
+
26
+ /**
27
+ * True when an argument names a PDF: a local file ending in .pdf, or a URL whose
28
+ * path ends in .pdf. A URL that hides the extension comes back from the fetch as
29
+ * an HTML page with no pictures, and the reader says to pass the file instead.
30
+ * @param {string} input
31
+ * @returns {boolean}
32
+ */
33
+ export function isPdfInput(input) {
34
+ try {
35
+ return /\.pdf$/i.test(new URL(input).pathname);
36
+ } catch {
37
+ return /\.pdf$/i.test(String(input).split(/[\\/]/).pop() ?? "");
38
+ }
39
+ }
40
+
41
+ /**
42
+ * The PDF's bytes, from a local file (`projectDir`-relative) or an http(s) URL.
43
+ * `host` names the cache folder, `label` is what a note calls the file.
44
+ * @param {string} input
45
+ * @param {{ projectDir?: string }} [options]
46
+ * @returns {Promise<{ bytes: Uint8Array, host: string, label: string }>}
47
+ */
48
+ export async function readPdfSource(input, { projectDir } = {}) {
49
+ // `new URL` accepts "C:\path" as a `c:` URL, so only http(s)/file are treated
50
+ // as URLs; anything else is a filesystem path.
51
+ let url = null;
52
+ try {
53
+ const parsed = new URL(input);
54
+ if (/^(https?|file):$/.test(parsed.protocol)) url = parsed;
55
+ } catch {
56
+ url = null;
57
+ }
58
+ if (!url && /^[a-z][a-z0-9+.-]*:\/\//i.test(input)) {
59
+ throw new TablefactsError(`"${input}" is not an http(s) URL or a file path.`, "EUSAGE");
60
+ }
61
+ if (url?.protocol === "file:") return readLocalPdf(fileURLToPath(url));
62
+ if (url) {
63
+ const res = await get(input, "PDF");
64
+ const type = (res.headers.get("content-type") ?? "").toLowerCase();
65
+ const declared = Number(res.headers.get("content-length"));
66
+ if (declared > MAX_BYTES) throw new TablefactsError(`${input} is ${(declared / 1048576).toFixed(1)} MB; menu PDFs are limited to 25 MB.`, "EFAILED");
67
+ const bytes = new Uint8Array(await res.arrayBuffer());
68
+ if (!type.includes("pdf") && !hasPdfMagic(bytes)) {
69
+ throw new TablefactsError(`${input} is ${type || "not a PDF"}: this source reads PDFs and web pages, not other files.`, "EFAILED");
70
+ }
71
+ if (bytes.length > MAX_BYTES) throw new TablefactsError(`${input} is ${(bytes.length / 1048576).toFixed(1)} MB; menu PDFs are limited to 25 MB.`, "EFAILED");
72
+ return { bytes, host: url.hostname, label: input };
73
+ }
74
+
75
+ return readLocalPdf(resolveIn(projectDir, input));
76
+ }
77
+
78
+ /** The bytes of a local PDF, with the magic-bytes and size checks in one place. */
79
+ async function readLocalPdf(file) {
80
+ const bytes = new Uint8Array(
81
+ await readFile(file).catch((error) => {
82
+ throw new TablefactsError(`Could not read ${file}: ${error.message}.`, "EUSAGE", { cause: error });
83
+ }),
84
+ );
85
+ if (!hasPdfMagic(bytes)) throw new TablefactsError(`${file} is not a PDF.`, "EUSAGE");
86
+ if (bytes.length > MAX_BYTES) throw new TablefactsError(`${file} is ${(bytes.length / 1048576).toFixed(1)} MB; menu PDFs are limited to 25 MB.`, "EUSAGE");
87
+ return { bytes, host: "local", label: file };
88
+ }
89
+
90
+ /**
91
+ * Opens a PDF for reading. `close()` releases pdfjs's worker; call it when the
92
+ * run is done with the file. `textCache` keeps a page's text from being read twice.
93
+ * @param {Uint8Array} bytes
94
+ * @returns {Promise<any>} the open document handle
95
+ */
96
+ export async function openPdf(bytes) {
97
+ const { pdfjs, standardFontDataUrl, cMapUrl } = await loadPdfjs();
98
+ const task = pdfjs.getDocument({
99
+ data: bytes instanceof Uint8Array ? bytes : new Uint8Array(bytes),
100
+ useSystemFonts: false,
101
+ isEvalSupported: false,
102
+ standardFontDataUrl,
103
+ cMapUrl,
104
+ cMapPacked: true,
105
+ verbosity: 0,
106
+ });
107
+ let doc;
108
+ try {
109
+ doc = await task.promise;
110
+ } catch (error) {
111
+ await task.destroy().catch(() => {});
112
+ if (error?.name === "PasswordException") throw new TablefactsError("The PDF is password-protected. Remove the password and try again.", "EFAILED", { cause: error });
113
+ throw new TablefactsError(`The PDF could not be opened: ${error?.message ?? error}.`, "EFAILED", { cause: error });
114
+ }
115
+ const ops = Object.fromEntries(Object.entries(pdfjs.OPS).map(([name, code]) => [code, name]));
116
+ return {
117
+ pdfjs,
118
+ doc,
119
+ numPages: doc.numPages,
120
+ ops,
121
+ textCache: new Map(),
122
+ opened: new Map(),
123
+ close: () => task.destroy().catch(() => {}),
124
+ };
125
+ }
126
+
127
+ const openPage = async (source, number) => {
128
+ if (!source.opened.has(number)) source.opened.set(number, source.doc.getPage(number));
129
+ return source.opened.get(number);
130
+ };
131
+
132
+ /**
133
+ * One page's text and its pieces. `items` are pdfjs's text items, kept for
134
+ * matching a product name to a spot on the page when placing product photos.
135
+ * @returns {Promise<{ page: any, text: string, items: any[] }>}
136
+ */
137
+ export async function readPdfText(source, number) {
138
+ if (source.textCache.has(number)) return source.textCache.get(number);
139
+ const page = await openPage(source, number);
140
+ const content = await page.getTextContent();
141
+ const items = content.items.filter((item) => typeof item.str === "string");
142
+ const result = { page, text: textFromItems(items), items };
143
+ source.textCache.set(number, result);
144
+ return result;
145
+ }
146
+
147
+ /** Every page that has text worth transcribing, and how much. Used by `--list`. */
148
+ export async function inspectPdf(source) {
149
+ const pages = [];
150
+ for (let number = 1; number <= source.numPages; number++) {
151
+ const { text } = await readPdfText(source, number);
152
+ pages.push({ number, hasText: text.trim().length > 0, chars: text.trim().length });
153
+ }
154
+ return pages;
155
+ }
156
+
157
+ /**
158
+ * The text items flattened to `{ str, x, y, width, height }`, which both the
159
+ * text builder and the product-photo matcher understand.
160
+ */
161
+ export const positionedItems = (items) =>
162
+ (items ?? [])
163
+ .filter((item) => typeof item.str === "string" && item.str.trim() !== "")
164
+ .map((item) => {
165
+ const transform = item.transform ?? [];
166
+ return {
167
+ str: item.str,
168
+ x: item.x ?? transform[4] ?? 0,
169
+ y: item.y ?? transform[5] ?? 0,
170
+ width: Math.abs(item.width ?? 0),
171
+ height: Math.abs(item.height ?? 0) || Math.hypot(transform[2] ?? 0, transform[3] ?? 0) || 10,
172
+ };
173
+ });
174
+
175
+ /**
176
+ * Rebuilds readable text from pdfjs text items: pieces on the same line are
177
+ * joined in reading order, a wide vertical gap becomes a blank line so the model
178
+ * still sees section breaks. A page with two clear columns is read column by
179
+ * column, so the two columns' rows are not merged. Pure, so it is tested without
180
+ * a PDF.
181
+ * @param {{ str?: string, transform?: number[], width?: number, height?: number }[]} items
182
+ * @returns {string}
183
+ */
184
+ export function textFromItems(items) {
185
+ const pieces = positionedItems(items).sort((a, b) => b.y - a.y || a.x - b.x);
186
+ const columns = splitColumns(pieces);
187
+ if (columns) return columns.map((column) => assemble(column)).filter(Boolean).join("\n\n");
188
+ return assemble(pieces);
189
+ }
190
+
191
+ const hasLetter = (text) => /\p{L}/u.test(text);
192
+ const hasDigit = (text) => /\d/.test(text);
193
+
194
+ /**
195
+ * Two columns when a wide vertical corridor splits the text and both sides look
196
+ * like a menu column (left-aligned, with headings and with prices). A single
197
+ * column that right- or left-aligns its prices has a corridor too, but then one
198
+ * side holds only prices, so it is not a column and the text is left alone.
199
+ * Returns [left, right] or null.
200
+ */
201
+ function splitColumns(pieces) {
202
+ if (pieces.length < 6) return null;
203
+ const minX = Math.min(...pieces.map((piece) => piece.x));
204
+ const maxX = Math.max(...pieces.map((piece) => piece.x + piece.width));
205
+ const width = maxX - minX;
206
+ if (width <= 0) return null;
207
+
208
+ const intervals = pieces.map((piece) => [piece.x, piece.x + piece.width]).sort((a, b) => a[0] - b[0]);
209
+ const merged = [];
210
+ for (const [start, end] of intervals) {
211
+ const last = merged.at(-1);
212
+ if (last && start <= last[1]) last[1] = Math.max(last[1], end);
213
+ else merged.push([start, end]);
214
+ }
215
+ const threshold = Math.max(width * 0.05, 24);
216
+ let corridor = null;
217
+ for (let i = 0; i + 1 < merged.length; i++) {
218
+ const gap = merged[i + 1][0] - merged[i][1];
219
+ if (gap >= threshold && (!corridor || gap > corridor.gap)) corridor = { gap, at: merged[i][1] + gap / 2 };
220
+ }
221
+ if (!corridor) return null;
222
+
223
+ const left = pieces.filter((piece) => piece.x + piece.width <= corridor.at);
224
+ const right = pieces.filter((piece) => piece.x >= corridor.at);
225
+ const enough = [left, right].every((side) => side.length >= Math.max(3, pieces.length * 0.25));
226
+ if (!enough) return null;
227
+ const isColumn = (side) => {
228
+ const starts = side.map((piece) => piece.x);
229
+ const ends = side.map((piece) => piece.x + piece.width);
230
+ const spread = Math.max(...starts) - Math.min(...starts);
231
+ const span = Math.max(...ends) - Math.min(...starts) || 1;
232
+ return spread <= span * 0.5 && side.some((piece) => hasLetter(piece.str) && !hasDigit(piece.str)) && side.some((piece) => hasDigit(piece.str));
233
+ };
234
+ return isColumn(left) && isColumn(right) ? [left, right] : null;
235
+ }
236
+
237
+ /** One column (or a single-column page) as lines, top to bottom. */
238
+ function assemble(pieces) {
239
+ const sorted = pieces.slice().sort((a, b) => b.y - a.y || a.x - b.x);
240
+
241
+ const lines = [];
242
+ for (const piece of sorted) {
243
+ const last = lines.at(-1);
244
+ if (last && Math.abs(last.y - piece.y) <= Math.max(last.h, piece.height) * 0.6) {
245
+ last.parts.push(piece);
246
+ last.y = (last.y * (last.parts.length - 1) + piece.y) / last.parts.length;
247
+ last.h = Math.max(last.h, piece.height);
248
+ } else {
249
+ lines.push({ y: piece.y, h: piece.height, parts: [piece] });
250
+ }
251
+ }
252
+ if (!lines.length) return "";
253
+
254
+ const heights = lines.map((line) => line.h).sort((a, b) => a - b);
255
+ const median = heights[Math.floor(heights.length / 2)] || 10;
256
+ let out = "";
257
+ let previous = null;
258
+ for (const line of lines) {
259
+ const text = line.parts
260
+ .slice()
261
+ .sort((a, b) => a.x - b.x)
262
+ .map((part) => part.str)
263
+ .join(" ")
264
+ .replace(/\s+/g, " ")
265
+ .trim();
266
+ if (!text) continue;
267
+ if (previous && previous.y - line.y > median * 1.8) out += "\n";
268
+ out += `${text}\n`;
269
+ previous = line;
270
+ }
271
+ return out.trimEnd();
272
+ }
273
+
274
+ /** A page's rendered pixels, the context, and the viewport they were rendered with. */
275
+ export async function renderPdfPage(source, page, { scale = 2, recordImages = false } = {}) {
276
+ let viewport = page.getViewport({ scale });
277
+ // A page with a huge MediaBox would allocate an enormous canvas: cap the
278
+ // pixels and render smaller, so a stray PDF cannot exhaust memory.
279
+ const pixels = viewport.width * viewport.height;
280
+ if (pixels > MAX_RENDER_PIXELS) {
281
+ scale *= Math.sqrt(MAX_RENDER_PIXELS / pixels);
282
+ viewport = page.getViewport({ scale });
283
+ }
284
+ try {
285
+ const { canvas, context } = source.doc.canvasFactory.create(Math.ceil(viewport.width), Math.ceil(viewport.height), false);
286
+ await page.render({ canvas, canvasContext: context, viewport, recordImages }).promise;
287
+ return { canvas, context, viewport, scale };
288
+ } catch (error) {
289
+ throw renderError(error);
290
+ }
291
+ }
292
+
293
+ /** Releases a rendered canvas's native memory. */
294
+ export function destroyPage(source, { canvas, context } = {}) {
295
+ if (canvas) source.doc.canvasFactory.destroy({ canvas, context });
296
+ }
297
+
298
+ /**
299
+ * The rectangle each image occupies on a page render. pdfjs records this itself
300
+ * when rendering with `recordImages` (three points per image, fractions of the
301
+ * canvas, top-left origin): it is clip- and group-aware, unlike walking the
302
+ * operator list by hand.
303
+ */
304
+ export function imageRects(canvas, coordinates) {
305
+ const out = [];
306
+ const coords = coordinates ?? [];
307
+ for (let i = 0; i + 5 < coords.length; i += 6) {
308
+ const xs = [coords[i] * canvas.width, coords[i + 2] * canvas.width, coords[i + 4] * canvas.width];
309
+ const ys = [coords[i + 1] * canvas.height, coords[i + 3] * canvas.height, coords[i + 5] * canvas.height];
310
+ const left = Math.max(0, Math.min(...xs));
311
+ const top = Math.max(0, Math.min(...ys));
312
+ const right = Math.min(canvas.width, Math.max(...xs));
313
+ const bottom = Math.min(canvas.height, Math.max(...ys));
314
+ if (right - left >= 1 && bottom - top >= 1) out.push({ x: left, y: top, width: right - left, height: bottom - top });
315
+ }
316
+ return out;
317
+ }
318
+
319
+ /** Crops a pixel rectangle out of a page render, as a PNG buffer. */
320
+ export function cropPdfPixels(source, canvas, { x, y, width, height }, { padding = 0 } = {}) {
321
+ const left = Math.max(0, Math.floor(x) - padding);
322
+ const top = Math.max(0, Math.floor(y) - padding);
323
+ const w = Math.max(1, Math.min(canvas.width - left, Math.ceil(width) + padding * 2));
324
+ const h = Math.max(1, Math.min(canvas.height - top, Math.ceil(height) + padding * 2));
325
+ const { canvas: out, context } = source.doc.canvasFactory.create(w, h, false);
326
+ context.drawImage(canvas, left, top, w, h, 0, 0, w, h);
327
+ const png = out.toBuffer("image/png");
328
+ source.doc.canvasFactory.destroy({ canvas: out, context });
329
+ return png;
330
+ }
@@ -0,0 +1,57 @@
1
+ // pdfjs-dist is an optional peer dependency: only PDF menus need it, so it loads
2
+ // on use (the way lib/playwright.mjs loads Playwright). Reading text needs only
3
+ // pdfjs-dist; rendering a page for a scan or a product-photo screenshot also
4
+ // needs the canvas pdfjs-dist loads itself (@napi-rs/canvas).
5
+ import { createRequire } from "node:module";
6
+ import { dirname, join } from "node:path";
7
+ import { pathToFileURL } from "node:url";
8
+ import { TablefactsError } from "../../lib/errors.mjs";
9
+
10
+ const require = createRequire(import.meta.url);
11
+
12
+ const INSTALL = "npm install pdfjs-dist @napi-rs/canvas";
13
+
14
+ /** True when the throw is the module being absent (rather than a real load failure). */
15
+ const missing = (error, name) => error?.code === "ERR_MODULE_NOT_FOUND" && String(error?.message ?? "").includes(name);
16
+
17
+ const notInstalled = (error) =>
18
+ new TablefactsError(`pdfjs-dist is not installed. Reading a PDF menu needs it. Install it with: ${INSTALL}`, "EDEPENDENCY", { cause: error });
19
+
20
+ // The were-loaded flag keeps the successful load out of every call's try/catch.
21
+ let pdfjsModule;
22
+
23
+ /**
24
+ * The pdfjs module plus the asset URLs it needs (standard fonts and CMaps) so
25
+ * text is extracted and pages render without the warnings Node otherwise logs.
26
+ * @returns {Promise<{ pdfjs: any, standardFontDataUrl: string, cMapUrl: string }>}
27
+ */
28
+ export async function loadPdfjs() {
29
+ if (pdfjsModule) return pdfjsModule;
30
+ try {
31
+ const pdfjs = await import("pdfjs-dist/legacy/build/pdf.mjs");
32
+ // Resolved from this file, so it finds the copy the project installed.
33
+ const root = dirname(require.resolve("pdfjs-dist/package.json"));
34
+ pdfjsModule = {
35
+ pdfjs,
36
+ standardFontDataUrl: pathToFileURL(join(root, "standard_fonts") + "/").href,
37
+ cMapUrl: pathToFileURL(join(root, "cmaps") + "/").href,
38
+ };
39
+ return pdfjsModule;
40
+ } catch (error) {
41
+ if (missing(error, "pdfjs-dist")) throw notInstalled(error);
42
+ throw error;
43
+ }
44
+ }
45
+
46
+ /** A rendering failure caused by the canvas package being absent becomes EDEPENDENCY with the install hint. */
47
+ export function renderError(error) {
48
+ const message = String(error?.message ?? "");
49
+ if (missing(error, "@napi-rs/canvas") || (/canvas/i.test(message) && /cannot find module|not installed|err_module|napi/i.test(message))) {
50
+ return new TablefactsError(
51
+ `Rendering a PDF page needs the canvas package. Install it with: ${INSTALL}`,
52
+ "EDEPENDENCY",
53
+ { cause: error },
54
+ );
55
+ }
56
+ return error;
57
+ }
@@ -13,7 +13,8 @@ const headers = { "user-agent": "cannario-menu-sync/1.0 (restaurant menu importe
13
13
  const IMAGE = /\.(jpe?g|png|webp)(\?|$)/i;
14
14
  const MAX_BYTES = 5 * 1024 * 1024; // the Claude API refuses larger images; Gemini and Groq take 20 MB, so this is the tightest limit
15
15
 
16
- async function get(url, what) {
16
+ /** A fetch with retries on server errors; shared with the PDF reader. */
17
+ export async function get(url, what) {
17
18
  let failure;
18
19
  for (let attempt = 1; attempt <= 3; attempt++) {
19
20
  try {
@@ -71,7 +72,13 @@ export async function discoverPages(inputs, options) {
71
72
  const found = await Promise.allSettled(
72
73
  inputs.map(async (input) => {
73
74
  if (IMAGE.test(new URL(input).pathname)) return [{ url: input, alt: "" }];
74
- const html = await (await get(input, "page")).text();
75
+ const res = await get(input, "page");
76
+ // The argument was not seen as a PDF (its address does not end in .pdf):
77
+ // say so instead of reporting "no images".
78
+ if ((res.headers.get("content-type") ?? "").toLowerCase().includes("pdf")) {
79
+ throw new TablefactsError(`${input} serves a PDF whose address does not end in .pdf. Download it and pass the file instead.`, "EFAILED");
80
+ }
81
+ const html = await res.text();
75
82
  const images = findImages(html, input, options);
76
83
  if (!images.length) {
77
84
  throw new TablefactsError(