tablefacts 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +1 -1
- package/CHANGELOG.md +22 -0
- package/README.md +30 -10
- package/package.json +13 -2
- package/src/lib/types.mjs +9 -2
- package/src/menu/README.md +28 -6
- package/src/menu/lib/run.mjs +3 -0
- package/src/menu/raw/config.mjs +6 -2
- package/src/menu/raw/extract.mjs +36 -15
- package/src/menu/raw/images.mjs +109 -0
- package/src/menu/raw/import.mjs +327 -51
- package/src/menu/raw/normalize.mjs +15 -4
- package/src/menu/raw/pdf.mjs +330 -0
- package/src/menu/raw/pdfjs.mjs +57 -0
- package/src/menu/raw/source.mjs +9 -2
- package/src/menu/raw/vision.mjs +124 -65
- package/types/lib/types.d.mts +41 -3
- package/types/menu/lib/run.d.mts +3 -0
- package/types/menu/raw/config.d.mts +1 -0
- package/types/menu/raw/images.d.mts +27 -0
- package/types/menu/raw/import.d.mts +33 -7
- package/types/menu/raw/normalize.d.mts +7 -1
- package/types/menu/raw/pdf.d.mts +98 -0
- package/types/menu/raw/pdfjs.d.mts +12 -0
- package/types/menu/raw/source.d.mts +2 -0
- package/types/menu/raw/vision.d.mts +11 -1
package/src/menu/raw/import.mjs
CHANGED
|
@@ -1,17 +1,62 @@
|
|
|
1
|
-
// Menus that are only pictures, as a library: find the pages, have a
|
|
2
|
-
// model transcribe them (
|
|
3
|
-
|
|
1
|
+
// Menus that are only pictures or a PDF, as a library: find the pages, have a
|
|
2
|
+
// model transcribe them (a PDF page from its text, a picture or scan with a
|
|
3
|
+
// vision model), normalize and import. Product photos printed on a PDF page are
|
|
4
|
+
// screenshotted too, when asked for.
|
|
5
|
+
import { mkdir, readFile, writeFile } from "node:fs/promises";
|
|
4
6
|
import { join } from "node:path";
|
|
5
7
|
import { resolveEnv } from "../../lib/env.mjs";
|
|
6
8
|
import { optionError, TablefactsError } from "../../lib/errors.mjs";
|
|
7
9
|
import { normalizeLog } from "../../lib/log.mjs";
|
|
10
|
+
import { resolveIn, workDirIn } from "../../lib/project.mjs";
|
|
8
11
|
import { importMenu } from "../lib/import.mjs";
|
|
12
|
+
import { slugify } from "../lib/menu.mjs";
|
|
9
13
|
import defaultConfig from "./config.mjs";
|
|
14
|
+
import { matchPlacements } from "./images.mjs";
|
|
10
15
|
import { normalizePages } from "./normalize.mjs";
|
|
11
|
-
import {
|
|
12
|
-
|
|
16
|
+
import {
|
|
17
|
+
cropPdfPixels,
|
|
18
|
+
destroyPage,
|
|
19
|
+
imageRects,
|
|
20
|
+
inspectPdf,
|
|
21
|
+
isPdfInput,
|
|
22
|
+
openPdf,
|
|
23
|
+
positionedItems,
|
|
24
|
+
readPdfSource,
|
|
25
|
+
readPdfText,
|
|
26
|
+
renderPdfPage,
|
|
27
|
+
} from "./pdf.mjs";
|
|
28
|
+
import { discoverPages, downloadPage, pageId } from "./source.mjs";
|
|
29
|
+
import { defaultProvider, providers, readPage, readText } from "./vision.mjs";
|
|
13
30
|
|
|
14
31
|
const providerNames = Object.keys(providers).join(", ");
|
|
32
|
+
// A page with fewer characters than this is a page number or a logo, not a menu:
|
|
33
|
+
// it is read as a picture instead (a scanned page often has a stray text layer).
|
|
34
|
+
const MIN_PAGE_TEXT = 40;
|
|
35
|
+
|
|
36
|
+
/** True when a transcription was read with photo boxes: every item then carries a `box` key (often null). A page read as having no items needs no re-read. */
|
|
37
|
+
const hasBoxField = (transcription) => {
|
|
38
|
+
const items = (transcription.sections ?? []).flatMap((section) => section.items ?? []);
|
|
39
|
+
return items.length === 0 || items.some((item) => item && Object.hasOwn(item, "box"));
|
|
40
|
+
};
|
|
41
|
+
|
|
42
|
+
/** A page longer than this is truncated before the model sees it, to bound cost. */
|
|
43
|
+
const MAX_PAGE_TEXT = 20000;
|
|
44
|
+
/** The tightest vision limit across providers; a bigger page render must be lowered. */
|
|
45
|
+
const MAX_PAGE_IMAGE = 5 * 1024 * 1024;
|
|
46
|
+
|
|
47
|
+
/** Text items moved from PDF user space into the render's pixel space, so they and the crop rectangles can be compared. */
|
|
48
|
+
const canvasItems = (items, viewport) => {
|
|
49
|
+
const [a, b, c, d, e, f] = viewport.transform;
|
|
50
|
+
const scaleX = Math.hypot(a, b);
|
|
51
|
+
const scaleY = Math.hypot(c, d);
|
|
52
|
+
return items.map((item) => ({
|
|
53
|
+
str: item.str,
|
|
54
|
+
x: a * item.x + c * item.y + e,
|
|
55
|
+
y: b * item.x + d * item.y + f,
|
|
56
|
+
width: item.width * scaleX,
|
|
57
|
+
height: item.height * scaleY,
|
|
58
|
+
}));
|
|
59
|
+
};
|
|
15
60
|
|
|
16
61
|
/** "1,3-5" (or a list of numbers) to Set {1,3,4,5}. */
|
|
17
62
|
function pageSet(only) {
|
|
@@ -25,78 +70,309 @@ function pageSet(only) {
|
|
|
25
70
|
return wanted;
|
|
26
71
|
}
|
|
27
72
|
|
|
28
|
-
/**
|
|
29
|
-
|
|
73
|
+
/**
|
|
74
|
+
* Every page found, and the numbered ones `only` selects. A PDF argument
|
|
75
|
+
* contributes one page per PDF page; anything else is scanned for menu pictures.
|
|
76
|
+
* `close()` releases the open PDFs.
|
|
77
|
+
*/
|
|
78
|
+
export async function findPages({ urls = [], only, minWidth = 500, projectDir, config = defaultConfig } = {}) {
|
|
30
79
|
if (!urls.length && !config.url) throw new TablefactsError("No menu page to read: pass urls, or set url in the raw menu config.", "ECONFIG");
|
|
31
|
-
const
|
|
80
|
+
const inputs = urls.length ? urls : [config.url];
|
|
81
|
+
const pages = [];
|
|
82
|
+
const sources = [];
|
|
83
|
+
const close = async () => {
|
|
84
|
+
for (const source of sources) await source.close();
|
|
85
|
+
};
|
|
86
|
+
try {
|
|
87
|
+
// Consecutive web/image arguments are fetched together (discoverPages runs
|
|
88
|
+
// them in parallel); PDFs in between are opened in order.
|
|
89
|
+
const web = [];
|
|
90
|
+
const flushWeb = async () => {
|
|
91
|
+
if (!web.length) return;
|
|
92
|
+
for (const image of await discoverPages([...web], { minWidth: Number(minWidth) })) pages.push({ kind: "image", ...image });
|
|
93
|
+
web.length = 0;
|
|
94
|
+
};
|
|
95
|
+
for (const input of inputs) {
|
|
96
|
+
if (isPdfInput(input)) {
|
|
97
|
+
await flushWeb();
|
|
98
|
+
const { bytes, host, label } = await readPdfSource(input, { projectDir });
|
|
99
|
+
const source = await openPdf(bytes);
|
|
100
|
+
sources.push(source);
|
|
101
|
+
for (const info of await inspectPdf(source)) {
|
|
102
|
+
pages.push({ kind: "pdf", source, host, label, pageNumber: info.number, url: `${label}#${info.number}`, chars: info.chars });
|
|
103
|
+
}
|
|
104
|
+
} else {
|
|
105
|
+
web.push(input);
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
await flushWeb();
|
|
109
|
+
} catch (error) {
|
|
110
|
+
await close();
|
|
111
|
+
throw error;
|
|
112
|
+
}
|
|
113
|
+
const numbered = pages.map((page, i) => ({ ...page, number: i + 1 }));
|
|
32
114
|
const wanted = only === undefined || only === "" ? null : pageSet(only);
|
|
33
|
-
const chosen =
|
|
34
|
-
if (!chosen.length)
|
|
35
|
-
|
|
115
|
+
const chosen = numbered.filter((page) => !wanted || wanted.has(page.number));
|
|
116
|
+
if (!chosen.length) {
|
|
117
|
+
await close();
|
|
118
|
+
throw optionError("only", `\`only\` ${only} matches none of the ${numbered.length} pages found.`);
|
|
119
|
+
}
|
|
120
|
+
return { pages: numbered, chosen, close };
|
|
36
121
|
}
|
|
37
122
|
|
|
38
123
|
/**
|
|
39
|
-
* The
|
|
124
|
+
* The pages `--list` prints, numbered, including PDF pages with their text size.
|
|
40
125
|
* @param {import('../../lib/types.mjs').ListMenuImagesOptions} [options]
|
|
41
126
|
* @returns {Promise<import('../../lib/types.mjs').MenuImage[]>}
|
|
42
127
|
*/
|
|
43
128
|
export async function listMenuImages(options = {}) {
|
|
44
|
-
|
|
129
|
+
const found = await findPages(options);
|
|
130
|
+
try {
|
|
131
|
+
return found.chosen.map((page) =>
|
|
132
|
+
page.kind === "pdf"
|
|
133
|
+
? { number: page.number, url: page.url, alt: "", kind: "pdf", chars: page.chars }
|
|
134
|
+
: { number: page.number, url: page.url, alt: page.alt ?? "", kind: "image" },
|
|
135
|
+
);
|
|
136
|
+
} finally {
|
|
137
|
+
await found.close();
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
/** A crop of the page render tied to a page number; `id` makes it unique in a run. */
|
|
142
|
+
const crop = (id, png, extra = {}) => ({ id, png, ...extra });
|
|
143
|
+
|
|
144
|
+
/**
|
|
145
|
+
* The images placed on a digital PDF page, as crops of its render, filtered
|
|
146
|
+
* against the page size so icons and a full-page background are left out.
|
|
147
|
+
*/
|
|
148
|
+
function placedCrops(source, canvas, page, pageNumber) {
|
|
149
|
+
const area = canvas.width * canvas.height;
|
|
150
|
+
const rects = imageRects(canvas, page.imageCoordinates)
|
|
151
|
+
// A dish photo is a modest part of the page; a near-full-page or banner
|
|
152
|
+
// image is a background, so it is dropped and the model is asked instead.
|
|
153
|
+
.filter((rect) => rect.width >= canvas.width * 0.04 && rect.height >= canvas.height * 0.04 && rect.width <= canvas.width * 0.8 && rect.height <= canvas.height * 0.8 && rect.width * rect.height <= area * 0.5)
|
|
154
|
+
// One photo can be recorded as several overlapping pieces: keep the largest.
|
|
155
|
+
.sort((a, b) => b.width * b.height - a.width * a.height)
|
|
156
|
+
.filter((rect, _i, all) => !all.some((other) => other !== rect && other.width * other.height > rect.width * rect.height && overlaps(other, rect)));
|
|
157
|
+
return rects.map((rect, i) =>
|
|
158
|
+
crop(`${pageNumber}:p${i}`, cropPdfPixels(source, canvas, rect, { padding: 2 }), { rect: { x0: rect.x, y0: rect.y, x1: rect.x + rect.width, y1: rect.y + rect.height } }),
|
|
159
|
+
);
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
const overlaps = (a, b) => {
|
|
163
|
+
const x = Math.min(a.x + a.width, b.x + b.width) - Math.max(a.x, b.x);
|
|
164
|
+
const y = Math.min(a.y + a.height, b.y + b.height) - Math.max(a.y, b.y);
|
|
165
|
+
return x > 0 && y > 0 && x * y > a.width * a.height * 0.9;
|
|
166
|
+
};
|
|
167
|
+
|
|
168
|
+
/** The photos a scanned page's model put boxes around, as crops of its render. */
|
|
169
|
+
function boxedCrops(source, canvas, transcription, pageNumber) {
|
|
170
|
+
const crops = [];
|
|
171
|
+
let i = 0;
|
|
172
|
+
for (const section of transcription.sections ?? []) {
|
|
173
|
+
for (const item of section.items ?? []) {
|
|
174
|
+
const box = Array.isArray(item.box) && item.box.length === 4 ? item.box.map(Number) : null;
|
|
175
|
+
if (!box || !box.every((n) => Number.isFinite(n))) continue;
|
|
176
|
+
const [x, y, w, h] = box;
|
|
177
|
+
const png = cropPdfPixels(source, canvas, { x: x * canvas.width, y: y * canvas.height, width: w * canvas.width, height: h * canvas.height }, { padding: 2 });
|
|
178
|
+
crops.push(crop(`${pageNumber}:b${i++}`, png, { name: item.name }));
|
|
179
|
+
}
|
|
180
|
+
}
|
|
181
|
+
return crops;
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
/** The title line the run prints, naming the pages and who read them. */
|
|
185
|
+
function menuTitle(chosen, reading, provider, model, currency) {
|
|
186
|
+
const pdfs = chosen.filter((page) => page.kind === "pdf").length;
|
|
187
|
+
const kind = pdfs === chosen.length ? "PDF" : pdfs === 0 ? "Picture" : "Picture and PDF";
|
|
188
|
+
const from = [...new Set(chosen.map((page) => (page.kind === "pdf" ? page.host : new URL(page.url).hostname)))].join(", ");
|
|
189
|
+
return `${kind} menu: ${chosen.length} pages from ${from} (${reading} read with ${providers[provider].label} ${model}, ${chosen.length - reading} from the saved transcriptions), prices in ${currency}`;
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
/** Saves the matched crops and fills `image_url` when a base URL was given. */
|
|
193
|
+
async function saveProductImages(matches, { imageDir, imageBaseUrl, projectDir }) {
|
|
194
|
+
const dir = resolveIn(projectDir, imageDir ?? workDirIn(projectDir, "menu-images"));
|
|
195
|
+
await mkdir(dir, { recursive: true });
|
|
196
|
+
const base = imageBaseUrl ? imageBaseUrl.replace(/\/*$/, "/") : null;
|
|
197
|
+
const notes = [];
|
|
198
|
+
let index = 0;
|
|
199
|
+
for (const { placement, crop: image } of matches) {
|
|
200
|
+
// Page-prefixed so re-running a different `--only` range does not overwrite earlier files.
|
|
201
|
+
const file = `p${placement.page}-${String(index + 1).padStart(3, "0")}-${slugify(placement.name) || "foto"}.png`;
|
|
202
|
+
await writeFile(join(dir, file), image.png);
|
|
203
|
+
if (base) for (const product of placement.products) product.image_url = base + file;
|
|
204
|
+
index++;
|
|
205
|
+
}
|
|
206
|
+
notes.push(`Saved ${matches.length} product photo(s) to ${dir}.`);
|
|
207
|
+
if (matches.length) {
|
|
208
|
+
notes.push(
|
|
209
|
+
base
|
|
210
|
+
? `image_url points at ${base}<file name>; publish that folder there, or the site shows a broken picture.`
|
|
211
|
+
: "image_url was left empty: upload the photos and re-run with `imageBaseUrl` (`--image-base-url`) to fill it.",
|
|
212
|
+
);
|
|
213
|
+
}
|
|
214
|
+
return notes;
|
|
45
215
|
}
|
|
46
216
|
|
|
47
217
|
/** Reads the pages and normalizes them: `{ menu, notes, title }`, ready for importMenu. */
|
|
48
|
-
export async function fetchImageMenu({ urls = [], only, provider, model, minWidth = 500, refresh = false, apiKey, env, projectDir, config = defaultConfig, log: logOption } = {}) {
|
|
218
|
+
export async function fetchImageMenu({ urls = [], only, provider, model, minWidth = 500, refresh = false, apiKey, env, projectDir, config = defaultConfig, imageDir, imageBaseUrl, imageBoxes = "auto", log: logOption } = {}) {
|
|
49
219
|
const log = normalizeLog(logOption);
|
|
50
220
|
provider = String(provider || resolveEnv(env).MENU_VISION_PROVIDER || defaultProvider).toLowerCase();
|
|
51
221
|
if (!Object.hasOwn(providers, provider)) throw optionError("provider", `Unknown provider "${provider}" (from \`provider\` or MENU_VISION_PROVIDER). Use one of: ${providerNames}.`, "ECONFIG");
|
|
52
222
|
model ||= providers[provider].defaultModel;
|
|
223
|
+
// The base URL the saved photos will be published at; it must be a real https
|
|
224
|
+
// URL, because the database only stores https image links.
|
|
225
|
+
if (imageBaseUrl) {
|
|
226
|
+
let parsed = null;
|
|
227
|
+
try {
|
|
228
|
+
parsed = new URL(imageBaseUrl);
|
|
229
|
+
} catch {
|
|
230
|
+
parsed = null;
|
|
231
|
+
}
|
|
232
|
+
if (parsed?.protocol !== "https:") throw optionError("imageBaseUrl", "`imageBaseUrl` must be an https URL, because the database only stores https image links.", "EUSAGE");
|
|
233
|
+
}
|
|
234
|
+
const extractImages = Boolean(imageDir || imageBaseUrl);
|
|
235
|
+
const scale = Number(config.imageScale) > 0 ? Number(config.imageScale) : 2;
|
|
236
|
+
if (!["auto", "always"].includes(imageBoxes)) throw optionError("imageBoxes", "`imageBoxes` must be \"auto\" or \"always\".", "EUSAGE");
|
|
237
|
+
// `always` reads every PDF page as a picture so the model can box every dish's
|
|
238
|
+
// photo, even when the page also places some photos separately; `auto` only
|
|
239
|
+
// does that when the page has no separately placed photo.
|
|
240
|
+
const alwaysBoxes = imageBoxes === "always";
|
|
53
241
|
|
|
54
|
-
const
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
242
|
+
const found = await findPages({ urls, only, minWidth, projectDir, config });
|
|
243
|
+
try {
|
|
244
|
+
const transcriptions = new Array(found.chosen.length);
|
|
245
|
+
const pageImages = new Map();
|
|
246
|
+
let next = 0;
|
|
247
|
+
let reading = 0;
|
|
248
|
+
const worker = async () => {
|
|
249
|
+
while (next < found.chosen.length) {
|
|
250
|
+
const slot = next++;
|
|
251
|
+
const page = found.chosen[slot];
|
|
252
|
+
const host = page.kind === "pdf" ? page.host : new URL(page.url).hostname;
|
|
253
|
+
const id = page.kind === "pdf" ? pageId(`${page.label}#${page.pageNumber}`) : pageId(page.url);
|
|
254
|
+
const dir = join(workDirIn(projectDir, "cache"), host);
|
|
255
|
+
await mkdir(dir, { recursive: true });
|
|
256
|
+
const cachedFile = join(dir, `${id}.json`);
|
|
257
|
+
const saved = refresh
|
|
258
|
+
? null
|
|
259
|
+
: await readFile(cachedFile, "utf8").catch((error) => {
|
|
260
|
+
if (error.code === "ENOENT") return null;
|
|
261
|
+
throw error;
|
|
262
|
+
});
|
|
263
|
+
|
|
264
|
+
let transcription = null;
|
|
265
|
+
if (saved !== null) {
|
|
266
|
+
try {
|
|
267
|
+
transcription = JSON.parse(saved);
|
|
268
|
+
} catch (error) {
|
|
269
|
+
throw new TablefactsError(`The saved transcription ${cachedFile} is not valid JSON (fix it, delete it, or pass \`refresh\`).`, "EFAILED", { cause: error });
|
|
270
|
+
}
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
const isPdf = page.kind === "pdf";
|
|
274
|
+
const read = isPdf ? await readPdfText(page.source, page.pageNumber) : null;
|
|
275
|
+
|
|
276
|
+
// When photos are wanted, render the page first: if it has no separately
|
|
277
|
+
// placed photo, the model must look at the page itself (a flattened or
|
|
278
|
+
// vector design), so the page is read as a picture with boxes rather
|
|
279
|
+
// than from its text.
|
|
280
|
+
let render = null;
|
|
281
|
+
let placed = null;
|
|
282
|
+
if (extractImages && isPdf) {
|
|
283
|
+
const rendered = await renderPdfPage(page.source, read.page, { scale, recordImages: true });
|
|
284
|
+
render = { page: read.page, ...rendered };
|
|
285
|
+
placed = placedCrops(page.source, render.canvas, read.page, page.number);
|
|
286
|
+
}
|
|
287
|
+
const useVision = isPdf && (page.chars < MIN_PAGE_TEXT || (extractImages && (alwaysBoxes || placed.length === 0)));
|
|
288
|
+
// A page read before photos were asked for has no box on its items: read
|
|
289
|
+
// it again so the model can point at each dish's photo. A page whose
|
|
290
|
+
// items all carry a (null) box was already read with boxes.
|
|
291
|
+
if (useVision && extractImages && transcription && !hasBoxField(transcription)) transcription = null;
|
|
292
|
+
|
|
293
|
+
if (transcription === null) {
|
|
294
|
+
reading++;
|
|
295
|
+
log(`Reading page ${page.number}...`);
|
|
296
|
+
if (page.kind === "image") {
|
|
297
|
+
const local = await downloadPage(page, host, { projectDir });
|
|
298
|
+
transcription = await readPage(local, { provider, model, apiKey, env });
|
|
299
|
+
} else if (!useVision) {
|
|
300
|
+
let { text } = read;
|
|
301
|
+
if (text.length > MAX_PAGE_TEXT) {
|
|
302
|
+
log(`Page ${page.number} has ${text.length} characters of text; only the first ${MAX_PAGE_TEXT} were read.`, "warn");
|
|
303
|
+
text = text.slice(0, MAX_PAGE_TEXT);
|
|
304
|
+
}
|
|
305
|
+
transcription = await readText({ text }, { provider, model, apiKey, env });
|
|
306
|
+
} else {
|
|
307
|
+
if (!render) render = { page: read.page, ...(await renderPdfPage(page.source, read.page, { scale })) };
|
|
308
|
+
const png = render.canvas.toBuffer("image/png");
|
|
309
|
+
if (png.length > MAX_PAGE_IMAGE) {
|
|
310
|
+
throw new TablefactsError(`Page ${page.number} rendered to ${(png.length / 1048576).toFixed(1)} MB, more than the vision APIs take. Lower \`imageScale\` in the raw config and try again.`, "EFAILED");
|
|
311
|
+
}
|
|
312
|
+
const image = join(dir, `${id}.png`);
|
|
313
|
+
await writeFile(image, png);
|
|
314
|
+
transcription = await readPage({ file: image, mediaType: "image/png" }, { provider, model, apiKey, env, boxes: extractImages });
|
|
315
|
+
}
|
|
316
|
+
await writeFile(cachedFile, JSON.stringify(transcription, null, 2) + "\n");
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
if (extractImages && isPdf) {
|
|
320
|
+
pageImages.set(
|
|
321
|
+
page.number,
|
|
322
|
+
useVision
|
|
323
|
+
? { width: render.canvas.width, items: [], placed: [], direct: boxedCrops(page.source, render.canvas, transcription, page.number) }
|
|
324
|
+
: { width: render.canvas.width, items: canvasItems(positionedItems(read.items), render.viewport), placed, direct: [] },
|
|
325
|
+
);
|
|
326
|
+
}
|
|
327
|
+
if (render) destroyPage(page.source, render);
|
|
328
|
+
transcriptions[slot] = transcription;
|
|
73
329
|
}
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
330
|
+
};
|
|
331
|
+
await Promise.all(Array.from({ length: Math.min(3, found.chosen.length) }, worker));
|
|
332
|
+
|
|
333
|
+
const { menu, notes, currency, placements } = normalizePages(transcriptions.map((t, i) => ({ ...t, number: found.chosen[i].number })), config);
|
|
334
|
+
const hasPdf = found.chosen.some((page) => page.kind === "pdf");
|
|
335
|
+
if (extractImages && hasPdf) {
|
|
336
|
+
const { matches, notes: imageNotes } = matchPlacements(placements, pageImages);
|
|
337
|
+
notes.push(...imageNotes, ...(await saveProductImages(matches, { imageDir, imageBaseUrl, projectDir })));
|
|
338
|
+
} else if (hasPdf) {
|
|
339
|
+
notes.push("To also save the photos printed on a PDF page, pass `imageDir` (CLI: --images <folder>).");
|
|
78
340
|
}
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
};
|
|
341
|
+
notes.unshift(`Prices were read from the menu pages: check them against the original before writing. Transcriptions are saved in .tablefacts/cache/ (edit a .json there to fix a page, then run again).`);
|
|
342
|
+
return {
|
|
343
|
+
menu,
|
|
344
|
+
notes,
|
|
345
|
+
title: menuTitle(found.chosen, reading, provider, model, currency),
|
|
346
|
+
// The restaurant's own table set, so a shared database is never touched by accident.
|
|
347
|
+
tablePrefix: config.tablePrefix,
|
|
348
|
+
};
|
|
349
|
+
} finally {
|
|
350
|
+
await found.close();
|
|
351
|
+
}
|
|
91
352
|
}
|
|
92
353
|
|
|
93
354
|
/**
|
|
94
|
-
* Reads the menu from pictures
|
|
355
|
+
* Reads the menu from pictures or a PDF and imports it. Takes importMenu's options too.
|
|
95
356
|
* @param {import('../../lib/types.mjs').ImportImageMenuOptions} [options]
|
|
96
357
|
* @returns {Promise<import('../../lib/types.mjs').ImportResult>}
|
|
97
358
|
*/
|
|
98
359
|
export async function importImageMenu({ urls, only, provider, model, minWidth, refresh, apiKey, env, config = defaultConfig, ...importOptions } = {}) {
|
|
99
|
-
const fetched = await fetchImageMenu({
|
|
360
|
+
const fetched = await fetchImageMenu({
|
|
361
|
+
urls,
|
|
362
|
+
only,
|
|
363
|
+
provider,
|
|
364
|
+
model,
|
|
365
|
+
minWidth,
|
|
366
|
+
refresh,
|
|
367
|
+
apiKey,
|
|
368
|
+
env,
|
|
369
|
+
config,
|
|
370
|
+
imageDir: importOptions.imageDir,
|
|
371
|
+
imageBaseUrl: importOptions.imageBaseUrl,
|
|
372
|
+
imageBoxes: importOptions.imageBoxes,
|
|
373
|
+
projectDir: importOptions.projectDir,
|
|
374
|
+
log: importOptions.log,
|
|
375
|
+
});
|
|
100
376
|
// An explicit `tablePrefix` (or the CLI's --table-prefix) wins over the config's.
|
|
101
377
|
return importMenu({ ...fetched, ...importOptions, tablePrefix: importOptions.tablePrefix ?? config.tablePrefix, env });
|
|
102
378
|
}
|
|
@@ -30,10 +30,14 @@ const listNames = (names) => names.slice(0, 6).join(", ") + (names.length > 6 ?
|
|
|
30
30
|
* product per column, "Name (Botella)", because a product has one price.
|
|
31
31
|
* @param {{ number?: number, notes?: string[], sections?: any[] }[]} pages transcriptions, one per page
|
|
32
32
|
* @param {import('../../lib/types.mjs').RawConfig} config
|
|
33
|
-
* @returns {{ menu: import('../../lib/types.mjs').Menu, notes: string[], currency: string }}
|
|
33
|
+
* @returns {{ menu: import('../../lib/types.mjs').Menu, notes: string[], currency: string, placements: { page: number, name: string, box: number[] | null, products: import('../../lib/types.mjs').MenuProduct[] }[] }}
|
|
34
34
|
*/
|
|
35
35
|
export function normalizePages(pages, config) {
|
|
36
36
|
const notes = [];
|
|
37
|
+
// What each item's products were built from, so a printed photo can be attached
|
|
38
|
+
// afterwards (raw/images.mjs). `box` is the model's rectangle on the page, when
|
|
39
|
+
// the page was read from a picture and product photos were requested.
|
|
40
|
+
const placements = [];
|
|
37
41
|
const currency = String(config.currency ?? "").toUpperCase();
|
|
38
42
|
if (!isCurrency(currency)) throw new TablefactsError(`config.currency "${config.currency}" is not a currency code (such as COP or USD).`, "ECONFIG");
|
|
39
43
|
|
|
@@ -90,15 +94,22 @@ export function normalizePages(pages, config) {
|
|
|
90
94
|
}
|
|
91
95
|
const variants = prices.length > 1 && prices.every((p) => p.label) ? prices : prices.slice(0, 1);
|
|
92
96
|
if (prices.length > 1 && variants.length === 1) unlabeled.push(itemName);
|
|
97
|
+
const created = [];
|
|
93
98
|
for (const { value, label: variant } of variants) {
|
|
94
|
-
|
|
99
|
+
const product = {
|
|
95
100
|
name: variants.length > 1 ? `${itemName} (${variant})` : itemName,
|
|
96
101
|
description: cleanText(item.description) || null,
|
|
97
102
|
price: value,
|
|
98
103
|
currency,
|
|
99
104
|
image_url: null,
|
|
100
105
|
recommended: false,
|
|
101
|
-
}
|
|
106
|
+
};
|
|
107
|
+
products.push(product);
|
|
108
|
+
created.push(product);
|
|
109
|
+
}
|
|
110
|
+
if (created.length) {
|
|
111
|
+
const box = Array.isArray(item.box) && item.box.length === 4 && item.box.every((n) => Number.isFinite(n)) ? item.box.map(Number) : null;
|
|
112
|
+
placements.push({ page: page.number ?? index + 1, name: itemName, box, products: created });
|
|
102
113
|
}
|
|
103
114
|
}
|
|
104
115
|
previous = target;
|
|
@@ -116,5 +127,5 @@ export function normalizePages(pages, config) {
|
|
|
116
127
|
sections: [...c.sections].filter(([, list]) => list.length).map(([name, list]) => ({ name, products: list })),
|
|
117
128
|
}))
|
|
118
129
|
.filter((c) => c.sections.length);
|
|
119
|
-
return { menu, notes, currency };
|
|
130
|
+
return { menu, notes, currency, placements };
|
|
120
131
|
}
|