tablefacts 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (98) hide show
  1. package/.env.example +14 -0
  2. package/CHANGELOG.md +12 -0
  3. package/LICENSE +21 -0
  4. package/README.md +313 -0
  5. package/bin/tablefacts.mjs +28 -0
  6. package/package.json +77 -0
  7. package/src/index.mjs +52 -0
  8. package/src/instagram/README.md +135 -0
  9. package/src/instagram/download.mjs +62 -0
  10. package/src/instagram/index.mjs +355 -0
  11. package/src/instagram/links.mjs +55 -0
  12. package/src/instagram/record.mjs +18 -0
  13. package/src/lib/edge.mjs +49 -0
  14. package/src/lib/env.mjs +35 -0
  15. package/src/lib/errors.mjs +39 -0
  16. package/src/lib/files.mjs +10 -0
  17. package/src/lib/images.mjs +20 -0
  18. package/src/lib/log.mjs +17 -0
  19. package/src/lib/photos.mjs +42 -0
  20. package/src/lib/playwright.mjs +13 -0
  21. package/src/lib/project.mjs +42 -0
  22. package/src/lib/text.mjs +7 -0
  23. package/src/lib/types.mjs +247 -0
  24. package/src/menu/README.md +97 -0
  25. package/src/menu/cluvi/config.mjs +33 -0
  26. package/src/menu/cluvi/extract.mjs +24 -0
  27. package/src/menu/cluvi/import.mjs +30 -0
  28. package/src/menu/cluvi/source.mjs +156 -0
  29. package/src/menu/index.mjs +9 -0
  30. package/src/menu/lib/db.mjs +119 -0
  31. package/src/menu/lib/import.mjs +126 -0
  32. package/src/menu/lib/menu.mjs +95 -0
  33. package/src/menu/lib/run.mjs +83 -0
  34. package/src/menu/raw/config.mjs +38 -0
  35. package/src/menu/raw/extract.mjs +72 -0
  36. package/src/menu/raw/import.mjs +99 -0
  37. package/src/menu/raw/normalize.mjs +120 -0
  38. package/src/menu/raw/source.mjs +116 -0
  39. package/src/menu/raw/vision.mjs +252 -0
  40. package/src/research/README.md +69 -0
  41. package/src/research/index.mjs +147 -0
  42. package/src/research/lib/google.mjs +93 -0
  43. package/src/research/lib/hours.mjs +109 -0
  44. package/src/research/lib/merge.mjs +119 -0
  45. package/src/research/lib/osm.mjs +49 -0
  46. package/src/research/lib/report.mjs +118 -0
  47. package/src/research/lib/social.mjs +49 -0
  48. package/src/research/lib/util.mjs +104 -0
  49. package/src/research/lib/website.mjs +285 -0
  50. package/src/research/research.mjs +67 -0
  51. package/src/tripadvisor/README.md +78 -0
  52. package/src/tripadvisor/index.mjs +178 -0
  53. package/src/tripadvisor/links.mjs +78 -0
  54. package/src/tripadvisor/photos.mjs +55 -0
  55. package/types/index.d.mts +65 -0
  56. package/types/instagram/download.d.mts +1 -0
  57. package/types/instagram/index.d.mts +13 -0
  58. package/types/instagram/links.d.mts +14 -0
  59. package/types/instagram/record.d.mts +1 -0
  60. package/types/lib/edge.d.mts +14 -0
  61. package/types/lib/env.d.mts +12 -0
  62. package/types/lib/errors.d.mts +25 -0
  63. package/types/lib/files.d.mts +1 -0
  64. package/types/lib/images.d.mts +6 -0
  65. package/types/lib/log.d.mts +5 -0
  66. package/types/lib/photos.d.mts +30 -0
  67. package/types/lib/playwright.d.mts +1677 -0
  68. package/types/lib/project.d.mts +32 -0
  69. package/types/lib/text.d.mts +4 -0
  70. package/types/lib/types.d.mts +668 -0
  71. package/types/menu/cluvi/config.d.mts +13 -0
  72. package/types/menu/cluvi/extract.d.mts +2 -0
  73. package/types/menu/cluvi/import.d.mts +35 -0
  74. package/types/menu/cluvi/source.d.mts +37 -0
  75. package/types/menu/index.d.mts +6 -0
  76. package/types/menu/lib/db.d.mts +23 -0
  77. package/types/menu/lib/import.d.mts +11 -0
  78. package/types/menu/lib/menu.d.mts +24 -0
  79. package/types/menu/lib/run.d.mts +27 -0
  80. package/types/menu/raw/config.d.mts +18 -0
  81. package/types/menu/raw/extract.d.mts +2 -0
  82. package/types/menu/raw/import.d.mts +64 -0
  83. package/types/menu/raw/normalize.d.mts +29 -0
  84. package/types/menu/raw/source.d.mts +19 -0
  85. package/types/menu/raw/vision.d.mts +13 -0
  86. package/types/research/index.d.mts +9 -0
  87. package/types/research/lib/google.d.mts +12 -0
  88. package/types/research/lib/hours.d.mts +22 -0
  89. package/types/research/lib/merge.d.mts +87 -0
  90. package/types/research/lib/osm.d.mts +36 -0
  91. package/types/research/lib/report.d.mts +6 -0
  92. package/types/research/lib/social.d.mts +118 -0
  93. package/types/research/lib/util.d.mts +45 -0
  94. package/types/research/lib/website.d.mts +283 -0
  95. package/types/research/research.d.mts +1 -0
  96. package/types/tripadvisor/index.d.mts +11 -0
  97. package/types/tripadvisor/links.d.mts +13 -0
  98. package/types/tripadvisor/photos.d.mts +1 -0
@@ -0,0 +1,252 @@
1
+ // Reads one menu page picture with a vision model and returns its sections and
2
+ // dishes as data. The model transcribes; it does not interpret: prices come back
3
+ // as the text printed on the page ("$95.000") and normalize.mjs turns them into
4
+ // numbers, so a misjudged thousands separator cannot silently turn 95.000
5
+ // into 95.
6
+ //
7
+ // Anthropic, Gemini and Groq can read the pages. A provider is only how to send
8
+ // the picture and how to get the transcription out of the answer; the schema,
9
+ // the instructions, the retries and the checks are shared, so all of them return
10
+ // the same `{ sections, notes }`.
11
+ import { readFileSync } from "node:fs";
12
+ import { resolveEnv } from "../../lib/env.mjs";
13
+ import { optionError, TablefactsError } from "../../lib/errors.mjs";
14
+
15
+ const TOOL = "record_menu_page";
16
+ const MAX_TOKENS = 16000; // Groq's model stops at 16,384
17
+
18
+ // Every object is closed and every property required: Groq's strict mode
19
+ // demands both, and Anthropic and Gemini accept them.
20
+ const schema = {
21
+ type: "object",
22
+ additionalProperties: false,
23
+ properties: {
24
+ sections: {
25
+ type: "array",
26
+ description: "Every menu section visible on the page, top to bottom (left column before right).",
27
+ items: {
28
+ type: "object",
29
+ additionalProperties: false,
30
+ properties: {
31
+ title: {
32
+ type: ["string", "null"],
33
+ description: "The section heading as printed (e.g. ENTRADAS, GIN). null when the page starts with dishes that continue the previous page's section, with no heading of their own.",
34
+ },
35
+ group: {
36
+ type: "string",
37
+ enum: ["food", "drink", "other"],
38
+ description: "food: dishes, sides, desserts. drink: cocktails, wine, beer, spirits, coffee, soft drinks. other: anything not sold from the menu (thanks, QR code, chef's note).",
39
+ },
40
+ items: {
41
+ type: "array",
42
+ items: {
43
+ type: "object",
44
+ additionalProperties: false,
45
+ properties: {
46
+ name: { type: "string" },
47
+ description: { type: ["string", "null"], description: "Ingredients or details printed under or beside the name, joined into one line; null if none." },
48
+ prices: {
49
+ type: "array",
50
+ description: "One entry per price printed for this item, left to right.",
51
+ items: {
52
+ type: "object",
53
+ additionalProperties: false,
54
+ properties: {
55
+ text: { type: "string", description: "The price exactly as printed, e.g. \"$95.000\" or \"12,5\"." },
56
+ label: {
57
+ type: ["string", "null"],
58
+ description: "What this price is for when the page says so, in the page's language: a size, a serving, or what a column icon stands for (a bottle icon is \"Botella\", a glass icon is \"Copa\" or \"Trago\"). null when there is a single price.",
59
+ },
60
+ },
61
+ required: ["text", "label"],
62
+ },
63
+ },
64
+ },
65
+ required: ["name", "description", "prices"],
66
+ },
67
+ },
68
+ },
69
+ required: ["title", "group", "items"],
70
+ },
71
+ },
72
+ notes: {
73
+ type: "array",
74
+ items: { type: "string" },
75
+ description: "Anything a person should check: text too small or blurred to read with confidence, an item you could not place, a price you are unsure of. Empty when the page is clear.",
76
+ },
77
+ },
78
+ required: ["sections", "notes"],
79
+ };
80
+
81
+ const intro = "You are transcribing one page of a restaurant menu from a picture, to load it into a database.";
82
+
83
+ const rules = `Rules:
84
+ - Transcribe, do not improve. Keep the language of the page, its spelling and accents. Never translate, invent, or fill in an item, ingredient or price that is not visible.
85
+ - Dish and drink names printed in capitals only as a style (LOMO AL GRILL) are written with normal capitalization (Lomo al Grill); keep capitals that are part of the name (MOM Gin, VSPR).
86
+ - Join a description that wraps over several lines into one line. Drop the dotted or underscore leaders between a name and its price.
87
+ - A price is only a price: copy the digits and separators as printed, drop nothing, add nothing. Currency symbols may stay.
88
+ - When a section has several price columns, give every item one price per column in the same order, and label each by the column heading or icon.
89
+ - Boxes and panels on one page (a side-dishes panel under the main list) are sections of their own.
90
+ - Decorative borders, logos, page numbers and chef's notes are not items. A page with no dishes or drinks (a thank-you card) returns no sections with items.
91
+ - If something is hard to read, give your best reading and say so in notes.`;
92
+
93
+ // Anthropic is made to call a tool. Gemini and Groq answer in JSON whose shape
94
+ // their server enforces; the schema goes in the prompt as well, because the
95
+ // descriptions in it only help if the model sees them.
96
+ const asToolCall = `${intro} Call ${TOOL} exactly once.\n\n${rules}`;
97
+ const asJson = `${intro} Answer with one JSON object that follows this JSON Schema, and nothing else.\n\n${rules}\n\nJSON Schema:\n${JSON.stringify(schema)}`;
98
+
99
+ const unusable = (why) => new TablefactsError(`The model did not return a usable transcription: ${why}.`, "EFAILED");
100
+
101
+ /** A JSON answer that stops in the middle was cut off by the token limit. */
102
+ function parseJson(text, cutOff) {
103
+ try {
104
+ return JSON.parse(text);
105
+ } catch {
106
+ throw unusable(cutOff ? "its answer was cut off (token limit)" : "its answer was not valid JSON");
107
+ }
108
+ }
109
+
110
+ // `request` builds the call for one picture (`data` is its base64) and `read`
111
+ // returns the transcription object from the answer, or throws why there is none.
112
+ /** @type {Record<string, import('../../lib/types.mjs').VisionProvider>} */
113
+ export const providers = {
114
+ anthropic: {
115
+ label: "Anthropic",
116
+ keyName: "ANTHROPIC_API_KEY",
117
+ defaultModel: "claude-sonnet-5-5",
118
+ request: ({ model, apiKey, data, mediaType }) => ({
119
+ url: "https://api.anthropic.com/v1/messages",
120
+ headers: { "x-api-key": apiKey, "anthropic-version": "2023-06-01" },
121
+ body: {
122
+ model,
123
+ max_tokens: MAX_TOKENS,
124
+ tools: [{ name: TOOL, description: "Record the sections and items transcribed from the menu page.", input_schema: schema }],
125
+ tool_choice: { type: "tool", name: TOOL },
126
+ messages: [
127
+ {
128
+ role: "user",
129
+ content: [
130
+ { type: "image", source: { type: "base64", media_type: mediaType, data } },
131
+ { type: "text", text: asToolCall },
132
+ ],
133
+ },
134
+ ],
135
+ },
136
+ }),
137
+ read(response) {
138
+ const call = response.content?.find((block) => block.type === "tool_use" && block.name === TOOL);
139
+ if (!Array.isArray(call?.input?.sections)) {
140
+ throw unusable(response.stop_reason === "max_tokens" ? "its answer was cut off (max_tokens)" : `it answered without calling ${TOOL}`);
141
+ }
142
+ return call.input;
143
+ },
144
+ },
145
+
146
+ gemini: {
147
+ label: "Gemini",
148
+ keyName: "GEMINI_API_KEY",
149
+ defaultModel: "gemini-3.8-flash",
150
+ // generateContent, not the Interactions API the guides now lead with: that
151
+ // one is in beta and its schema has already changed once, while Google
152
+ // keeps generateContent as the path for stable use.
153
+ request: ({ model, apiKey, data, mediaType }) => ({
154
+ url: `https://generativelanguage.googleapis.com/v1beta/models/${encodeURIComponent(model)}:generateContent`,
155
+ headers: { "x-goog-api-key": apiKey },
156
+ body: {
157
+ contents: [{ role: "user", parts: [{ text: asJson }, { inlineData: { mimeType: mediaType, data } }] }],
158
+ // Thinking models count their thoughts against the limit: leave room.
159
+ generationConfig: { responseMimeType: "application/json", responseJsonSchema: schema, maxOutputTokens: 32000 },
160
+ },
161
+ }),
162
+ read(response) {
163
+ const candidate = response.candidates?.[0];
164
+ const text = (candidate?.content?.parts ?? []).filter((part) => !part.thought).map((part) => part.text ?? "").join("");
165
+ if (!text) {
166
+ const blocked = response.promptFeedback?.blockReason;
167
+ throw unusable(blocked ? `Gemini blocked the page (${blocked})` : `it answered with no text (${candidate?.finishReason ?? "no candidate"})`);
168
+ }
169
+ return parseJson(text, candidate.finishReason === "MAX_TOKENS");
170
+ },
171
+ },
172
+
173
+ groq: {
174
+ label: "Groq",
175
+ keyName: "GROQ_API_KEY",
176
+ // The only vision model Groq lists (October 2026), and a preview one: when
177
+ // it is retired, `model` names its successor.
178
+ defaultModel: "qwen/qwen3.8-27b",
179
+ request: ({ model, apiKey, data, mediaType }) => ({
180
+ url: "https://api.groq.com/openai/v1/chat/completions",
181
+ headers: { authorization: `Bearer ${apiKey}` },
182
+ body: {
183
+ model,
184
+ messages: [{ role: "user", content: [{ type: "text", text: asJson }, { type: "image_url", image_url: { url: `data:${mediaType};base64,${data}` } }] }],
185
+ response_format: { type: "json_schema", json_schema: { name: TOOL, strict: true, schema } },
186
+ // Reading a page needs no reasoning, and none may leak into the JSON.
187
+ reasoning_effort: "none",
188
+ reasoning_format: "hidden",
189
+ max_completion_tokens: MAX_TOKENS,
190
+ },
191
+ }),
192
+ read(response) {
193
+ const choice = response.choices?.[0];
194
+ const text = choice?.message?.content;
195
+ if (!text) throw unusable(`it answered with no text (${choice?.finish_reason ?? "no choice"})`);
196
+ return parseJson(text, choice.finish_reason === "length");
197
+ },
198
+ },
199
+ };
200
+
201
+ export const defaultProvider = "anthropic";
202
+
203
+ const ATTEMPTS = 4;
204
+ const RETRY_STATUSES = [408, 429, 500, 502, 503, 504, 529];
205
+ const LONGEST_WAIT = 60; // seconds
206
+
207
+ /** Seconds the service asks us to wait: Retry-After (Groq, Anthropic) or the RetryInfo in a Gemini error. */
208
+ function askedWait(res, detail) {
209
+ return [Number(res.headers.get("retry-after")), Number(detail.match(/"retryDelay":\s*"([\d.]+)s"/)?.[1])].find((seconds) => seconds > 0);
210
+ }
211
+
212
+ async function callApi(label, { url, headers, body }) {
213
+ let failure;
214
+ const payload = JSON.stringify(body);
215
+ for (let attempt = 1; attempt <= ATTEMPTS; attempt++) {
216
+ let wait = attempt * 2000;
217
+ try {
218
+ const res = await fetch(url, {
219
+ method: "POST",
220
+ headers: { "content-type": "application/json", ...headers },
221
+ body: payload,
222
+ signal: AbortSignal.timeout(180_000),
223
+ });
224
+ if (res.ok) return await res.json();
225
+ const detail = await res.text();
226
+ failure = Object.assign(new TablefactsError(`${label} API ${res.status}: ${detail.slice(0, 300)}`, "EFAILED"), { status: res.status });
227
+ if (!RETRY_STATUSES.includes(res.status)) break; // a bad key or request will not fix itself
228
+ const asked = askedWait(res, detail);
229
+ if (asked > LONGEST_WAIT) break; // a quota that refills in hours is not worth waiting for here
230
+ if (asked) wait = asked * 1000;
231
+ } catch (error) {
232
+ failure = error;
233
+ }
234
+ if (attempt < ATTEMPTS) await new Promise((resolve) => setTimeout(resolve, wait));
235
+ }
236
+ throw failure;
237
+ }
238
+
239
+ /** `file` is a downloaded picture; returns `{ sections, notes }` as the schema above describes. */
240
+ export async function readPage({ file, mediaType }, { provider = defaultProvider, model, apiKey, env } = {}) {
241
+ const reader = Object.hasOwn(providers, provider) ? providers[provider] : null;
242
+ const names = Object.keys(providers).join(", ");
243
+ if (!reader) throw optionError("provider", `"${provider}" is not a provider the pages can be read with (${names}).`, "ECONFIG");
244
+ apiKey ??= resolveEnv(env)[reader.keyName];
245
+ if (!apiKey) {
246
+ throw optionError("provider", `${reader.keyName} is not set. Add it to .env (see .env.example); the menu pages are read with ${reader.label}. To use another provider, pass \`provider\` (${names}).`, "ECONFIG");
247
+ }
248
+ const request = reader.request({ model: model ?? reader.defaultModel, apiKey, data: readFileSync(file).toString("base64"), mediaType });
249
+ const answer = reader.read(await callApi(reader.label, request));
250
+ if (!Array.isArray(answer?.sections)) throw unusable('its answer has no "sections" list');
251
+ return { sections: answer.sections, notes: Array.isArray(answer.notes) ? answer.notes : [] };
252
+ }
@@ -0,0 +1,69 @@
1
+ # Restaurant research
2
+
3
+ Give it a name and a place, it reads the public web and writes what the template needs.
4
+
5
+ ```bash
6
+ tablefacts research "Casa Luna" "Medellín, Colombia" --country CO
7
+ tablefacts research "Casa Luna" "Medellín" --website https://casaluna.co --instagram casaluna --photos 6
8
+ ```
9
+
10
+ Output goes to `.tablefacts/research/<slug>/` (git-ignored, it holds third-party data):
11
+
12
+ | File | What it is |
13
+ | --- | --- |
14
+ | `report.md` | Read first. Facts table in BRIEF.md's order (value, source, confidence, where it goes), disagreements, hours as `hoursRows` for both languages, links (menu, delivery, hub), material for the copy, what to ask the client |
15
+ | `profile.json` | The same with provenance, plus each source's raw result |
16
+ | `setup-answers.txt` | One line per prompt of `npm run setup`; blank keeps the placeholder |
17
+ | `photos/` | Only with `--photos n`. Reference, not assets: ask the client for originals |
18
+
19
+ From code, `research()` returns the same data (`profile`, `notes`, `photos`, `outDir`, `files`, the paths absolute);
20
+ see the [main README](../../README.md#use-it-as-a-library). `out` (relative paths resolve against `projectDir`),
21
+ `projectDir`, `env` (where `GOOGLE_PLACES_API_KEY` is read from, `process.env` by default), `googleKey` and
22
+ `log(message, level)` are options; it is silent without `log`. A missing `name` throws a `TablefactsError`
23
+ (`EUSAGE`, `err.option === 'name'`; the CLI exits `2`). A source that fails never throws: it is a note in the report.
24
+
25
+ ## Sources
26
+
27
+ | Source | Gives | Notes |
28
+ | --- | --- | --- |
29
+ | Google Maps (Places API) | address, map point, phone, hours, price, cuisine types, photos, status | Needs `GOOGLE_PLACES_API_KEY` in `.env` (Places API (New) enabled). Phone and hours fields are billed at the Enterprise rate; one search per run |
30
+ | OpenStreetMap (Nominatim) | address, map point, sometimes phone, hours, website, socials | No key. Often stale or sparse |
31
+ | Website | JSON-LD, contact page, WhatsApp, reserve platform, menu links, images, logo, theme colour, page text | Falls back to Playwright on 403 or client-rendered pages (`--render` forces it) |
32
+ | Instagram | handle, followers, bio, bio link | Public share card only. Usually login-walled |
33
+ | TripAdvisor | JSON-LD (address, phone, cuisine, price, rating) | Usually bot-blocked: the link is kept for a manual read |
34
+ | Linktree and similar | WhatsApp, reserve, menu, delivery links | Found from the website, Instagram bio or `--linktree` |
35
+
36
+ Sources find each other (website → Instagram, TripAdvisor, hub). Order of trust per field is in `lib/merge.mjs`; two sources agreeing is "high", one is "medium", a guess is "low".
37
+
38
+ ## How it runs
39
+
40
+ - **Concurrent lookups.** Google and OpenStreetMap run together; then the website is read. Instagram, TripAdvisor and
41
+ the link-in-bio page start as soon as their address is known (given, or found on the website), so a slow source does
42
+ not hold up the others. Report notes keep a fixed order whatever finishes first. Photo downloads run four at a time.
43
+ - **One shared browser.** Playwright's Chromium is launched at most once per run, only when a page needs rendering
44
+ (`--render`, a 403, a client-rendered page), is shared by the website, TripAdvisor and link-in-bio readers, and is
45
+ closed at the end even on failure.
46
+ - **Fetch cap.** Pages are read up to 2 MB each; larger bodies are truncated, so a huge page cannot exhaust memory.
47
+
48
+ ## Limits
49
+
50
+ - It does not log in anywhere or get past bot protection. A blocked source is reported, not worked around.
51
+ - It finds facts, not copy, logos or licensed photos.
52
+ - A wrong match (a sister branch, a closed place) is the main risk: read the Warnings and "Sources disagree" first.
53
+
54
+ ## For agents
55
+
56
+ Use it first, whenever you are given only a restaurant's name and place (or little else).
57
+
58
+ 1. **Run it** from the repo root: `tablefacts research "<name>" "<city, country>" --country <ISO>`. Add `--website`, `--instagram`, `--tripadvisor` or `--linktree` for anything the user gave you; a user-supplied value is trusted over a search result. It takes under a minute, needs no prompts and prints the output folder.
59
+ 2. **Read `out/<slug>/report.md`**, in this order: Warnings, "What each source did", Facts, "Sources disagree". Confirm the match is the right restaurant (name, address, not closed) before using anything. If it matched the wrong place, rerun with a more specific location or `--website`.
60
+ 3. **Treat every value as unconfirmed.** Fill `BRIEF.md` with the value and its source, never as client-confirmed. Anything under "not found" stays a placeholder and goes in your hand-off as an open question. Do not invent it.
61
+ 4. **Apply the facts**: review `setup-answers.txt`, run `npm run setup < .tablefacts/research/<slug>/setup-answers.txt`, read `git diff`. Setup does not cover `phone`, `email`, `MENU_LOCALE`, `brand.tagline`, `visit.hoursRows` or `jsonld.ts`: take those from the report by hand. Hours are ready to paste. Add street, phone and hours to `jsonld.ts` only for values the client confirmed.
62
+ 5. **Use the leads**: a Cluvi menu link means `tablefacts menu cluvi`, a PDF means `tablefacts menu raw` (`src/menu/README.md`). The theme colour and logo candidates are hints for `tokens.css` and `public/logo.svg`, not decisions.
63
+ 6. **Blocked sources** (Instagram, TripAdvisor) are normal. Open the link yourself, or ask the user for the bio and hours. Do not try to get around a login wall.
64
+
65
+ `profile.json` has the same data as JSON for scripting: `fields.<name>` is `{ value, source, confidence, alternatives? }`, `hours.rows.{es,en}` are `hoursRows`, and `raw` holds each source's untouched result. Field names: `name street locality region country coordinates phone whatsapp email instagram facebook tiktok tripadvisor website reserveUrl priceRange mapsUrl descriptor cuisines menuLocale`.
66
+
67
+ Do not commit `out/`, and never paste the Google key anywhere but `.env`. Photos it downloads belong to the restaurant or the photographer: use them as reference, not as site assets.
68
+
69
+ Code map: `research.mjs` runs the sources in order and writes the files; `lib/google.mjs`, `osm.mjs`, `website.mjs`, `social.mjs` read one source each; `merge.mjs` holds the per-field trust order; `hours.mjs` normalises hours; `report.mjs` writes the report and the setup answers. A new source is a reader returning plain data, a line in `research.mjs` and its candidates in `merge.mjs`.
@@ -0,0 +1,147 @@
1
+ // Programmatic entry for the research pipeline: gathers what the public web says
2
+ // about a restaurant and writes profile.json, report.md and setup-answers.txt.
3
+ // Silent by default and never exits the process; the CLI lives in research.mjs.
4
+ import { mkdir, writeFile } from "node:fs/promises";
5
+ import { join, resolve } from "node:path";
6
+ import { resolveEnv } from "../lib/env.mjs";
7
+ import { optionError } from "../lib/errors.mjs";
8
+ import { normalizeLog } from "../lib/log.mjs";
9
+ import { resolveIn, workDirIn } from "../lib/project.mjs";
10
+ import { downloadPhotos, searchGoogle } from "./lib/google.mjs";
11
+ import { buildProfile } from "./lib/merge.mjs";
12
+ import { searchOsm } from "./lib/osm.mjs";
13
+ import { renderReport, setupAnswers } from "./lib/report.mjs";
14
+ import { readHub, readInstagram, readTripadvisor } from "./lib/social.mjs";
15
+ import { mapPool, slugify } from "./lib/util.mjs";
16
+ import { createBrowser, scrapeSite } from "./lib/website.mjs";
17
+
18
+ /**
19
+ * Researches a restaurant from public sources and writes profile.json, report.md and
20
+ * setup-answers.txt. Silent unless `log` is given; a source that fails becomes a note in the
21
+ * report, not an error. Throws a TablefactsError with code 'EUSAGE' (and `option` 'name') when
22
+ * `name` is missing.
23
+ * @param {import('../lib/types.mjs').ResearchOptions} options
24
+ * @returns {Promise<import('../lib/types.mjs').ResearchResult>}
25
+ */
26
+ export async function research(options = {}) {
27
+ const { name, location = "", country, website, instagram: instagramOpt, tripadvisor: tripadvisorOpt, linktree, photos: photosOpt = 0, render = false, google: useGoogle = true } = options;
28
+ if (typeof name !== "string" || !name.trim()) throw optionError("name", "research: `name` is required");
29
+ const log = normalizeLog(options.log);
30
+ const env = resolveEnv(options.env);
31
+
32
+ const slug = slugify(name);
33
+ const outDir = options.out ? resolveIn(options.projectDir, options.out) : workDirIn(options.projectDir, "research", slug);
34
+ const query = { name, location, slug, country, website, instagram: instagramOpt, tripadvisor: tripadvisorOpt, linktree };
35
+ const notes = [];
36
+
37
+ /** Runs one source; a failure is a note in the report, never the end of the run. */
38
+ async function source(label, fn, sink = notes) {
39
+ try {
40
+ return (await fn()) ?? null;
41
+ } catch (e) {
42
+ sink.push(`${label}: failed (${e.message})`);
43
+ log(` ${label}: failed (${e.message})`, "warn");
44
+ return null;
45
+ }
46
+ }
47
+
48
+ const key = useGoogle === false ? "" : (options.googleKey ?? env.GOOGLE_PLACES_API_KEY);
49
+ log(`Researching "${name}"${location ? ` in ${location}` : ""}`, "info");
50
+
51
+ // One Chromium for the whole run, launched only if a page needs rendering.
52
+ const browser = createBrowser();
53
+ let site, instagram, hub, tripadvisor, taUrl, handle, google, osm;
54
+ try {
55
+ // Google and OpenStreetMap only need the name and place, so they run together; their notes are added in this order afterwards.
56
+ const googleNotes = [];
57
+ const osmNotes = [];
58
+ if (!key) notes.push("Google Places: skipped (no GOOGLE_PLACES_API_KEY in .env, and no `googleKey`). Address, hours and phone come from OpenStreetMap and the website instead.");
59
+ const [googleResult, osmResult] = await Promise.all([
60
+ key ? source("Google Places", () => searchGoogle({ name, location, country, key }), googleNotes) : null,
61
+ source("OpenStreetMap", () => searchOsm({ name, location, country }), osmNotes),
62
+ ]);
63
+ if (googleResult) googleNotes.push(googleResult.place ? `Google Places: matched "${googleResult.place.name}", ${googleResult.place.formattedAddress}` : `Google Places: no result named like "${name}". Closest: ${googleResult.candidates.map((c) => `${c.name} (${c.address})`).join("; ") || "none"}`);
64
+ if (osmResult) osmNotes.push(osmResult.place ? `OpenStreetMap: matched "${osmResult.place.name}"` : `OpenStreetMap: no match. Closest: ${osmResult.candidates.map((c) => c.name || c.address).join("; ") || "none"}`);
65
+ notes.push(...googleNotes, ...osmNotes);
66
+ google = googleResult?.place ?? null;
67
+ osm = osmResult?.place ?? null;
68
+
69
+ const websiteUrl = website ?? google?.website ?? osm?.website;
70
+ site = websiteUrl ? await source("Website", () => scrapeSite(websiteUrl, { renderJs: render, browser })) : null;
71
+ if (site) notes.push(`Website: read ${site.pages.length} page(s) of ${site.url}${site.pages.some((p) => p.rendered) ? " (rendered with Playwright)" : ""}`);
72
+ else if (!websiteUrl) notes.push("Website: none found. Pass the website (--website / `website`) if the restaurant has one.");
73
+
74
+ // Follow the leads the first sources gave. What is already known (handle, TripAdvisor link, hub link)
75
+ // starts right away; the rest waits only for the source that can supply it.
76
+ handle = (instagramOpt ?? site?.links.instagram[0] ?? osm?.instagram ?? "").replace(/^https?:\/\/(www\.)?instagram\.com\//i, "").replace(/^@/, "").split(/[/?#]/)[0];
77
+ taUrl = tripadvisorOpt ?? site?.links.tripadvisor[0] ?? (site?.jsonld?.sameAs ?? []).find((u) => /tripadvisor\./.test(u));
78
+ const directHubUrl = linktree ?? site?.links.hubs[0];
79
+ const readHubAt = (url) => source("Link-in-bio", () => readHub(url, { browser }));
80
+
81
+ const instagramP = handle ? source("Instagram", () => readInstagram(handle)) : null;
82
+ const tripadvisorP = taUrl ? source("TripAdvisor", () => readTripadvisor(taUrl, { browser })) : null;
83
+ // The hub URL comes from the options or the website, else from the Instagram bio.
84
+ const hubTask = (async () => {
85
+ let url = directHubUrl;
86
+ if (!url) {
87
+ const ig = await instagramP;
88
+ url = ig?.links?.find((l) => /linktr|beacons|bio\.link|lnk\.bio|taplink/.test(l));
89
+ }
90
+ return url ? { url, result: await readHubAt(url) } : { url: null, result: null };
91
+ })();
92
+
93
+ instagram = instagramP ? await instagramP : null;
94
+ const hubRead = await hubTask;
95
+ hub = hubRead.result;
96
+ const hubUrl = hubRead.url;
97
+ if (!taUrl) {
98
+ // Links found on the hub can add the pages the website did not link to.
99
+ taUrl = hub?.links?.tripadvisor?.[0];
100
+ if (taUrl) tripadvisor = await source("TripAdvisor", () => readTripadvisor(taUrl, { browser }));
101
+ } else {
102
+ tripadvisor = await tripadvisorP;
103
+ }
104
+
105
+ // Notes keep the order of the sequential flow: Instagram, link-in-bio, TripAdvisor.
106
+ if (instagram) notes.push(instagram.blocked ? `Instagram @${handle}: not readable (${instagram.blocked}). Add the bio by hand, or use \`tablefacts photos instagram\` for posts.` : `Instagram @${handle}: profile card read`);
107
+ else if (!handle) notes.push("Instagram: no handle found. Pass the handle (--instagram / `instagram`).");
108
+ if (hub) notes.push(hub.blocked ? `Link-in-bio ${hubUrl}: not readable (${hub.blocked})` : `Link-in-bio: read ${hub.url}`);
109
+ if (tripadvisor) notes.push(tripadvisor.blocked ? `TripAdvisor ${taUrl}: blocked (${tripadvisor.blocked}). The link is kept; read it by hand.` : "TripAdvisor: page read");
110
+ else notes.push("TripAdvisor: no link found. Pass the page (--tripadvisor / `tripadvisor`).");
111
+ } finally {
112
+ await browser.close();
113
+ }
114
+
115
+ const profile = buildProfile({ query, google, osm, site, hub: hub?.blocked ? null : hub, instagram, tripadvisor: tripadvisor?.blocked ? null : tripadvisor });
116
+ if (tripadvisor?.blocked && taUrl && !profile.fields.tripadvisor) profile.fields.tripadvisor = { value: taUrl, source: "website", confidence: "medium" };
117
+
118
+ // Optional photo downloads: reference material, never wired into the site.
119
+ const photos = [];
120
+ const wanted = Number(photosOpt ?? 0);
121
+ if (wanted > 0) {
122
+ await mkdir(join(outDir, "photos"), { recursive: true });
123
+ const save = (file, buf) => writeFile(join(outDir, "photos", file), buf);
124
+ if (google && key) photos.push(...(await source("Google photos", () => downloadPhotos(google, key, wanted, save)) ?? []));
125
+ const images = (site?.images ?? []).filter((x) => /\.(jpe?g|png|webp)(\?|$)/i.test(x.url)).slice(0, wanted);
126
+ const saved = await mapPool(images, 4, async (img, i) => {
127
+ try {
128
+ const res = await fetch(img.url, { signal: AbortSignal.timeout(20000) });
129
+ if (!res.ok) return null;
130
+ const buf = Buffer.from(await res.arrayBuffer());
131
+ if (buf.length < 15000) return null;
132
+ const file = `site-${String(i + 1).padStart(2, "0")}${img.url.match(/\.(jpe?g|png|webp)/i)?.[0].toLowerCase() ?? ".jpg"}`;
133
+ await save(file, buf);
134
+ return { file, source: img.url, alt: img.alt };
135
+ } catch { return null; /* one bad image should not stop the rest */ }
136
+ });
137
+ photos.push(...saved.filter(Boolean));
138
+ }
139
+
140
+ const files = { profile: resolve(outDir, "profile.json"), report: resolve(outDir, "report.md"), setupAnswers: resolve(outDir, "setup-answers.txt") };
141
+ await mkdir(outDir, { recursive: true });
142
+ await writeFile(files.profile, JSON.stringify({ ...profile, raw: { google, osm, site, hub, instagram, tripadvisor }, photos }, null, 2));
143
+ await writeFile(files.report, renderReport(profile, { notes, photos }));
144
+ await writeFile(files.setupAnswers, setupAnswers(profile));
145
+
146
+ return { profile, notes, photos, outDir, files };
147
+ }
@@ -0,0 +1,93 @@
1
+ // Google Places API (New): the best single source for address, map point,
2
+ // phone, hours and photos. Needs GOOGLE_PLACES_API_KEY in .env.
3
+ import { fromGoogle } from "./hours.mjs";
4
+ import { mapPool, sameText } from "./util.mjs";
5
+
6
+ const FIELDS = [
7
+ "id", "displayName", "formattedAddress", "addressComponents", "location", "nationalPhoneNumber",
8
+ "internationalPhoneNumber", "websiteUri", "googleMapsUri", "regularOpeningHours", "priceLevel",
9
+ "rating", "userRatingCount", "types", "primaryTypeDisplayName", "editorialSummary", "photos",
10
+ "businessStatus", "reservable",
11
+ ].map((f) => `places.${f}`).join(",");
12
+
13
+ const GENERIC_TYPES = new Set(["restaurant", "fast_food_restaurant", "family_restaurant", "meal_takeaway", "meal_delivery", "food_court", "buffet_restaurant"]);
14
+ const PRICE = { PRICE_LEVEL_INEXPENSIVE: "$", PRICE_LEVEL_MODERATE: "$$", PRICE_LEVEL_EXPENSIVE: "$$$", PRICE_LEVEL_VERY_EXPENSIVE: "$$$$" };
15
+
16
+ const component = (place, ...types) => {
17
+ for (const t of types) {
18
+ const c = place.addressComponents?.find((x) => x.types?.includes(t));
19
+ if (c) return c;
20
+ }
21
+ return null;
22
+ };
23
+
24
+ /** "italian_restaurant" -> "Italian", "steak_house" -> "Steakhouse". */
25
+ function cuisinesFromTypes(types = []) {
26
+ const out = [];
27
+ for (const t of types) {
28
+ if (GENERIC_TYPES.has(t)) continue;
29
+ const m = t.match(/^(.+)_restaurant$/);
30
+ if (m) out.push(m[1].split("_").map((w) => w[0].toUpperCase() + w.slice(1)).join(" "));
31
+ else if (t === "steak_house") out.push("Steakhouse");
32
+ }
33
+ return out;
34
+ }
35
+
36
+ function normalize(p) {
37
+ return {
38
+ source: "google",
39
+ id: p.id,
40
+ name: p.displayName?.text ?? "",
41
+ formattedAddress: p.formattedAddress ?? "",
42
+ // The first comma-separated part is the street line in every format Google uses.
43
+ street: (p.formattedAddress ?? "").split(",")[0].trim(),
44
+ locality: (component(p, "locality", "administrative_area_level_2", "sublocality")?.longText) ?? "",
45
+ region: component(p, "administrative_area_level_1")?.longText ?? "",
46
+ country: component(p, "country")?.shortText ?? "",
47
+ postcode: component(p, "postal_code")?.longText ?? "",
48
+ coords: p.location ? { lat: p.location.latitude, lng: p.location.longitude } : null,
49
+ phone: p.internationalPhoneNumber ?? "",
50
+ nationalPhone: p.nationalPhoneNumber ?? "",
51
+ website: p.websiteUri ?? "",
52
+ mapsUrl: p.googleMapsUri ?? "",
53
+ hours: fromGoogle(p.regularOpeningHours?.periods),
54
+ priceRange: PRICE[p.priceLevel] ?? "",
55
+ rating: p.rating ?? null,
56
+ ratingCount: p.userRatingCount ?? null,
57
+ cuisines: cuisinesFromTypes(p.types),
58
+ types: p.types ?? [],
59
+ descriptor: p.primaryTypeDisplayName?.text ?? "",
60
+ summary: p.editorialSummary?.text ?? "",
61
+ status: p.businessStatus ?? "",
62
+ reservable: p.reservable ?? null,
63
+ photos: (p.photos ?? []).map((ph) => ({ name: ph.name, width: ph.widthPx, height: ph.heightPx, authors: (ph.authorAttributions ?? []).map((a) => a.displayName) })),
64
+ };
65
+ }
66
+
67
+ /** Returns { place, candidates }: `place` is the first result whose name matches, or null. */
68
+ export async function searchGoogle({ name, location, country, key }) {
69
+ const res = await fetch("https://places.googleapis.com/v1/places:searchText", {
70
+ method: "POST",
71
+ headers: { "content-type": "application/json", "x-goog-api-key": key, "x-goog-fieldmask": FIELDS },
72
+ body: JSON.stringify({ textQuery: `${name} ${location}`.trim(), maxResultCount: 5, ...(country && { regionCode: country }) }),
73
+ signal: AbortSignal.timeout(15000),
74
+ });
75
+ if (!res.ok) throw new Error(`Google Places ${res.status}: ${(await res.text()).slice(0, 300)}`);
76
+ const places = ((await res.json()).places ?? []).map(normalize);
77
+ return {
78
+ place: places.find((p) => sameText(p.name, name)) ?? null,
79
+ candidates: places.map((p) => ({ name: p.name, address: p.formattedAddress, status: p.status })),
80
+ };
81
+ }
82
+
83
+ /** Downloads up to `count` photos into `dir`, returns what was saved. The key stays out of the result. */
84
+ export async function downloadPhotos(place, key, count, save) {
85
+ const saved = await mapPool(place.photos.slice(0, count), 4, async (ph, i) => {
86
+ const res = await fetch(`https://places.googleapis.com/v1/${ph.name}/media?maxWidthPx=2400&key=${key}`, { signal: AbortSignal.timeout(30000) });
87
+ if (!res.ok) return null;
88
+ const file = `google-${String(i + 1).padStart(2, "0")}.jpg`;
89
+ await save(file, Buffer.from(await res.arrayBuffer()));
90
+ return { file, width: ph.width, height: ph.height, authors: ph.authors };
91
+ });
92
+ return saved.filter(Boolean);
93
+ }