tablefacts 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (98) hide show
  1. package/.env.example +14 -0
  2. package/CHANGELOG.md +12 -0
  3. package/LICENSE +21 -0
  4. package/README.md +313 -0
  5. package/bin/tablefacts.mjs +28 -0
  6. package/package.json +77 -0
  7. package/src/index.mjs +52 -0
  8. package/src/instagram/README.md +135 -0
  9. package/src/instagram/download.mjs +62 -0
  10. package/src/instagram/index.mjs +355 -0
  11. package/src/instagram/links.mjs +55 -0
  12. package/src/instagram/record.mjs +18 -0
  13. package/src/lib/edge.mjs +49 -0
  14. package/src/lib/env.mjs +35 -0
  15. package/src/lib/errors.mjs +39 -0
  16. package/src/lib/files.mjs +10 -0
  17. package/src/lib/images.mjs +20 -0
  18. package/src/lib/log.mjs +17 -0
  19. package/src/lib/photos.mjs +42 -0
  20. package/src/lib/playwright.mjs +13 -0
  21. package/src/lib/project.mjs +42 -0
  22. package/src/lib/text.mjs +7 -0
  23. package/src/lib/types.mjs +247 -0
  24. package/src/menu/README.md +97 -0
  25. package/src/menu/cluvi/config.mjs +33 -0
  26. package/src/menu/cluvi/extract.mjs +24 -0
  27. package/src/menu/cluvi/import.mjs +30 -0
  28. package/src/menu/cluvi/source.mjs +156 -0
  29. package/src/menu/index.mjs +9 -0
  30. package/src/menu/lib/db.mjs +119 -0
  31. package/src/menu/lib/import.mjs +126 -0
  32. package/src/menu/lib/menu.mjs +95 -0
  33. package/src/menu/lib/run.mjs +83 -0
  34. package/src/menu/raw/config.mjs +38 -0
  35. package/src/menu/raw/extract.mjs +72 -0
  36. package/src/menu/raw/import.mjs +99 -0
  37. package/src/menu/raw/normalize.mjs +120 -0
  38. package/src/menu/raw/source.mjs +116 -0
  39. package/src/menu/raw/vision.mjs +252 -0
  40. package/src/research/README.md +69 -0
  41. package/src/research/index.mjs +147 -0
  42. package/src/research/lib/google.mjs +93 -0
  43. package/src/research/lib/hours.mjs +109 -0
  44. package/src/research/lib/merge.mjs +119 -0
  45. package/src/research/lib/osm.mjs +49 -0
  46. package/src/research/lib/report.mjs +118 -0
  47. package/src/research/lib/social.mjs +49 -0
  48. package/src/research/lib/util.mjs +104 -0
  49. package/src/research/lib/website.mjs +285 -0
  50. package/src/research/research.mjs +67 -0
  51. package/src/tripadvisor/README.md +78 -0
  52. package/src/tripadvisor/index.mjs +178 -0
  53. package/src/tripadvisor/links.mjs +78 -0
  54. package/src/tripadvisor/photos.mjs +55 -0
  55. package/types/index.d.mts +65 -0
  56. package/types/instagram/download.d.mts +1 -0
  57. package/types/instagram/index.d.mts +13 -0
  58. package/types/instagram/links.d.mts +14 -0
  59. package/types/instagram/record.d.mts +1 -0
  60. package/types/lib/edge.d.mts +14 -0
  61. package/types/lib/env.d.mts +12 -0
  62. package/types/lib/errors.d.mts +25 -0
  63. package/types/lib/files.d.mts +1 -0
  64. package/types/lib/images.d.mts +6 -0
  65. package/types/lib/log.d.mts +5 -0
  66. package/types/lib/photos.d.mts +30 -0
  67. package/types/lib/playwright.d.mts +1677 -0
  68. package/types/lib/project.d.mts +32 -0
  69. package/types/lib/text.d.mts +4 -0
  70. package/types/lib/types.d.mts +668 -0
  71. package/types/menu/cluvi/config.d.mts +13 -0
  72. package/types/menu/cluvi/extract.d.mts +2 -0
  73. package/types/menu/cluvi/import.d.mts +35 -0
  74. package/types/menu/cluvi/source.d.mts +37 -0
  75. package/types/menu/index.d.mts +6 -0
  76. package/types/menu/lib/db.d.mts +23 -0
  77. package/types/menu/lib/import.d.mts +11 -0
  78. package/types/menu/lib/menu.d.mts +24 -0
  79. package/types/menu/lib/run.d.mts +27 -0
  80. package/types/menu/raw/config.d.mts +18 -0
  81. package/types/menu/raw/extract.d.mts +2 -0
  82. package/types/menu/raw/import.d.mts +64 -0
  83. package/types/menu/raw/normalize.d.mts +29 -0
  84. package/types/menu/raw/source.d.mts +19 -0
  85. package/types/menu/raw/vision.d.mts +13 -0
  86. package/types/research/index.d.mts +9 -0
  87. package/types/research/lib/google.d.mts +12 -0
  88. package/types/research/lib/hours.d.mts +22 -0
  89. package/types/research/lib/merge.d.mts +87 -0
  90. package/types/research/lib/osm.d.mts +36 -0
  91. package/types/research/lib/report.d.mts +6 -0
  92. package/types/research/lib/social.d.mts +118 -0
  93. package/types/research/lib/util.d.mts +45 -0
  94. package/types/research/lib/website.d.mts +283 -0
  95. package/types/research/research.d.mts +1 -0
  96. package/types/tripadvisor/index.d.mts +11 -0
  97. package/types/tripadvisor/links.d.mts +13 -0
  98. package/types/tripadvisor/photos.d.mts +1 -0
@@ -0,0 +1,285 @@
1
+ // Reads a restaurant's own web pages (and link-in-bio hubs like Linktree, and a
2
+ // TripAdvisor page when it lets us in) with plain fetch and regexes: no HTML
3
+ // parser dependency. Falls back to Playwright when a page renders client-side.
4
+ import { loadPlaywright } from "../../lib/playwright.mjs";
5
+ import { fetchText, sleep, BROWSER_UA } from "./util.mjs";
6
+ import { fromOsm, fromSpec } from "./hours.mjs";
7
+
8
+ const ENT = { amp: "&", lt: "<", gt: ">", quot: '"', apos: "'", nbsp: " " };
9
+ const decode = (s) =>
10
+ s.replace(/&(#x[0-9a-f]+|#\d+|[a-z]+);/gi, (m, e) => {
11
+ if (e[0] === "#") {
12
+ const n = e[1].toLowerCase() === "x" ? parseInt(e.slice(2), 16) : Number(e.slice(1));
13
+ return Number.isFinite(n) ? String.fromCodePoint(n) : m;
14
+ }
15
+ return ENT[e.toLowerCase()] ?? m;
16
+ });
17
+
18
+ const attrs = (tag) => {
19
+ const out = {};
20
+ for (const m of tag.matchAll(/([a-zA-Z_:][-\w:.]*)\s*(?:=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'>]+)))?/g))
21
+ out[m[1].toLowerCase()] = decode(m[2] ?? m[3] ?? m[4] ?? "");
22
+ return out;
23
+ };
24
+ const abs = (u, base) => { try { return new URL(u, base).href; } catch { return null; } };
25
+ const strip = (s) => decode(s.replace(/<[^>]+>/g, " ")).replace(/\s+/g, " ").trim();
26
+ const host = (u) => { try { return new URL(u).hostname.replace(/^www\./, ""); } catch { return ""; } };
27
+
28
+ const IG_RESERVED = new Set(["p", "reel", "reels", "explore", "accounts", "tv", "stories", "share", "direct", "about", "legal", "web", "developer"]);
29
+ const RESERVE = /(opentable|resy\.com|thefork|eltenedor|exploretock|sevenrooms|covermanager|quandoo|tablein|mesa247|bookatable|tablecheck|resos\.com|reservandonos|agendapro|fudo\.)/i;
30
+ const DELIVERY = /(rappi|ubereats|pedidosya|doordash|grubhub|domicilios\.com|didi-food|glovoapp)/i;
31
+ const HUBS = /(linktr\.ee|beacons\.ai|bio\.link|lnk\.bio|linktree\.com|taplink|campsite\.bio|solo\.to|linkin\.bio|flow\.page)/i;
32
+ const DAY_RE = /\b(lun(es)?|mar(tes)?|mi[eé](rcoles)?|jue(ves)?|vie(rnes)?|s[aá]b(ado)?|dom(ingo)?|mon(day)?|tue(sday)?|wed(nesday)?|thu(rsday)?|fri(day)?|sat(urday)?|sun(day)?)\b/i;
33
+ const TIME_RE = /\d{1,2}([:.]\d{2})?\s*(am|pm|h|hrs)?\s*(-|–|—|a|to|hasta)\s*\d{1,2}([:.]\d{2})?/i;
34
+ const BUSINESS_TYPE = /Restaurant|FoodEstablishment|BarOrPub|CafeOrCoffeeShop|Bakery|NightClub|LocalBusiness|Hotel/;
35
+
36
+ function visibleLines(html) {
37
+ const body = html
38
+ .replace(/<!--[\s\S]*?-->/g, "")
39
+ .replace(/<(script|style|noscript|svg|template|head)\b[\s\S]*?<\/\1>/gi, "");
40
+ return decode(body.replace(/<\/(p|div|li|h[1-6]|tr|section|article|header|footer|ul|ol|table)>|<br\s*\/?>/gi, "\n").replace(/<[^>]+>/g, " "))
41
+ .split("\n").map((l) => l.replace(/\s+/g, " ").trim()).filter(Boolean);
42
+ }
43
+
44
+ function jsonLd(html) {
45
+ const nodes = [];
46
+ for (const m of html.matchAll(/<script[^>]+application\/ld\+json[^>]*>([\s\S]*?)<\/script>/gi)) {
47
+ try {
48
+ const data = JSON.parse(decode(m[1]).trim());
49
+ for (const n of [data].flat()) nodes.push(...(n["@graph"] ? n["@graph"] : [n]));
50
+ } catch { /* malformed block: skip it */ }
51
+ }
52
+ const biz = nodes.find((n) => BUSINESS_TYPE.test([n["@type"]].flat().join(" ")));
53
+ if (!biz) return null;
54
+ const addr = biz.address && typeof biz.address === "object" ? biz.address : {};
55
+ const text = (v) => (typeof v === "string" ? v : v?.name ?? v?.url ?? "");
56
+ return {
57
+ name: text(biz.name),
58
+ telephone: text(biz.telephone),
59
+ email: text(biz.email),
60
+ priceRange: text(biz.priceRange),
61
+ cuisines: [biz.servesCuisine].flat().filter((c) => typeof c === "string"),
62
+ address: { street: addr.streetAddress ?? "", locality: addr.addressLocality ?? "", region: addr.addressRegion ?? "", postcode: addr.postalCode ?? "", country: typeof addr.addressCountry === "string" ? addr.addressCountry : addr.addressCountry?.name ?? "" },
63
+ geo: biz.geo?.latitude ? { lat: Number(biz.geo.latitude), lng: Number(biz.geo.longitude) } : null,
64
+ sameAs: [biz.sameAs].flat().filter((u) => typeof u === "string"),
65
+ logo: text(biz.logo),
66
+ rating: biz.aggregateRating ? { value: Number(biz.aggregateRating.ratingValue), count: Number(biz.aggregateRating.reviewCount ?? biz.aggregateRating.ratingCount) || null } : null,
67
+ hours: fromSpec(biz.openingHoursSpecification) ?? fromOsm([biz.openingHours].flat().filter(Boolean).join("; ")),
68
+ };
69
+ }
70
+
71
+ /** Pulls every fact a page holds. `base` resolves relative links. */
72
+ export function analyzeHtml(html, base, lines = visibleLines(html)) {
73
+ const meta = {};
74
+ for (const m of html.matchAll(/<meta\b([^>]*)>/gi)) {
75
+ const a = attrs(m[1]);
76
+ const key = (a.property || a.name || "").toLowerCase();
77
+ if (key && a.content && !(key in meta)) meta[key] = a.content;
78
+ }
79
+ const anchors = [...html.matchAll(/<a\b([^>]*)>([\s\S]*?)<\/a>/gi)]
80
+ .map((m) => ({ href: abs(attrs(m[1]).href ?? "", base), text: strip(m[2]) }))
81
+ .filter((a) => a.href);
82
+ // Hubs and single-page apps keep their links in JSON blobs, not in <a> tags.
83
+ const hrefs = new Set(anchors.map((a) => a.href));
84
+ for (const m of html.matchAll(/"(?:url|href|link)"\s*:\s*"(https?:[^"]+)"/g)) {
85
+ const u = m[1].replace(/\\u0026/g, "&").replace(/\\\//g, "/");
86
+ if (!hrefs.has(u)) {
87
+ hrefs.add(u);
88
+ anchors.push({ href: u, text: "" });
89
+ }
90
+ }
91
+
92
+ const links = { tel: [], mail: [], whatsapp: [], reserve: [], menu: [], delivery: [], maps: [], waze: [], facebook: [], tiktok: [], tripadvisor: [], hubs: [], instagram: [] };
93
+ const ig = new Map();
94
+ const push = (k, v) => { if (v && !links[k].includes(v)) links[k].push(v); };
95
+ for (const { href, text } of anchors) {
96
+ let u;
97
+ try { u = new URL(href); } catch { continue; }
98
+ const h = u.hostname.replace(/^www\./, "");
99
+ if (u.protocol === "tel:") push("tel", decodeURIComponent(href.slice(4)).trim());
100
+ else if (u.protocol === "mailto:") push("mail", decodeURIComponent(href.slice(7).split("?")[0]).trim());
101
+ else if (/(^|\.)wa\.me$|whatsapp\.com$|wa\.link$/.test(h)) {
102
+ const num = (u.pathname.match(/^\/(\d{7,})/)?.[1]) ?? u.searchParams.get("phone");
103
+ push("whatsapp", num ? num.replace(/\D/g, "") : href);
104
+ } else if (/(^|\.)instagram\.com$/.test(h)) {
105
+ const handle = u.pathname.split("/")[1]?.toLowerCase();
106
+ if (handle && !IG_RESERVED.has(handle) && /^[a-z0-9._]+$/.test(handle)) ig.set(handle, (ig.get(handle) ?? 0) + 1);
107
+ } else if (/facebook\.com$|fb\.com$/.test(h)) push("facebook", href);
108
+ else if (/tiktok\.com$/.test(h)) push("tiktok", href);
109
+ else if (/tripadvisor\./.test(h)) push("tripadvisor", href);
110
+ else if (/maps\.app\.goo\.gl|google\.[a-z.]+\/maps|goo\.gl\/maps/.test(href)) push("maps", href);
111
+ else if (/waze\.com/.test(h)) push("waze", href);
112
+ else if (HUBS.test(h)) push("hubs", href);
113
+ else if (RESERVE.test(href)) push("reserve", href);
114
+ else if (DELIVERY.test(href)) push("delivery", href);
115
+ else if (/\.pdf($|\?)/i.test(u.pathname) || /\b(menu|men[uú]|carta)\b/i.test(`${text} ${u.pathname}`)) push("menu", href);
116
+ }
117
+ links.instagram = [...ig].sort((a, b) => b[1] - a[1]).map(([h]) => h);
118
+
119
+ const images = [];
120
+ const imageKeys = new Set();
121
+ const addImage = (url, alt = "", extra = {}) => {
122
+ if (!url || /^data:/.test(url) || /(icon|sprite|favicon|pixel|spacer|avatar|badge|payment|flag|tripadvisor|facebook|instagram|whatsapp)\b/i.test(url)) return;
123
+ const key = url.split("?")[0];
124
+ if (!imageKeys.has(key)) {
125
+ imageKeys.add(key);
126
+ images.push({ url, alt, ...extra });
127
+ }
128
+ };
129
+ addImage(abs(meta["og:image"], base), "og:image");
130
+ const logos = [];
131
+ for (const m of html.matchAll(/<img\b([^>]*)>/gi)) {
132
+ const a = attrs(m[1]);
133
+ const srcset = (a.srcset || a["data-srcset"] || "").split(",").map((s) => s.trim().split(/\s+/)).filter((p) => p[0]);
134
+ const best = srcset.sort((x, y) => parseInt(y[1] ?? "0") - parseInt(x[1] ?? "0"))[0]?.[0];
135
+ const url = abs(best || a.src || a["data-src"] || a["data-lazy-src"], base);
136
+ if (!url) continue;
137
+ if (/logo/i.test(`${a.src} ${a.class} ${a.id} ${a.alt}`)) logos.push({ url, kind: "img" });
138
+ else addImage(url, a.alt, { width: Number(a.width) || undefined, height: Number(a.height) || undefined });
139
+ }
140
+ for (const m of html.matchAll(/url\(\s*['"]?([^'")]+\.(?:jpe?g|png|webp))['"]?\s*\)/gi)) addImage(abs(m[1], base));
141
+ for (const m of html.matchAll(/<link\b([^>]*)>/gi)) {
142
+ const a = attrs(m[1]);
143
+ if (/apple-touch-icon|icon/.test(a.rel ?? "") && a.href) logos.push({ url: abs(a.href, base), kind: a.rel });
144
+ }
145
+
146
+ const ld = jsonLd(html);
147
+ if (ld?.logo) logos.unshift({ url: abs(ld.logo, base), kind: "json-ld" });
148
+ return {
149
+ url: base,
150
+ title: strip(html.match(/<title[^>]*>([\s\S]*?)<\/title>/i)?.[1] ?? ""),
151
+ lang: (html.match(/<html[^>]*\blang=["']?([a-zA-Z-]+)/i)?.[1] ?? "").toLowerCase(),
152
+ meta: { siteName: meta["og:site_name"] ?? "", title: meta["og:title"] ?? "", description: meta["og:description"] ?? meta.description ?? "", themeColor: meta["theme-color"] ?? "" },
153
+ jsonld: ld, links, images, logos,
154
+ textLength: lines.join(" ").length,
155
+ headings: [...html.matchAll(/<h[1-3][^>]*>([\s\S]*?)<\/h[1-3]>/gi)].map((m) => strip(m[1])).filter((t) => t && t.length < 140),
156
+ paragraphs: lines.filter((l) => l.length >= 70 && !/[©]|cookie/i.test(l)).slice(0, 12),
157
+ hoursText: lines.filter((l) => l.length <= 160 && ((DAY_RE.test(l) && TIME_RE.test(l)) || /^(horarios?|opening hours|hours)\b/i.test(l))),
158
+ anchors: anchors.filter((a) => /^https?:/.test(a.href)),
159
+ };
160
+ }
161
+
162
+ /**
163
+ * One lazily launched Chromium shared by every render of a run. `get()` launches on first use;
164
+ * `close()` is safe to call when nothing was launched.
165
+ */
166
+ export function createBrowser() {
167
+ let launching = null;
168
+ return {
169
+ get() {
170
+ launching ??= loadPlaywright().then(({ chromium }) => chromium.launch());
171
+ return launching;
172
+ },
173
+ async close() {
174
+ if (!launching) return;
175
+ const pending = launching;
176
+ launching = null;
177
+ try { await (await pending).close(); } catch { /* it never launched */ }
178
+ },
179
+ };
180
+ }
181
+
182
+ // `strict` (the caller asked for rendering) reports a missing Playwright; the automatic fallbacks stay silent.
183
+ async function render(url, strict, browser) {
184
+ let page;
185
+ try {
186
+ page = await (await browser.get()).newPage({ userAgent: BROWSER_UA });
187
+ await page.goto(url, { waitUntil: "networkidle", timeout: 25000 });
188
+ return await page.content();
189
+ } catch (e) {
190
+ if (strict && e.code === "EDEPENDENCY") throw e;
191
+ return null;
192
+ } finally {
193
+ await page?.close().catch(() => {});
194
+ }
195
+ }
196
+
197
+ /** One page, analysed. Returns { page, blocked } and never throws on HTTP errors. `browser` is a shared createBrowser(); without one a private browser is used and closed. */
198
+ export async function readPage(url, { renderJs = false, browser } = {}) {
199
+ const own = browser ? null : createBrowser();
200
+ const shared = browser ?? own;
201
+ try {
202
+ let r;
203
+ try { r = await fetchText(url); } catch (e) { return { page: null, blocked: `${e.message}` }; }
204
+ if (!r.ok) {
205
+ // Bot walls answer 403/429/503 to a plain fetch; a real browser often gets through.
206
+ const html = [403, 429, 503].includes(r.status) ? await render(url, false, shared) : null;
207
+ return html ? { page: { ...analyzeHtml(html, url), rendered: true } } : { page: null, blocked: `HTTP ${r.status}` };
208
+ }
209
+ // Decide "thin" from the visible text first, so a thin page is parsed only once (rendered) and a full one once (as fetched).
210
+ const lines = visibleLines(r.text);
211
+ if (renderJs || lines.join(" ").length < 300) {
212
+ const html = await render(r.url, renderJs, shared);
213
+ if (html) return { page: { ...analyzeHtml(html, r.url), rendered: true } };
214
+ }
215
+ return { page: analyzeHtml(r.text, r.url, lines) };
216
+ } finally {
217
+ await own?.close();
218
+ }
219
+ }
220
+
221
+ const PAGE_HINT = /contact|ubica|visit|about|nosotros|historia|story|nuestra|menu|carta|reserv|evento|event|horario|hours|donde|find/i;
222
+
223
+ /** The restaurant's own site: the home page plus a few contact/about/menu pages, merged. */
224
+ export async function scrapeSite(startUrl, { renderJs = false, maxPages = 6, browser } = {}) {
225
+ const own = browser ? null : createBrowser();
226
+ try {
227
+ return await scrape(startUrl, { renderJs, maxPages, browser: browser ?? own });
228
+ } finally {
229
+ await own?.close();
230
+ }
231
+ }
232
+
233
+ async function scrape(startUrl, { renderJs, maxPages, browser }) {
234
+ const first = await readPage(startUrl, { renderJs, browser });
235
+ if (!first.page) throw new Error(`could not read ${startUrl} (${first.blocked})`);
236
+ const pages = [first.page];
237
+ const seen = new Set([first.page.url.split("#")[0]]);
238
+ const same = host(first.page.url);
239
+ const next = first.page.anchors
240
+ .filter((a) => host(a.href) === same && !/\.(pdf|jpe?g|png|webp|svg|zip)($|\?)/i.test(a.href) && PAGE_HINT.test(`${a.text} ${new URL(a.href).pathname}`))
241
+ .map((a) => a.href.split("#")[0]);
242
+ const queue = [];
243
+ for (const url of next) {
244
+ if (seen.has(url)) continue;
245
+ seen.add(url);
246
+ queue.push(url);
247
+ }
248
+ // Batches of 3 pages, started 200 ms apart: it is the restaurant's own site, so no hammering.
249
+ for (let i = 0; i < queue.length && pages.length < maxPages; ) {
250
+ const batch = queue.slice(i, i + Math.min(3, maxPages - pages.length));
251
+ i += batch.length;
252
+ const read = await Promise.all(
253
+ batch.map(async (url, n) => {
254
+ await sleep(n * 200);
255
+ return readPage(url, { renderJs, browser });
256
+ }),
257
+ );
258
+ for (const { page } of read) if (page) pages.push(page);
259
+ }
260
+
261
+ const uniq = (list, key = (x) => x) => {
262
+ const keys = new Set();
263
+ return list.filter((x) => {
264
+ const k = key(x);
265
+ if (keys.has(k)) return false;
266
+ keys.add(k);
267
+ return true;
268
+ });
269
+ };
270
+ const links = {};
271
+ for (const k of Object.keys(pages[0].links)) links[k] = uniq(pages.flatMap((p) => p.links[k]));
272
+ return {
273
+ url: first.page.url,
274
+ pages: pages.map((p) => ({ url: p.url, title: p.title, rendered: !!p.rendered })),
275
+ lang: first.page.lang,
276
+ meta: first.page.meta,
277
+ jsonld: pages.map((p) => p.jsonld).find(Boolean) ?? null,
278
+ links,
279
+ images: uniq(pages.flatMap((p) => p.images), (i) => i.url.split("?")[0]),
280
+ logos: uniq(pages.flatMap((p) => p.logos), (l) => l.url),
281
+ headings: uniq(pages.flatMap((p) => p.headings)).slice(0, 30),
282
+ paragraphs: uniq(pages.flatMap((p) => p.paragraphs)).slice(0, 20),
283
+ hoursText: uniq(pages.flatMap((p) => p.hoursText)).slice(0, 12),
284
+ };
285
+ }
@@ -0,0 +1,67 @@
1
+ // Gathers what the public web says about a restaurant, to fill the template.
2
+ // Usage: tablefacts research "Restaurant name" "City, Country" [options]
3
+ import { parseArgs } from "node:util";
4
+ import { research } from "./index.mjs";
5
+ import { loadEnv } from "../lib/env.mjs";
6
+ import { cliMessage, exitCodeFor } from "../lib/errors.mjs";
7
+ import { consoleLog } from "../lib/log.mjs";
8
+
9
+ const HELP = `Research a restaurant from public sources and write a profile for the template.
10
+
11
+ tablefacts research "<name>" "<city, country>" [options]
12
+
13
+ Sources: Google Maps (Places API), OpenStreetMap, the restaurant's website,
14
+ Instagram, TripAdvisor and link-in-bio pages (Linktree and similar). They find
15
+ each other: the website or Google leads to Instagram, TripAdvisor and the hub.
16
+
17
+ Options:
18
+ --country <ISO> country code, narrows the search (CO, MX, US...)
19
+ --website <url> the restaurant's site, if the search does not find it
20
+ --instagram <handle> Instagram handle or URL
21
+ --tripadvisor <url> TripAdvisor page
22
+ --linktree <url> Linktree or other link-in-bio page
23
+ --photos <n> also download up to n Google and n website photos (reference only)
24
+ --render render the website with Playwright (sites built in JavaScript)
25
+ --no-google skip Google even if GOOGLE_PLACES_API_KEY is set
26
+ --out <dir> output folder (default .tablefacts/research/<slug>)
27
+ -h, --help this text
28
+
29
+ Writes profile.json, report.md and setup-answers.txt. GOOGLE_PLACES_API_KEY goes
30
+ in .env (see .env.example); without it OpenStreetMap and the web pages
31
+ still work, with fewer facts.`;
32
+
33
+ const { values, positionals } = parseArgs({
34
+ allowPositionals: true,
35
+ options: {
36
+ country: { type: "string" }, website: { type: "string" }, instagram: { type: "string" },
37
+ tripadvisor: { type: "string" }, linktree: { type: "string" }, photos: { type: "string" },
38
+ render: { type: "boolean" }, "no-google": { type: "boolean" }, out: { type: "string" },
39
+ help: { type: "boolean", short: "h" },
40
+ },
41
+ });
42
+ if (values.help || !positionals.length) {
43
+ console.log(HELP);
44
+ process.exit(values.help ? 0 : 1);
45
+ }
46
+
47
+ loadEnv();
48
+ const [name, location = ""] = positionals;
49
+ try {
50
+ const { profile, outDir } = await research({
51
+ name, location, country: values.country, website: values.website, instagram: values.instagram,
52
+ tripadvisor: values.tripadvisor, linktree: values.linktree, render: values.render,
53
+ google: !values["no-google"], photos: Number(values.photos ?? 0), out: values.out, log: consoleLog,
54
+ });
55
+ const found = Object.keys(profile.fields).filter((k) => profile.fields[k]);
56
+ console.log(`
57
+ Found ${found.length} fields: ${found.join(", ")}`);
58
+ for (const w of profile.warnings) console.log(`Warning: ${w}`);
59
+ console.log(`
60
+ Wrote ${outDir}
61
+ report.md read this first
62
+ profile.json everything, with sources
63
+ setup-answers.txt npm run setup < that file`);
64
+ } catch (e) {
65
+ console.error(cliMessage(e, { name: "<name>" }));
66
+ process.exitCode = exitCodeFor(e);
67
+ }
@@ -0,0 +1,78 @@
1
+ # TripAdvisor photos
2
+
3
+ Downloads the photos on a restaurant's TripAdvisor page, driven by Playwright in **the user's own Edge window**. Use it when the restaurant has no originals to send and the research report (`tablefacts research`) found a TripAdvisor link.
4
+
5
+ **Rights first.** Most TripAdvisor photos were uploaded by guests, who keep the rights to them; TripAdvisor's terms also forbid scraping. Ask the restaurant for its originals first. Use these photos as reference, or only where the restaurant confirms it may use them, and say in the hand-off where each one came from.
6
+
7
+ ## Status
8
+
9
+ The URL parsing, size upgrade and file naming are unit-tested (`tests/tripadvisor-links.test.ts`).
10
+
11
+ **Verified once** on Maki Bar Medellín (tripadvisor.co, October 2026): Edge started by the script, no check shown, the slideshow opened and showed "1 de 16", and all 16 photos saved as valid JPEGs (originals, 100 KB to 3.7 MB); a re-run skipped all 16. **Not verified:** other country domains, restaurants with hundreds of photos, a run where the DataDome check does appear, and the English/other-language page (the selector is a `data-section-signature` attribute, not text, so it should hold). TripAdvisor changes its markup: do a `--dry-run --debug` first and fix `openGallery` and `collect` in `photos.mjs` from the saved HTML if it finds nothing or the wrong photos.
12
+
13
+ **Check the count.** The slideshow says "N de M"; the tool should report M photos. A page-wide scrape was tried first and returned 35 URLs, most from other places ("keep planning", nearby attractions, other hotels), so the tool reads only the restaurant's own carousel and slideshow.
14
+
15
+ ## The browser
16
+
17
+ TripAdvisor guards its pages with DataDome. Like `tablefacts photos instagram`, the script does not drive a browser of its own: it **attaches to an ordinary Edge window** started with remote debugging (`--cdp http://localhost:9222`, the default) and opens its own tab. If nothing is listening, it starts Edge with the profile `C:\ig-edge` (`--edge-dir` changes it), the same window `photos:instagram` uses. See [the Instagram README](../instagram/README.md) for starting it by hand and its Windows notes.
18
+
19
+ Rules for agents:
20
+
21
+ - **Never try to bypass, solve or fake the check** (no stealth plugins, header tricks, solvers). The script only waits for the real page to appear (2 minutes). If a "verify you are human" box shows, **the user ticks it**. Tell them when you are waiting.
22
+ - **Never kill Edge or its processes.** Leave the window open between runs.
23
+
24
+ ## Run
25
+
26
+ ```bash
27
+ # list what it finds, write nothing, keep the page HTML to fix selectors
28
+ tablefacts photos tripadvisor --dry-run --debug --out <folder> <restaurant link>
29
+
30
+ # for real
31
+ tablefacts photos tripadvisor --out <folder> <restaurant link>
32
+ tablefacts photos tripadvisor --out <folder> --max 30 <link> <another link>
33
+ ```
34
+
35
+ | Flag | Meaning |
36
+ | --- | --- |
37
+ | `--out <folder>` | required; created if missing. Use a scratch folder outside the site's public folder, then copy what you pick |
38
+ | `--cdp <url>` | the Edge debugging address (default `http://localhost:9222`) |
39
+ | `--edge-dir <dir>` | profile folder of the Edge it starts (default `C:\ig-edge`) |
40
+ | `--max <n>` | stop after n photos per restaurant |
41
+ | `--dry-run` | print the photo URLs found, save nothing. Do this first |
42
+ | `--debug` | save the page HTML in `<out>/_debug/<id>.html` |
43
+
44
+ The link is the restaurant's own page (`…/Restaurant_Review-g…-d…-Reviews-….html`, any country domain); the query string is dropped. Output files are `<d-id>-<photo path>-<name>.jpg`, so a re-run skips what is already there. The summary prints saved, already there and failed; the exit code is non-zero if anything failed.
45
+
46
+ ## From code
47
+
48
+ ```js
49
+ import { downloadTripadvisor } from 'tablefacts'
50
+
51
+ const { saved, skipped, failed, found } = await downloadTripadvisor({
52
+ links: ['https://www.tripadvisor.com/Restaurant_Review-g...-d...-Reviews-....html'],
53
+ out: 'photos/tripadvisor', // relative paths resolve against `projectDir`
54
+ max: 30,
55
+ dryRun: true, // `found` ([{ name, url }]) is only present in a dry run
56
+ projectDir: '/path/to/project',
57
+ log: (message, level) => console.log(level ?? 'info', message),
58
+ })
59
+ for (const { item, reason } of failed) console.error(item, reason) // the restaurant link that failed
60
+ ```
61
+
62
+ `out` resolves against `projectDir` (default `TABLEFACTS_PROJECT`, then the current folder). `log(message, level)` gets
63
+ `'info'`, `'warn'` (skipped links) or `'error'`; without it nothing is printed. No valid link or no `out` throws a
64
+ `TablefactsError` with code `EUSAGE` and `err.option` (the CLI exits `2`); a missing Playwright is `EDEPENDENCY`; a
65
+ page that fails is in `failed` (the CLI exits `1`).
66
+
67
+ ## What the script does
68
+
69
+ 1. Opens the restaurant page and waits for the DataDome challenge to go away (an iframe from `captcha-delivery.com`, or a page without an `h1`).
70
+ 2. Clicks the first photo of the restaurant's carousel, `[data-section-signature="photo_viewer"]`, which opens the slideshow (`openGallery`).
71
+ 3. Presses the right arrow key through the slideshow and, at each step, reads the big photo on screen (`visibleSlides`, then `findPhotos` for `dynamic-media-cdn.tripadvisor.com` / `media-cdn.tripadvisor.com` `/media/photo-<size>/…`). It stops after two steps with nothing new. Without a carousel it reads only that element, never the rest of the page.
72
+ 4. Tries each photo's original (`photo-o`), then `photo-w`, `photo-l`, then the size the page showed (`sizeCandidates`), and saves the first that is an image over 5 KB. Reviewer avatars, maps and logos are skipped.
73
+
74
+ Only what the page loads is reachable: a restaurant with hundreds of photos may give a fraction of them.
75
+
76
+ ## After the download
77
+
78
+ Look at every image. TripAdvisor mixes dishes with receipts, menus and people. Copy the chosen files into your site's public folder, and record real pixel `width`/`height` and an `alt` text for each.
@@ -0,0 +1,178 @@
1
+ // Library entry for downloading the photos of a restaurant's TripAdvisor page, in the user's own Edge window.
2
+ // Silent unless a `log` function is given; importing this file loads nothing heavy (playwright is loaded on use).
3
+ import { writeFile } from 'node:fs/promises'
4
+ import { join } from 'node:path'
5
+ import { DEFAULT_CDP, ensureEdge } from '../lib/edge.mjs'
6
+ import { TablefactsError, optionError } from '../lib/errors.mjs'
7
+ import { assertImageResponse, writeImage } from '../lib/images.mjs'
8
+ import { normalizeLog } from '../lib/log.mjs'
9
+ import { DELAY_MS, newSummary, prepareOut, storeFile } from '../lib/photos.mjs'
10
+ import { loadPlaywright } from '../lib/playwright.mjs'
11
+ import { fileName, findPhotos, normalizeRestaurant, sizeCandidates } from './links.mjs'
12
+
13
+ const MIN_BYTES = 5000
14
+
15
+ // TripAdvisor's bot protection (DataDome) puts a challenge page, often in an iframe from
16
+ // captcha-delivery.com, in front of the restaurant. In a real window it passes alone or after a
17
+ // tick by the user. We only wait for the real page (it has an h1) to appear, never try to solve it.
18
+ const CHALLENGE = 'iframe[src*="captcha-delivery.com"], iframe[src*="datadome"]'
19
+
20
+ async function waitForCheck(page, log) {
21
+ const passed = () => page.evaluate((sel) => !document.querySelector(sel) && !!document.querySelector('h1'), CHALLENGE)
22
+ if (await passed()) return
23
+ log(' waiting for the TripAdvisor check: tick it in the browser window if it asks (2 min)')
24
+ await page
25
+ .waitForFunction((sel) => !document.querySelector(sel) && !!document.querySelector('h1'), CHALLENGE, { timeout: 120000 })
26
+ .catch(() => {
27
+ throw new TablefactsError('the TripAdvisor check was not passed: tick it in the window, or use a normal Edge window', 'EFAILED')
28
+ })
29
+ }
30
+
31
+ // The restaurant's own photos are in the carousel under data-section-signature="photo_viewer";
32
+ // clicking a photo opens a slideshow ("1 de 16 en Todas las fotos"). The rest of the page (nearby
33
+ // places, "keep planning", reviews of other places) also carries TripAdvisor photos that are NOT
34
+ // this restaurant's, so nothing outside the carousel and the slideshow is ever read.
35
+ const VIEWER = '[data-section-signature="photo_viewer"]'
36
+
37
+ async function openGallery(page) {
38
+ const first = page.locator(`${VIEWER} button`).first()
39
+ if (!(await first.count())) return false
40
+ await first.click({ timeout: 4000 }).catch(() => {})
41
+ await page.waitForTimeout(2000)
42
+ return true
43
+ }
44
+
45
+ // The big photos on screen now: the slideshow's current slide (and the carousel behind it).
46
+ const visibleSlides = (page) =>
47
+ page.evaluate(() =>
48
+ [...document.querySelectorAll('img')]
49
+ .filter((img) => {
50
+ const r = img.getBoundingClientRect()
51
+ return r.width > 400 && r.height > 300 && r.right > 0 && r.left < innerWidth && r.bottom > 0 && r.top < innerHeight
52
+ })
53
+ .map((img) => img.src)
54
+ .join('\n'),
55
+ )
56
+
57
+ // Steps through the slideshow with the arrow key, reading each slide, until two steps in a row show
58
+ // nothing new (the end or a wrap-around) or `max` photos are known. Without a slideshow, reads only
59
+ // the carousel on the page.
60
+ async function collect(page, max, opened) {
61
+ const found = new Map()
62
+ const add = (text) => {
63
+ for (const photo of findPhotos(text)) if (!found.has(photo.key)) found.set(photo.key, photo)
64
+ }
65
+ if (!opened) {
66
+ add(await page.locator(VIEWER).first().evaluate((el) => el.outerHTML).catch(() => ''))
67
+ return [...found.values()].slice(0, max)
68
+ }
69
+ let idle = 0
70
+ for (let step = 0; step < 300 && idle < 2 && found.size < max; step++) {
71
+ const before = found.size
72
+ add(await visibleSlides(page))
73
+ idle = found.size === before ? idle + 1 : 0
74
+ await page.keyboard.press('ArrowRight')
75
+ await page.waitForTimeout(1000)
76
+ }
77
+ return [...found.values()].slice(0, max)
78
+ }
79
+
80
+ // Tries the biggest size first; not every photo exists in every size.
81
+ async function save(context, photo, file) {
82
+ let last = 'no size worked'
83
+ for (const url of sizeCandidates(photo)) {
84
+ try {
85
+ const res = await context.request.get(url)
86
+ const bytes = await res.body()
87
+ assertImageResponse({ ok: res.ok(), status: res.status(), contentType: res.headers()['content-type'], bytes, minBytes: MIN_BYTES })
88
+ await writeImage(file, bytes)
89
+ return
90
+ } catch (err) {
91
+ last = err.message
92
+ }
93
+ }
94
+ throw new TablefactsError(last, 'EFAILED')
95
+ }
96
+
97
+ /**
98
+ * Downloads the photos of TripAdvisor restaurant pages through the user's own Edge.
99
+ * Resolves to a summary whose `failed` lists `{ item, reason }` and whose `found` is present only
100
+ * in a dry run. Throws a TablefactsError on invalid arguments (code 'EUSAGE', with `option`), a
101
+ * missing playwright ('EDEPENDENCY') or a browser that cannot be reached.
102
+ * `out` resolves against `projectDir` (default: TABLEFACTS_PROJECT or the current folder).
103
+ * `log(message, level)` gets level 'info', 'warn' or 'error'.
104
+ * @param {import('../lib/types.mjs').TripadvisorOptions} options
105
+ * @returns {Promise<import('../lib/types.mjs').PhotoSummary>}
106
+ */
107
+ export async function downloadTripadvisor({
108
+ links = [],
109
+ out,
110
+ cdp = DEFAULT_CDP,
111
+ edgeDir,
112
+ max = Infinity,
113
+ projectDir,
114
+ dryRun = false,
115
+ debug = false,
116
+ log: logOption,
117
+ } = {}) {
118
+ const log = normalizeLog(logOption)
119
+ if (!out) throw optionError('out', '`out` is required')
120
+ const restaurants = []
121
+ for (const raw of links) {
122
+ const r = normalizeRestaurant(raw)
123
+ if (r) restaurants.push(r)
124
+ else log(`Skipping, not a TripAdvisor restaurant link: ${raw}`, 'warn')
125
+ }
126
+ if (!restaurants.length) throw optionError('links', 'no valid TripAdvisor restaurant links in `links`')
127
+
128
+ const { outDir, debugDir } = await prepareOut({ out, dryRun, debug, projectDir })
129
+ const summary = newSummary({ dryRun })
130
+
131
+ const { chromium } = await loadPlaywright()
132
+ let close
133
+ try {
134
+ await ensureEdge(cdp, edgeDir)
135
+ const browser = await chromium.connectOverCDP(cdp)
136
+ const context = browser.contexts()[0]
137
+ let page
138
+ close = async () => {
139
+ try {
140
+ await page?.close()
141
+ } finally {
142
+ await browser.close()
143
+ }
144
+ }
145
+ page = await context.newPage()
146
+
147
+ for (const [i, restaurant] of restaurants.entries()) {
148
+ log(`[${i + 1}/${restaurants.length}] ${restaurant.url}`)
149
+ try {
150
+ await page.goto(restaurant.url, { waitUntil: 'domcontentloaded' })
151
+ await waitForCheck(page, log)
152
+ await page.getByRole('button', { name: /accept|aceptar|i agree/i }).first().click({ timeout: 2000 }).catch(() => {})
153
+ const opened = await openGallery(page)
154
+ log(opened ? ' photo slideshow opened' : ' no photo carousel found; reading nothing else on the page', opened ? 'info' : 'warn')
155
+ const photos = await collect(page, max, opened)
156
+ if (debugDir) await writeFile(join(debugDir, `${restaurant.id}.html`), await page.content())
157
+ if (!photos.length) throw new TablefactsError('no photos found (markup changed, or the check page is still showing); run with `debug`', 'EFAILED')
158
+ log(` ${photos.length} photos`)
159
+ for (const photo of photos) {
160
+ const name = `${restaurant.id}-${fileName(photo)}`
161
+ try {
162
+ await storeFile({ outDir, name, url: photo.url, dryRun, summary, log, save: ({ file }) => save(context, photo, file) })
163
+ } catch (err) {
164
+ summary.failed.push({ item: name, reason: err.message })
165
+ log(` ${name} failed: ${err.message}`, 'error')
166
+ }
167
+ }
168
+ } catch (err) {
169
+ summary.failed.push({ item: restaurant.url, reason: err.message })
170
+ log(` failed: ${err.message}`, 'error')
171
+ }
172
+ if (i < restaurants.length - 1) await page.waitForTimeout(DELAY_MS)
173
+ }
174
+ } finally {
175
+ await close?.().catch(() => {})
176
+ }
177
+ return summary
178
+ }