tablefacts 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (98) hide show
  1. package/.env.example +14 -0
  2. package/CHANGELOG.md +12 -0
  3. package/LICENSE +21 -0
  4. package/README.md +313 -0
  5. package/bin/tablefacts.mjs +28 -0
  6. package/package.json +77 -0
  7. package/src/index.mjs +52 -0
  8. package/src/instagram/README.md +135 -0
  9. package/src/instagram/download.mjs +62 -0
  10. package/src/instagram/index.mjs +355 -0
  11. package/src/instagram/links.mjs +55 -0
  12. package/src/instagram/record.mjs +18 -0
  13. package/src/lib/edge.mjs +49 -0
  14. package/src/lib/env.mjs +35 -0
  15. package/src/lib/errors.mjs +39 -0
  16. package/src/lib/files.mjs +10 -0
  17. package/src/lib/images.mjs +20 -0
  18. package/src/lib/log.mjs +17 -0
  19. package/src/lib/photos.mjs +42 -0
  20. package/src/lib/playwright.mjs +13 -0
  21. package/src/lib/project.mjs +42 -0
  22. package/src/lib/text.mjs +7 -0
  23. package/src/lib/types.mjs +247 -0
  24. package/src/menu/README.md +97 -0
  25. package/src/menu/cluvi/config.mjs +33 -0
  26. package/src/menu/cluvi/extract.mjs +24 -0
  27. package/src/menu/cluvi/import.mjs +30 -0
  28. package/src/menu/cluvi/source.mjs +156 -0
  29. package/src/menu/index.mjs +9 -0
  30. package/src/menu/lib/db.mjs +119 -0
  31. package/src/menu/lib/import.mjs +126 -0
  32. package/src/menu/lib/menu.mjs +95 -0
  33. package/src/menu/lib/run.mjs +83 -0
  34. package/src/menu/raw/config.mjs +38 -0
  35. package/src/menu/raw/extract.mjs +72 -0
  36. package/src/menu/raw/import.mjs +99 -0
  37. package/src/menu/raw/normalize.mjs +120 -0
  38. package/src/menu/raw/source.mjs +116 -0
  39. package/src/menu/raw/vision.mjs +252 -0
  40. package/src/research/README.md +69 -0
  41. package/src/research/index.mjs +147 -0
  42. package/src/research/lib/google.mjs +93 -0
  43. package/src/research/lib/hours.mjs +109 -0
  44. package/src/research/lib/merge.mjs +119 -0
  45. package/src/research/lib/osm.mjs +49 -0
  46. package/src/research/lib/report.mjs +118 -0
  47. package/src/research/lib/social.mjs +49 -0
  48. package/src/research/lib/util.mjs +104 -0
  49. package/src/research/lib/website.mjs +285 -0
  50. package/src/research/research.mjs +67 -0
  51. package/src/tripadvisor/README.md +78 -0
  52. package/src/tripadvisor/index.mjs +178 -0
  53. package/src/tripadvisor/links.mjs +78 -0
  54. package/src/tripadvisor/photos.mjs +55 -0
  55. package/types/index.d.mts +65 -0
  56. package/types/instagram/download.d.mts +1 -0
  57. package/types/instagram/index.d.mts +13 -0
  58. package/types/instagram/links.d.mts +14 -0
  59. package/types/instagram/record.d.mts +1 -0
  60. package/types/lib/edge.d.mts +14 -0
  61. package/types/lib/env.d.mts +12 -0
  62. package/types/lib/errors.d.mts +25 -0
  63. package/types/lib/files.d.mts +1 -0
  64. package/types/lib/images.d.mts +6 -0
  65. package/types/lib/log.d.mts +5 -0
  66. package/types/lib/photos.d.mts +30 -0
  67. package/types/lib/playwright.d.mts +1677 -0
  68. package/types/lib/project.d.mts +32 -0
  69. package/types/lib/text.d.mts +4 -0
  70. package/types/lib/types.d.mts +668 -0
  71. package/types/menu/cluvi/config.d.mts +13 -0
  72. package/types/menu/cluvi/extract.d.mts +2 -0
  73. package/types/menu/cluvi/import.d.mts +35 -0
  74. package/types/menu/cluvi/source.d.mts +37 -0
  75. package/types/menu/index.d.mts +6 -0
  76. package/types/menu/lib/db.d.mts +23 -0
  77. package/types/menu/lib/import.d.mts +11 -0
  78. package/types/menu/lib/menu.d.mts +24 -0
  79. package/types/menu/lib/run.d.mts +27 -0
  80. package/types/menu/raw/config.d.mts +18 -0
  81. package/types/menu/raw/extract.d.mts +2 -0
  82. package/types/menu/raw/import.d.mts +64 -0
  83. package/types/menu/raw/normalize.d.mts +29 -0
  84. package/types/menu/raw/source.d.mts +19 -0
  85. package/types/menu/raw/vision.d.mts +13 -0
  86. package/types/research/index.d.mts +9 -0
  87. package/types/research/lib/google.d.mts +12 -0
  88. package/types/research/lib/hours.d.mts +22 -0
  89. package/types/research/lib/merge.d.mts +87 -0
  90. package/types/research/lib/osm.d.mts +36 -0
  91. package/types/research/lib/report.d.mts +6 -0
  92. package/types/research/lib/social.d.mts +118 -0
  93. package/types/research/lib/util.d.mts +45 -0
  94. package/types/research/lib/website.d.mts +283 -0
  95. package/types/research/research.d.mts +1 -0
  96. package/types/tripadvisor/index.d.mts +11 -0
  97. package/types/tripadvisor/links.d.mts +13 -0
  98. package/types/tripadvisor/photos.d.mts +1 -0
@@ -0,0 +1,109 @@
1
+ // Opening hours in one shape, whatever the source: week[0..6] (Monday first),
2
+ // each day a list of ["HH:MM", "HH:MM"] slots; a closed day is []. The template
3
+ // shows them as `visit.hoursRows`, consecutive identical days merged.
4
+
5
+ const pad = (n) => String(n).padStart(2, "0");
6
+ const hm = (h, m = 0) => `${pad(h)}:${pad(m)}`;
7
+ const emptyWeek = () => Array.from({ length: 7 }, () => []);
8
+ const allDay = () => Array.from({ length: 7 }, () => [["00:00", "24:00"]]);
9
+
10
+ /** Google `regularOpeningHours.periods` (day 0 = Sunday). */
11
+ export function fromGoogle(periods = []) {
12
+ if (!periods.length) return null;
13
+ if (periods.length === 1 && !periods[0].close) return { week: allDay() };
14
+ const week = emptyWeek();
15
+ for (const p of periods) {
16
+ if (!p.open || !p.close) continue;
17
+ week[(p.open.day + 6) % 7].push([hm(p.open.hour, p.open.minute ?? 0), hm(p.close.hour, p.close.minute ?? 0)]);
18
+ }
19
+ week.forEach((d) => d.sort());
20
+ return { week };
21
+ }
22
+
23
+ const DAY_INDEX = { mo: 0, tu: 1, we: 2, th: 3, fr: 4, sa: 5, su: 6 };
24
+
25
+ function expandDays(text) {
26
+ const out = [];
27
+ for (const part of text.split(",").map((s) => s.trim()).filter(Boolean)) {
28
+ const [a, b] = part.split("-").map((s) => DAY_INDEX[s.trim().toLowerCase().slice(0, 2)]);
29
+ if (a === undefined || (part.includes("-") && b === undefined)) return null;
30
+ if (b === undefined) out.push(a);
31
+ else for (let d = a; ; d = (d + 1) % 7) { out.push(d); if (d === b) break; }
32
+ }
33
+ return out.length ? out : null;
34
+ }
35
+
36
+ /** OpenStreetMap / schema.org `openingHours` syntax: "Mo-Th 12:00-22:00; Fr,Sa 12:00-00:00; Su off". */
37
+ export function fromOsm(text) {
38
+ if (!text) return null;
39
+ if (/^\s*24\/7\s*$/.test(text)) return { week: allDay() };
40
+ const week = emptyWeek();
41
+ let any = false;
42
+ let partial = false;
43
+ for (const rule of text.split(";").map((s) => s.trim()).filter(Boolean)) {
44
+ const m = !/\b(PH|SH|week|sunrise|sunset|Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)\b|[\[\]()"]/i.test(rule) &&
45
+ rule.match(/^([A-Za-z,\- ]+?)?\s*((?:\d{1,2}:\d{2}\s*-\s*\d{1,2}:\d{2}\s*,?\s*)+|off|closed)$/i);
46
+ const days = m ? (m[1] ? expandDays(m[1]) : [0, 1, 2, 3, 4, 5, 6]) : null;
47
+ if (!m || !days) { partial = true; continue; }
48
+ const slots = /^(off|closed)$/i.test(m[2])
49
+ ? []
50
+ : [...m[2].matchAll(/(\d{1,2}):(\d{2})\s*-\s*(\d{1,2}):(\d{2})/g)].map((x) => [hm(+x[1], +x[2]), hm(+x[3], +x[4])]);
51
+ for (const d of days) week[d] = slots;
52
+ any = true;
53
+ }
54
+ return any ? { week, partial } : null;
55
+ }
56
+
57
+ const EN_DAYS = ["monday", "tuesday", "wednesday", "thursday", "friday", "saturday", "sunday"];
58
+
59
+ /** schema.org `openingHoursSpecification` entries. */
60
+ export function fromSpec(specs = []) {
61
+ const week = emptyWeek();
62
+ let any = false;
63
+ for (const s of [specs].flat()) {
64
+ if (!s || !s.opens || !s.closes) continue;
65
+ for (const d of [s.dayOfWeek].flat()) {
66
+ const i = EN_DAYS.indexOf(String(d ?? "").split("/").pop().toLowerCase());
67
+ if (i < 0) continue;
68
+ week[i].push([String(s.opens).slice(0, 5), String(s.closes).slice(0, 5)]);
69
+ any = true;
70
+ }
71
+ }
72
+ week.forEach((d) => d.sort());
73
+ return any ? { week } : null;
74
+ }
75
+
76
+ const signature = (slots) => slots.map((s) => s.join("-")).join(",");
77
+ export const weekSignature = (week) => week.map(signature).join("|");
78
+
79
+ const NAMES = {
80
+ es: { days: ["lunes", "martes", "miércoles", "jueves", "viernes", "sábado", "domingo"], and: "y", to: "a", closed: "Cerrado", allDay: "24 horas", every: "Todos los días" },
81
+ en: { days: EN_DAYS, and: "and", to: "to", closed: "Closed", allDay: "24 hours", every: "Every day" },
82
+ };
83
+ const cap = (s) => s[0].toUpperCase() + s.slice(1);
84
+
85
+ /** Rows for `visit.hoursRows`: { label, value }, matching the template's wording. */
86
+ export function formatRows(week, lang) {
87
+ const t = NAMES[lang];
88
+ const runs = [];
89
+ week.forEach((slots, d) => {
90
+ const last = runs.at(-1);
91
+ if (last && signature(last.slots) === signature(slots)) last.days.push(d);
92
+ else runs.push({ days: [d], slots });
93
+ });
94
+ if (runs.every((r) => !r.slots.length)) return [];
95
+ return runs.map(({ days, slots }) => {
96
+ const a = t.days[days[0]];
97
+ const label =
98
+ days.length === 7 ? t.every
99
+ : days.length === 1 ? cap(a)
100
+ : days.length === 2 ? `${cap(a)} ${t.and} ${t.days[days[1]]}`
101
+ : `${cap(a)} ${t.to} ${t.days[days.at(-1)]}`;
102
+ const value = !slots.length
103
+ ? t.closed
104
+ : signature(slots) === "00:00-24:00"
105
+ ? t.allDay
106
+ : slots.map(([o, c]) => `${o} ${t.to} ${c}`).join(", ");
107
+ return { label, value };
108
+ });
109
+ }
@@ -0,0 +1,119 @@
1
+ // Reconciles what each source said into one profile. Every field keeps where it
2
+ // came from, how sure we are, and what the other sources said instead: BRIEF.md
3
+ // warns that Google, TripAdvisor and the restaurant's own site disagree.
4
+ import { digits, fold, km, samePhone, sameText } from "./util.mjs";
5
+ import { formatRows, weekSignature } from "./hours.mjs";
6
+
7
+ const ES = new Set(["CO", "MX", "ES", "AR", "CL", "PE", "EC", "UY", "VE", "BO", "PY", "CR", "PA", "DO", "GT", "HN", "SV", "NI", "CU", "PR"]);
8
+ const present = (v) => v !== undefined && v !== null && v !== "" && !(Array.isArray(v) && !v.length);
9
+ const handleOf = (v) => String(v ?? "").replace(/^https?:\/\/(www\.)?instagram\.com\//i, "").replace(/^@/, "").split(/[/?#]/)[0].toLowerCase();
10
+
11
+ /**
12
+ * First candidate wins (list them most trusted first). Confidence: high when two
13
+ * sources agree, medium for one, low for a guess (`low: true`).
14
+ */
15
+ function choose(candidates, same = (a, b) => fold(a) === fold(b)) {
16
+ const list = candidates.filter((c) => present(c.value));
17
+ if (!list.length) return null;
18
+ const first = list[0];
19
+ const agree = list.filter((c) => same(c.value, first.value));
20
+ const alternatives = list.filter((c) => !same(c.value, first.value)).map(({ value, source }) => ({ value, source }));
21
+ const sources = [...new Set(agree.map((c) => c.source))];
22
+ return {
23
+ value: first.value,
24
+ source: sources.join(" + "),
25
+ confidence: first.low ? "low" : sources.length > 1 ? "high" : "medium",
26
+ ...(alternatives.length && { alternatives }),
27
+ };
28
+ }
29
+
30
+ const cleanUrl = (u) => {
31
+ try {
32
+ const x = new URL(u);
33
+ [...x.searchParams.keys()].filter((k) => /^utm_|^fbclid|^gclid/.test(k)).forEach((k) => x.searchParams.delete(k));
34
+ return x.href.replace(/\/$/, "");
35
+ } catch { return u; }
36
+ };
37
+
38
+ export function buildProfile({ query, google, osm, site, hub, instagram, tripadvisor }) {
39
+ const g = google ?? {};
40
+ const o = osm ?? {};
41
+ const ld = site?.jsonld ?? null;
42
+ const ta = tripadvisor?.jsonld ?? null;
43
+ const warnings = [];
44
+ const ok = (v) => (v ? v : undefined);
45
+ const hubLinks = hub?.links ?? {};
46
+ const links = (k) => [...(site?.links?.[k] ?? []), ...(hubLinks[k] ?? [])];
47
+
48
+ if (g.status && g.status !== "OPERATIONAL") warnings.push(`Google lists this place as ${g.status}.`);
49
+ if (g.coords && o.coords && km(g.coords, o.coords) > 0.5) warnings.push("Google and OpenStreetMap place it more than 500 m apart: check they are the same restaurant.");
50
+ if (!google && !osm) warnings.push("No Google or OpenStreetMap match: facts rely on the website and social pages only.");
51
+
52
+ const coordSame = (a, b) => km(a, b) < 0.15;
53
+ const phoneSame = samePhone;
54
+ const igLinks = [...(site?.links?.instagram ?? []), ...(hubLinks.instagram ?? [])];
55
+
56
+ const fields = {
57
+ name: choose([{ value: g.name, source: "google" }, { value: ld?.name, source: "website" }, { value: ta?.name, source: "tripadvisor" }, { value: o.name, source: "osm" }, { value: site?.meta.siteName, source: "website" }], sameText),
58
+ street: choose([{ value: g.street, source: "google" }, { value: ld?.address.street, source: "website" }, { value: ta?.address.street, source: "tripadvisor" }, { value: o.street, source: "osm" }], sameText),
59
+ locality: choose([{ value: g.locality, source: "google" }, { value: ld?.address.locality, source: "website" }, { value: o.locality, source: "osm" }], sameText),
60
+ region: choose([{ value: g.region, source: "google" }, { value: ld?.address.region, source: "website" }, { value: o.region, source: "osm" }], sameText),
61
+ country: choose([{ value: query.country?.toUpperCase(), source: "you" }, { value: g.country, source: "google" }, { value: o.country, source: "osm" }, { value: ld?.address.country.length === 2 ? ld.address.country.toUpperCase() : "", source: "website" }]),
62
+ coordinates: choose([{ value: g.coords, source: "google" }, { value: o.coords, source: "osm" }, { value: ld?.geo, source: "website" }], coordSame),
63
+ phone: choose([{ value: g.phone, source: "google" }, { value: ok(links("tel")[0]), source: "website" }, { value: ld?.telephone, source: "website" }, { value: ta?.telephone, source: "tripadvisor" }, { value: instagram?.phones?.[0], source: "instagram" }, { value: o.phone, source: "osm" }], phoneSame),
64
+ email: choose([{ value: ld?.email, source: "website" }, { value: links("mail").find((m) => !/sentry|wixpress|example|domain\./i.test(m)), source: "website" }, { value: o.email, source: "osm" }]),
65
+ instagram: choose([{ value: query.instagram && handleOf(query.instagram), source: "you" }, ...igLinks.map((h) => ({ value: h, source: "website" })), ...(ld?.sameAs ?? []).filter((u) => /instagram\.com/.test(u)).map((u) => ({ value: handleOf(u), source: "website" })), { value: o.instagram && handleOf(o.instagram), source: "osm" }]),
66
+ facebook: choose([{ value: links("facebook")[0], source: "website" }, { value: o.facebook, source: "osm" }]),
67
+ tiktok: choose([{ value: links("tiktok")[0], source: "website" }]),
68
+ tripadvisor: choose([{ value: query.tripadvisor, source: "you" }, { value: links("tripadvisor")[0], source: "website" }, { value: (ld?.sameAs ?? []).find((u) => /tripadvisor\./.test(u)), source: "website" }]),
69
+ website: choose([{ value: query.website && cleanUrl(query.website), source: "you" }, { value: g.website && cleanUrl(g.website), source: "google" }, { value: o.website && cleanUrl(o.website), source: "osm" }]),
70
+ reserveUrl: choose([...links("reserve").map((u) => ({ value: u, source: "website" }))]),
71
+ priceRange: choose([{ value: g.priceRange, source: "google" }, { value: ld?.priceRange, source: "website" }, { value: ta?.priceRange, source: "tripadvisor" }]),
72
+ mapsUrl: choose([{ value: g.mapsUrl, source: "google" }, { value: links("maps")[0], source: "website" }]),
73
+ descriptor: choose([{ value: g.descriptor, source: "google" }]),
74
+ };
75
+
76
+ // WhatsApp: an explicit link wins; otherwise the phone is only a guess.
77
+ const wa = [...links("whatsapp"), o.whatsapp].map(digits).find((d) => d.length >= 7);
78
+ const phoneDisplay = (n) => [g.phone, ...(links("tel")), ld?.telephone].find((p) => p && p.trim().startsWith("+") && samePhone(p, n)) ?? `+${digits(n)}`;
79
+ if (wa) fields.whatsapp = { value: wa, display: phoneDisplay(wa), source: site?.links?.whatsapp?.length ? "website" : hubLinks.whatsapp?.length ? "link-in-bio" : "osm", confidence: "medium" };
80
+ else if (fields.phone) fields.whatsapp = { value: digits(fields.phone.value), display: fields.phone.value, source: fields.phone.source, confidence: "low", note: "Guess: the phone number. Confirm that it takes WhatsApp." };
81
+
82
+ const cuisines = [...new Set([...(g.cuisines ?? []), ...(ld?.cuisines ?? []), ...(ta?.cuisines ?? []), ...(o.cuisines ?? [])].map((c) => c.trim()).filter(Boolean))];
83
+ if (cuisines.length) {
84
+ const from = [g.cuisines?.length && "google", ld?.cuisines?.length && "website", ta?.cuisines?.length && "tripadvisor", o.cuisines?.length && "osm"].filter(Boolean);
85
+ fields.cuisines = { value: cuisines, source: from.join(" + "), confidence: from.length > 1 ? "high" : "medium" };
86
+ }
87
+
88
+ const lang = (site?.lang ?? "").slice(0, 2);
89
+ const country = fields.country?.value;
90
+ const menuLocale = ["es", "en"].includes(lang) ? lang : ES.has(country) ? "es" : country ? "en" : "";
91
+ if (menuLocale) fields.menuLocale = { value: menuLocale, source: ["es", "en"].includes(lang) ? "website language" : "country", confidence: "low", note: "Guess. The language the dish names are written in decides this." };
92
+
93
+ // Hours: the restaurant's own markup first, then Google, then OSM.
94
+ const hourSources = [
95
+ { source: "website", week: ld?.hours?.week },
96
+ { source: "tripadvisor", week: ta?.hours?.week },
97
+ { source: "google", week: g.hours?.week },
98
+ { source: "osm", week: o.hours?.week, partial: o.hours?.partial },
99
+ ].filter((h) => h.week);
100
+ const hours = hourSources.length
101
+ ? {
102
+ source: hourSources[0].source,
103
+ week: hourSources[0].week,
104
+ rows: { es: formatRows(hourSources[0].week, "es"), en: formatRows(hourSources[0].week, "en") },
105
+ confidence: hourSources.filter((h) => weekSignature(h.week) === weekSignature(hourSources[0].week)).length > 1 ? "high" : "medium",
106
+ conflicts: hourSources.filter((h) => weekSignature(h.week) !== weekSignature(hourSources[0].week)).map((h) => ({ source: h.source, rows: formatRows(h.week, "en") })),
107
+ }
108
+ : null;
109
+
110
+ return {
111
+ query, generatedAt: new Date().toISOString(), warnings, fields, hours,
112
+ textHours: [...(site?.hoursText ?? [])],
113
+ links: { menu: links("menu"), delivery: links("delivery"), reserve: links("reserve"), waze: links("waze"), hubs: [...new Set([...(site?.links?.hubs ?? []), ...(instagram?.links ?? []).filter((l) => /linktr|beacons|bio\.link|lnk\.bio|taplink/.test(l))])] },
114
+ ratings: [g.rating && { source: "google", value: g.rating, count: g.ratingCount }, ld?.rating && { source: "website", ...ld.rating }, ta?.rating && { source: "tripadvisor", ...ta.rating }].filter(Boolean),
115
+ images: { logos: [...(site?.logos ?? []), ...(hub?.logos ?? [])].filter((l) => l.url), site: site?.images ?? [], hub: hub?.images ?? [], googlePhotos: g.photos?.length ?? 0, instagramImage: instagram?.image ?? "", themeColor: site?.meta.themeColor ?? "" },
116
+ copySources: { googleSummary: g.summary ?? "", siteDescription: site?.meta.description ?? "", instagramBio: instagram?.bio ?? "", tripadvisor: tripadvisor?.description ?? "", headings: site?.headings ?? [], paragraphs: site?.paragraphs ?? [] },
117
+ social: { instagram: instagram ? { handle: instagram.handle, followers: instagram.followers, posts: instagram.posts, blocked: instagram.blocked } : null },
118
+ };
119
+ }
@@ -0,0 +1,49 @@
1
+ // OpenStreetMap through Nominatim: no key, so it is the fallback when Google is
2
+ // not configured and a second opinion when it is. Policy: identify yourself and
3
+ // stay under one request per second (this makes one).
4
+ import { fetchText, sameText } from "./util.mjs";
5
+ import { fromOsm } from "./hours.mjs";
6
+
7
+ function normalize(p) {
8
+ const a = p.address ?? {};
9
+ const x = p.extratags ?? {};
10
+ const tag = (...keys) => keys.map((k) => x[k]).find(Boolean) ?? "";
11
+ return {
12
+ source: "osm",
13
+ name: p.name ?? p.namedetails?.name ?? "",
14
+ street: [a.road, a.house_number].filter(Boolean).join(" "),
15
+ locality: (a.city ?? a.town ?? a.village ?? a.municipality ?? "").replace(/^Perímetro Urbano\s+/i, ""),
16
+ region: a.state ?? "",
17
+ country: (a.country_code ?? "").toUpperCase(),
18
+ postcode: a.postcode ?? "",
19
+ coords: { lat: Number(p.lat), lng: Number(p.lon) },
20
+ phone: tag("phone", "contact:phone"),
21
+ whatsapp: tag("contact:whatsapp"),
22
+ email: tag("email", "contact:email"),
23
+ website: tag("website", "contact:website"),
24
+ instagram: tag("contact:instagram"),
25
+ facebook: tag("contact:facebook"),
26
+ cuisines: (x.cuisine ?? "").split(";").map((c) => c.trim().replace(/_/g, " ")).filter(Boolean).map((c) => c[0].toUpperCase() + c.slice(1)),
27
+ hours: fromOsm(x.opening_hours),
28
+ hoursRaw: x.opening_hours ?? "",
29
+ osmUrl: p.osm_type && p.osm_id ? `https://www.openstreetmap.org/${p.osm_type}/${p.osm_id}` : "",
30
+ };
31
+ }
32
+
33
+ export async function searchOsm({ name, location, country }) {
34
+ const params = new URLSearchParams({
35
+ q: `${name}, ${location}`.replace(/,\s*$/, ""),
36
+ format: "jsonv2", addressdetails: "1", extratags: "1", namedetails: "1", limit: "5",
37
+ ...(country && { countrycodes: country.toLowerCase() }),
38
+ });
39
+ const r = await fetchText(`https://nominatim.openstreetmap.org/search?${params}`, {
40
+ headers: { "user-agent": "cannario-research/1.0 (restaurant site template)", accept: "application/json" },
41
+ });
42
+ if (!r.ok) throw new Error(`Nominatim ${r.status}`);
43
+ const list = JSON.parse(r.text);
44
+ const hit = list.find((p) => sameText(p.name ?? p.namedetails?.name ?? "", name));
45
+ return {
46
+ place: hit ? normalize(hit) : null,
47
+ candidates: list.map((p) => ({ name: p.name ?? "", address: p.display_name })),
48
+ };
49
+ }
@@ -0,0 +1,118 @@
1
+ // Turns a profile into what a person (or an agent) acts on: a report that mirrors
2
+ // BRIEF.md's table, and the answers file `npm run setup` reads from stdin.
3
+ const val = (f) => {
4
+ if (!f) return "";
5
+ const v = f.value;
6
+ if (Array.isArray(v)) return v.join(", ");
7
+ if (v && typeof v === "object") return `${v.lat}, ${v.lng}`;
8
+ return String(v);
9
+ };
10
+
11
+ const ROWS = [
12
+ ["Legal and display name", "name", "`site.name`"],
13
+ ["Street address", "street", "`site.street`"],
14
+ ["City", "locality", "`site.locality`"],
15
+ ["Region", "region", "`site.region`"],
16
+ ["Country (ISO)", "country", "`site.country`"],
17
+ ["Map point", "coordinates", "`site.coordinates`"],
18
+ ["WhatsApp number", "whatsapp", "`site.whatsapp`, `whatsappDisplay`"],
19
+ ["Landline or reservations number", "phone", "`site.phone` (hide it if it is the WhatsApp number)"],
20
+ ["Email", "email", "`site.email`"],
21
+ ["Instagram", "instagram", "`site.instagram`, `instagramHandle`"],
22
+ ["Reservations platform", "reserveUrl", "`site.reserveUrl` (WhatsApp link if none)"],
23
+ ["Cuisine(s)", "cuisines", "`site.cuisines`"],
24
+ ["Price range", "priceRange", "BRIEF only"],
25
+ ["Menu language", "menuLocale", "`MENU_LOCALE`"],
26
+ ["Current website", "website", "reference (the new `site.url` is the new domain)"],
27
+ ["TripAdvisor", "tripadvisor", "reference"],
28
+ ["Facebook", "facebook", "reference"],
29
+ ["TikTok", "tiktok", "reference"],
30
+ ["Google Maps", "mapsUrl", "reference"],
31
+ ["Descriptor (tagline idea)", "descriptor", "`brand.tagline`, both languages"],
32
+ ];
33
+
34
+ const cell = (s) => String(s).replace(/\|/g, "\\|");
35
+
36
+ export function renderReport(p, { notes, photos }) {
37
+ const out = [];
38
+ const q = p.query;
39
+ out.push(`# Research: ${q.name}${q.location ? `, ${q.location}` : ""}`, "", `Generated ${p.generatedAt}. Everything here is **unconfirmed**: check it with the client and copy the answers into \`BRIEF.md\`. Sources disagree often, and the confidence column says how many agreed.`, "");
40
+ if (p.warnings.length) out.push("## Warnings", "", ...p.warnings.map((w) => `- ${w}`), "");
41
+ if (notes.length) out.push("## What each source did", "", ...notes.map((n) => `- ${n}`), "");
42
+
43
+ out.push("## Facts", "", "| Item | Value | Source | Confidence | Goes in |", "| --- | --- | --- | --- | --- |");
44
+ for (const [label, key, goes] of ROWS) {
45
+ const f = p.fields[key];
46
+ out.push(`| ${label} | ${f ? cell(val(f)) : "_not found_"} | ${f?.source ?? ""} | ${f?.confidence ?? ""} | ${goes} |`);
47
+ }
48
+ out.push("");
49
+
50
+ const conflicts = Object.entries(p.fields).filter(([, f]) => f?.alternatives?.length);
51
+ if (conflicts.length) {
52
+ out.push("## Sources disagree", "");
53
+ for (const [key, f] of conflicts) out.push(`- **${key}**: using \`${val(f)}\` (${f.source}); others say ${f.alternatives.map((a) => `\`${val({ value: a.value })}\` (${a.source})`).join(", ")}`);
54
+ out.push("");
55
+ }
56
+ for (const key of ["whatsapp", "menuLocale"]) if (p.fields[key]?.note) out.push(`> ${key}: ${p.fields[key].note}`, "");
57
+
58
+ out.push("## Opening hours", "");
59
+ if (p.hours) {
60
+ out.push(`From ${p.hours.source} (${p.hours.confidence}). Paste into \`visit.hoursRows\`:`, "");
61
+ for (const lang of ["es", "en"]) out.push(`\`${lang}\``, "```ts", "hoursRows: [", ...p.hours.rows[lang].map((r) => ` { label: ${JSON.stringify(r.label)}, value: ${JSON.stringify(r.value)} },`), "],", "```", "");
62
+ for (const c of p.hours.conflicts) out.push(`- ${c.source} disagrees: ${c.rows.map((r) => `${r.label} ${r.value}`).join("; ")}`);
63
+ } else out.push("No structured hours found.");
64
+ if (p.textHours.length) out.push("", "Hours text found on the website (verify, then use):", ...p.textHours.map((t) => `- ${t}`));
65
+ out.push("");
66
+
67
+ out.push("## Links worth following", "");
68
+ const list = (label, arr) => arr.length && out.push(`- **${label}:** ${arr.map((u) => u).join(", ")}`);
69
+ list("Menu", p.links.menu);
70
+ list("Delivery", p.links.delivery);
71
+ list("Link-in-bio", p.links.hubs);
72
+ list("Waze", p.links.waze);
73
+ if (p.links.menu.some((u) => /cluvi/i.test(u))) out.push("- The menu is on Cluvi: use `tablefacts menu cluvi` (see `src/menu/README.md` in the tablefacts package).");
74
+ else if (p.links.menu.some((u) => /\.pdf/i.test(u))) out.push("- The menu is a PDF: `tablefacts menu raw` reads it.");
75
+ out.push("");
76
+
77
+ if (p.ratings.length) out.push("## Ratings (context only)", "", ...p.ratings.map((r) => `- ${r.source}: ${r.value}${r.count ? ` (${r.count} reviews)` : ""}`), "");
78
+ if (p.social.instagram && !p.social.instagram.blocked) out.push(`Instagram @${p.social.instagram.handle}: ${p.social.instagram.followers} followers, ${p.social.instagram.posts} posts.`, "");
79
+
80
+ out.push("## Images", "");
81
+ out.push(`- Website images found: ${p.images.site.length}; link-in-bio: ${p.images.hub.length}; Google photos available: ${p.images.googlePhotos}.`);
82
+ if (p.images.logos[0]) out.push(`- Logo candidates: ${p.images.logos.slice(0, 4).map((l) => l.url).join(", ")}`);
83
+ if (p.images.themeColor) out.push(`- Site theme colour: \`${p.images.themeColor}\` (a hint for \`tokens.css\`).`);
84
+ if (photos.length) out.push(`- Downloaded to \`photos/\` (${photos.length}). **They are reference, not assets**: the restaurant or the photographer owns them. Ask the client for originals, or get written permission, before shipping any.`);
85
+ out.push("");
86
+
87
+ out.push("## Material for the copy", "", "Facts and tone only. Rewrite it in the template's voice in both languages; do not paste it.", "");
88
+ const c = p.copySources;
89
+ for (const [label, text] of [["Google summary", c.googleSummary], ["Site description", c.siteDescription], ["Instagram bio", c.instagramBio], ["TripAdvisor", c.tripadvisor]]) if (text) out.push(`- **${label}:** ${text}`);
90
+ if (c.headings.length) out.push(`- **Site headings:** ${c.headings.slice(0, 12).join(" / ")}`);
91
+ for (const t of c.paragraphs.slice(0, 6)) out.push(` - ${t}`);
92
+ out.push("");
93
+
94
+ const missing = ROWS.filter(([, key]) => !p.fields[key] && ["name", "street", "coordinates", "whatsapp", "instagram", "reserveUrl", "cuisines"].includes(key)).map(([l]) => l);
95
+ out.push("## Ask the client", "", ...[...missing, "Opening hours incl. holidays", "Logo as SVG and original photos", "The story, signature dishes and events", "Production domain"].filter((v, i, a) => a.indexOf(v) === i && (v !== "Opening hours incl. holidays" || true)).map((m) => `- ${m}`), "");
96
+ out.push("## Next", "", "```bash", `npm run setup < .tablefacts/research/${q.slug}/setup-answers.txt`, "git diff # review before keeping it", "```", "", "Blank lines in that file keep the template's current value, so anything not found stays a placeholder.", "");
97
+ return out.join("\n");
98
+ }
99
+
100
+ /** One line per prompt of `data/scripts/setup.mjs`, in its order. Blank keeps the current value. */
101
+ export function setupAnswers(p) {
102
+ const f = p.fields;
103
+ const ig = f.instagram?.value;
104
+ return [
105
+ val(f.name),
106
+ "", // production URL: the new domain, not the current website
107
+ f.reserveUrl?.value ?? "",
108
+ f.whatsapp?.display ?? "",
109
+ ig ? `@${ig}` : "",
110
+ val(f.street),
111
+ val(f.locality),
112
+ val(f.region),
113
+ val(f.country),
114
+ f.coordinates?.value?.lat ?? "",
115
+ f.coordinates?.value?.lng ?? "",
116
+ val(f.cuisines),
117
+ ].join("\n") + "\n";
118
+ }
@@ -0,0 +1,49 @@
1
+ // Instagram, TripAdvisor and link-in-bio hubs. All three are best-effort: they
2
+ // sit behind login walls or bot protection, so each one returns what it could
3
+ // read plus a `blocked` reason, and the report says what is missing.
4
+ import { fetchText } from "./util.mjs";
5
+ import { readPage } from "./website.mjs";
6
+
7
+ /**
8
+ * Public Instagram profile metadata. Without a login the page only exposes the
9
+ * share card: "1,234 Followers, 56 Following, 789 Posts - Name (@handle) on
10
+ * Instagram: "bio"". The bio often carries the phone, the area and a hub link.
11
+ */
12
+ export async function readInstagram(handle) {
13
+ const url = `https://www.instagram.com/${handle}/`;
14
+ let r;
15
+ try { r = await fetchText(url, { headers: { accept: "text/html" } }); } catch (e) { return { handle, url, blocked: e.message }; }
16
+ if (!r.ok) return { handle, url, blocked: `HTTP ${r.status}` };
17
+ const meta = (key) => r.text.match(new RegExp(`<meta[^>]+(?:property|name)=["']${key}["'][^>]+content=["']([^"']*)["']`, "i"))?.[1] ?? "";
18
+ const desc = meta("og:description").replace(/&quot;/g, '"').replace(/&#039;|&#x27;/g, "'").replace(/&amp;/g, "&");
19
+ const title = meta("og:title");
20
+ if (!desc && !title) return { handle, url, blocked: "login wall (no public metadata)" };
21
+ const stats = desc.match(/([\d.,KMkm]+)\s+Followers?,\s*([\d.,KMkm]+)\s+Following,\s*([\d.,KMkm]+)\s+Posts?/i);
22
+ const bio = desc.match(/on Instagram:\s*["“]([\s\S]*?)["”]?\s*$/i)?.[1]?.trim() ?? "";
23
+ const links = bio.match(/https?:\/\/[^\s)]+|\b[\w-]+\.(?:com|co|link|bio|ee)\/[\w./-]+/gi) ?? [];
24
+ return {
25
+ handle, url,
26
+ displayName: title.replace(/\s*\(@.*$/, "").trim(),
27
+ followers: stats?.[1] ?? "",
28
+ posts: stats?.[3] ?? "",
29
+ bio,
30
+ phones: [...bio.matchAll(/\+?\d[\d\s().-]{7,}\d/g)].map((m) => m[0].trim()),
31
+ links: links.map((l) => (/^https?:/.test(l) ? l : `https://${l}`)),
32
+ image: meta("og:image"),
33
+ };
34
+ }
35
+
36
+ /** TripAdvisor: schema.org JSON-LD when the page loads; DataDome often blocks it. */
37
+ export async function readTripadvisor(url, { browser } = {}) {
38
+ const { page, blocked } = await readPage(url, { browser });
39
+ if (!page) return { url, blocked };
40
+ if (!page.jsonld && page.textLength < 500) return { url, blocked: "bot protection (page has no data)" };
41
+ return { url: page.url, jsonld: page.jsonld, description: page.meta.description, hoursText: page.hoursText, images: page.images.slice(0, 10) };
42
+ }
43
+
44
+ /** A link-in-bio hub (Linktree, Beacons, bio.link): its links, classified like a website's. */
45
+ export async function readHub(url, { browser } = {}) {
46
+ const { page, blocked } = await readPage(url, { renderJs: false, browser });
47
+ if (!page) return { url, blocked };
48
+ return { url: page.url, title: page.title, meta: page.meta, links: page.links, anchors: page.anchors.filter((a) => a.text).slice(0, 30), logos: page.logos, images: page.images.slice(0, 10) };
49
+ }
@@ -0,0 +1,104 @@
1
+ // Small helpers shared by the research sources: HTTP, name matching.
2
+ import { fold, slugify as slug } from "../../lib/text.mjs";
3
+
4
+ export { fold };
5
+
6
+ export const BROWSER_UA =
7
+ "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/130.0 Safari/537.36 cannario-research";
8
+
9
+ export const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
10
+
11
+ /** Runs `fn(item, index)` over `items` with at most `size` in flight; results keep the order of `items`. */
12
+ export async function mapPool(items, size, fn) {
13
+ const results = new Array(items.length);
14
+ let next = 0;
15
+ const worker = async () => {
16
+ while (next < items.length) {
17
+ const i = next++;
18
+ results[i] = await fn(items[i], i);
19
+ }
20
+ };
21
+ await Promise.all(Array.from({ length: Math.min(size, items.length) }, worker));
22
+ return results;
23
+ }
24
+
25
+ export const MAX_BODY_BYTES = 2 * 1024 * 1024;
26
+ const BINARY_TYPE = /^\s*(image|video|audio|font)\/|application\/(pdf|zip|gzip|octet-stream|vnd\.[\w.+-]*(?:excel|powerpoint|ms-)[\w.+-]*)/i;
27
+
28
+ async function readCapped(res, cap) {
29
+ const declared = Number(res.headers.get("content-length"));
30
+ const reader = res.body?.getReader?.();
31
+ if (!reader) {
32
+ // No stream (a minimal Response-like): read it whole, then cut.
33
+ const text = await res.text();
34
+ return text.length > cap ? { text: text.slice(0, cap), truncated: true } : { text, truncated: false };
35
+ }
36
+ const chunks = [];
37
+ let size = 0;
38
+ let truncated = Number.isFinite(declared) && declared > cap;
39
+ while (size < cap) {
40
+ const { done, value } = await reader.read();
41
+ if (done) break;
42
+ chunks.push(value);
43
+ size += value.length;
44
+ }
45
+ if (size >= cap) {
46
+ truncated = true;
47
+ await reader.cancel().catch(() => {});
48
+ }
49
+ const buf = Buffer.concat(chunks.map((c) => Buffer.from(c.buffer, c.byteOffset, c.byteLength))).subarray(0, cap);
50
+ return { text: buf.toString("utf8"), truncated };
51
+ }
52
+
53
+ /**
54
+ * GETs a URL as text. The body is capped at 2 MB (`truncated: true` when it was cut, never an error) and a
55
+ * clearly binary content-type (image, pdf, zip...) is not read at all (`skipped: true`, empty text).
56
+ */
57
+ export async function fetchText(url, { headers = {}, timeout = 15000, maxBytes = MAX_BODY_BYTES } = {}) {
58
+ const res = await fetch(url, {
59
+ headers: { "user-agent": BROWSER_UA, "accept-language": "es,en;q=0.8", accept: "text/html,application/json,*/*", ...headers },
60
+ redirect: "follow",
61
+ signal: AbortSignal.timeout(timeout),
62
+ });
63
+ const type = res.headers.get("content-type") ?? "";
64
+ const head = { ok: res.ok, status: res.status, url: res.url, type };
65
+ if (BINARY_TYPE.test(type)) {
66
+ await res.body?.cancel?.().catch?.(() => {});
67
+ return { ...head, text: "", skipped: true };
68
+ }
69
+ const { text, truncated } = await readCapped(res, maxBytes);
70
+ return truncated ? { ...head, text, truncated: true } : { ...head, text };
71
+ }
72
+
73
+ const STOP = new Set(["restaurante", "restaurant", "bar", "the", "el", "la", "los", "las", "de", "del", "y", "and", "cafe", "gastro", "grill", "bistro", "calle", "carrera", "cra", "cl", "av", "avenida"]);
74
+ export const tokens = (s) => fold(s).split(/[^a-z0-9]+/).filter((t) => t && !STOP.has(t));
75
+
76
+ /** Loose text match: at least 60% of the shorter side's words appear in the other. */
77
+ export function sameText(a, b) {
78
+ const A = new Set(tokens(a));
79
+ const B = new Set(tokens(b));
80
+ if (!A.size || !B.size) return fold(a).trim() === fold(b).trim();
81
+ let hit = 0;
82
+ for (const t of A) if (B.has(t)) hit++;
83
+ return hit / Math.min(A.size, B.size) >= 0.6;
84
+ }
85
+
86
+ export const slugify = (s) => slug(s) || "restaurant";
87
+
88
+ export const digits = (s) => String(s ?? "").replace(/\D/g, "");
89
+
90
+ /** Same phone number written differently (spaces, +country, leading 0). */
91
+ export const samePhone = (a, b) => {
92
+ const x = digits(a);
93
+ const y = digits(b);
94
+ return x.length >= 7 && y.length >= 7 && x.slice(-9) === y.slice(-9);
95
+ };
96
+
97
+ /** Distance in km between two { lat, lng } points. */
98
+ export function km(a, b) {
99
+ const rad = (d) => (d * Math.PI) / 180;
100
+ const dLat = rad(b.lat - a.lat);
101
+ const dLng = rad(b.lng - a.lng);
102
+ const h = Math.sin(dLat / 2) ** 2 + Math.cos(rad(a.lat)) * Math.cos(rad(b.lat)) * Math.sin(dLng / 2) ** 2;
103
+ return 12742 * Math.asin(Math.sqrt(h));
104
+ }