tablefacts 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. package/.env.example +2 -0
  2. package/CHANGELOG.md +39 -0
  3. package/README.md +37 -9
  4. package/package.json +2 -2
  5. package/src/lib/types.mjs +7 -2
  6. package/src/menu/README.md +37 -11
  7. package/src/menu/cluvi/config.mjs +6 -0
  8. package/src/menu/cluvi/import.mjs +6 -2
  9. package/src/menu/lib/db.mjs +81 -17
  10. package/src/menu/lib/import.mjs +20 -3
  11. package/src/menu/lib/run.mjs +15 -0
  12. package/src/menu/lib/tables.mjs +44 -0
  13. package/src/menu/raw/config.mjs +6 -0
  14. package/src/menu/raw/import.mjs +5 -2
  15. package/src/research/README.md +23 -13
  16. package/src/research/index.mjs +52 -17
  17. package/src/research/lib/merge.mjs +11 -10
  18. package/src/research/lib/report.mjs +72 -19
  19. package/src/research/lib/search.mjs +127 -0
  20. package/src/research/lib/social.mjs +137 -26
  21. package/src/research/lib/util.mjs +22 -0
  22. package/src/research/lib/website.mjs +3 -14
  23. package/src/research/research.mjs +12 -8
  24. package/types/lib/types.d.mts +29 -3
  25. package/types/menu/cluvi/config.d.mts +1 -0
  26. package/types/menu/cluvi/import.d.mts +2 -0
  27. package/types/menu/lib/db.d.mts +31 -3
  28. package/types/menu/lib/import.d.mts +1 -1
  29. package/types/menu/lib/run.d.mts +3 -0
  30. package/types/menu/lib/tables.d.mts +18 -0
  31. package/types/menu/raw/config.d.mts +1 -0
  32. package/types/menu/raw/import.d.mts +3 -0
  33. package/types/research/lib/merge.d.mts +4 -1
  34. package/types/research/lib/report.d.mts +14 -1
  35. package/types/research/lib/search.d.mts +58 -0
  36. package/types/research/lib/social.d.mts +32 -31
  37. package/types/research/lib/util.d.mts +5 -0
  38. package/types/research/lib/website.d.mts +20 -20
@@ -0,0 +1,44 @@
1
+ // The table names a menu import writes. Several restaurants can share one
2
+ // Supabase database, so each restaurant's tables carry its own prefix
3
+ // (`cannario_menu_categories`, `makibar_menu_categories`, ...). The prefix is
4
+ // restaurant-specific and lives in the source's `config.mjs`; the CLI's
5
+ // `--table-prefix` overrides it for one run.
6
+ import { optionError } from "../../lib/errors.mjs";
7
+
8
+ // The only prefixes accepted: empty, or lowercase letters, digits and
9
+ // underscores ending in "_". A value that passed this allow-list is safe to
10
+ // build the table names from; nothing else is ever put into the SQL text.
11
+ const PREFIX = /^[a-z][a-z0-9_]*_$|^$/;
12
+
13
+ /**
14
+ * `tablePrefix` when it is empty or a safe `<name>_` prefix (an ECONFIG error otherwise).
15
+ * @param {string} [tablePrefix]
16
+ * @returns {string}
17
+ */
18
+ export function validateTablePrefix(tablePrefix) {
19
+ const value = tablePrefix ?? "";
20
+ if (!PREFIX.test(value)) {
21
+ throw optionError(
22
+ "tablePrefix",
23
+ `${JSON.stringify(value)} is not a valid table prefix: \`tablePrefix\` must be empty or lowercase letters, digits and underscores ending in "_", such as "makibar_".`,
24
+ "ECONFIG",
25
+ );
26
+ }
27
+ return value;
28
+ }
29
+
30
+ /**
31
+ * The three `public` menu tables for a prefix, e.g.
32
+ * `{ categories: "public.makibar_menu_categories", sections: "public.makibar_menu_sections", products: "public.makibar_menu_products" }`.
33
+ * The default (empty prefix) is the unprefixed `public.menu_*` set.
34
+ * @param {string} [tablePrefix]
35
+ * @returns {{ categories: string, sections: string, products: string }}
36
+ */
37
+ export function menuTables(tablePrefix) {
38
+ const prefix = validateTablePrefix(tablePrefix);
39
+ return {
40
+ categories: `public.${prefix}menu_categories`,
41
+ sections: `public.${prefix}menu_sections`,
42
+ products: `public.${prefix}menu_products`,
43
+ };
44
+ }
@@ -1,6 +1,12 @@
1
1
  // Everything restaurant-specific about the image-menu import. Edit this file,
2
2
  // not the other scripts, when the menu is organised differently.
3
3
  export default {
4
+ // This restaurant's tables in a shared Supabase database: the import writes
5
+ // public.mombasa_menu_categories / _menu_sections / _menu_products. Change
6
+ // it (or pass `--table-prefix`) when importing another restaurant; leave it
7
+ // empty only when this restaurant owns the unprefixed menu_* tables.
8
+ tablePrefix: "mombasa_",
9
+
4
10
  // The page that shows the menu pictures, or direct image URLs.
5
11
  // `tablefacts menu raw <url> [<url>...]` overrides it for one run.
6
12
  url: "https://www.mombasa.co/carta-restaurante-espanol/",
@@ -85,6 +85,8 @@ export async function fetchImageMenu({ urls = [], only, provider, model, minWidt
85
85
  menu,
86
86
  notes,
87
87
  title: `Image menu: ${chosen.length} pages from ${host} (${reading} read with ${providers[provider].label} ${model}, ${chosen.length - reading} from the saved transcriptions), prices in ${currency}`,
88
+ // The restaurant's own table set, so a shared database is never touched by accident.
89
+ tablePrefix: config.tablePrefix,
88
90
  };
89
91
  }
90
92
 
@@ -93,7 +95,8 @@ export async function fetchImageMenu({ urls = [], only, provider, model, minWidt
93
95
  * @param {import('../../lib/types.mjs').ImportImageMenuOptions} [options]
94
96
  * @returns {Promise<import('../../lib/types.mjs').ImportResult>}
95
97
  */
96
- export async function importImageMenu({ urls, only, provider, model, minWidth, refresh, apiKey, env, config, ...importOptions } = {}) {
98
+ export async function importImageMenu({ urls, only, provider, model, minWidth, refresh, apiKey, env, config = defaultConfig, ...importOptions } = {}) {
97
99
  const fetched = await fetchImageMenu({ urls, only, provider, model, minWidth, refresh, apiKey, env, config, projectDir: importOptions.projectDir, log: importOptions.log });
98
- return importMenu({ ...fetched, ...importOptions, env });
100
+ // An explicit `tablePrefix` (or the CLI's --table-prefix) wins over the config's.
101
+ return importMenu({ ...fetched, ...importOptions, tablePrefix: importOptions.tablePrefix ?? config.tablePrefix, env });
99
102
  }
@@ -28,42 +28,52 @@ see the [main README](../../README.md#use-it-as-a-library). `out` (relative path
28
28
  | --- | --- | --- |
29
29
  | Google Maps (Places API) | address, map point, phone, hours, price, cuisine types, photos, status | Needs `GOOGLE_PLACES_API_KEY` in `.env` (Places API (New) enabled). Phone and hours fields are billed at the Enterprise rate; one search per run |
30
30
  | OpenStreetMap (Nominatim) | address, map point, sometimes phone, hours, website, socials | No key. Often stale or sparse |
31
+ | Web search (DuckDuckGo) | candidate website, Instagram, TripAdvisor, Maps and hub links | No key. Runs only when Google and OpenStreetMap find no place or no website, so the reported "name-only finds nothing" case now has a fallback |
31
32
  | Website | JSON-LD, contact page, WhatsApp, reserve platform, menu links, images, logo, theme colour, page text | Falls back to Playwright on 403 or client-rendered pages (`--render` forces it) |
32
- | Instagram | handle, followers, bio, bio link | Public share card only. Usually login-walled |
33
- | TripAdvisor | JSON-LD (address, phone, cuisine, price, rating) | Usually bot-blocked: the link is kept for a manual read |
33
+ | Instagram | handle, followers, bio, bio link | Tries the public share card, the web profile API, oEmbed and finally a search-engine snippet of the bio. Requests identify honestly and never log in; often all surfaces are walled |
34
+ | TripAdvisor | JSON-LD (address, phone, cuisine, price, rating) | Usually bot-blocked. Playwright is tried on a 403 or a thin page like any website, but DataDome still wins. The name and city are read from the URL slug without fetching, and the link is kept for a manual read (stated at the top of the report) |
34
35
  | Linktree and similar | WhatsApp, reserve, menu, delivery links | Found from the website, Instagram bio or `--linktree` |
35
36
 
36
- Sources find each other (website → Instagram, TripAdvisor, hub). Order of trust per field is in `lib/merge.mjs`; two sources agreeing is "high", one is "medium", a guess is "low".
37
+ Sources find each other (web search or website → Instagram, TripAdvisor, hub). Order of trust per field is in `lib/merge.mjs`; two sources agreeing is "high", one is "medium", a guess is "low".
37
38
 
38
39
  ## How it runs
39
40
 
40
41
  - **Concurrent lookups.** Google and OpenStreetMap run together; then the website is read. Instagram, TripAdvisor and
41
- the link-in-bio page start as soon as their address is known (given, or found on the website), so a slow source does
42
- not hold up the others. Report notes keep a fixed order whatever finishes first. Photo downloads run four at a time.
42
+ the link-in-bio page start as soon as their address is known (given, or found on the website or web search), so a slow
43
+ source does not hold up the others. Report notes keep a fixed order whatever finishes first. Photo downloads run four at a time.
44
+ - **Web-search fallback.** When Google and OpenStreetMap find no place or no website, a key-free DuckDuckGo query
45
+ runs and its candidate links feed the rest of the pipeline. It is one extra request, only when it is needed.
43
46
  - **One shared browser.** Playwright's Chromium is launched at most once per run, only when a page needs rendering
44
47
  (`--render`, a 403, a client-rendered page), is shared by the website, TripAdvisor and link-in-bio readers, and is
45
48
  closed at the end even on failure.
46
49
  - **Fetch cap.** Pages are read up to 2 MB each; larger bodies are truncated, so a huge page cannot exhaust memory.
47
50
 
51
+ ## Runs and re-runs
52
+
53
+ Output goes to `.tablefacts/research/<slug>/`; two spellings of the same name ("Makibar" and "Maki Bar") make two
54
+ folders. A similar-name run is noted in the report, and `.tablefacts/research/latest.json` always points at the newest
55
+ report, so reading every folder cannot resurface a stale one.
56
+
48
57
  ## Limits
49
58
 
50
59
  - It does not log in anywhere or get past bot protection. A blocked source is reported, not worked around.
51
60
  - It finds facts, not copy, logos or licensed photos.
52
- - A wrong match (a sister branch, a closed place) is the main risk: read the Warnings and "Sources disagree" first.
61
+ - A wrong match (a sister branch, a closed place) is the main risk: read the Warnings and "Sources disagree" first. When Google returns several places with the same name, the report says so and asks which branch.
62
+ - Only public surfaces are tried for Instagram/TripAdvisor. When they are walled, the report keeps the link and says so at the top; it does not defeat the wall.
53
63
 
54
64
  ## For agents
55
65
 
56
- Use it first, whenever you are given only a restaurant's name and place (or little else).
66
+ Use it first, whenever you are given only a restaurant's name and place (or little else). If Google and OpenStreetMap find no place or no website, a key-free web search runs automatically, so a name-only call is no longer an empty report.
57
67
 
58
- 1. **Run it** from the repo root: `tablefacts research "<name>" "<city, country>" --country <ISO>`. Add `--website`, `--instagram`, `--tripadvisor` or `--linktree` for anything the user gave you; a user-supplied value is trusted over a search result. It takes under a minute, needs no prompts and prints the output folder.
59
- 2. **Read `out/<slug>/report.md`**, in this order: Warnings, "What each source did", Facts, "Sources disagree". Confirm the match is the right restaurant (name, address, not closed) before using anything. If it matched the wrong place, rerun with a more specific location or `--website`.
68
+ 1. **Run it** from the repo root: `tablefacts research "<name>" "<city, country>" --country <ISO>`. Add `--website`, `--instagram`, `--tripadvisor` or `--linktree` for anything the user gave you; a user-supplied value is trusted over a search result. Uploading a Google Maps link as `--website` (or pasting it in the report as `mapsUrl`) unlocks coordinates; ask for it first when the report is thin. It takes under a minute, needs no prompts and prints the output folder.
69
+ 2. **Read `out/<slug>/report.md`**, in this order: the kept-links notice and Warnings, "What each source did", Facts, "Sources disagree", "Ask the client". Confirm the match is the right restaurant (name, address, not closed) before using anything. If it matched the wrong place, rerun with a more specific location or `--website`.
60
70
  3. **Treat every value as unconfirmed.** Fill `BRIEF.md` with the value and its source, never as client-confirmed. Anything under "not found" stays a placeholder and goes in your hand-off as an open question. Do not invent it.
61
- 4. **Apply the facts**: review `setup-answers.txt`, run `npm run setup < .tablefacts/research/<slug>/setup-answers.txt`, read `git diff`. Setup does not cover `phone`, `email`, `MENU_LOCALE`, `brand.tagline`, `visit.hoursRows` or `jsonld.ts`: take those from the report by hand. Hours are ready to paste. Add street, phone and hours to `jsonld.ts` only for values the client confirmed.
71
+ 4. **Apply the facts**: review `setup-answers.txt`, run `npm run setup < .tablefacts/research/<slug>/setup-answers.txt`, read `git diff`. A low-confidence guess (marked "low (guess)") is left blank so setup keeps the template's placeholder; it is never written in as a fact. Setup does not cover `phone`, `email`, `MENU_LOCALE`, `brand.tagline`, `visit.hoursRows` or `jsonld.ts`: take those from the report by hand. Hours are ready to paste. Add street, phone and hours to `jsonld.ts` only for values the client confirmed.
62
72
  5. **Use the leads**: a Cluvi menu link means `tablefacts menu cluvi`, a PDF means `tablefacts menu raw` (`src/menu/README.md`). The theme colour and logo candidates are hints for `tokens.css` and `public/logo.svg`, not decisions.
63
- 6. **Blocked sources** (Instagram, TripAdvisor) are normal. Open the link yourself, or ask the user for the bio and hours. Do not try to get around a login wall.
73
+ 6. **Blocked sources** (Instagram, TripAdvisor) are normal. The top of the report names any link that was kept unread; open it yourself, or ask the user for the bio and hours. Do not try to get around a login wall. `.tablefacts/research/latest.json` points at the newest run when the same place was researched under a different spelling.
64
74
 
65
- `profile.json` has the same data as JSON for scripting: `fields.<name>` is `{ value, source, confidence, alternatives? }`, `hours.rows.{es,en}` are `hoursRows`, and `raw` holds each source's untouched result. Field names: `name street locality region country coordinates phone whatsapp email instagram facebook tiktok tripadvisor website reserveUrl priceRange mapsUrl descriptor cuisines menuLocale`.
75
+ `profile.json` has the same data as JSON for scripting: `fields.<name>` is `{ value, source, confidence, alternatives? }`, `hours.rows.{es,en}` are `hoursRows`, and `raw` holds each source's untouched result (including `search`). Field names: `name street locality region country coordinates phone whatsapp email instagram facebook tiktok tripadvisor website reserveUrl priceRange mapsUrl descriptor cuisines menuLocale`.
66
76
 
67
77
  Do not commit `out/`, and never paste the Google key anywhere but `.env`. Photos it downloads belong to the restaurant or the photographer: use them as reference, not as site assets.
68
78
 
69
- Code map: `research.mjs` runs the sources in order and writes the files; `lib/google.mjs`, `osm.mjs`, `website.mjs`, `social.mjs` read one source each; `merge.mjs` holds the per-field trust order; `hours.mjs` normalises hours; `report.mjs` writes the report and the setup answers. A new source is a reader returning plain data, a line in `research.mjs` and its candidates in `merge.mjs`.
79
+ Code map: `research.mjs` runs the sources in order and writes the files; `lib/google.mjs`, `osm.mjs`, `search.mjs`, `website.mjs`, `social.mjs` read one source each; `merge.mjs` holds the per-field trust order; `hours.mjs` normalises hours; `report.mjs` writes the report and the setup answers. A new source is a reader returning plain data, a line in `research.mjs` and its candidates in `merge.mjs`.
@@ -1,8 +1,8 @@
1
1
  // Programmatic entry for the research pipeline: gathers what the public web says
2
2
  // about a restaurant and writes profile.json, report.md and setup-answers.txt.
3
3
  // Silent by default and never exits the process; the CLI lives in research.mjs.
4
- import { mkdir, writeFile } from "node:fs/promises";
5
- import { join, resolve } from "node:path";
4
+ import { mkdir, readdir, writeFile } from "node:fs/promises";
5
+ import { dirname, join, resolve } from "node:path";
6
6
  import { resolveEnv } from "../lib/env.mjs";
7
7
  import { optionError } from "../lib/errors.mjs";
8
8
  import { normalizeLog } from "../lib/log.mjs";
@@ -11,8 +11,9 @@ import { downloadPhotos, searchGoogle } from "./lib/google.mjs";
11
11
  import { buildProfile } from "./lib/merge.mjs";
12
12
  import { searchOsm } from "./lib/osm.mjs";
13
13
  import { renderReport, setupAnswers } from "./lib/report.mjs";
14
+ import { searchWeb } from "./lib/search.mjs";
14
15
  import { readHub, readInstagram, readTripadvisor } from "./lib/social.mjs";
15
- import { mapPool, slugify } from "./lib/util.mjs";
16
+ import { mapPool, sameText, slugify } from "./lib/util.mjs";
16
17
  import { createBrowser, scrapeSite } from "./lib/website.mjs";
17
18
 
18
19
  /**
@@ -31,6 +32,7 @@ export async function research(options = {}) {
31
32
 
32
33
  const slug = slugify(name);
33
34
  const outDir = options.out ? resolveIn(options.projectDir, options.out) : workDirIn(options.projectDir, "research", slug);
35
+ const researchDir = dirname(outDir);
34
36
  const query = { name, location, slug, country, website, instagram: instagramOpt, tripadvisor: tripadvisorOpt, linktree };
35
37
  const notes = [];
36
38
 
@@ -50,7 +52,7 @@ export async function research(options = {}) {
50
52
 
51
53
  // One Chromium for the whole run, launched only if a page needs rendering.
52
54
  const browser = createBrowser();
53
- let site, instagram, hub, tripadvisor, taUrl, handle, google, osm;
55
+ let site, instagram, hub, tripadvisor, taUrl, handle, google, osm, search, branches = [];
54
56
  try {
55
57
  // Google and OpenStreetMap only need the name and place, so they run together; their notes are added in this order afterwards.
56
58
  const googleNotes = [];
@@ -65,20 +67,33 @@ export async function research(options = {}) {
65
67
  notes.push(...googleNotes, ...osmNotes);
66
68
  google = googleResult?.place ?? null;
67
69
  osm = osmResult?.place ?? null;
70
+ // Several Google places with the same name are likely branches of one chain; the template assumes one address.
71
+ branches = (googleResult?.candidates ?? []).filter((c) => sameText(c.name, name));
68
72
 
69
- const websiteUrl = website ?? google?.website ?? osm?.website;
73
+ // No website yet: a key-free web search turns a bare name + place into candidate links.
74
+ if (!website && !google?.website && !osm?.website) {
75
+ search = await source("Web search", () => searchWeb({ name, location }));
76
+ if (search) {
77
+ const found = search.candidates.length ? ` (${search.candidates.slice(0, 2).map((c) => c.title || c.url).join(", ")})` : "";
78
+ notes.push(`Web search: ${search.results.length ? `${search.results.length} result(s) for "${search.query}"${found}` : `no results for "${search.query}"`}`);
79
+ }
80
+ }
81
+
82
+ const websiteUrl = website ?? google?.website ?? osm?.website ?? search?.website;
70
83
  site = websiteUrl ? await source("Website", () => scrapeSite(websiteUrl, { renderJs: render, browser })) : null;
71
- if (site) notes.push(`Website: read ${site.pages.length} page(s) of ${site.url}${site.pages.some((p) => p.rendered) ? " (rendered with Playwright)" : ""}`);
84
+ if (site) notes.push(`Website: read ${site.pages.length} page(s) of ${site.url}${site.pages.some((p) => p.rendered) ? " (rendered with Playwright)" : ""}${!website && !google?.website && !osm?.website ? " (found by web search)" : ""}`);
72
85
  else if (!websiteUrl) notes.push("Website: none found. Pass the website (--website / `website`) if the restaurant has one.");
73
86
 
74
87
  // Follow the leads the first sources gave. What is already known (handle, TripAdvisor link, hub link)
75
88
  // starts right away; the rest waits only for the source that can supply it.
76
- handle = (instagramOpt ?? site?.links.instagram[0] ?? osm?.instagram ?? "").replace(/^https?:\/\/(www\.)?instagram\.com\//i, "").replace(/^@/, "").split(/[/?#]/)[0];
77
- taUrl = tripadvisorOpt ?? site?.links.tripadvisor[0] ?? (site?.jsonld?.sameAs ?? []).find((u) => /tripadvisor\./.test(u));
78
- const directHubUrl = linktree ?? site?.links.hubs[0];
89
+ handle = (instagramOpt ?? site?.links.instagram[0] ?? osm?.instagram ?? search?.instagram ?? "").replace(/^https?:\/\/(www\.)?instagram\.com\//i, "").replace(/^@/, "").split(/[/?#]/)[0];
90
+ taUrl = tripadvisorOpt ?? site?.links.tripadvisor[0] ?? (site?.jsonld?.sameAs ?? []).find((u) => /tripadvisor\./.test(u)) ?? search?.tripadvisor;
91
+ const directHubUrl = linktree ?? site?.links.hubs[0] ?? search?.hubs?.[0];
79
92
  const readHubAt = (url) => source("Link-in-bio", () => readHub(url, { browser }));
80
93
 
81
- const instagramP = handle ? source("Instagram", () => readInstagram(handle)) : null;
94
+ // The search snippet for the profile is a last resort when Instagram walls the bio.
95
+ const igSnippet = search?.results?.find((r) => /instagram\.com\//i.test(r.url) && r.url.toLowerCase().includes(`/${handle.toLowerCase()}`))?.snippet ?? "";
96
+ const instagramP = handle ? source("Instagram", () => readInstagram(handle, { snippet: igSnippet })) : null;
82
97
  const tripadvisorP = taUrl ? source("TripAdvisor", () => readTripadvisor(taUrl, { browser })) : null;
83
98
  // The hub URL comes from the options or the website, else from the Instagram bio.
84
99
  const hubTask = (async () => {
@@ -103,17 +118,27 @@ export async function research(options = {}) {
103
118
  }
104
119
 
105
120
  // Notes keep the order of the sequential flow: Instagram, link-in-bio, TripAdvisor.
106
- if (instagram) notes.push(instagram.blocked ? `Instagram @${handle}: not readable (${instagram.blocked}). Add the bio by hand, or use \`tablefacts photos instagram\` for posts.` : `Instagram @${handle}: profile card read`);
121
+ if (instagram) notes.push(instagram.blocked ? `Instagram @${handle}: not readable (${instagram.blocked}). Add the bio by hand, or use \`tablefacts photos instagram\` for posts.` : `Instagram @${handle}: profile read (${instagram.surface ?? "share card"})`);
107
122
  else if (!handle) notes.push("Instagram: no handle found. Pass the handle (--instagram / `instagram`).");
108
123
  if (hub) notes.push(hub.blocked ? `Link-in-bio ${hubUrl}: not readable (${hub.blocked})` : `Link-in-bio: read ${hub.url}`);
109
- if (tripadvisor) notes.push(tripadvisor.blocked ? `TripAdvisor ${taUrl}: blocked (${tripadvisor.blocked}). The link is kept; read it by hand.` : "TripAdvisor: page read");
110
- else notes.push("TripAdvisor: no link found. Pass the page (--tripadvisor / `tripadvisor`).");
124
+ if (tripadvisor) {
125
+ // The URL slug is a hint the source kept for a manual read; it never becomes a field on its own.
126
+ const hint = tripadvisor.slug ? ` URL suggests "${tripadvisor.slug.name}"${tripadvisor.slug.location ? `, ${tripadvisor.slug.location}` : ""}.` : "";
127
+ notes.push(tripadvisor.blocked ? `TripAdvisor ${taUrl}: blocked (${tripadvisor.blocked}). The link is kept; read it by hand.${hint}` : "TripAdvisor: page read");
128
+ } else notes.push("TripAdvisor: no link found. Pass the page (--tripadvisor / `tripadvisor`).");
111
129
  } finally {
112
130
  await browser.close();
113
131
  }
114
132
 
115
- const profile = buildProfile({ query, google, osm, site, hub: hub?.blocked ? null : hub, instagram, tripadvisor: tripadvisor?.blocked ? null : tripadvisor });
116
- if (tripadvisor?.blocked && taUrl && !profile.fields.tripadvisor) profile.fields.tripadvisor = { value: taUrl, source: "website", confidence: "medium" };
133
+ const profile = buildProfile({ query, google, osm, site, hub: hub?.blocked ? null : hub, instagram, tripadvisor, search });
134
+ const taOrigin = tripadvisorOpt ? "you" : (hub?.links?.tripadvisor ?? []).includes(taUrl) ? "link-in-bio" : search?.tripadvisor === taUrl ? "search" : "website";
135
+ if (tripadvisor?.blocked && taUrl && !profile.fields.tripadvisor) profile.fields.tripadvisor = { value: taUrl, source: taOrigin, confidence: "medium" };
136
+ if (branches.length > 1) profile.warnings.push(`Google returns ${branches.length} places named like "${name}" (${branches.map((b) => b.address).filter(Boolean).join("; ")}). This looks like a chain: confirm which branch this site is for.`);
137
+
138
+ // A link a source could not read is stated at the top of the report, not only in the source list.
139
+ const kept = [];
140
+ if (tripadvisor?.blocked && taUrl) kept.push({ label: "TripAdvisor", url: taUrl, reason: tripadvisor.blocked });
141
+ if (instagram?.blocked && handle) kept.push({ label: `Instagram @${handle}`, url: `https://www.instagram.com/${handle}/`, reason: instagram.blocked });
117
142
 
118
143
  // Optional photo downloads: reference material, never wired into the site.
119
144
  const photos = [];
@@ -137,11 +162,21 @@ export async function research(options = {}) {
137
162
  photos.push(...saved.filter(Boolean));
138
163
  }
139
164
 
165
+ // Two names for one place ("Makibar" and "Maki Bar") used to make two folders, and reading them all showed
166
+ // a stale report. Note the sibling run and keep a pointer to the newest, so the last run is the one to read.
167
+ const loose = slug.replace(/-/g, "");
168
+ const siblings = options.out ? [] : await readdir(researchDir, { withFileTypes: true })
169
+ .then((entries) => entries.filter((e) => e.isDirectory() && e.name.replace(/-/g, "") === loose && join(researchDir, e.name) !== outDir).map((e) => e.name))
170
+ .catch(() => []);
171
+ if (siblings.length) notes.push(`Other runs with a similar name exist (${siblings.join(", ")}); \`latest.json\` points at the newest. Older folders may be stale.`);
172
+
140
173
  const files = { profile: resolve(outDir, "profile.json"), report: resolve(outDir, "report.md"), setupAnswers: resolve(outDir, "setup-answers.txt") };
141
174
  await mkdir(outDir, { recursive: true });
142
- await writeFile(files.profile, JSON.stringify({ ...profile, raw: { google, osm, site, hub, instagram, tripadvisor }, photos }, null, 2));
143
- await writeFile(files.report, renderReport(profile, { notes, photos }));
175
+ await writeFile(files.profile, JSON.stringify({ ...profile, raw: { google, osm, site, hub, instagram, tripadvisor, search }, photos }, null, 2));
176
+ await writeFile(files.report, renderReport(profile, { notes, photos, kept }));
144
177
  await writeFile(files.setupAnswers, setupAnswers(profile));
178
+ // Only the default layout gets the "newest" pointer; a custom --out is the caller's own tree.
179
+ if (!options.out) await writeFile(join(researchDir, "latest.json"), JSON.stringify({ slug, name, location, generatedAt: profile.generatedAt, outDir, report: files.report }, null, 2));
145
180
 
146
181
  return { profile, notes, photos, outDir, files };
147
182
  }
@@ -35,11 +35,12 @@ const cleanUrl = (u) => {
35
35
  } catch { return u; }
36
36
  };
37
37
 
38
- export function buildProfile({ query, google, osm, site, hub, instagram, tripadvisor }) {
38
+ export function buildProfile({ query, google, osm, site, hub, instagram, tripadvisor, search }) {
39
39
  const g = google ?? {};
40
40
  const o = osm ?? {};
41
41
  const ld = site?.jsonld ?? null;
42
42
  const ta = tripadvisor?.jsonld ?? null;
43
+ const taSlug = tripadvisor?.slug ?? null;
43
44
  const warnings = [];
44
45
  const ok = (v) => (v ? v : undefined);
45
46
  const hubLinks = hub?.links ?? {};
@@ -54,7 +55,7 @@ export function buildProfile({ query, google, osm, site, hub, instagram, tripadv
54
55
  const igLinks = [...(site?.links?.instagram ?? []), ...(hubLinks.instagram ?? [])];
55
56
 
56
57
  const fields = {
57
- name: choose([{ value: g.name, source: "google" }, { value: ld?.name, source: "website" }, { value: ta?.name, source: "tripadvisor" }, { value: o.name, source: "osm" }, { value: site?.meta.siteName, source: "website" }], sameText),
58
+ name: choose([{ value: g.name, source: "google" }, { value: ld?.name, source: "website" }, { value: ta?.name, source: "tripadvisor" }, { value: o.name, source: "osm" }, { value: site?.meta.siteName, source: "website" }, { value: taSlug?.name, source: "tripadvisor (URL)", low: true }], sameText),
58
59
  street: choose([{ value: g.street, source: "google" }, { value: ld?.address.street, source: "website" }, { value: ta?.address.street, source: "tripadvisor" }, { value: o.street, source: "osm" }], sameText),
59
60
  locality: choose([{ value: g.locality, source: "google" }, { value: ld?.address.locality, source: "website" }, { value: o.locality, source: "osm" }], sameText),
60
61
  region: choose([{ value: g.region, source: "google" }, { value: ld?.address.region, source: "website" }, { value: o.region, source: "osm" }], sameText),
@@ -62,14 +63,14 @@ export function buildProfile({ query, google, osm, site, hub, instagram, tripadv
62
63
  coordinates: choose([{ value: g.coords, source: "google" }, { value: o.coords, source: "osm" }, { value: ld?.geo, source: "website" }], coordSame),
63
64
  phone: choose([{ value: g.phone, source: "google" }, { value: ok(links("tel")[0]), source: "website" }, { value: ld?.telephone, source: "website" }, { value: ta?.telephone, source: "tripadvisor" }, { value: instagram?.phones?.[0], source: "instagram" }, { value: o.phone, source: "osm" }], phoneSame),
64
65
  email: choose([{ value: ld?.email, source: "website" }, { value: links("mail").find((m) => !/sentry|wixpress|example|domain\./i.test(m)), source: "website" }, { value: o.email, source: "osm" }]),
65
- instagram: choose([{ value: query.instagram && handleOf(query.instagram), source: "you" }, ...igLinks.map((h) => ({ value: h, source: "website" })), ...(ld?.sameAs ?? []).filter((u) => /instagram\.com/.test(u)).map((u) => ({ value: handleOf(u), source: "website" })), { value: o.instagram && handleOf(o.instagram), source: "osm" }]),
66
- facebook: choose([{ value: links("facebook")[0], source: "website" }, { value: o.facebook, source: "osm" }]),
67
- tiktok: choose([{ value: links("tiktok")[0], source: "website" }]),
68
- tripadvisor: choose([{ value: query.tripadvisor, source: "you" }, { value: links("tripadvisor")[0], source: "website" }, { value: (ld?.sameAs ?? []).find((u) => /tripadvisor\./.test(u)), source: "website" }]),
69
- website: choose([{ value: query.website && cleanUrl(query.website), source: "you" }, { value: g.website && cleanUrl(g.website), source: "google" }, { value: o.website && cleanUrl(o.website), source: "osm" }]),
66
+ instagram: choose([{ value: query.instagram && handleOf(query.instagram), source: "you" }, ...igLinks.map((h) => ({ value: h, source: "website" })), ...(ld?.sameAs ?? []).filter((u) => /instagram\.com/.test(u)).map((u) => ({ value: handleOf(u), source: "website" })), { value: o.instagram && handleOf(o.instagram), source: "osm" }, { value: search?.instagram && handleOf(search.instagram), source: "search" }]),
67
+ facebook: choose([{ value: links("facebook")[0], source: "website" }, { value: o.facebook, source: "osm" }, { value: search?.facebook && cleanUrl(search.facebook), source: "search" }]),
68
+ tiktok: choose([{ value: links("tiktok")[0], source: "website" }, { value: search?.tiktok && cleanUrl(search.tiktok), source: "search" }]),
69
+ tripadvisor: choose([{ value: query.tripadvisor, source: "you" }, { value: links("tripadvisor")[0], source: "website" }, { value: (ld?.sameAs ?? []).find((u) => /tripadvisor\./.test(u)), source: "website" }, { value: search?.tripadvisor && cleanUrl(search.tripadvisor), source: "search" }]),
70
+ website: choose([{ value: query.website && cleanUrl(query.website), source: "you" }, { value: g.website && cleanUrl(g.website), source: "google" }, { value: o.website && cleanUrl(o.website), source: "osm" }, { value: search?.website && cleanUrl(search.website), source: "search" }]),
70
71
  reserveUrl: choose([...links("reserve").map((u) => ({ value: u, source: "website" }))]),
71
72
  priceRange: choose([{ value: g.priceRange, source: "google" }, { value: ld?.priceRange, source: "website" }, { value: ta?.priceRange, source: "tripadvisor" }]),
72
- mapsUrl: choose([{ value: g.mapsUrl, source: "google" }, { value: links("maps")[0], source: "website" }]),
73
+ mapsUrl: choose([{ value: g.mapsUrl, source: "google" }, { value: links("maps")[0], source: "website" }, { value: search?.mapsUrl && cleanUrl(search.mapsUrl), source: "search" }]),
73
74
  descriptor: choose([{ value: g.descriptor, source: "google" }]),
74
75
  };
75
76
 
@@ -110,10 +111,10 @@ export function buildProfile({ query, google, osm, site, hub, instagram, tripadv
110
111
  return {
111
112
  query, generatedAt: new Date().toISOString(), warnings, fields, hours,
112
113
  textHours: [...(site?.hoursText ?? [])],
113
- links: { menu: links("menu"), delivery: links("delivery"), reserve: links("reserve"), waze: links("waze"), hubs: [...new Set([...(site?.links?.hubs ?? []), ...(instagram?.links ?? []).filter((l) => /linktr|beacons|bio\.link|lnk\.bio|taplink/.test(l))])] },
114
+ links: { menu: links("menu"), delivery: links("delivery"), reserve: links("reserve"), waze: links("waze"), hubs: [...new Set([...(site?.links?.hubs ?? []), ...(search?.hubs ?? []), ...(instagram?.links ?? []).filter((l) => /linktr|beacons|bio\.link|lnk\.bio|taplink/.test(l))])] },
114
115
  ratings: [g.rating && { source: "google", value: g.rating, count: g.ratingCount }, ld?.rating && { source: "website", ...ld.rating }, ta?.rating && { source: "tripadvisor", ...ta.rating }].filter(Boolean),
115
116
  images: { logos: [...(site?.logos ?? []), ...(hub?.logos ?? [])].filter((l) => l.url), site: site?.images ?? [], hub: hub?.images ?? [], googlePhotos: g.photos?.length ?? 0, instagramImage: instagram?.image ?? "", themeColor: site?.meta.themeColor ?? "" },
116
117
  copySources: { googleSummary: g.summary ?? "", siteDescription: site?.meta.description ?? "", instagramBio: instagram?.bio ?? "", tripadvisor: tripadvisor?.description ?? "", headings: site?.headings ?? [], paragraphs: site?.paragraphs ?? [] },
117
- social: { instagram: instagram ? { handle: instagram.handle, followers: instagram.followers, posts: instagram.posts, blocked: instagram.blocked } : null },
118
+ social: { instagram: instagram ? { handle: instagram.handle, followers: instagram.followers, posts: instagram.posts, blocked: instagram.blocked, partial: instagram.partial, surface: instagram.surface } : null },
118
119
  };
119
120
  }
@@ -33,17 +33,54 @@ const ROWS = [
33
33
 
34
34
  const cell = (s) => String(s).replace(/\|/g, "\\|");
35
35
 
36
- export function renderReport(p, { notes, photos }) {
36
+ // Web text lands in a file agents are told to act on: collapse newlines and
37
+ // neutralise backticks so third-party text cannot forge headings or a code fence.
38
+ const flat = (s) => String(s ?? "").replace(/\s+/g, " ").replace(/`/g, "'").trim();
39
+
40
+ /**
41
+ * Splits the fields into what was discovered from a source, what was only echoed
42
+ * back from the user's own input, and low-confidence guesses, so a summary can
43
+ * say "N facts, M from your input" instead of counting all of them as found.
44
+ */
45
+ export function fieldSummary(p) {
46
+ const held = Object.entries(p.fields ?? {}).filter(([, f]) => {
47
+ const v = f?.value;
48
+ return v !== undefined && v !== null && v !== "" && !(Array.isArray(v) && !v.length);
49
+ });
50
+ const parts = (f) => String(f.source ?? "").split(" + ").map((s) => s.trim()).filter(Boolean);
51
+ return {
52
+ discovered: held.filter(([, f]) => f.confidence !== "low" && parts(f).some((s) => s !== "you")).map(([k]) => k),
53
+ echoed: held.filter(([, f]) => {
54
+ const sources = parts(f);
55
+ return f.confidence !== "low" && sources.length > 0 && sources.every((s) => s === "you");
56
+ }).map(([k]) => k),
57
+ guessed: held.filter(([, f]) => f.confidence === "low").map(([k]) => k),
58
+ };
59
+ }
60
+
61
+ /** The CLI summary: what came from a source, what was only echoed, what was a low guess. */
62
+ export function summaryLines(p) {
63
+ const { discovered, echoed, guessed } = fieldSummary(p);
64
+ const list = (a) => (a.length ? a.join(", ") : "none");
65
+ const lines = [`Found ${discovered.length} fact(s) from sources: ${list(discovered)}`];
66
+ if (echoed.length) lines.push(`Echoed ${echoed.length} from your input: ${list(echoed)}`);
67
+ if (guessed.length) lines.push(`Not counted (low-confidence guess): ${list(guessed)}`);
68
+ return lines;
69
+ }
70
+
71
+ export function renderReport(p, { notes, photos, kept = [] }) {
37
72
  const out = [];
38
73
  const q = p.query;
39
- out.push(`# Research: ${q.name}${q.location ? `, ${q.location}` : ""}`, "", `Generated ${p.generatedAt}. Everything here is **unconfirmed**: check it with the client and copy the answers into \`BRIEF.md\`. Sources disagree often, and the confidence column says how many agreed.`, "");
74
+ out.push(`# Research: ${flat(q.name)}${q.location ? `, ${flat(q.location)}` : ""}`, "", `Generated ${p.generatedAt}. Everything here is **unconfirmed**: check it with the client and copy the answers into \`BRIEF.md\`. Sources disagree often, and the confidence column says how many agreed.`, "");
75
+ for (const k of kept) out.push(`> **Kept for a manual read:** ${k.label} (${k.url}) — could not be read (${k.reason}). Nothing was extracted from it; open it yourself.`, "");
40
76
  if (p.warnings.length) out.push("## Warnings", "", ...p.warnings.map((w) => `- ${w}`), "");
41
77
  if (notes.length) out.push("## What each source did", "", ...notes.map((n) => `- ${n}`), "");
42
78
 
43
79
  out.push("## Facts", "", "| Item | Value | Source | Confidence | Goes in |", "| --- | --- | --- | --- | --- |");
44
80
  for (const [label, key, goes] of ROWS) {
45
81
  const f = p.fields[key];
46
- out.push(`| ${label} | ${f ? cell(val(f)) : "_not found_"} | ${f?.source ?? ""} | ${f?.confidence ?? ""} | ${goes} |`);
82
+ const confidence = f ? (f.confidence === "low" ? "low (guess)" : f.confidence) : "";
83
+ out.push(`| ${label} | ${f ? cell(val(f)) : "_not found_"} | ${f?.source ?? ""} | ${confidence} | ${goes} |`);
47
84
  }
48
85
  out.push("");
49
86
 
@@ -75,7 +112,12 @@ export function renderReport(p, { notes, photos }) {
75
112
  out.push("");
76
113
 
77
114
  if (p.ratings.length) out.push("## Ratings (context only)", "", ...p.ratings.map((r) => `- ${r.source}: ${r.value}${r.count ? ` (${r.count} reviews)` : ""}`), "");
78
- if (p.social.instagram && !p.social.instagram.blocked) out.push(`Instagram @${p.social.instagram.handle}: ${p.social.instagram.followers} followers, ${p.social.instagram.posts} posts.`, "");
115
+ const ig = p.social.instagram;
116
+ if (ig && !ig.blocked) {
117
+ const counts = ig.followers || ig.posts ? `${ig.followers || "?"} followers, ${ig.posts || "?"} posts.` : "";
118
+ const note = ig.partial ? "Bio from a search snippet (the profile itself was walled)." : counts ? "" : "Profile card read; followers and bio are not public.";
119
+ out.push(`Instagram @${ig.handle}: ${[counts, note].filter(Boolean).join(" ")}`.trim(), "");
120
+ }
79
121
 
80
122
  out.push("## Images", "");
81
123
  out.push(`- Website images found: ${p.images.site.length}; link-in-bio: ${p.images.hub.length}; Google photos available: ${p.images.googlePhotos}.`);
@@ -86,13 +128,21 @@ export function renderReport(p, { notes, photos }) {
86
128
 
87
129
  out.push("## Material for the copy", "", "Facts and tone only. Rewrite it in the template's voice in both languages; do not paste it.", "");
88
130
  const c = p.copySources;
89
- for (const [label, text] of [["Google summary", c.googleSummary], ["Site description", c.siteDescription], ["Instagram bio", c.instagramBio], ["TripAdvisor", c.tripadvisor]]) if (text) out.push(`- **${label}:** ${text}`);
90
- if (c.headings.length) out.push(`- **Site headings:** ${c.headings.slice(0, 12).join(" / ")}`);
91
- for (const t of c.paragraphs.slice(0, 6)) out.push(` - ${t}`);
131
+ for (const [label, text] of [["Google summary", c.googleSummary], ["Site description", c.siteDescription], ["Instagram bio", c.instagramBio], ["TripAdvisor", c.tripadvisor]]) if (text) out.push(`- **${label}:** ${flat(text)}`);
132
+ if (c.headings.length) out.push(`- **Site headings:** ${c.headings.slice(0, 12).map(flat).join(" / ")}`);
133
+ for (const t of c.paragraphs.slice(0, 6)) out.push(` - ${flat(t)}`);
92
134
  out.push("");
93
135
 
94
136
  const missing = ROWS.filter(([, key]) => !p.fields[key] && ["name", "street", "coordinates", "whatsapp", "instagram", "reserveUrl", "cuisines"].includes(key)).map(([l]) => l);
95
- out.push("## Ask the client", "", ...[...missing, "Opening hours incl. holidays", "Logo as SVG and original photos", "The story, signature dishes and events", "Production domain"].filter((v, i, a) => a.indexOf(v) === i && (v !== "Opening hours incl. holidays" || true)).map((m) => `- ${m}`), "");
137
+ const ask = !p.fields.website && !p.fields.mapsUrl
138
+ ? "Send the restaurant's website or a Google Maps link — that unlocks the address, map point and place ID."
139
+ : !p.fields.coordinates
140
+ ? "A Google Maps link would pin the exact location."
141
+ : !p.fields.whatsapp
142
+ ? "Ask for the WhatsApp number taken with reservations, and confirm it accepts messages."
143
+ : "Ask for the current website and social profiles to cross-check the rest.";
144
+ const asks = ["Opening hours incl. holidays", "Logo as SVG and original photos", "The story, signature dishes and events", "Production domain"];
145
+ out.push("## Ask the client", "", `- **Best next question:** ${ask}`, ...[...new Set([...missing, ...asks])].map((m) => `- ${m}`), "");
96
146
  out.push("## Next", "", "```bash", `npm run setup < .tablefacts/research/${q.slug}/setup-answers.txt`, "git diff # review before keeping it", "```", "", "Blank lines in that file keep the template's current value, so anything not found stays a placeholder.", "");
97
147
  return out.join("\n");
98
148
  }
@@ -100,19 +150,22 @@ export function renderReport(p, { notes, photos }) {
100
150
  /** One line per prompt of `data/scripts/setup.mjs`, in its order. Blank keeps the current value. */
101
151
  export function setupAnswers(p) {
102
152
  const f = p.fields;
103
- const ig = f.instagram?.value;
153
+ // A low-confidence guess stays blank so setup keeps the template's placeholder,
154
+ // instead of a guess being written in as though it were a discovered fact.
155
+ const sure = (field) => (field && field.confidence !== "low" ? field : null);
156
+ const ig = sure(f.instagram)?.value;
104
157
  return [
105
- val(f.name),
158
+ val(sure(f.name)),
106
159
  "", // production URL: the new domain, not the current website
107
- f.reserveUrl?.value ?? "",
108
- f.whatsapp?.display ?? "",
160
+ sure(f.reserveUrl)?.value ?? "",
161
+ sure(f.whatsapp)?.display ?? "",
109
162
  ig ? `@${ig}` : "",
110
- val(f.street),
111
- val(f.locality),
112
- val(f.region),
113
- val(f.country),
114
- f.coordinates?.value?.lat ?? "",
115
- f.coordinates?.value?.lng ?? "",
116
- val(f.cuisines),
163
+ val(sure(f.street)),
164
+ val(sure(f.locality)),
165
+ val(sure(f.region)),
166
+ val(sure(f.country)),
167
+ sure(f.coordinates)?.value?.lat ?? "",
168
+ sure(f.coordinates)?.value?.lng ?? "",
169
+ val(sure(f.cuisines)),
117
170
  ].join("\n") + "\n";
118
171
  }
@@ -0,0 +1,127 @@
1
+ // Key-free web search, used when Google Places and OpenStreetMap find no place
2
+ // (or no website): a bare name + place otherwise yields an empty report. It only
3
+ // surfaces candidate links (website, Instagram, TripAdvisor, Maps, hub); the rest
4
+ // of the pipeline follows them. DuckDuckGo's Lite endpoint is tried first (the
5
+ // same results in simpler HTML, no JavaScript), then the html endpoint. No key
6
+ // and no login. A bot wall makes the source fail like any other.
7
+ import { fetchText, fold, sameText, tokens, decodeHtml, instagramHandle } from "./util.mjs";
8
+
9
+ const DDG = [
10
+ "https://lite.duckduckgo.com/lite/",
11
+ "https://html.duckduckgo.com/html/",
12
+ ];
13
+
14
+ // Hosts that are directories or social pages, never the restaurant's own site.
15
+ const NOT_WEBSITE =
16
+ /(duckduckgo|bing|google|youtube|facebook|instagram|tiktok|twitter|(^|\.)x\.com$|tripadvisor|yelp|wikipedia|foursquare|mapquest|waze|trip\.com|expedia|booking|airbnb|justdial|zomato|rappi|ubereats|pedidosya|doordash|grubhub|glovo|opentable|thefork|eltenedor|exploretock|sevenrooms|covermanager|quandoo|linktr|beacons|linkin\.bio|lnk\.bio|taplink|wa\.me|whatsapp)/i;
17
+ const INSTAGRAM = /(^|\.)instagram\.com$/i;
18
+ const TRIPADVISOR = /tripadvisor\./i;
19
+ const FACEBOOK = /(^|\.)facebook\.com$|(^|\.)fb\.com$/i;
20
+ const TIKTOK = /(^|\.)tiktok\.com$/i;
21
+ const MAPS = /maps\.app\.goo\.gl|(^|\.)google\.[a-z.]+\/maps|maps\.google\.com|goo\.gl\/maps/i;
22
+ const HUBS = /(linktr\.ee|beacons\.ai|bio\.link|lnk\.bio|linktree\.com|taplink|campsite\.bio|solo\.to|linkin\.bio|flow\.page)/i;
23
+
24
+ const textOf = (s) => decodeHtml(String(s).replace(/<[^>]+>/g, " ")).replace(/\s+/g, " ").trim();
25
+
26
+ /** Unwraps DuckDuckGo's `//duckduckgo.com/l/?uddg=<url>` redirect to the real result URL. */
27
+ export function resultUrl(href) {
28
+ if (!href) return null;
29
+ let u;
30
+ try { u = new URL(href, "https://html.duckduckgo.com"); } catch { return null; }
31
+ if (/(^|\.)duckduckgo\.com$/i.test(u.hostname)) {
32
+ const target = u.searchParams.get("uddg");
33
+ if (!target) return null;
34
+ try { return new URL(target).href; } catch { return null; }
35
+ }
36
+ return u.protocol === "http:" || u.protocol === "https:" ? u.href : null;
37
+ }
38
+
39
+ // One pass over the page, matching a result link or a snippet cell in order, so a
40
+ // result with no snippet cannot borrow the next result's text.
41
+ const RESULT_TOKEN = /<a\b([^>]*\b(?:result__a|result-link)\b[^>]*)>([\s\S]*?)<\/a>|<([a-z]+)\b[^>]*\b(?:result__snippet|result-snippet)\b[^>]*>([\s\S]*?)<\/\3>/gi;
42
+
43
+ /** Pulls { url, title, snippet } out of DuckDuckGo's HTML result page (no DOM parser). */
44
+ export function parseSearchResults(html) {
45
+ const out = [];
46
+ const seen = new Set();
47
+ let current = -1;
48
+ for (const m of html.matchAll(RESULT_TOKEN)) {
49
+ if (m[1] !== undefined) {
50
+ const href = m[1].match(/href=["']([^"']+)["']/i)?.[1] ?? "";
51
+ const url = resultUrl(href);
52
+ if (!url || seen.has(url)) { current = -1; continue; }
53
+ seen.add(url);
54
+ out.push({ url, title: textOf(m[2]), snippet: "" });
55
+ current = out.length - 1;
56
+ } else if (current >= 0 && !out[current].snippet) {
57
+ out[current].snippet = textOf(m[4]);
58
+ }
59
+ }
60
+ return out;
61
+ }
62
+
63
+ const hostOf = (url) => { try { return new URL(url).hostname.replace(/^www\./, ""); } catch { return ""; } };
64
+
65
+ /**
66
+ * Sorts results into the links the pipeline can follow. `website` is the best
67
+ * non-directory result whose title/host looks like the restaurant's name; a
68
+ * result that only matches the place is kept as a weaker candidate.
69
+ */
70
+ export function classifyResults(results, { name = "" } = {}) {
71
+ const links = { website: [], instagram: [], tripadvisor: [], facebook: [], tiktok: [], maps: [], hubs: [] };
72
+ const add = (k, v) => { if (v && !links[k].includes(v)) links[k].push(v); };
73
+ const candidates = [];
74
+ for (const { url, title, snippet } of results) {
75
+ const host = hostOf(url);
76
+ if (INSTAGRAM.test(host)) {
77
+ const handle = instagramHandle(url);
78
+ if (handle) add("instagram", handle);
79
+ } else if (TRIPADVISOR.test(host)) add("tripadvisor", url);
80
+ else if (FACEBOOK.test(host)) add("facebook", url);
81
+ else if (TIKTOK.test(host)) add("tiktok", url);
82
+ else if (MAPS.test(url)) add("maps", url);
83
+ else if (HUBS.test(host)) add("hubs", url);
84
+ else if (NOT_WEBSITE.test(host)) continue;
85
+ else {
86
+ const strong = sameText(name, title) || tokens(name).some((t) => fold(host).includes(t));
87
+ const score = tokens(name).filter((t) => fold(`${title} ${host} ${snippet}`).includes(t)).length;
88
+ candidates.push({ url, title, snippet, score, strong });
89
+ }
90
+ }
91
+ candidates.sort((a, b) => Number(b.strong) - Number(a.strong) || b.score - a.score);
92
+ for (const c of candidates) add("website", c.url);
93
+ return {
94
+ website: candidates.find((c) => c.strong)?.url ?? candidates[0]?.url ?? "",
95
+ instagram: links.instagram[0] ?? "",
96
+ tripadvisor: links.tripadvisor[0] ?? "",
97
+ facebook: links.facebook[0] ?? "",
98
+ tiktok: links.tiktok[0] ?? "",
99
+ mapsUrl: links.maps[0] ?? "",
100
+ hubs: links.hubs,
101
+ candidates: candidates.slice(0, 5).map(({ url, title }) => ({ url, title })),
102
+ };
103
+ }
104
+
105
+ /**
106
+ * Runs the search for a restaurant and returns its candidate links. Tries every
107
+ * endpoint until one returns results; if all answer but none has results, the
108
+ * empty result is returned so the caller still says "no results" rather than
109
+ * "failed". Throws only when no endpoint answers.
110
+ */
111
+ export async function searchWeb({ name, location = "" }) {
112
+ const query = [name, location, "restaurant"].filter(Boolean).join(" ");
113
+ const qs = new URLSearchParams({ q: query }).toString();
114
+ let lastError, empty;
115
+ for (const base of DDG) {
116
+ let r;
117
+ try { r = await fetchText(`${base}?${qs}`, { headers: { accept: "text/html" } }); }
118
+ catch (e) { lastError = e; continue; }
119
+ if (!r.ok) { lastError = new Error(`DuckDuckGo ${r.status}`); continue; }
120
+ const results = parseSearchResults(r.text);
121
+ const found = { source: "search", url: r.url, query, results, ...classifyResults(results, { name }) };
122
+ if (results.length) return found;
123
+ empty = found;
124
+ }
125
+ if (empty) return empty;
126
+ throw lastError ?? new Error("DuckDuckGo search failed");
127
+ }