tablefacts 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +2 -0
- package/CHANGELOG.md +39 -0
- package/README.md +37 -9
- package/package.json +2 -2
- package/src/lib/types.mjs +7 -2
- package/src/menu/README.md +37 -11
- package/src/menu/cluvi/config.mjs +6 -0
- package/src/menu/cluvi/import.mjs +6 -2
- package/src/menu/lib/db.mjs +81 -17
- package/src/menu/lib/import.mjs +20 -3
- package/src/menu/lib/run.mjs +15 -0
- package/src/menu/lib/tables.mjs +44 -0
- package/src/menu/raw/config.mjs +6 -0
- package/src/menu/raw/import.mjs +5 -2
- package/src/research/README.md +23 -13
- package/src/research/index.mjs +52 -17
- package/src/research/lib/merge.mjs +11 -10
- package/src/research/lib/report.mjs +72 -19
- package/src/research/lib/search.mjs +127 -0
- package/src/research/lib/social.mjs +137 -26
- package/src/research/lib/util.mjs +22 -0
- package/src/research/lib/website.mjs +3 -14
- package/src/research/research.mjs +12 -8
- package/types/lib/types.d.mts +29 -3
- package/types/menu/cluvi/config.d.mts +1 -0
- package/types/menu/cluvi/import.d.mts +2 -0
- package/types/menu/lib/db.d.mts +31 -3
- package/types/menu/lib/import.d.mts +1 -1
- package/types/menu/lib/run.d.mts +3 -0
- package/types/menu/lib/tables.d.mts +18 -0
- package/types/menu/raw/config.d.mts +1 -0
- package/types/menu/raw/import.d.mts +3 -0
- package/types/research/lib/merge.d.mts +4 -1
- package/types/research/lib/report.d.mts +14 -1
- package/types/research/lib/search.d.mts +58 -0
- package/types/research/lib/social.d.mts +32 -31
- package/types/research/lib/util.d.mts +5 -0
- package/types/research/lib/website.d.mts +20 -20
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
// The table names a menu import writes. Several restaurants can share one
|
|
2
|
+
// Supabase database, so each restaurant's tables carry its own prefix
|
|
3
|
+
// (`cannario_menu_categories`, `makibar_menu_categories`, ...). The prefix is
|
|
4
|
+
// restaurant-specific and lives in the source's `config.mjs`; the CLI's
|
|
5
|
+
// `--table-prefix` overrides it for one run.
|
|
6
|
+
import { optionError } from "../../lib/errors.mjs";
|
|
7
|
+
|
|
8
|
+
// The only prefixes accepted: empty, or lowercase letters, digits and
|
|
9
|
+
// underscores ending in "_". A value that passed this allow-list is safe to
|
|
10
|
+
// build the table names from; nothing else is ever put into the SQL text.
|
|
11
|
+
const PREFIX = /^[a-z][a-z0-9_]*_$|^$/;
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* `tablePrefix` when it is empty or a safe `<name>_` prefix (an ECONFIG error otherwise).
|
|
15
|
+
* @param {string} [tablePrefix]
|
|
16
|
+
* @returns {string}
|
|
17
|
+
*/
|
|
18
|
+
export function validateTablePrefix(tablePrefix) {
|
|
19
|
+
const value = tablePrefix ?? "";
|
|
20
|
+
if (!PREFIX.test(value)) {
|
|
21
|
+
throw optionError(
|
|
22
|
+
"tablePrefix",
|
|
23
|
+
`${JSON.stringify(value)} is not a valid table prefix: \`tablePrefix\` must be empty or lowercase letters, digits and underscores ending in "_", such as "makibar_".`,
|
|
24
|
+
"ECONFIG",
|
|
25
|
+
);
|
|
26
|
+
}
|
|
27
|
+
return value;
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* The three `public` menu tables for a prefix, e.g.
|
|
32
|
+
* `{ categories: "public.makibar_menu_categories", sections: "public.makibar_menu_sections", products: "public.makibar_menu_products" }`.
|
|
33
|
+
* The default (empty prefix) is the unprefixed `public.menu_*` set.
|
|
34
|
+
* @param {string} [tablePrefix]
|
|
35
|
+
* @returns {{ categories: string, sections: string, products: string }}
|
|
36
|
+
*/
|
|
37
|
+
export function menuTables(tablePrefix) {
|
|
38
|
+
const prefix = validateTablePrefix(tablePrefix);
|
|
39
|
+
return {
|
|
40
|
+
categories: `public.${prefix}menu_categories`,
|
|
41
|
+
sections: `public.${prefix}menu_sections`,
|
|
42
|
+
products: `public.${prefix}menu_products`,
|
|
43
|
+
};
|
|
44
|
+
}
|
package/src/menu/raw/config.mjs
CHANGED
|
@@ -1,6 +1,12 @@
|
|
|
1
1
|
// Everything restaurant-specific about the image-menu import. Edit this file,
|
|
2
2
|
// not the other scripts, when the menu is organised differently.
|
|
3
3
|
export default {
|
|
4
|
+
// This restaurant's tables in a shared Supabase database: the import writes
|
|
5
|
+
// public.mombasa_menu_categories / _menu_sections / _menu_products. Change
|
|
6
|
+
// it (or pass `--table-prefix`) when importing another restaurant; leave it
|
|
7
|
+
// empty only when this restaurant owns the unprefixed menu_* tables.
|
|
8
|
+
tablePrefix: "mombasa_",
|
|
9
|
+
|
|
4
10
|
// The page that shows the menu pictures, or direct image URLs.
|
|
5
11
|
// `tablefacts menu raw <url> [<url>...]` overrides it for one run.
|
|
6
12
|
url: "https://www.mombasa.co/carta-restaurante-espanol/",
|
package/src/menu/raw/import.mjs
CHANGED
|
@@ -85,6 +85,8 @@ export async function fetchImageMenu({ urls = [], only, provider, model, minWidt
|
|
|
85
85
|
menu,
|
|
86
86
|
notes,
|
|
87
87
|
title: `Image menu: ${chosen.length} pages from ${host} (${reading} read with ${providers[provider].label} ${model}, ${chosen.length - reading} from the saved transcriptions), prices in ${currency}`,
|
|
88
|
+
// The restaurant's own table set, so a shared database is never touched by accident.
|
|
89
|
+
tablePrefix: config.tablePrefix,
|
|
88
90
|
};
|
|
89
91
|
}
|
|
90
92
|
|
|
@@ -93,7 +95,8 @@ export async function fetchImageMenu({ urls = [], only, provider, model, minWidt
|
|
|
93
95
|
* @param {import('../../lib/types.mjs').ImportImageMenuOptions} [options]
|
|
94
96
|
* @returns {Promise<import('../../lib/types.mjs').ImportResult>}
|
|
95
97
|
*/
|
|
96
|
-
export async function importImageMenu({ urls, only, provider, model, minWidth, refresh, apiKey, env, config, ...importOptions } = {}) {
|
|
98
|
+
export async function importImageMenu({ urls, only, provider, model, minWidth, refresh, apiKey, env, config = defaultConfig, ...importOptions } = {}) {
|
|
97
99
|
const fetched = await fetchImageMenu({ urls, only, provider, model, minWidth, refresh, apiKey, env, config, projectDir: importOptions.projectDir, log: importOptions.log });
|
|
98
|
-
|
|
100
|
+
// An explicit `tablePrefix` (or the CLI's --table-prefix) wins over the config's.
|
|
101
|
+
return importMenu({ ...fetched, ...importOptions, tablePrefix: importOptions.tablePrefix ?? config.tablePrefix, env });
|
|
99
102
|
}
|
package/src/research/README.md
CHANGED
|
@@ -28,42 +28,52 @@ see the [main README](../../README.md#use-it-as-a-library). `out` (relative path
|
|
|
28
28
|
| --- | --- | --- |
|
|
29
29
|
| Google Maps (Places API) | address, map point, phone, hours, price, cuisine types, photos, status | Needs `GOOGLE_PLACES_API_KEY` in `.env` (Places API (New) enabled). Phone and hours fields are billed at the Enterprise rate; one search per run |
|
|
30
30
|
| OpenStreetMap (Nominatim) | address, map point, sometimes phone, hours, website, socials | No key. Often stale or sparse |
|
|
31
|
+
| Web search (DuckDuckGo) | candidate website, Instagram, TripAdvisor, Maps and hub links | No key. Runs only when Google and OpenStreetMap find no place or no website, so the reported "name-only finds nothing" case now has a fallback |
|
|
31
32
|
| Website | JSON-LD, contact page, WhatsApp, reserve platform, menu links, images, logo, theme colour, page text | Falls back to Playwright on 403 or client-rendered pages (`--render` forces it) |
|
|
32
|
-
| Instagram | handle, followers, bio, bio link |
|
|
33
|
-
| TripAdvisor | JSON-LD (address, phone, cuisine, price, rating) | Usually bot-blocked
|
|
33
|
+
| Instagram | handle, followers, bio, bio link | Tries the public share card, the web profile API, oEmbed and finally a search-engine snippet of the bio. Requests identify honestly and never log in; often all surfaces are walled |
|
|
34
|
+
| TripAdvisor | JSON-LD (address, phone, cuisine, price, rating) | Usually bot-blocked. Playwright is tried on a 403 or a thin page like any website, but DataDome still wins. The name and city are read from the URL slug without fetching, and the link is kept for a manual read (stated at the top of the report) |
|
|
34
35
|
| Linktree and similar | WhatsApp, reserve, menu, delivery links | Found from the website, Instagram bio or `--linktree` |
|
|
35
36
|
|
|
36
|
-
Sources find each other (website → Instagram, TripAdvisor, hub). Order of trust per field is in `lib/merge.mjs`; two sources agreeing is "high", one is "medium", a guess is "low".
|
|
37
|
+
Sources find each other (web search or website → Instagram, TripAdvisor, hub). Order of trust per field is in `lib/merge.mjs`; two sources agreeing is "high", one is "medium", a guess is "low".
|
|
37
38
|
|
|
38
39
|
## How it runs
|
|
39
40
|
|
|
40
41
|
- **Concurrent lookups.** Google and OpenStreetMap run together; then the website is read. Instagram, TripAdvisor and
|
|
41
|
-
the link-in-bio page start as soon as their address is known (given, or found on the website), so a slow
|
|
42
|
-
not hold up the others. Report notes keep a fixed order whatever finishes first. Photo downloads run four at a time.
|
|
42
|
+
the link-in-bio page start as soon as their address is known (given, or found on the website or web search), so a slow
|
|
43
|
+
source does not hold up the others. Report notes keep a fixed order whatever finishes first. Photo downloads run four at a time.
|
|
44
|
+
- **Web-search fallback.** When Google and OpenStreetMap find no place or no website, a key-free DuckDuckGo query
|
|
45
|
+
runs and its candidate links feed the rest of the pipeline. It is one extra request, only when it is needed.
|
|
43
46
|
- **One shared browser.** Playwright's Chromium is launched at most once per run, only when a page needs rendering
|
|
44
47
|
(`--render`, a 403, a client-rendered page), is shared by the website, TripAdvisor and link-in-bio readers, and is
|
|
45
48
|
closed at the end even on failure.
|
|
46
49
|
- **Fetch cap.** Pages are read up to 2 MB each; larger bodies are truncated, so a huge page cannot exhaust memory.
|
|
47
50
|
|
|
51
|
+
## Runs and re-runs
|
|
52
|
+
|
|
53
|
+
Output goes to `.tablefacts/research/<slug>/`; two spellings of the same name ("Makibar" and "Maki Bar") make two
|
|
54
|
+
folders. A similar-name run is noted in the report, and `.tablefacts/research/latest.json` always points at the newest
|
|
55
|
+
report, so reading every folder cannot resurface a stale one.
|
|
56
|
+
|
|
48
57
|
## Limits
|
|
49
58
|
|
|
50
59
|
- It does not log in anywhere or get past bot protection. A blocked source is reported, not worked around.
|
|
51
60
|
- It finds facts, not copy, logos or licensed photos.
|
|
52
|
-
- A wrong match (a sister branch, a closed place) is the main risk: read the Warnings and "Sources disagree" first.
|
|
61
|
+
- A wrong match (a sister branch, a closed place) is the main risk: read the Warnings and "Sources disagree" first. When Google returns several places with the same name, the report says so and asks which branch.
|
|
62
|
+
- Only public surfaces are tried for Instagram/TripAdvisor. When they are walled, the report keeps the link and says so at the top; it does not defeat the wall.
|
|
53
63
|
|
|
54
64
|
## For agents
|
|
55
65
|
|
|
56
|
-
Use it first, whenever you are given only a restaurant's name and place (or little else).
|
|
66
|
+
Use it first, whenever you are given only a restaurant's name and place (or little else). If Google and OpenStreetMap find no place or no website, a key-free web search runs automatically, so a name-only call is no longer an empty report.
|
|
57
67
|
|
|
58
|
-
1. **Run it** from the repo root: `tablefacts research "<name>" "<city, country>" --country <ISO>`. Add `--website`, `--instagram`, `--tripadvisor` or `--linktree` for anything the user gave you; a user-supplied value is trusted over a search result. It takes under a minute, needs no prompts and prints the output folder.
|
|
59
|
-
2. **Read `out/<slug>/report.md`**, in this order: Warnings, "What each source did", Facts, "Sources disagree". Confirm the match is the right restaurant (name, address, not closed) before using anything. If it matched the wrong place, rerun with a more specific location or `--website`.
|
|
68
|
+
1. **Run it** from the repo root: `tablefacts research "<name>" "<city, country>" --country <ISO>`. Add `--website`, `--instagram`, `--tripadvisor` or `--linktree` for anything the user gave you; a user-supplied value is trusted over a search result. Uploading a Google Maps link as `--website` (or pasting it in the report as `mapsUrl`) unlocks coordinates; ask for it first when the report is thin. It takes under a minute, needs no prompts and prints the output folder.
|
|
69
|
+
2. **Read `out/<slug>/report.md`**, in this order: the kept-links notice and Warnings, "What each source did", Facts, "Sources disagree", "Ask the client". Confirm the match is the right restaurant (name, address, not closed) before using anything. If it matched the wrong place, rerun with a more specific location or `--website`.
|
|
60
70
|
3. **Treat every value as unconfirmed.** Fill `BRIEF.md` with the value and its source, never as client-confirmed. Anything under "not found" stays a placeholder and goes in your hand-off as an open question. Do not invent it.
|
|
61
|
-
4. **Apply the facts**: review `setup-answers.txt`, run `npm run setup < .tablefacts/research/<slug>/setup-answers.txt`, read `git diff`. Setup does not cover `phone`, `email`, `MENU_LOCALE`, `brand.tagline`, `visit.hoursRows` or `jsonld.ts`: take those from the report by hand. Hours are ready to paste. Add street, phone and hours to `jsonld.ts` only for values the client confirmed.
|
|
71
|
+
4. **Apply the facts**: review `setup-answers.txt`, run `npm run setup < .tablefacts/research/<slug>/setup-answers.txt`, read `git diff`. A low-confidence guess (marked "low (guess)") is left blank so setup keeps the template's placeholder; it is never written in as a fact. Setup does not cover `phone`, `email`, `MENU_LOCALE`, `brand.tagline`, `visit.hoursRows` or `jsonld.ts`: take those from the report by hand. Hours are ready to paste. Add street, phone and hours to `jsonld.ts` only for values the client confirmed.
|
|
62
72
|
5. **Use the leads**: a Cluvi menu link means `tablefacts menu cluvi`, a PDF means `tablefacts menu raw` (`src/menu/README.md`). The theme colour and logo candidates are hints for `tokens.css` and `public/logo.svg`, not decisions.
|
|
63
|
-
6. **Blocked sources** (Instagram, TripAdvisor) are normal.
|
|
73
|
+
6. **Blocked sources** (Instagram, TripAdvisor) are normal. The top of the report names any link that was kept unread; open it yourself, or ask the user for the bio and hours. Do not try to get around a login wall. `.tablefacts/research/latest.json` points at the newest run when the same place was researched under a different spelling.
|
|
64
74
|
|
|
65
|
-
`profile.json` has the same data as JSON for scripting: `fields.<name>` is `{ value, source, confidence, alternatives? }`, `hours.rows.{es,en}` are `hoursRows`, and `raw` holds each source's untouched result. Field names: `name street locality region country coordinates phone whatsapp email instagram facebook tiktok tripadvisor website reserveUrl priceRange mapsUrl descriptor cuisines menuLocale`.
|
|
75
|
+
`profile.json` has the same data as JSON for scripting: `fields.<name>` is `{ value, source, confidence, alternatives? }`, `hours.rows.{es,en}` are `hoursRows`, and `raw` holds each source's untouched result (including `search`). Field names: `name street locality region country coordinates phone whatsapp email instagram facebook tiktok tripadvisor website reserveUrl priceRange mapsUrl descriptor cuisines menuLocale`.
|
|
66
76
|
|
|
67
77
|
Do not commit `out/`, and never paste the Google key anywhere but `.env`. Photos it downloads belong to the restaurant or the photographer: use them as reference, not as site assets.
|
|
68
78
|
|
|
69
|
-
Code map: `research.mjs` runs the sources in order and writes the files; `lib/google.mjs`, `osm.mjs`, `website.mjs`, `social.mjs` read one source each; `merge.mjs` holds the per-field trust order; `hours.mjs` normalises hours; `report.mjs` writes the report and the setup answers. A new source is a reader returning plain data, a line in `research.mjs` and its candidates in `merge.mjs`.
|
|
79
|
+
Code map: `research.mjs` runs the sources in order and writes the files; `lib/google.mjs`, `osm.mjs`, `search.mjs`, `website.mjs`, `social.mjs` read one source each; `merge.mjs` holds the per-field trust order; `hours.mjs` normalises hours; `report.mjs` writes the report and the setup answers. A new source is a reader returning plain data, a line in `research.mjs` and its candidates in `merge.mjs`.
|
package/src/research/index.mjs
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
// Programmatic entry for the research pipeline: gathers what the public web says
|
|
2
2
|
// about a restaurant and writes profile.json, report.md and setup-answers.txt.
|
|
3
3
|
// Silent by default and never exits the process; the CLI lives in research.mjs.
|
|
4
|
-
import { mkdir, writeFile } from "node:fs/promises";
|
|
5
|
-
import { join, resolve } from "node:path";
|
|
4
|
+
import { mkdir, readdir, writeFile } from "node:fs/promises";
|
|
5
|
+
import { dirname, join, resolve } from "node:path";
|
|
6
6
|
import { resolveEnv } from "../lib/env.mjs";
|
|
7
7
|
import { optionError } from "../lib/errors.mjs";
|
|
8
8
|
import { normalizeLog } from "../lib/log.mjs";
|
|
@@ -11,8 +11,9 @@ import { downloadPhotos, searchGoogle } from "./lib/google.mjs";
|
|
|
11
11
|
import { buildProfile } from "./lib/merge.mjs";
|
|
12
12
|
import { searchOsm } from "./lib/osm.mjs";
|
|
13
13
|
import { renderReport, setupAnswers } from "./lib/report.mjs";
|
|
14
|
+
import { searchWeb } from "./lib/search.mjs";
|
|
14
15
|
import { readHub, readInstagram, readTripadvisor } from "./lib/social.mjs";
|
|
15
|
-
import { mapPool, slugify } from "./lib/util.mjs";
|
|
16
|
+
import { mapPool, sameText, slugify } from "./lib/util.mjs";
|
|
16
17
|
import { createBrowser, scrapeSite } from "./lib/website.mjs";
|
|
17
18
|
|
|
18
19
|
/**
|
|
@@ -31,6 +32,7 @@ export async function research(options = {}) {
|
|
|
31
32
|
|
|
32
33
|
const slug = slugify(name);
|
|
33
34
|
const outDir = options.out ? resolveIn(options.projectDir, options.out) : workDirIn(options.projectDir, "research", slug);
|
|
35
|
+
const researchDir = dirname(outDir);
|
|
34
36
|
const query = { name, location, slug, country, website, instagram: instagramOpt, tripadvisor: tripadvisorOpt, linktree };
|
|
35
37
|
const notes = [];
|
|
36
38
|
|
|
@@ -50,7 +52,7 @@ export async function research(options = {}) {
|
|
|
50
52
|
|
|
51
53
|
// One Chromium for the whole run, launched only if a page needs rendering.
|
|
52
54
|
const browser = createBrowser();
|
|
53
|
-
let site, instagram, hub, tripadvisor, taUrl, handle, google, osm;
|
|
55
|
+
let site, instagram, hub, tripadvisor, taUrl, handle, google, osm, search, branches = [];
|
|
54
56
|
try {
|
|
55
57
|
// Google and OpenStreetMap only need the name and place, so they run together; their notes are added in this order afterwards.
|
|
56
58
|
const googleNotes = [];
|
|
@@ -65,20 +67,33 @@ export async function research(options = {}) {
|
|
|
65
67
|
notes.push(...googleNotes, ...osmNotes);
|
|
66
68
|
google = googleResult?.place ?? null;
|
|
67
69
|
osm = osmResult?.place ?? null;
|
|
70
|
+
// Several Google places with the same name are likely branches of one chain; the template assumes one address.
|
|
71
|
+
branches = (googleResult?.candidates ?? []).filter((c) => sameText(c.name, name));
|
|
68
72
|
|
|
69
|
-
|
|
73
|
+
// No website yet: a key-free web search turns a bare name + place into candidate links.
|
|
74
|
+
if (!website && !google?.website && !osm?.website) {
|
|
75
|
+
search = await source("Web search", () => searchWeb({ name, location }));
|
|
76
|
+
if (search) {
|
|
77
|
+
const found = search.candidates.length ? ` (${search.candidates.slice(0, 2).map((c) => c.title || c.url).join(", ")})` : "";
|
|
78
|
+
notes.push(`Web search: ${search.results.length ? `${search.results.length} result(s) for "${search.query}"${found}` : `no results for "${search.query}"`}`);
|
|
79
|
+
}
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
const websiteUrl = website ?? google?.website ?? osm?.website ?? search?.website;
|
|
70
83
|
site = websiteUrl ? await source("Website", () => scrapeSite(websiteUrl, { renderJs: render, browser })) : null;
|
|
71
|
-
if (site) notes.push(`Website: read ${site.pages.length} page(s) of ${site.url}${site.pages.some((p) => p.rendered) ? " (rendered with Playwright)" : ""}`);
|
|
84
|
+
if (site) notes.push(`Website: read ${site.pages.length} page(s) of ${site.url}${site.pages.some((p) => p.rendered) ? " (rendered with Playwright)" : ""}${!website && !google?.website && !osm?.website ? " (found by web search)" : ""}`);
|
|
72
85
|
else if (!websiteUrl) notes.push("Website: none found. Pass the website (--website / `website`) if the restaurant has one.");
|
|
73
86
|
|
|
74
87
|
// Follow the leads the first sources gave. What is already known (handle, TripAdvisor link, hub link)
|
|
75
88
|
// starts right away; the rest waits only for the source that can supply it.
|
|
76
|
-
handle = (instagramOpt ?? site?.links.instagram[0] ?? osm?.instagram ?? "").replace(/^https?:\/\/(www\.)?instagram\.com\//i, "").replace(/^@/, "").split(/[/?#]/)[0];
|
|
77
|
-
taUrl = tripadvisorOpt ?? site?.links.tripadvisor[0] ?? (site?.jsonld?.sameAs ?? []).find((u) => /tripadvisor\./.test(u));
|
|
78
|
-
const directHubUrl = linktree ?? site?.links.hubs[0];
|
|
89
|
+
handle = (instagramOpt ?? site?.links.instagram[0] ?? osm?.instagram ?? search?.instagram ?? "").replace(/^https?:\/\/(www\.)?instagram\.com\//i, "").replace(/^@/, "").split(/[/?#]/)[0];
|
|
90
|
+
taUrl = tripadvisorOpt ?? site?.links.tripadvisor[0] ?? (site?.jsonld?.sameAs ?? []).find((u) => /tripadvisor\./.test(u)) ?? search?.tripadvisor;
|
|
91
|
+
const directHubUrl = linktree ?? site?.links.hubs[0] ?? search?.hubs?.[0];
|
|
79
92
|
const readHubAt = (url) => source("Link-in-bio", () => readHub(url, { browser }));
|
|
80
93
|
|
|
81
|
-
|
|
94
|
+
// The search snippet for the profile is a last resort when Instagram walls the bio.
|
|
95
|
+
const igSnippet = search?.results?.find((r) => /instagram\.com\//i.test(r.url) && r.url.toLowerCase().includes(`/${handle.toLowerCase()}`))?.snippet ?? "";
|
|
96
|
+
const instagramP = handle ? source("Instagram", () => readInstagram(handle, { snippet: igSnippet })) : null;
|
|
82
97
|
const tripadvisorP = taUrl ? source("TripAdvisor", () => readTripadvisor(taUrl, { browser })) : null;
|
|
83
98
|
// The hub URL comes from the options or the website, else from the Instagram bio.
|
|
84
99
|
const hubTask = (async () => {
|
|
@@ -103,17 +118,27 @@ export async function research(options = {}) {
|
|
|
103
118
|
}
|
|
104
119
|
|
|
105
120
|
// Notes keep the order of the sequential flow: Instagram, link-in-bio, TripAdvisor.
|
|
106
|
-
if (instagram) notes.push(instagram.blocked ? `Instagram @${handle}: not readable (${instagram.blocked}). Add the bio by hand, or use \`tablefacts photos instagram\` for posts.` : `Instagram @${handle}: profile card
|
|
121
|
+
if (instagram) notes.push(instagram.blocked ? `Instagram @${handle}: not readable (${instagram.blocked}). Add the bio by hand, or use \`tablefacts photos instagram\` for posts.` : `Instagram @${handle}: profile read (${instagram.surface ?? "share card"})`);
|
|
107
122
|
else if (!handle) notes.push("Instagram: no handle found. Pass the handle (--instagram / `instagram`).");
|
|
108
123
|
if (hub) notes.push(hub.blocked ? `Link-in-bio ${hubUrl}: not readable (${hub.blocked})` : `Link-in-bio: read ${hub.url}`);
|
|
109
|
-
if (tripadvisor)
|
|
110
|
-
|
|
124
|
+
if (tripadvisor) {
|
|
125
|
+
// The URL slug is a hint the source kept for a manual read; it never becomes a field on its own.
|
|
126
|
+
const hint = tripadvisor.slug ? ` URL suggests "${tripadvisor.slug.name}"${tripadvisor.slug.location ? `, ${tripadvisor.slug.location}` : ""}.` : "";
|
|
127
|
+
notes.push(tripadvisor.blocked ? `TripAdvisor ${taUrl}: blocked (${tripadvisor.blocked}). The link is kept; read it by hand.${hint}` : "TripAdvisor: page read");
|
|
128
|
+
} else notes.push("TripAdvisor: no link found. Pass the page (--tripadvisor / `tripadvisor`).");
|
|
111
129
|
} finally {
|
|
112
130
|
await browser.close();
|
|
113
131
|
}
|
|
114
132
|
|
|
115
|
-
const profile = buildProfile({ query, google, osm, site, hub: hub?.blocked ? null : hub, instagram, tripadvisor
|
|
116
|
-
|
|
133
|
+
const profile = buildProfile({ query, google, osm, site, hub: hub?.blocked ? null : hub, instagram, tripadvisor, search });
|
|
134
|
+
const taOrigin = tripadvisorOpt ? "you" : (hub?.links?.tripadvisor ?? []).includes(taUrl) ? "link-in-bio" : search?.tripadvisor === taUrl ? "search" : "website";
|
|
135
|
+
if (tripadvisor?.blocked && taUrl && !profile.fields.tripadvisor) profile.fields.tripadvisor = { value: taUrl, source: taOrigin, confidence: "medium" };
|
|
136
|
+
if (branches.length > 1) profile.warnings.push(`Google returns ${branches.length} places named like "${name}" (${branches.map((b) => b.address).filter(Boolean).join("; ")}). This looks like a chain: confirm which branch this site is for.`);
|
|
137
|
+
|
|
138
|
+
// A link a source could not read is stated at the top of the report, not only in the source list.
|
|
139
|
+
const kept = [];
|
|
140
|
+
if (tripadvisor?.blocked && taUrl) kept.push({ label: "TripAdvisor", url: taUrl, reason: tripadvisor.blocked });
|
|
141
|
+
if (instagram?.blocked && handle) kept.push({ label: `Instagram @${handle}`, url: `https://www.instagram.com/${handle}/`, reason: instagram.blocked });
|
|
117
142
|
|
|
118
143
|
// Optional photo downloads: reference material, never wired into the site.
|
|
119
144
|
const photos = [];
|
|
@@ -137,11 +162,21 @@ export async function research(options = {}) {
|
|
|
137
162
|
photos.push(...saved.filter(Boolean));
|
|
138
163
|
}
|
|
139
164
|
|
|
165
|
+
// Two names for one place ("Makibar" and "Maki Bar") used to make two folders, and reading them all showed
|
|
166
|
+
// a stale report. Note the sibling run and keep a pointer to the newest, so the last run is the one to read.
|
|
167
|
+
const loose = slug.replace(/-/g, "");
|
|
168
|
+
const siblings = options.out ? [] : await readdir(researchDir, { withFileTypes: true })
|
|
169
|
+
.then((entries) => entries.filter((e) => e.isDirectory() && e.name.replace(/-/g, "") === loose && join(researchDir, e.name) !== outDir).map((e) => e.name))
|
|
170
|
+
.catch(() => []);
|
|
171
|
+
if (siblings.length) notes.push(`Other runs with a similar name exist (${siblings.join(", ")}); \`latest.json\` points at the newest. Older folders may be stale.`);
|
|
172
|
+
|
|
140
173
|
const files = { profile: resolve(outDir, "profile.json"), report: resolve(outDir, "report.md"), setupAnswers: resolve(outDir, "setup-answers.txt") };
|
|
141
174
|
await mkdir(outDir, { recursive: true });
|
|
142
|
-
await writeFile(files.profile, JSON.stringify({ ...profile, raw: { google, osm, site, hub, instagram, tripadvisor }, photos }, null, 2));
|
|
143
|
-
await writeFile(files.report, renderReport(profile, { notes, photos }));
|
|
175
|
+
await writeFile(files.profile, JSON.stringify({ ...profile, raw: { google, osm, site, hub, instagram, tripadvisor, search }, photos }, null, 2));
|
|
176
|
+
await writeFile(files.report, renderReport(profile, { notes, photos, kept }));
|
|
144
177
|
await writeFile(files.setupAnswers, setupAnswers(profile));
|
|
178
|
+
// Only the default layout gets the "newest" pointer; a custom --out is the caller's own tree.
|
|
179
|
+
if (!options.out) await writeFile(join(researchDir, "latest.json"), JSON.stringify({ slug, name, location, generatedAt: profile.generatedAt, outDir, report: files.report }, null, 2));
|
|
145
180
|
|
|
146
181
|
return { profile, notes, photos, outDir, files };
|
|
147
182
|
}
|
|
@@ -35,11 +35,12 @@ const cleanUrl = (u) => {
|
|
|
35
35
|
} catch { return u; }
|
|
36
36
|
};
|
|
37
37
|
|
|
38
|
-
export function buildProfile({ query, google, osm, site, hub, instagram, tripadvisor }) {
|
|
38
|
+
export function buildProfile({ query, google, osm, site, hub, instagram, tripadvisor, search }) {
|
|
39
39
|
const g = google ?? {};
|
|
40
40
|
const o = osm ?? {};
|
|
41
41
|
const ld = site?.jsonld ?? null;
|
|
42
42
|
const ta = tripadvisor?.jsonld ?? null;
|
|
43
|
+
const taSlug = tripadvisor?.slug ?? null;
|
|
43
44
|
const warnings = [];
|
|
44
45
|
const ok = (v) => (v ? v : undefined);
|
|
45
46
|
const hubLinks = hub?.links ?? {};
|
|
@@ -54,7 +55,7 @@ export function buildProfile({ query, google, osm, site, hub, instagram, tripadv
|
|
|
54
55
|
const igLinks = [...(site?.links?.instagram ?? []), ...(hubLinks.instagram ?? [])];
|
|
55
56
|
|
|
56
57
|
const fields = {
|
|
57
|
-
name: choose([{ value: g.name, source: "google" }, { value: ld?.name, source: "website" }, { value: ta?.name, source: "tripadvisor" }, { value: o.name, source: "osm" }, { value: site?.meta.siteName, source: "website" }], sameText),
|
|
58
|
+
name: choose([{ value: g.name, source: "google" }, { value: ld?.name, source: "website" }, { value: ta?.name, source: "tripadvisor" }, { value: o.name, source: "osm" }, { value: site?.meta.siteName, source: "website" }, { value: taSlug?.name, source: "tripadvisor (URL)", low: true }], sameText),
|
|
58
59
|
street: choose([{ value: g.street, source: "google" }, { value: ld?.address.street, source: "website" }, { value: ta?.address.street, source: "tripadvisor" }, { value: o.street, source: "osm" }], sameText),
|
|
59
60
|
locality: choose([{ value: g.locality, source: "google" }, { value: ld?.address.locality, source: "website" }, { value: o.locality, source: "osm" }], sameText),
|
|
60
61
|
region: choose([{ value: g.region, source: "google" }, { value: ld?.address.region, source: "website" }, { value: o.region, source: "osm" }], sameText),
|
|
@@ -62,14 +63,14 @@ export function buildProfile({ query, google, osm, site, hub, instagram, tripadv
|
|
|
62
63
|
coordinates: choose([{ value: g.coords, source: "google" }, { value: o.coords, source: "osm" }, { value: ld?.geo, source: "website" }], coordSame),
|
|
63
64
|
phone: choose([{ value: g.phone, source: "google" }, { value: ok(links("tel")[0]), source: "website" }, { value: ld?.telephone, source: "website" }, { value: ta?.telephone, source: "tripadvisor" }, { value: instagram?.phones?.[0], source: "instagram" }, { value: o.phone, source: "osm" }], phoneSame),
|
|
64
65
|
email: choose([{ value: ld?.email, source: "website" }, { value: links("mail").find((m) => !/sentry|wixpress|example|domain\./i.test(m)), source: "website" }, { value: o.email, source: "osm" }]),
|
|
65
|
-
instagram: choose([{ value: query.instagram && handleOf(query.instagram), source: "you" }, ...igLinks.map((h) => ({ value: h, source: "website" })), ...(ld?.sameAs ?? []).filter((u) => /instagram\.com/.test(u)).map((u) => ({ value: handleOf(u), source: "website" })), { value: o.instagram && handleOf(o.instagram), source: "osm" }]),
|
|
66
|
-
facebook: choose([{ value: links("facebook")[0], source: "website" }, { value: o.facebook, source: "osm" }]),
|
|
67
|
-
tiktok: choose([{ value: links("tiktok")[0], source: "website" }]),
|
|
68
|
-
tripadvisor: choose([{ value: query.tripadvisor, source: "you" }, { value: links("tripadvisor")[0], source: "website" }, { value: (ld?.sameAs ?? []).find((u) => /tripadvisor\./.test(u)), source: "website" }]),
|
|
69
|
-
website: choose([{ value: query.website && cleanUrl(query.website), source: "you" }, { value: g.website && cleanUrl(g.website), source: "google" }, { value: o.website && cleanUrl(o.website), source: "osm" }]),
|
|
66
|
+
instagram: choose([{ value: query.instagram && handleOf(query.instagram), source: "you" }, ...igLinks.map((h) => ({ value: h, source: "website" })), ...(ld?.sameAs ?? []).filter((u) => /instagram\.com/.test(u)).map((u) => ({ value: handleOf(u), source: "website" })), { value: o.instagram && handleOf(o.instagram), source: "osm" }, { value: search?.instagram && handleOf(search.instagram), source: "search" }]),
|
|
67
|
+
facebook: choose([{ value: links("facebook")[0], source: "website" }, { value: o.facebook, source: "osm" }, { value: search?.facebook && cleanUrl(search.facebook), source: "search" }]),
|
|
68
|
+
tiktok: choose([{ value: links("tiktok")[0], source: "website" }, { value: search?.tiktok && cleanUrl(search.tiktok), source: "search" }]),
|
|
69
|
+
tripadvisor: choose([{ value: query.tripadvisor, source: "you" }, { value: links("tripadvisor")[0], source: "website" }, { value: (ld?.sameAs ?? []).find((u) => /tripadvisor\./.test(u)), source: "website" }, { value: search?.tripadvisor && cleanUrl(search.tripadvisor), source: "search" }]),
|
|
70
|
+
website: choose([{ value: query.website && cleanUrl(query.website), source: "you" }, { value: g.website && cleanUrl(g.website), source: "google" }, { value: o.website && cleanUrl(o.website), source: "osm" }, { value: search?.website && cleanUrl(search.website), source: "search" }]),
|
|
70
71
|
reserveUrl: choose([...links("reserve").map((u) => ({ value: u, source: "website" }))]),
|
|
71
72
|
priceRange: choose([{ value: g.priceRange, source: "google" }, { value: ld?.priceRange, source: "website" }, { value: ta?.priceRange, source: "tripadvisor" }]),
|
|
72
|
-
mapsUrl: choose([{ value: g.mapsUrl, source: "google" }, { value: links("maps")[0], source: "website" }]),
|
|
73
|
+
mapsUrl: choose([{ value: g.mapsUrl, source: "google" }, { value: links("maps")[0], source: "website" }, { value: search?.mapsUrl && cleanUrl(search.mapsUrl), source: "search" }]),
|
|
73
74
|
descriptor: choose([{ value: g.descriptor, source: "google" }]),
|
|
74
75
|
};
|
|
75
76
|
|
|
@@ -110,10 +111,10 @@ export function buildProfile({ query, google, osm, site, hub, instagram, tripadv
|
|
|
110
111
|
return {
|
|
111
112
|
query, generatedAt: new Date().toISOString(), warnings, fields, hours,
|
|
112
113
|
textHours: [...(site?.hoursText ?? [])],
|
|
113
|
-
links: { menu: links("menu"), delivery: links("delivery"), reserve: links("reserve"), waze: links("waze"), hubs: [...new Set([...(site?.links?.hubs ?? []), ...(instagram?.links ?? []).filter((l) => /linktr|beacons|bio\.link|lnk\.bio|taplink/.test(l))])] },
|
|
114
|
+
links: { menu: links("menu"), delivery: links("delivery"), reserve: links("reserve"), waze: links("waze"), hubs: [...new Set([...(site?.links?.hubs ?? []), ...(search?.hubs ?? []), ...(instagram?.links ?? []).filter((l) => /linktr|beacons|bio\.link|lnk\.bio|taplink/.test(l))])] },
|
|
114
115
|
ratings: [g.rating && { source: "google", value: g.rating, count: g.ratingCount }, ld?.rating && { source: "website", ...ld.rating }, ta?.rating && { source: "tripadvisor", ...ta.rating }].filter(Boolean),
|
|
115
116
|
images: { logos: [...(site?.logos ?? []), ...(hub?.logos ?? [])].filter((l) => l.url), site: site?.images ?? [], hub: hub?.images ?? [], googlePhotos: g.photos?.length ?? 0, instagramImage: instagram?.image ?? "", themeColor: site?.meta.themeColor ?? "" },
|
|
116
117
|
copySources: { googleSummary: g.summary ?? "", siteDescription: site?.meta.description ?? "", instagramBio: instagram?.bio ?? "", tripadvisor: tripadvisor?.description ?? "", headings: site?.headings ?? [], paragraphs: site?.paragraphs ?? [] },
|
|
117
|
-
social: { instagram: instagram ? { handle: instagram.handle, followers: instagram.followers, posts: instagram.posts, blocked: instagram.blocked } : null },
|
|
118
|
+
social: { instagram: instagram ? { handle: instagram.handle, followers: instagram.followers, posts: instagram.posts, blocked: instagram.blocked, partial: instagram.partial, surface: instagram.surface } : null },
|
|
118
119
|
};
|
|
119
120
|
}
|
|
@@ -33,17 +33,54 @@ const ROWS = [
|
|
|
33
33
|
|
|
34
34
|
const cell = (s) => String(s).replace(/\|/g, "\\|");
|
|
35
35
|
|
|
36
|
-
|
|
36
|
+
// Web text lands in a file agents are told to act on: collapse newlines and
|
|
37
|
+
// neutralise backticks so third-party text cannot forge headings or a code fence.
|
|
38
|
+
const flat = (s) => String(s ?? "").replace(/\s+/g, " ").replace(/`/g, "'").trim();
|
|
39
|
+
|
|
40
|
+
/**
|
|
41
|
+
* Splits the fields into what was discovered from a source, what was only echoed
|
|
42
|
+
* back from the user's own input, and low-confidence guesses, so a summary can
|
|
43
|
+
* say "N facts, M from your input" instead of counting all of them as found.
|
|
44
|
+
*/
|
|
45
|
+
export function fieldSummary(p) {
|
|
46
|
+
const held = Object.entries(p.fields ?? {}).filter(([, f]) => {
|
|
47
|
+
const v = f?.value;
|
|
48
|
+
return v !== undefined && v !== null && v !== "" && !(Array.isArray(v) && !v.length);
|
|
49
|
+
});
|
|
50
|
+
const parts = (f) => String(f.source ?? "").split(" + ").map((s) => s.trim()).filter(Boolean);
|
|
51
|
+
return {
|
|
52
|
+
discovered: held.filter(([, f]) => f.confidence !== "low" && parts(f).some((s) => s !== "you")).map(([k]) => k),
|
|
53
|
+
echoed: held.filter(([, f]) => {
|
|
54
|
+
const sources = parts(f);
|
|
55
|
+
return f.confidence !== "low" && sources.length > 0 && sources.every((s) => s === "you");
|
|
56
|
+
}).map(([k]) => k),
|
|
57
|
+
guessed: held.filter(([, f]) => f.confidence === "low").map(([k]) => k),
|
|
58
|
+
};
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/** The CLI summary: what came from a source, what was only echoed, what was a low guess. */
|
|
62
|
+
export function summaryLines(p) {
|
|
63
|
+
const { discovered, echoed, guessed } = fieldSummary(p);
|
|
64
|
+
const list = (a) => (a.length ? a.join(", ") : "none");
|
|
65
|
+
const lines = [`Found ${discovered.length} fact(s) from sources: ${list(discovered)}`];
|
|
66
|
+
if (echoed.length) lines.push(`Echoed ${echoed.length} from your input: ${list(echoed)}`);
|
|
67
|
+
if (guessed.length) lines.push(`Not counted (low-confidence guess): ${list(guessed)}`);
|
|
68
|
+
return lines;
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
export function renderReport(p, { notes, photos, kept = [] }) {
|
|
37
72
|
const out = [];
|
|
38
73
|
const q = p.query;
|
|
39
|
-
out.push(`# Research: ${q.name}${q.location ? `, ${q.location}` : ""}`, "", `Generated ${p.generatedAt}. Everything here is **unconfirmed**: check it with the client and copy the answers into \`BRIEF.md\`. Sources disagree often, and the confidence column says how many agreed.`, "");
|
|
74
|
+
out.push(`# Research: ${flat(q.name)}${q.location ? `, ${flat(q.location)}` : ""}`, "", `Generated ${p.generatedAt}. Everything here is **unconfirmed**: check it with the client and copy the answers into \`BRIEF.md\`. Sources disagree often, and the confidence column says how many agreed.`, "");
|
|
75
|
+
for (const k of kept) out.push(`> **Kept for a manual read:** ${k.label} (${k.url}) — could not be read (${k.reason}). Nothing was extracted from it; open it yourself.`, "");
|
|
40
76
|
if (p.warnings.length) out.push("## Warnings", "", ...p.warnings.map((w) => `- ${w}`), "");
|
|
41
77
|
if (notes.length) out.push("## What each source did", "", ...notes.map((n) => `- ${n}`), "");
|
|
42
78
|
|
|
43
79
|
out.push("## Facts", "", "| Item | Value | Source | Confidence | Goes in |", "| --- | --- | --- | --- | --- |");
|
|
44
80
|
for (const [label, key, goes] of ROWS) {
|
|
45
81
|
const f = p.fields[key];
|
|
46
|
-
|
|
82
|
+
const confidence = f ? (f.confidence === "low" ? "low (guess)" : f.confidence) : "";
|
|
83
|
+
out.push(`| ${label} | ${f ? cell(val(f)) : "_not found_"} | ${f?.source ?? ""} | ${confidence} | ${goes} |`);
|
|
47
84
|
}
|
|
48
85
|
out.push("");
|
|
49
86
|
|
|
@@ -75,7 +112,12 @@ export function renderReport(p, { notes, photos }) {
|
|
|
75
112
|
out.push("");
|
|
76
113
|
|
|
77
114
|
if (p.ratings.length) out.push("## Ratings (context only)", "", ...p.ratings.map((r) => `- ${r.source}: ${r.value}${r.count ? ` (${r.count} reviews)` : ""}`), "");
|
|
78
|
-
|
|
115
|
+
const ig = p.social.instagram;
|
|
116
|
+
if (ig && !ig.blocked) {
|
|
117
|
+
const counts = ig.followers || ig.posts ? `${ig.followers || "?"} followers, ${ig.posts || "?"} posts.` : "";
|
|
118
|
+
const note = ig.partial ? "Bio from a search snippet (the profile itself was walled)." : counts ? "" : "Profile card read; followers and bio are not public.";
|
|
119
|
+
out.push(`Instagram @${ig.handle}: ${[counts, note].filter(Boolean).join(" ")}`.trim(), "");
|
|
120
|
+
}
|
|
79
121
|
|
|
80
122
|
out.push("## Images", "");
|
|
81
123
|
out.push(`- Website images found: ${p.images.site.length}; link-in-bio: ${p.images.hub.length}; Google photos available: ${p.images.googlePhotos}.`);
|
|
@@ -86,13 +128,21 @@ export function renderReport(p, { notes, photos }) {
|
|
|
86
128
|
|
|
87
129
|
out.push("## Material for the copy", "", "Facts and tone only. Rewrite it in the template's voice in both languages; do not paste it.", "");
|
|
88
130
|
const c = p.copySources;
|
|
89
|
-
for (const [label, text] of [["Google summary", c.googleSummary], ["Site description", c.siteDescription], ["Instagram bio", c.instagramBio], ["TripAdvisor", c.tripadvisor]]) if (text) out.push(`- **${label}:** ${text}`);
|
|
90
|
-
if (c.headings.length) out.push(`- **Site headings:** ${c.headings.slice(0, 12).join(" / ")}`);
|
|
91
|
-
for (const t of c.paragraphs.slice(0, 6)) out.push(` - ${t}`);
|
|
131
|
+
for (const [label, text] of [["Google summary", c.googleSummary], ["Site description", c.siteDescription], ["Instagram bio", c.instagramBio], ["TripAdvisor", c.tripadvisor]]) if (text) out.push(`- **${label}:** ${flat(text)}`);
|
|
132
|
+
if (c.headings.length) out.push(`- **Site headings:** ${c.headings.slice(0, 12).map(flat).join(" / ")}`);
|
|
133
|
+
for (const t of c.paragraphs.slice(0, 6)) out.push(` - ${flat(t)}`);
|
|
92
134
|
out.push("");
|
|
93
135
|
|
|
94
136
|
const missing = ROWS.filter(([, key]) => !p.fields[key] && ["name", "street", "coordinates", "whatsapp", "instagram", "reserveUrl", "cuisines"].includes(key)).map(([l]) => l);
|
|
95
|
-
|
|
137
|
+
const ask = !p.fields.website && !p.fields.mapsUrl
|
|
138
|
+
? "Send the restaurant's website or a Google Maps link — that unlocks the address, map point and place ID."
|
|
139
|
+
: !p.fields.coordinates
|
|
140
|
+
? "A Google Maps link would pin the exact location."
|
|
141
|
+
: !p.fields.whatsapp
|
|
142
|
+
? "Ask for the WhatsApp number taken with reservations, and confirm it accepts messages."
|
|
143
|
+
: "Ask for the current website and social profiles to cross-check the rest.";
|
|
144
|
+
const asks = ["Opening hours incl. holidays", "Logo as SVG and original photos", "The story, signature dishes and events", "Production domain"];
|
|
145
|
+
out.push("## Ask the client", "", `- **Best next question:** ${ask}`, ...[...new Set([...missing, ...asks])].map((m) => `- ${m}`), "");
|
|
96
146
|
out.push("## Next", "", "```bash", `npm run setup < .tablefacts/research/${q.slug}/setup-answers.txt`, "git diff # review before keeping it", "```", "", "Blank lines in that file keep the template's current value, so anything not found stays a placeholder.", "");
|
|
97
147
|
return out.join("\n");
|
|
98
148
|
}
|
|
@@ -100,19 +150,22 @@ export function renderReport(p, { notes, photos }) {
|
|
|
100
150
|
/** One line per prompt of `data/scripts/setup.mjs`, in its order. Blank keeps the current value. */
|
|
101
151
|
export function setupAnswers(p) {
|
|
102
152
|
const f = p.fields;
|
|
103
|
-
|
|
153
|
+
// A low-confidence guess stays blank so setup keeps the template's placeholder,
|
|
154
|
+
// instead of a guess being written in as though it were a discovered fact.
|
|
155
|
+
const sure = (field) => (field && field.confidence !== "low" ? field : null);
|
|
156
|
+
const ig = sure(f.instagram)?.value;
|
|
104
157
|
return [
|
|
105
|
-
val(f.name),
|
|
158
|
+
val(sure(f.name)),
|
|
106
159
|
"", // production URL: the new domain, not the current website
|
|
107
|
-
f.reserveUrl?.value ?? "",
|
|
108
|
-
f.whatsapp?.display ?? "",
|
|
160
|
+
sure(f.reserveUrl)?.value ?? "",
|
|
161
|
+
sure(f.whatsapp)?.display ?? "",
|
|
109
162
|
ig ? `@${ig}` : "",
|
|
110
|
-
val(f.street),
|
|
111
|
-
val(f.locality),
|
|
112
|
-
val(f.region),
|
|
113
|
-
val(f.country),
|
|
114
|
-
f.coordinates?.value?.lat ?? "",
|
|
115
|
-
f.coordinates?.value?.lng ?? "",
|
|
116
|
-
val(f.cuisines),
|
|
163
|
+
val(sure(f.street)),
|
|
164
|
+
val(sure(f.locality)),
|
|
165
|
+
val(sure(f.region)),
|
|
166
|
+
val(sure(f.country)),
|
|
167
|
+
sure(f.coordinates)?.value?.lat ?? "",
|
|
168
|
+
sure(f.coordinates)?.value?.lng ?? "",
|
|
169
|
+
val(sure(f.cuisines)),
|
|
117
170
|
].join("\n") + "\n";
|
|
118
171
|
}
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
// Key-free web search, used when Google Places and OpenStreetMap find no place
|
|
2
|
+
// (or no website): a bare name + place otherwise yields an empty report. It only
|
|
3
|
+
// surfaces candidate links (website, Instagram, TripAdvisor, Maps, hub); the rest
|
|
4
|
+
// of the pipeline follows them. DuckDuckGo's Lite endpoint is tried first (the
|
|
5
|
+
// same results in simpler HTML, no JavaScript), then the html endpoint. No key
|
|
6
|
+
// and no login. A bot wall makes the source fail like any other.
|
|
7
|
+
import { fetchText, fold, sameText, tokens, decodeHtml, instagramHandle } from "./util.mjs";
|
|
8
|
+
|
|
9
|
+
const DDG = [
|
|
10
|
+
"https://lite.duckduckgo.com/lite/",
|
|
11
|
+
"https://html.duckduckgo.com/html/",
|
|
12
|
+
];
|
|
13
|
+
|
|
14
|
+
// Hosts that are directories or social pages, never the restaurant's own site.
|
|
15
|
+
const NOT_WEBSITE =
|
|
16
|
+
/(duckduckgo|bing|google|youtube|facebook|instagram|tiktok|twitter|(^|\.)x\.com$|tripadvisor|yelp|wikipedia|foursquare|mapquest|waze|trip\.com|expedia|booking|airbnb|justdial|zomato|rappi|ubereats|pedidosya|doordash|grubhub|glovo|opentable|thefork|eltenedor|exploretock|sevenrooms|covermanager|quandoo|linktr|beacons|linkin\.bio|lnk\.bio|taplink|wa\.me|whatsapp)/i;
|
|
17
|
+
const INSTAGRAM = /(^|\.)instagram\.com$/i;
|
|
18
|
+
const TRIPADVISOR = /tripadvisor\./i;
|
|
19
|
+
const FACEBOOK = /(^|\.)facebook\.com$|(^|\.)fb\.com$/i;
|
|
20
|
+
const TIKTOK = /(^|\.)tiktok\.com$/i;
|
|
21
|
+
const MAPS = /maps\.app\.goo\.gl|(^|\.)google\.[a-z.]+\/maps|maps\.google\.com|goo\.gl\/maps/i;
|
|
22
|
+
const HUBS = /(linktr\.ee|beacons\.ai|bio\.link|lnk\.bio|linktree\.com|taplink|campsite\.bio|solo\.to|linkin\.bio|flow\.page)/i;
|
|
23
|
+
|
|
24
|
+
const textOf = (s) => decodeHtml(String(s).replace(/<[^>]+>/g, " ")).replace(/\s+/g, " ").trim();
|
|
25
|
+
|
|
26
|
+
/** Unwraps DuckDuckGo's `//duckduckgo.com/l/?uddg=<url>` redirect to the real result URL. */
|
|
27
|
+
export function resultUrl(href) {
|
|
28
|
+
if (!href) return null;
|
|
29
|
+
let u;
|
|
30
|
+
try { u = new URL(href, "https://html.duckduckgo.com"); } catch { return null; }
|
|
31
|
+
if (/(^|\.)duckduckgo\.com$/i.test(u.hostname)) {
|
|
32
|
+
const target = u.searchParams.get("uddg");
|
|
33
|
+
if (!target) return null;
|
|
34
|
+
try { return new URL(target).href; } catch { return null; }
|
|
35
|
+
}
|
|
36
|
+
return u.protocol === "http:" || u.protocol === "https:" ? u.href : null;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
// One pass over the page, matching a result link or a snippet cell in order, so a
|
|
40
|
+
// result with no snippet cannot borrow the next result's text.
|
|
41
|
+
const RESULT_TOKEN = /<a\b([^>]*\b(?:result__a|result-link)\b[^>]*)>([\s\S]*?)<\/a>|<([a-z]+)\b[^>]*\b(?:result__snippet|result-snippet)\b[^>]*>([\s\S]*?)<\/\3>/gi;
|
|
42
|
+
|
|
43
|
+
/** Pulls { url, title, snippet } out of DuckDuckGo's HTML result page (no DOM parser). */
|
|
44
|
+
export function parseSearchResults(html) {
|
|
45
|
+
const out = [];
|
|
46
|
+
const seen = new Set();
|
|
47
|
+
let current = -1;
|
|
48
|
+
for (const m of html.matchAll(RESULT_TOKEN)) {
|
|
49
|
+
if (m[1] !== undefined) {
|
|
50
|
+
const href = m[1].match(/href=["']([^"']+)["']/i)?.[1] ?? "";
|
|
51
|
+
const url = resultUrl(href);
|
|
52
|
+
if (!url || seen.has(url)) { current = -1; continue; }
|
|
53
|
+
seen.add(url);
|
|
54
|
+
out.push({ url, title: textOf(m[2]), snippet: "" });
|
|
55
|
+
current = out.length - 1;
|
|
56
|
+
} else if (current >= 0 && !out[current].snippet) {
|
|
57
|
+
out[current].snippet = textOf(m[4]);
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
return out;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
const hostOf = (url) => { try { return new URL(url).hostname.replace(/^www\./, ""); } catch { return ""; } };
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* Sorts results into the links the pipeline can follow. `website` is the best
|
|
67
|
+
* non-directory result whose title/host looks like the restaurant's name; a
|
|
68
|
+
* result that only matches the place is kept as a weaker candidate.
|
|
69
|
+
*/
|
|
70
|
+
export function classifyResults(results, { name = "" } = {}) {
|
|
71
|
+
const links = { website: [], instagram: [], tripadvisor: [], facebook: [], tiktok: [], maps: [], hubs: [] };
|
|
72
|
+
const add = (k, v) => { if (v && !links[k].includes(v)) links[k].push(v); };
|
|
73
|
+
const candidates = [];
|
|
74
|
+
for (const { url, title, snippet } of results) {
|
|
75
|
+
const host = hostOf(url);
|
|
76
|
+
if (INSTAGRAM.test(host)) {
|
|
77
|
+
const handle = instagramHandle(url);
|
|
78
|
+
if (handle) add("instagram", handle);
|
|
79
|
+
} else if (TRIPADVISOR.test(host)) add("tripadvisor", url);
|
|
80
|
+
else if (FACEBOOK.test(host)) add("facebook", url);
|
|
81
|
+
else if (TIKTOK.test(host)) add("tiktok", url);
|
|
82
|
+
else if (MAPS.test(url)) add("maps", url);
|
|
83
|
+
else if (HUBS.test(host)) add("hubs", url);
|
|
84
|
+
else if (NOT_WEBSITE.test(host)) continue;
|
|
85
|
+
else {
|
|
86
|
+
const strong = sameText(name, title) || tokens(name).some((t) => fold(host).includes(t));
|
|
87
|
+
const score = tokens(name).filter((t) => fold(`${title} ${host} ${snippet}`).includes(t)).length;
|
|
88
|
+
candidates.push({ url, title, snippet, score, strong });
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
candidates.sort((a, b) => Number(b.strong) - Number(a.strong) || b.score - a.score);
|
|
92
|
+
for (const c of candidates) add("website", c.url);
|
|
93
|
+
return {
|
|
94
|
+
website: candidates.find((c) => c.strong)?.url ?? candidates[0]?.url ?? "",
|
|
95
|
+
instagram: links.instagram[0] ?? "",
|
|
96
|
+
tripadvisor: links.tripadvisor[0] ?? "",
|
|
97
|
+
facebook: links.facebook[0] ?? "",
|
|
98
|
+
tiktok: links.tiktok[0] ?? "",
|
|
99
|
+
mapsUrl: links.maps[0] ?? "",
|
|
100
|
+
hubs: links.hubs,
|
|
101
|
+
candidates: candidates.slice(0, 5).map(({ url, title }) => ({ url, title })),
|
|
102
|
+
};
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* Runs the search for a restaurant and returns its candidate links. Tries every
|
|
107
|
+
* endpoint until one returns results; if all answer but none has results, the
|
|
108
|
+
* empty result is returned so the caller still says "no results" rather than
|
|
109
|
+
* "failed". Throws only when no endpoint answers.
|
|
110
|
+
*/
|
|
111
|
+
export async function searchWeb({ name, location = "" }) {
|
|
112
|
+
const query = [name, location, "restaurant"].filter(Boolean).join(" ");
|
|
113
|
+
const qs = new URLSearchParams({ q: query }).toString();
|
|
114
|
+
let lastError, empty;
|
|
115
|
+
for (const base of DDG) {
|
|
116
|
+
let r;
|
|
117
|
+
try { r = await fetchText(`${base}?${qs}`, { headers: { accept: "text/html" } }); }
|
|
118
|
+
catch (e) { lastError = e; continue; }
|
|
119
|
+
if (!r.ok) { lastError = new Error(`DuckDuckGo ${r.status}`); continue; }
|
|
120
|
+
const results = parseSearchResults(r.text);
|
|
121
|
+
const found = { source: "search", url: r.url, query, results, ...classifyResults(results, { name }) };
|
|
122
|
+
if (results.length) return found;
|
|
123
|
+
empty = found;
|
|
124
|
+
}
|
|
125
|
+
if (empty) return empty;
|
|
126
|
+
throw lastError ?? new Error("DuckDuckGo search failed");
|
|
127
|
+
}
|