tablefacts 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +3 -1
- package/CHANGELOG.md +61 -0
- package/README.md +67 -19
- package/package.json +14 -3
- package/src/lib/types.mjs +16 -4
- package/src/menu/README.md +63 -15
- package/src/menu/cluvi/config.mjs +6 -0
- package/src/menu/cluvi/import.mjs +6 -2
- package/src/menu/lib/db.mjs +81 -17
- package/src/menu/lib/import.mjs +20 -3
- package/src/menu/lib/run.mjs +18 -0
- package/src/menu/lib/tables.mjs +44 -0
- package/src/menu/raw/config.mjs +12 -2
- package/src/menu/raw/extract.mjs +36 -15
- package/src/menu/raw/images.mjs +109 -0
- package/src/menu/raw/import.mjs +330 -51
- package/src/menu/raw/normalize.mjs +15 -4
- package/src/menu/raw/pdf.mjs +330 -0
- package/src/menu/raw/pdfjs.mjs +57 -0
- package/src/menu/raw/source.mjs +9 -2
- package/src/menu/raw/vision.mjs +124 -65
- package/src/research/README.md +23 -13
- package/src/research/index.mjs +52 -17
- package/src/research/lib/merge.mjs +11 -10
- package/src/research/lib/report.mjs +72 -19
- package/src/research/lib/search.mjs +127 -0
- package/src/research/lib/social.mjs +137 -26
- package/src/research/lib/util.mjs +22 -0
- package/src/research/lib/website.mjs +3 -14
- package/src/research/research.mjs +12 -8
- package/types/lib/types.d.mts +70 -6
- package/types/menu/cluvi/config.d.mts +1 -0
- package/types/menu/cluvi/import.d.mts +2 -0
- package/types/menu/lib/db.d.mts +31 -3
- package/types/menu/lib/import.d.mts +1 -1
- package/types/menu/lib/run.d.mts +6 -0
- package/types/menu/lib/tables.d.mts +18 -0
- package/types/menu/raw/config.d.mts +2 -0
- package/types/menu/raw/images.d.mts +27 -0
- package/types/menu/raw/import.d.mts +36 -7
- package/types/menu/raw/normalize.d.mts +7 -1
- package/types/menu/raw/pdf.d.mts +98 -0
- package/types/menu/raw/pdfjs.d.mts +12 -0
- package/types/menu/raw/source.d.mts +2 -0
- package/types/menu/raw/vision.d.mts +11 -1
- package/types/research/lib/merge.d.mts +4 -1
- package/types/research/lib/report.d.mts +14 -1
- package/types/research/lib/search.d.mts +58 -0
- package/types/research/lib/social.d.mts +32 -31
- package/types/research/lib/util.d.mts +5 -0
- package/types/research/lib/website.d.mts +20 -20
|
@@ -33,17 +33,54 @@ const ROWS = [
|
|
|
33
33
|
|
|
34
34
|
const cell = (s) => String(s).replace(/\|/g, "\\|");
|
|
35
35
|
|
|
36
|
-
|
|
36
|
+
// Web text lands in a file agents are told to act on: collapse newlines and
|
|
37
|
+
// neutralise backticks so third-party text cannot forge headings or a code fence.
|
|
38
|
+
const flat = (s) => String(s ?? "").replace(/\s+/g, " ").replace(/`/g, "'").trim();
|
|
39
|
+
|
|
40
|
+
/**
|
|
41
|
+
* Splits the fields into what was discovered from a source, what was only echoed
|
|
42
|
+
* back from the user's own input, and low-confidence guesses, so a summary can
|
|
43
|
+
* say "N facts, M from your input" instead of counting all of them as found.
|
|
44
|
+
*/
|
|
45
|
+
export function fieldSummary(p) {
|
|
46
|
+
const held = Object.entries(p.fields ?? {}).filter(([, f]) => {
|
|
47
|
+
const v = f?.value;
|
|
48
|
+
return v !== undefined && v !== null && v !== "" && !(Array.isArray(v) && !v.length);
|
|
49
|
+
});
|
|
50
|
+
const parts = (f) => String(f.source ?? "").split(" + ").map((s) => s.trim()).filter(Boolean);
|
|
51
|
+
return {
|
|
52
|
+
discovered: held.filter(([, f]) => f.confidence !== "low" && parts(f).some((s) => s !== "you")).map(([k]) => k),
|
|
53
|
+
echoed: held.filter(([, f]) => {
|
|
54
|
+
const sources = parts(f);
|
|
55
|
+
return f.confidence !== "low" && sources.length > 0 && sources.every((s) => s === "you");
|
|
56
|
+
}).map(([k]) => k),
|
|
57
|
+
guessed: held.filter(([, f]) => f.confidence === "low").map(([k]) => k),
|
|
58
|
+
};
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/** The CLI summary: what came from a source, what was only echoed, what was a low guess. */
|
|
62
|
+
export function summaryLines(p) {
|
|
63
|
+
const { discovered, echoed, guessed } = fieldSummary(p);
|
|
64
|
+
const list = (a) => (a.length ? a.join(", ") : "none");
|
|
65
|
+
const lines = [`Found ${discovered.length} fact(s) from sources: ${list(discovered)}`];
|
|
66
|
+
if (echoed.length) lines.push(`Echoed ${echoed.length} from your input: ${list(echoed)}`);
|
|
67
|
+
if (guessed.length) lines.push(`Not counted (low-confidence guess): ${list(guessed)}`);
|
|
68
|
+
return lines;
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
export function renderReport(p, { notes, photos, kept = [] }) {
|
|
37
72
|
const out = [];
|
|
38
73
|
const q = p.query;
|
|
39
|
-
out.push(`# Research: ${q.name}${q.location ? `, ${q.location}` : ""}`, "", `Generated ${p.generatedAt}. Everything here is **unconfirmed**: check it with the client and copy the answers into \`BRIEF.md\`. Sources disagree often, and the confidence column says how many agreed.`, "");
|
|
74
|
+
out.push(`# Research: ${flat(q.name)}${q.location ? `, ${flat(q.location)}` : ""}`, "", `Generated ${p.generatedAt}. Everything here is **unconfirmed**: check it with the client and copy the answers into \`BRIEF.md\`. Sources disagree often, and the confidence column says how many agreed.`, "");
|
|
75
|
+
for (const k of kept) out.push(`> **Kept for a manual read:** ${k.label} (${k.url}) — could not be read (${k.reason}). Nothing was extracted from it; open it yourself.`, "");
|
|
40
76
|
if (p.warnings.length) out.push("## Warnings", "", ...p.warnings.map((w) => `- ${w}`), "");
|
|
41
77
|
if (notes.length) out.push("## What each source did", "", ...notes.map((n) => `- ${n}`), "");
|
|
42
78
|
|
|
43
79
|
out.push("## Facts", "", "| Item | Value | Source | Confidence | Goes in |", "| --- | --- | --- | --- | --- |");
|
|
44
80
|
for (const [label, key, goes] of ROWS) {
|
|
45
81
|
const f = p.fields[key];
|
|
46
|
-
|
|
82
|
+
const confidence = f ? (f.confidence === "low" ? "low (guess)" : f.confidence) : "";
|
|
83
|
+
out.push(`| ${label} | ${f ? cell(val(f)) : "_not found_"} | ${f?.source ?? ""} | ${confidence} | ${goes} |`);
|
|
47
84
|
}
|
|
48
85
|
out.push("");
|
|
49
86
|
|
|
@@ -75,7 +112,12 @@ export function renderReport(p, { notes, photos }) {
|
|
|
75
112
|
out.push("");
|
|
76
113
|
|
|
77
114
|
if (p.ratings.length) out.push("## Ratings (context only)", "", ...p.ratings.map((r) => `- ${r.source}: ${r.value}${r.count ? ` (${r.count} reviews)` : ""}`), "");
|
|
78
|
-
|
|
115
|
+
const ig = p.social.instagram;
|
|
116
|
+
if (ig && !ig.blocked) {
|
|
117
|
+
const counts = ig.followers || ig.posts ? `${ig.followers || "?"} followers, ${ig.posts || "?"} posts.` : "";
|
|
118
|
+
const note = ig.partial ? "Bio from a search snippet (the profile itself was walled)." : counts ? "" : "Profile card read; followers and bio are not public.";
|
|
119
|
+
out.push(`Instagram @${ig.handle}: ${[counts, note].filter(Boolean).join(" ")}`.trim(), "");
|
|
120
|
+
}
|
|
79
121
|
|
|
80
122
|
out.push("## Images", "");
|
|
81
123
|
out.push(`- Website images found: ${p.images.site.length}; link-in-bio: ${p.images.hub.length}; Google photos available: ${p.images.googlePhotos}.`);
|
|
@@ -86,13 +128,21 @@ export function renderReport(p, { notes, photos }) {
|
|
|
86
128
|
|
|
87
129
|
out.push("## Material for the copy", "", "Facts and tone only. Rewrite it in the template's voice in both languages; do not paste it.", "");
|
|
88
130
|
const c = p.copySources;
|
|
89
|
-
for (const [label, text] of [["Google summary", c.googleSummary], ["Site description", c.siteDescription], ["Instagram bio", c.instagramBio], ["TripAdvisor", c.tripadvisor]]) if (text) out.push(`- **${label}:** ${text}`);
|
|
90
|
-
if (c.headings.length) out.push(`- **Site headings:** ${c.headings.slice(0, 12).join(" / ")}`);
|
|
91
|
-
for (const t of c.paragraphs.slice(0, 6)) out.push(` - ${t}`);
|
|
131
|
+
for (const [label, text] of [["Google summary", c.googleSummary], ["Site description", c.siteDescription], ["Instagram bio", c.instagramBio], ["TripAdvisor", c.tripadvisor]]) if (text) out.push(`- **${label}:** ${flat(text)}`);
|
|
132
|
+
if (c.headings.length) out.push(`- **Site headings:** ${c.headings.slice(0, 12).map(flat).join(" / ")}`);
|
|
133
|
+
for (const t of c.paragraphs.slice(0, 6)) out.push(` - ${flat(t)}`);
|
|
92
134
|
out.push("");
|
|
93
135
|
|
|
94
136
|
const missing = ROWS.filter(([, key]) => !p.fields[key] && ["name", "street", "coordinates", "whatsapp", "instagram", "reserveUrl", "cuisines"].includes(key)).map(([l]) => l);
|
|
95
|
-
|
|
137
|
+
const ask = !p.fields.website && !p.fields.mapsUrl
|
|
138
|
+
? "Send the restaurant's website or a Google Maps link — that unlocks the address, map point and place ID."
|
|
139
|
+
: !p.fields.coordinates
|
|
140
|
+
? "A Google Maps link would pin the exact location."
|
|
141
|
+
: !p.fields.whatsapp
|
|
142
|
+
? "Ask for the WhatsApp number taken with reservations, and confirm it accepts messages."
|
|
143
|
+
: "Ask for the current website and social profiles to cross-check the rest.";
|
|
144
|
+
const asks = ["Opening hours incl. holidays", "Logo as SVG and original photos", "The story, signature dishes and events", "Production domain"];
|
|
145
|
+
out.push("## Ask the client", "", `- **Best next question:** ${ask}`, ...[...new Set([...missing, ...asks])].map((m) => `- ${m}`), "");
|
|
96
146
|
out.push("## Next", "", "```bash", `npm run setup < .tablefacts/research/${q.slug}/setup-answers.txt`, "git diff # review before keeping it", "```", "", "Blank lines in that file keep the template's current value, so anything not found stays a placeholder.", "");
|
|
97
147
|
return out.join("\n");
|
|
98
148
|
}
|
|
@@ -100,19 +150,22 @@ export function renderReport(p, { notes, photos }) {
|
|
|
100
150
|
/** One line per prompt of `data/scripts/setup.mjs`, in its order. Blank keeps the current value. */
|
|
101
151
|
export function setupAnswers(p) {
|
|
102
152
|
const f = p.fields;
|
|
103
|
-
|
|
153
|
+
// A low-confidence guess stays blank so setup keeps the template's placeholder,
|
|
154
|
+
// instead of a guess being written in as though it were a discovered fact.
|
|
155
|
+
const sure = (field) => (field && field.confidence !== "low" ? field : null);
|
|
156
|
+
const ig = sure(f.instagram)?.value;
|
|
104
157
|
return [
|
|
105
|
-
val(f.name),
|
|
158
|
+
val(sure(f.name)),
|
|
106
159
|
"", // production URL: the new domain, not the current website
|
|
107
|
-
f.reserveUrl?.value ?? "",
|
|
108
|
-
f.whatsapp?.display ?? "",
|
|
160
|
+
sure(f.reserveUrl)?.value ?? "",
|
|
161
|
+
sure(f.whatsapp)?.display ?? "",
|
|
109
162
|
ig ? `@${ig}` : "",
|
|
110
|
-
val(f.street),
|
|
111
|
-
val(f.locality),
|
|
112
|
-
val(f.region),
|
|
113
|
-
val(f.country),
|
|
114
|
-
f.coordinates?.value?.lat ?? "",
|
|
115
|
-
f.coordinates?.value?.lng ?? "",
|
|
116
|
-
val(f.cuisines),
|
|
163
|
+
val(sure(f.street)),
|
|
164
|
+
val(sure(f.locality)),
|
|
165
|
+
val(sure(f.region)),
|
|
166
|
+
val(sure(f.country)),
|
|
167
|
+
sure(f.coordinates)?.value?.lat ?? "",
|
|
168
|
+
sure(f.coordinates)?.value?.lng ?? "",
|
|
169
|
+
val(sure(f.cuisines)),
|
|
117
170
|
].join("\n") + "\n";
|
|
118
171
|
}
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
// Key-free web search, used when Google Places and OpenStreetMap find no place
|
|
2
|
+
// (or no website): a bare name + place otherwise yields an empty report. It only
|
|
3
|
+
// surfaces candidate links (website, Instagram, TripAdvisor, Maps, hub); the rest
|
|
4
|
+
// of the pipeline follows them. DuckDuckGo's Lite endpoint is tried first (the
|
|
5
|
+
// same results in simpler HTML, no JavaScript), then the html endpoint. No key
|
|
6
|
+
// and no login. A bot wall makes the source fail like any other.
|
|
7
|
+
import { fetchText, fold, sameText, tokens, decodeHtml, instagramHandle } from "./util.mjs";
|
|
8
|
+
|
|
9
|
+
const DDG = [
|
|
10
|
+
"https://lite.duckduckgo.com/lite/",
|
|
11
|
+
"https://html.duckduckgo.com/html/",
|
|
12
|
+
];
|
|
13
|
+
|
|
14
|
+
// Hosts that are directories or social pages, never the restaurant's own site.
|
|
15
|
+
const NOT_WEBSITE =
|
|
16
|
+
/(duckduckgo|bing|google|youtube|facebook|instagram|tiktok|twitter|(^|\.)x\.com$|tripadvisor|yelp|wikipedia|foursquare|mapquest|waze|trip\.com|expedia|booking|airbnb|justdial|zomato|rappi|ubereats|pedidosya|doordash|grubhub|glovo|opentable|thefork|eltenedor|exploretock|sevenrooms|covermanager|quandoo|linktr|beacons|linkin\.bio|lnk\.bio|taplink|wa\.me|whatsapp)/i;
|
|
17
|
+
const INSTAGRAM = /(^|\.)instagram\.com$/i;
|
|
18
|
+
const TRIPADVISOR = /tripadvisor\./i;
|
|
19
|
+
const FACEBOOK = /(^|\.)facebook\.com$|(^|\.)fb\.com$/i;
|
|
20
|
+
const TIKTOK = /(^|\.)tiktok\.com$/i;
|
|
21
|
+
const MAPS = /maps\.app\.goo\.gl|(^|\.)google\.[a-z.]+\/maps|maps\.google\.com|goo\.gl\/maps/i;
|
|
22
|
+
const HUBS = /(linktr\.ee|beacons\.ai|bio\.link|lnk\.bio|linktree\.com|taplink|campsite\.bio|solo\.to|linkin\.bio|flow\.page)/i;
|
|
23
|
+
|
|
24
|
+
const textOf = (s) => decodeHtml(String(s).replace(/<[^>]+>/g, " ")).replace(/\s+/g, " ").trim();
|
|
25
|
+
|
|
26
|
+
/** Unwraps DuckDuckGo's `//duckduckgo.com/l/?uddg=<url>` redirect to the real result URL. */
|
|
27
|
+
export function resultUrl(href) {
|
|
28
|
+
if (!href) return null;
|
|
29
|
+
let u;
|
|
30
|
+
try { u = new URL(href, "https://html.duckduckgo.com"); } catch { return null; }
|
|
31
|
+
if (/(^|\.)duckduckgo\.com$/i.test(u.hostname)) {
|
|
32
|
+
const target = u.searchParams.get("uddg");
|
|
33
|
+
if (!target) return null;
|
|
34
|
+
try { return new URL(target).href; } catch { return null; }
|
|
35
|
+
}
|
|
36
|
+
return u.protocol === "http:" || u.protocol === "https:" ? u.href : null;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
// One pass over the page, matching a result link or a snippet cell in order, so a
|
|
40
|
+
// result with no snippet cannot borrow the next result's text.
|
|
41
|
+
const RESULT_TOKEN = /<a\b([^>]*\b(?:result__a|result-link)\b[^>]*)>([\s\S]*?)<\/a>|<([a-z]+)\b[^>]*\b(?:result__snippet|result-snippet)\b[^>]*>([\s\S]*?)<\/\3>/gi;
|
|
42
|
+
|
|
43
|
+
/** Pulls { url, title, snippet } out of DuckDuckGo's HTML result page (no DOM parser). */
|
|
44
|
+
export function parseSearchResults(html) {
|
|
45
|
+
const out = [];
|
|
46
|
+
const seen = new Set();
|
|
47
|
+
let current = -1;
|
|
48
|
+
for (const m of html.matchAll(RESULT_TOKEN)) {
|
|
49
|
+
if (m[1] !== undefined) {
|
|
50
|
+
const href = m[1].match(/href=["']([^"']+)["']/i)?.[1] ?? "";
|
|
51
|
+
const url = resultUrl(href);
|
|
52
|
+
if (!url || seen.has(url)) { current = -1; continue; }
|
|
53
|
+
seen.add(url);
|
|
54
|
+
out.push({ url, title: textOf(m[2]), snippet: "" });
|
|
55
|
+
current = out.length - 1;
|
|
56
|
+
} else if (current >= 0 && !out[current].snippet) {
|
|
57
|
+
out[current].snippet = textOf(m[4]);
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
return out;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
const hostOf = (url) => { try { return new URL(url).hostname.replace(/^www\./, ""); } catch { return ""; } };
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* Sorts results into the links the pipeline can follow. `website` is the best
|
|
67
|
+
* non-directory result whose title/host looks like the restaurant's name; a
|
|
68
|
+
* result that only matches the place is kept as a weaker candidate.
|
|
69
|
+
*/
|
|
70
|
+
export function classifyResults(results, { name = "" } = {}) {
|
|
71
|
+
const links = { website: [], instagram: [], tripadvisor: [], facebook: [], tiktok: [], maps: [], hubs: [] };
|
|
72
|
+
const add = (k, v) => { if (v && !links[k].includes(v)) links[k].push(v); };
|
|
73
|
+
const candidates = [];
|
|
74
|
+
for (const { url, title, snippet } of results) {
|
|
75
|
+
const host = hostOf(url);
|
|
76
|
+
if (INSTAGRAM.test(host)) {
|
|
77
|
+
const handle = instagramHandle(url);
|
|
78
|
+
if (handle) add("instagram", handle);
|
|
79
|
+
} else if (TRIPADVISOR.test(host)) add("tripadvisor", url);
|
|
80
|
+
else if (FACEBOOK.test(host)) add("facebook", url);
|
|
81
|
+
else if (TIKTOK.test(host)) add("tiktok", url);
|
|
82
|
+
else if (MAPS.test(url)) add("maps", url);
|
|
83
|
+
else if (HUBS.test(host)) add("hubs", url);
|
|
84
|
+
else if (NOT_WEBSITE.test(host)) continue;
|
|
85
|
+
else {
|
|
86
|
+
const strong = sameText(name, title) || tokens(name).some((t) => fold(host).includes(t));
|
|
87
|
+
const score = tokens(name).filter((t) => fold(`${title} ${host} ${snippet}`).includes(t)).length;
|
|
88
|
+
candidates.push({ url, title, snippet, score, strong });
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
candidates.sort((a, b) => Number(b.strong) - Number(a.strong) || b.score - a.score);
|
|
92
|
+
for (const c of candidates) add("website", c.url);
|
|
93
|
+
return {
|
|
94
|
+
website: candidates.find((c) => c.strong)?.url ?? candidates[0]?.url ?? "",
|
|
95
|
+
instagram: links.instagram[0] ?? "",
|
|
96
|
+
tripadvisor: links.tripadvisor[0] ?? "",
|
|
97
|
+
facebook: links.facebook[0] ?? "",
|
|
98
|
+
tiktok: links.tiktok[0] ?? "",
|
|
99
|
+
mapsUrl: links.maps[0] ?? "",
|
|
100
|
+
hubs: links.hubs,
|
|
101
|
+
candidates: candidates.slice(0, 5).map(({ url, title }) => ({ url, title })),
|
|
102
|
+
};
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* Runs the search for a restaurant and returns its candidate links. Tries every
|
|
107
|
+
* endpoint until one returns results; if all answer but none has results, the
|
|
108
|
+
* empty result is returned so the caller still says "no results" rather than
|
|
109
|
+
* "failed". Throws only when no endpoint answers.
|
|
110
|
+
*/
|
|
111
|
+
export async function searchWeb({ name, location = "" }) {
|
|
112
|
+
const query = [name, location, "restaurant"].filter(Boolean).join(" ");
|
|
113
|
+
const qs = new URLSearchParams({ q: query }).toString();
|
|
114
|
+
let lastError, empty;
|
|
115
|
+
for (const base of DDG) {
|
|
116
|
+
let r;
|
|
117
|
+
try { r = await fetchText(`${base}?${qs}`, { headers: { accept: "text/html" } }); }
|
|
118
|
+
catch (e) { lastError = e; continue; }
|
|
119
|
+
if (!r.ok) { lastError = new Error(`DuckDuckGo ${r.status}`); continue; }
|
|
120
|
+
const results = parseSearchResults(r.text);
|
|
121
|
+
const found = { source: "search", url: r.url, query, results, ...classifyResults(results, { name }) };
|
|
122
|
+
if (results.length) return found;
|
|
123
|
+
empty = found;
|
|
124
|
+
}
|
|
125
|
+
if (empty) return empty;
|
|
126
|
+
throw lastError ?? new Error("DuckDuckGo search failed");
|
|
127
|
+
}
|
|
@@ -1,44 +1,155 @@
|
|
|
1
1
|
// Instagram, TripAdvisor and link-in-bio hubs. All three are best-effort: they
|
|
2
2
|
// sit behind login walls or bot protection, so each one returns what it could
|
|
3
3
|
// read plus a `blocked` reason, and the report says what is missing.
|
|
4
|
-
import { fetchText } from "./util.mjs";
|
|
4
|
+
import { fetchText, sleep, decodeHtml as decodeEntities } from "./util.mjs";
|
|
5
5
|
import { readPage } from "./website.mjs";
|
|
6
6
|
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
const
|
|
20
|
-
|
|
7
|
+
// Instagram's own web app sends this app id with its public profile request; no
|
|
8
|
+
// login is involved, and the request identifies honestly.
|
|
9
|
+
const IG_APP_ID = "936619743392459";
|
|
10
|
+
// A short gap between the Instagram surfaces: one walled profile should not become a burst.
|
|
11
|
+
const SURFACE_PAUSE = 300;
|
|
12
|
+
|
|
13
|
+
const bioLinks = (bio) => (bio.match(/https?:\/\/[^\s)]+|\b[\w-]+\.(?:com|co|link|bio|ee)\/[\w./-]+/gi) ?? [])
|
|
14
|
+
.map((l) => (/^https?:/.test(l) ? l : `https://${l}`));
|
|
15
|
+
const bioPhones = (bio) => [...bio.matchAll(/\+?\d[\d\s().-]{7,}\d/g)].map((m) => m[0].trim());
|
|
16
|
+
|
|
17
|
+
/** The public share card: "1,234 Followers, 56 Following, 789 Posts - Name (@handle) on Instagram: "bio"". */
|
|
18
|
+
function shareCard(html) {
|
|
19
|
+
const meta = (key) => html.match(new RegExp(`<meta[^>]+(?:property|name)=["']${key}["'][^>]+content=["']([^"']*)["']`, "i"))?.[1] ?? "";
|
|
20
|
+
const desc = decodeEntities(meta("og:description"));
|
|
21
|
+
const title = decodeEntities(meta("og:title"));
|
|
22
|
+
if (!desc && !title) return null;
|
|
23
|
+
return { title, desc, image: meta("og:image") };
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
function fromCard(handle, url, card, surface) {
|
|
27
|
+
const desc = card.desc;
|
|
21
28
|
const stats = desc.match(/([\d.,KMkm]+)\s+Followers?,\s*([\d.,KMkm]+)\s+Following,\s*([\d.,KMkm]+)\s+Posts?/i);
|
|
22
|
-
const
|
|
23
|
-
|
|
29
|
+
const wrapped = desc.match(/on Instagram:\s*["“]([\s\S]*?)["”]?\s*$/i)?.[1]?.trim();
|
|
30
|
+
// A bare bio snippet has no share-card wrapper; a stats-only line has no bio.
|
|
31
|
+
const bio = wrapped ?? (/Followers?,\s*[\d.,KMkm]+\s+Following/i.test(desc) ? "" : desc.trim());
|
|
24
32
|
return {
|
|
25
|
-
handle, url,
|
|
26
|
-
displayName: title.replace(/\s*\(@.*$/, "").trim(),
|
|
33
|
+
handle, url, surface,
|
|
34
|
+
displayName: card.title.replace(/\s*\(@.*$/, "").trim(),
|
|
27
35
|
followers: stats?.[1] ?? "",
|
|
28
36
|
posts: stats?.[3] ?? "",
|
|
29
37
|
bio,
|
|
30
|
-
phones:
|
|
31
|
-
links:
|
|
32
|
-
image:
|
|
38
|
+
phones: bioPhones(bio),
|
|
39
|
+
links: bioLinks(bio),
|
|
40
|
+
image: card.image,
|
|
41
|
+
};
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/** Fills in a missing bio or counts from the search-engine snippet, keeping whatever the surface itself gave. */
|
|
45
|
+
function withSnippet(result, snippet) {
|
|
46
|
+
if (!snippet) return result;
|
|
47
|
+
const merged = fromCard(result.handle, result.url, { title: result.displayName ?? "", desc: snippet, image: result.image ?? "" }, result.surface);
|
|
48
|
+
return {
|
|
49
|
+
...result,
|
|
50
|
+
followers: result.followers || merged.followers,
|
|
51
|
+
posts: result.posts || merged.posts,
|
|
52
|
+
bio: result.bio || merged.bio || snippet,
|
|
53
|
+
phones: result.phones?.length ? result.phones : merged.phones,
|
|
54
|
+
links: result.links?.length ? result.links : merged.links,
|
|
55
|
+
};
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/** The public web_profile_info endpoint the profile page calls. No login, but often walled. */
|
|
59
|
+
async function readProfileApi(handle) {
|
|
60
|
+
const url = `https://www.instagram.com/${handle}/`;
|
|
61
|
+
let r;
|
|
62
|
+
try { r = await fetchText(`https://www.instagram.com/api/v1/users/web_profile_info/?username=${encodeURIComponent(handle)}`, { headers: { accept: "application/json", "x-ig-app-id": IG_APP_ID } }); }
|
|
63
|
+
catch { return null; }
|
|
64
|
+
if (!r.ok) return null;
|
|
65
|
+
let user;
|
|
66
|
+
try { user = JSON.parse(r.text)?.data?.user; } catch { return null; }
|
|
67
|
+
if (!user) return null;
|
|
68
|
+
const bio = user.biography ?? "";
|
|
69
|
+
return {
|
|
70
|
+
handle, url, surface: "profile API",
|
|
71
|
+
displayName: user.full_name ?? "",
|
|
72
|
+
followers: user.edge_followed_by?.count != null ? String(user.edge_followed_by.count) : "",
|
|
73
|
+
posts: user.edge_owner_to_timeline_media?.count != null ? String(user.edge_owner_to_timeline_media.count) : "",
|
|
74
|
+
bio,
|
|
75
|
+
phones: bioPhones(bio),
|
|
76
|
+
links: bioLinks(bio),
|
|
77
|
+
image: user.profile_pic_url_hd ?? user.profile_pic_url ?? "",
|
|
33
78
|
};
|
|
34
79
|
}
|
|
35
80
|
|
|
36
|
-
/**
|
|
81
|
+
/** oEmbed (public, though Instagram now usually wants an app token): display name only. */
|
|
82
|
+
async function readOembed(handle, url) {
|
|
83
|
+
let r;
|
|
84
|
+
try { r = await fetchText(`https://api.instagram.com/oembed/?url=${encodeURIComponent(url)}`, { headers: { accept: "application/json" } }); }
|
|
85
|
+
catch { return null; }
|
|
86
|
+
if (!r.ok) return null;
|
|
87
|
+
let j;
|
|
88
|
+
try { j = JSON.parse(r.text); } catch { return null; }
|
|
89
|
+
if (!j?.author_name && !j?.title) return null;
|
|
90
|
+
return { handle, url, surface: "oembed", displayName: j.author_name ?? "", followers: "", posts: "", bio: "", phones: [], links: [], image: "" };
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/**
|
|
94
|
+
* Public Instagram profile metadata, trying the public surfaces in turn: the
|
|
95
|
+
* share card, the web profile API, oEmbed, and finally the bio a search engine
|
|
96
|
+
* indexed (`snippet`). Only honest requests are made; a walled profile is
|
|
97
|
+
* reported as blocked, never worked around. Returns `blocked` only when none of
|
|
98
|
+
* them had anything.
|
|
99
|
+
*/
|
|
100
|
+
export async function readInstagram(handle, { snippet = "" } = {}) {
|
|
101
|
+
const url = `https://www.instagram.com/${handle}/`;
|
|
102
|
+
let reason = "";
|
|
103
|
+
|
|
104
|
+
try {
|
|
105
|
+
const r = await fetchText(url, { headers: { accept: "text/html" } });
|
|
106
|
+
if (!r.ok) reason = `HTTP ${r.status}`;
|
|
107
|
+
else {
|
|
108
|
+
const card = shareCard(r.text);
|
|
109
|
+
if (card) return withSnippet(fromCard(handle, url, card, "share card"), snippet);
|
|
110
|
+
}
|
|
111
|
+
} catch (e) { reason = e.message; }
|
|
112
|
+
|
|
113
|
+
await sleep(SURFACE_PAUSE);
|
|
114
|
+
const fromApi = await readProfileApi(handle);
|
|
115
|
+
if (fromApi) return withSnippet(fromApi, snippet);
|
|
116
|
+
|
|
117
|
+
await sleep(SURFACE_PAUSE);
|
|
118
|
+
const oembed = await readOembed(handle, url);
|
|
119
|
+
if (oembed) return withSnippet(oembed, snippet);
|
|
120
|
+
|
|
121
|
+
if (snippet) return { ...fromCard(handle, url, { title: "", desc: snippet, image: "" }, "search snippet"), partial: true };
|
|
122
|
+
|
|
123
|
+
return { handle, url, blocked: reason || "login wall (no public metadata on any surface)" };
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/**
|
|
127
|
+
* Name and location out of a TripAdvisor restaurant URL slug, e.g.
|
|
128
|
+
* "...-Reviews-Maki_Bar_Medellin-Medellin_Antioquia_Department.html" gives
|
|
129
|
+
* { name: "Maki Bar Medellin", location: "Medellin Antioquia Department" }. The
|
|
130
|
+
* city/region boundary is ambiguous for multi-word cities, so the whole location
|
|
131
|
+
* is kept and the caller treats it as a hint, never as a confirmed locality.
|
|
132
|
+
*/
|
|
133
|
+
export function parseTripadvisorUrl(raw) {
|
|
134
|
+
let path;
|
|
135
|
+
try { path = new URL(raw).pathname; } catch { return null; }
|
|
136
|
+
const segment = path.split("/").find((s) => /-Reviews-|-ShowUserReviews-|-Management-/.test(s));
|
|
137
|
+
if (!segment) return null;
|
|
138
|
+
const after = segment.split(/-(?:Reviews|ShowUserReviews|Management)-/).pop()?.replace(/\.html?$/i, "") ?? "";
|
|
139
|
+
const [namePart = "", locationPart = ""] = after.split("-");
|
|
140
|
+
const words = (s) => decodeEntities(s).replace(/_/g, " ").replace(/\s+/g, " ").trim();
|
|
141
|
+
const name = words(namePart);
|
|
142
|
+
const location = words(locationPart);
|
|
143
|
+
return name ? { name, location } : null;
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/** TripAdvisor: schema.org JSON-LD when the page loads; DataDome often blocks it. The URL slug is always kept. */
|
|
37
147
|
export async function readTripadvisor(url, { browser } = {}) {
|
|
148
|
+
const slug = parseTripadvisorUrl(url);
|
|
38
149
|
const { page, blocked } = await readPage(url, { browser });
|
|
39
|
-
if (!page) return { url, blocked };
|
|
40
|
-
if (!page.jsonld && page.textLength < 500) return { url, blocked: "bot protection (page has no data)" };
|
|
41
|
-
return { url: page.url, jsonld: page.jsonld, description: page.meta.description, hoursText: page.hoursText, images: page.images.slice(0, 10) };
|
|
150
|
+
if (!page) return { url, slug, blocked };
|
|
151
|
+
if (!page.jsonld && page.textLength < 500) return { url, slug, blocked: "bot protection (page has no data)" };
|
|
152
|
+
return { url: page.url, slug, jsonld: page.jsonld, description: page.meta.description, hoursText: page.hoursText, images: page.images.slice(0, 10) };
|
|
42
153
|
}
|
|
43
154
|
|
|
44
155
|
/** A link-in-bio hub (Linktree, Beacons, bio.link): its links, classified like a website's. */
|
|
@@ -8,6 +8,28 @@ export const BROWSER_UA =
|
|
|
8
8
|
|
|
9
9
|
export const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
|
10
10
|
|
|
11
|
+
const HTML_ENT = { amp: "&", lt: "<", gt: ">", quot: '"', apos: "'", nbsp: " " };
|
|
12
|
+
|
|
13
|
+
/** Decodes the HTML entities a page's attributes or text may hold, named and numeric. */
|
|
14
|
+
export const decodeHtml = (s) => String(s).replace(/&(#x[0-9a-f]+|#\d+|[a-z]+);/gi, (m, e) => {
|
|
15
|
+
if (e[0] === "#") {
|
|
16
|
+
const n = e[1].toLowerCase() === "x" ? parseInt(e.slice(2), 16) : Number(e.slice(1));
|
|
17
|
+
return Number.isFinite(n) ? String.fromCodePoint(n) : m;
|
|
18
|
+
}
|
|
19
|
+
return HTML_ENT[e.toLowerCase()] ?? m;
|
|
20
|
+
});
|
|
21
|
+
|
|
22
|
+
// Instagram paths that are not a profile.
|
|
23
|
+
export const IG_RESERVED = new Set(["p", "reel", "reels", "explore", "accounts", "tv", "stories", "share", "direct", "about", "legal", "web", "developer"]);
|
|
24
|
+
|
|
25
|
+
/** The profile handle in an instagram.com URL, or "" for a post, reel or other non-profile path. */
|
|
26
|
+
export const instagramHandle = (url) => {
|
|
27
|
+
try {
|
|
28
|
+
const h = new URL(url).pathname.split("/")[1]?.toLowerCase();
|
|
29
|
+
return h && !IG_RESERVED.has(h) && /^[a-z0-9._]+$/.test(h) ? h : "";
|
|
30
|
+
} catch { return ""; }
|
|
31
|
+
};
|
|
32
|
+
|
|
11
33
|
/** Runs `fn(item, index)` over `items` with at most `size` in flight; results keep the order of `items`. */
|
|
12
34
|
export async function mapPool(items, size, fn) {
|
|
13
35
|
const results = new Array(items.length);
|
|
@@ -2,19 +2,9 @@
|
|
|
2
2
|
// TripAdvisor page when it lets us in) with plain fetch and regexes: no HTML
|
|
3
3
|
// parser dependency. Falls back to Playwright when a page renders client-side.
|
|
4
4
|
import { loadPlaywright } from "../../lib/playwright.mjs";
|
|
5
|
-
import { fetchText, sleep, BROWSER_UA } from "./util.mjs";
|
|
5
|
+
import { fetchText, sleep, BROWSER_UA, decodeHtml as decode, instagramHandle } from "./util.mjs";
|
|
6
6
|
import { fromOsm, fromSpec } from "./hours.mjs";
|
|
7
7
|
|
|
8
|
-
const ENT = { amp: "&", lt: "<", gt: ">", quot: '"', apos: "'", nbsp: " " };
|
|
9
|
-
const decode = (s) =>
|
|
10
|
-
s.replace(/&(#x[0-9a-f]+|#\d+|[a-z]+);/gi, (m, e) => {
|
|
11
|
-
if (e[0] === "#") {
|
|
12
|
-
const n = e[1].toLowerCase() === "x" ? parseInt(e.slice(2), 16) : Number(e.slice(1));
|
|
13
|
-
return Number.isFinite(n) ? String.fromCodePoint(n) : m;
|
|
14
|
-
}
|
|
15
|
-
return ENT[e.toLowerCase()] ?? m;
|
|
16
|
-
});
|
|
17
|
-
|
|
18
8
|
const attrs = (tag) => {
|
|
19
9
|
const out = {};
|
|
20
10
|
for (const m of tag.matchAll(/([a-zA-Z_:][-\w:.]*)\s*(?:=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'>]+)))?/g))
|
|
@@ -25,7 +15,6 @@ const abs = (u, base) => { try { return new URL(u, base).href; } catch { return
|
|
|
25
15
|
const strip = (s) => decode(s.replace(/<[^>]+>/g, " ")).replace(/\s+/g, " ").trim();
|
|
26
16
|
const host = (u) => { try { return new URL(u).hostname.replace(/^www\./, ""); } catch { return ""; } };
|
|
27
17
|
|
|
28
|
-
const IG_RESERVED = new Set(["p", "reel", "reels", "explore", "accounts", "tv", "stories", "share", "direct", "about", "legal", "web", "developer"]);
|
|
29
18
|
const RESERVE = /(opentable|resy\.com|thefork|eltenedor|exploretock|sevenrooms|covermanager|quandoo|tablein|mesa247|bookatable|tablecheck|resos\.com|reservandonos|agendapro|fudo\.)/i;
|
|
30
19
|
const DELIVERY = /(rappi|ubereats|pedidosya|doordash|grubhub|domicilios\.com|didi-food|glovoapp)/i;
|
|
31
20
|
const HUBS = /(linktr\.ee|beacons\.ai|bio\.link|lnk\.bio|linktree\.com|taplink|campsite\.bio|solo\.to|linkin\.bio|flow\.page)/i;
|
|
@@ -102,8 +91,8 @@ export function analyzeHtml(html, base, lines = visibleLines(html)) {
|
|
|
102
91
|
const num = (u.pathname.match(/^\/(\d{7,})/)?.[1]) ?? u.searchParams.get("phone");
|
|
103
92
|
push("whatsapp", num ? num.replace(/\D/g, "") : href);
|
|
104
93
|
} else if (/(^|\.)instagram\.com$/.test(h)) {
|
|
105
|
-
const handle =
|
|
106
|
-
if (handle
|
|
94
|
+
const handle = instagramHandle(href);
|
|
95
|
+
if (handle) ig.set(handle, (ig.get(handle) ?? 0) + 1);
|
|
107
96
|
} else if (/facebook\.com$|fb\.com$/.test(h)) push("facebook", href);
|
|
108
97
|
else if (/tiktok\.com$/.test(h)) push("tiktok", href);
|
|
109
98
|
else if (/tripadvisor\./.test(h)) push("tripadvisor", href);
|
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
// Usage: tablefacts research "Restaurant name" "City, Country" [options]
|
|
3
3
|
import { parseArgs } from "node:util";
|
|
4
4
|
import { research } from "./index.mjs";
|
|
5
|
+
import { summaryLines } from "./lib/report.mjs";
|
|
5
6
|
import { loadEnv } from "../lib/env.mjs";
|
|
6
7
|
import { cliMessage, exitCodeFor } from "../lib/errors.mjs";
|
|
7
8
|
import { consoleLog } from "../lib/log.mjs";
|
|
@@ -10,9 +11,11 @@ const HELP = `Research a restaurant from public sources and write a profile for
|
|
|
10
11
|
|
|
11
12
|
tablefacts research "<name>" "<city, country>" [options]
|
|
12
13
|
|
|
13
|
-
Sources: Google Maps (Places API), OpenStreetMap,
|
|
14
|
-
Instagram, TripAdvisor and link-in-bio pages (Linktree
|
|
15
|
-
each other: the website or Google leads to
|
|
14
|
+
Sources: Google Maps (Places API), OpenStreetMap, a key-free web search, the
|
|
15
|
+
restaurant's website, Instagram, TripAdvisor and link-in-bio pages (Linktree
|
|
16
|
+
and similar). They find each other: the search, website or Google leads to
|
|
17
|
+
Instagram, TripAdvisor and the hub. The web search runs when Google and
|
|
18
|
+
OpenStreetMap find nothing.
|
|
16
19
|
|
|
17
20
|
Options:
|
|
18
21
|
--country <ISO> country code, narrows the search (CO, MX, US...)
|
|
@@ -26,8 +29,9 @@ Options:
|
|
|
26
29
|
--out <dir> output folder (default .tablefacts/research/<slug>)
|
|
27
30
|
-h, --help this text
|
|
28
31
|
|
|
29
|
-
Writes profile.json, report.md and setup-answers.txt.
|
|
30
|
-
in
|
|
32
|
+
Writes profile.json, report.md and setup-answers.txt (and latest.json next to the
|
|
33
|
+
folder, in the default location). GOOGLE_PLACES_API_KEY goes in .env (see
|
|
34
|
+
.env.example); without it OpenStreetMap, the web search and the web pages
|
|
31
35
|
still work, with fewer facts.`;
|
|
32
36
|
|
|
33
37
|
const { values, positionals } = parseArgs({
|
|
@@ -52,9 +56,9 @@ try {
|
|
|
52
56
|
tripadvisor: values.tripadvisor, linktree: values.linktree, render: values.render,
|
|
53
57
|
google: !values["no-google"], photos: Number(values.photos ?? 0), out: values.out, log: consoleLog,
|
|
54
58
|
});
|
|
55
|
-
|
|
56
|
-
console.log(
|
|
57
|
-
|
|
59
|
+
// The lines are built in one place (summaryLines) so tests cover the exact wording the CLI prints.
|
|
60
|
+
console.log();
|
|
61
|
+
for (const line of summaryLines(profile)) console.log(line);
|
|
58
62
|
for (const w of profile.warnings) console.log(`Warning: ${w}`);
|
|
59
63
|
console.log(`
|
|
60
64
|
Wrote ${outDir}
|