tablefacts 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +2 -0
- package/CHANGELOG.md +39 -0
- package/README.md +37 -9
- package/package.json +2 -2
- package/src/lib/types.mjs +7 -2
- package/src/menu/README.md +37 -11
- package/src/menu/cluvi/config.mjs +6 -0
- package/src/menu/cluvi/import.mjs +6 -2
- package/src/menu/lib/db.mjs +81 -17
- package/src/menu/lib/import.mjs +20 -3
- package/src/menu/lib/run.mjs +15 -0
- package/src/menu/lib/tables.mjs +44 -0
- package/src/menu/raw/config.mjs +6 -0
- package/src/menu/raw/import.mjs +5 -2
- package/src/research/README.md +23 -13
- package/src/research/index.mjs +52 -17
- package/src/research/lib/merge.mjs +11 -10
- package/src/research/lib/report.mjs +72 -19
- package/src/research/lib/search.mjs +127 -0
- package/src/research/lib/social.mjs +137 -26
- package/src/research/lib/util.mjs +22 -0
- package/src/research/lib/website.mjs +3 -14
- package/src/research/research.mjs +12 -8
- package/types/lib/types.d.mts +29 -3
- package/types/menu/cluvi/config.d.mts +1 -0
- package/types/menu/cluvi/import.d.mts +2 -0
- package/types/menu/lib/db.d.mts +31 -3
- package/types/menu/lib/import.d.mts +1 -1
- package/types/menu/lib/run.d.mts +3 -0
- package/types/menu/lib/tables.d.mts +18 -0
- package/types/menu/raw/config.d.mts +1 -0
- package/types/menu/raw/import.d.mts +3 -0
- package/types/research/lib/merge.d.mts +4 -1
- package/types/research/lib/report.d.mts +14 -1
- package/types/research/lib/search.d.mts +58 -0
- package/types/research/lib/social.d.mts +32 -31
- package/types/research/lib/util.d.mts +5 -0
- package/types/research/lib/website.d.mts +20 -20
|
@@ -1,44 +1,155 @@
|
|
|
1
1
|
// Instagram, TripAdvisor and link-in-bio hubs. All three are best-effort: they
|
|
2
2
|
// sit behind login walls or bot protection, so each one returns what it could
|
|
3
3
|
// read plus a `blocked` reason, and the report says what is missing.
|
|
4
|
-
import { fetchText } from "./util.mjs";
|
|
4
|
+
import { fetchText, sleep, decodeHtml as decodeEntities } from "./util.mjs";
|
|
5
5
|
import { readPage } from "./website.mjs";
|
|
6
6
|
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
const
|
|
20
|
-
|
|
7
|
+
// Instagram's own web app sends this app id with its public profile request; no
|
|
8
|
+
// login is involved, and the request identifies honestly.
|
|
9
|
+
const IG_APP_ID = "936619743392459";
|
|
10
|
+
// A short gap between the Instagram surfaces: one walled profile should not become a burst.
|
|
11
|
+
const SURFACE_PAUSE = 300;
|
|
12
|
+
|
|
13
|
+
const bioLinks = (bio) => (bio.match(/https?:\/\/[^\s)]+|\b[\w-]+\.(?:com|co|link|bio|ee)\/[\w./-]+/gi) ?? [])
|
|
14
|
+
.map((l) => (/^https?:/.test(l) ? l : `https://${l}`));
|
|
15
|
+
const bioPhones = (bio) => [...bio.matchAll(/\+?\d[\d\s().-]{7,}\d/g)].map((m) => m[0].trim());
|
|
16
|
+
|
|
17
|
+
/** The public share card: "1,234 Followers, 56 Following, 789 Posts - Name (@handle) on Instagram: "bio"". */
|
|
18
|
+
function shareCard(html) {
|
|
19
|
+
const meta = (key) => html.match(new RegExp(`<meta[^>]+(?:property|name)=["']${key}["'][^>]+content=["']([^"']*)["']`, "i"))?.[1] ?? "";
|
|
20
|
+
const desc = decodeEntities(meta("og:description"));
|
|
21
|
+
const title = decodeEntities(meta("og:title"));
|
|
22
|
+
if (!desc && !title) return null;
|
|
23
|
+
return { title, desc, image: meta("og:image") };
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
function fromCard(handle, url, card, surface) {
|
|
27
|
+
const desc = card.desc;
|
|
21
28
|
const stats = desc.match(/([\d.,KMkm]+)\s+Followers?,\s*([\d.,KMkm]+)\s+Following,\s*([\d.,KMkm]+)\s+Posts?/i);
|
|
22
|
-
const
|
|
23
|
-
|
|
29
|
+
const wrapped = desc.match(/on Instagram:\s*["“]([\s\S]*?)["”]?\s*$/i)?.[1]?.trim();
|
|
30
|
+
// A bare bio snippet has no share-card wrapper; a stats-only line has no bio.
|
|
31
|
+
const bio = wrapped ?? (/Followers?,\s*[\d.,KMkm]+\s+Following/i.test(desc) ? "" : desc.trim());
|
|
24
32
|
return {
|
|
25
|
-
handle, url,
|
|
26
|
-
displayName: title.replace(/\s*\(@.*$/, "").trim(),
|
|
33
|
+
handle, url, surface,
|
|
34
|
+
displayName: card.title.replace(/\s*\(@.*$/, "").trim(),
|
|
27
35
|
followers: stats?.[1] ?? "",
|
|
28
36
|
posts: stats?.[3] ?? "",
|
|
29
37
|
bio,
|
|
30
|
-
phones:
|
|
31
|
-
links:
|
|
32
|
-
image:
|
|
38
|
+
phones: bioPhones(bio),
|
|
39
|
+
links: bioLinks(bio),
|
|
40
|
+
image: card.image,
|
|
41
|
+
};
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/** Fills in a missing bio or counts from the search-engine snippet, keeping whatever the surface itself gave. */
|
|
45
|
+
function withSnippet(result, snippet) {
|
|
46
|
+
if (!snippet) return result;
|
|
47
|
+
const merged = fromCard(result.handle, result.url, { title: result.displayName ?? "", desc: snippet, image: result.image ?? "" }, result.surface);
|
|
48
|
+
return {
|
|
49
|
+
...result,
|
|
50
|
+
followers: result.followers || merged.followers,
|
|
51
|
+
posts: result.posts || merged.posts,
|
|
52
|
+
bio: result.bio || merged.bio || snippet,
|
|
53
|
+
phones: result.phones?.length ? result.phones : merged.phones,
|
|
54
|
+
links: result.links?.length ? result.links : merged.links,
|
|
55
|
+
};
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/** The public web_profile_info endpoint the profile page calls. No login, but often walled. */
|
|
59
|
+
async function readProfileApi(handle) {
|
|
60
|
+
const url = `https://www.instagram.com/${handle}/`;
|
|
61
|
+
let r;
|
|
62
|
+
try { r = await fetchText(`https://www.instagram.com/api/v1/users/web_profile_info/?username=${encodeURIComponent(handle)}`, { headers: { accept: "application/json", "x-ig-app-id": IG_APP_ID } }); }
|
|
63
|
+
catch { return null; }
|
|
64
|
+
if (!r.ok) return null;
|
|
65
|
+
let user;
|
|
66
|
+
try { user = JSON.parse(r.text)?.data?.user; } catch { return null; }
|
|
67
|
+
if (!user) return null;
|
|
68
|
+
const bio = user.biography ?? "";
|
|
69
|
+
return {
|
|
70
|
+
handle, url, surface: "profile API",
|
|
71
|
+
displayName: user.full_name ?? "",
|
|
72
|
+
followers: user.edge_followed_by?.count != null ? String(user.edge_followed_by.count) : "",
|
|
73
|
+
posts: user.edge_owner_to_timeline_media?.count != null ? String(user.edge_owner_to_timeline_media.count) : "",
|
|
74
|
+
bio,
|
|
75
|
+
phones: bioPhones(bio),
|
|
76
|
+
links: bioLinks(bio),
|
|
77
|
+
image: user.profile_pic_url_hd ?? user.profile_pic_url ?? "",
|
|
33
78
|
};
|
|
34
79
|
}
|
|
35
80
|
|
|
36
|
-
/**
|
|
81
|
+
/** oEmbed (public, though Instagram now usually wants an app token): display name only. */
|
|
82
|
+
async function readOembed(handle, url) {
|
|
83
|
+
let r;
|
|
84
|
+
try { r = await fetchText(`https://api.instagram.com/oembed/?url=${encodeURIComponent(url)}`, { headers: { accept: "application/json" } }); }
|
|
85
|
+
catch { return null; }
|
|
86
|
+
if (!r.ok) return null;
|
|
87
|
+
let j;
|
|
88
|
+
try { j = JSON.parse(r.text); } catch { return null; }
|
|
89
|
+
if (!j?.author_name && !j?.title) return null;
|
|
90
|
+
return { handle, url, surface: "oembed", displayName: j.author_name ?? "", followers: "", posts: "", bio: "", phones: [], links: [], image: "" };
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/**
|
|
94
|
+
* Public Instagram profile metadata, trying the public surfaces in turn: the
|
|
95
|
+
* share card, the web profile API, oEmbed, and finally the bio a search engine
|
|
96
|
+
* indexed (`snippet`). Only honest requests are made; a walled profile is
|
|
97
|
+
* reported as blocked, never worked around. Returns `blocked` only when none of
|
|
98
|
+
* them had anything.
|
|
99
|
+
*/
|
|
100
|
+
export async function readInstagram(handle, { snippet = "" } = {}) {
|
|
101
|
+
const url = `https://www.instagram.com/${handle}/`;
|
|
102
|
+
let reason = "";
|
|
103
|
+
|
|
104
|
+
try {
|
|
105
|
+
const r = await fetchText(url, { headers: { accept: "text/html" } });
|
|
106
|
+
if (!r.ok) reason = `HTTP ${r.status}`;
|
|
107
|
+
else {
|
|
108
|
+
const card = shareCard(r.text);
|
|
109
|
+
if (card) return withSnippet(fromCard(handle, url, card, "share card"), snippet);
|
|
110
|
+
}
|
|
111
|
+
} catch (e) { reason = e.message; }
|
|
112
|
+
|
|
113
|
+
await sleep(SURFACE_PAUSE);
|
|
114
|
+
const fromApi = await readProfileApi(handle);
|
|
115
|
+
if (fromApi) return withSnippet(fromApi, snippet);
|
|
116
|
+
|
|
117
|
+
await sleep(SURFACE_PAUSE);
|
|
118
|
+
const oembed = await readOembed(handle, url);
|
|
119
|
+
if (oembed) return withSnippet(oembed, snippet);
|
|
120
|
+
|
|
121
|
+
if (snippet) return { ...fromCard(handle, url, { title: "", desc: snippet, image: "" }, "search snippet"), partial: true };
|
|
122
|
+
|
|
123
|
+
return { handle, url, blocked: reason || "login wall (no public metadata on any surface)" };
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/**
|
|
127
|
+
* Name and location out of a TripAdvisor restaurant URL slug, e.g.
|
|
128
|
+
* "...-Reviews-Maki_Bar_Medellin-Medellin_Antioquia_Department.html" gives
|
|
129
|
+
* { name: "Maki Bar Medellin", location: "Medellin Antioquia Department" }. The
|
|
130
|
+
* city/region boundary is ambiguous for multi-word cities, so the whole location
|
|
131
|
+
* is kept and the caller treats it as a hint, never as a confirmed locality.
|
|
132
|
+
*/
|
|
133
|
+
export function parseTripadvisorUrl(raw) {
|
|
134
|
+
let path;
|
|
135
|
+
try { path = new URL(raw).pathname; } catch { return null; }
|
|
136
|
+
const segment = path.split("/").find((s) => /-Reviews-|-ShowUserReviews-|-Management-/.test(s));
|
|
137
|
+
if (!segment) return null;
|
|
138
|
+
const after = segment.split(/-(?:Reviews|ShowUserReviews|Management)-/).pop()?.replace(/\.html?$/i, "") ?? "";
|
|
139
|
+
const [namePart = "", locationPart = ""] = after.split("-");
|
|
140
|
+
const words = (s) => decodeEntities(s).replace(/_/g, " ").replace(/\s+/g, " ").trim();
|
|
141
|
+
const name = words(namePart);
|
|
142
|
+
const location = words(locationPart);
|
|
143
|
+
return name ? { name, location } : null;
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/** TripAdvisor: schema.org JSON-LD when the page loads; DataDome often blocks it. The URL slug is always kept. */
|
|
37
147
|
export async function readTripadvisor(url, { browser } = {}) {
|
|
148
|
+
const slug = parseTripadvisorUrl(url);
|
|
38
149
|
const { page, blocked } = await readPage(url, { browser });
|
|
39
|
-
if (!page) return { url, blocked };
|
|
40
|
-
if (!page.jsonld && page.textLength < 500) return { url, blocked: "bot protection (page has no data)" };
|
|
41
|
-
return { url: page.url, jsonld: page.jsonld, description: page.meta.description, hoursText: page.hoursText, images: page.images.slice(0, 10) };
|
|
150
|
+
if (!page) return { url, slug, blocked };
|
|
151
|
+
if (!page.jsonld && page.textLength < 500) return { url, slug, blocked: "bot protection (page has no data)" };
|
|
152
|
+
return { url: page.url, slug, jsonld: page.jsonld, description: page.meta.description, hoursText: page.hoursText, images: page.images.slice(0, 10) };
|
|
42
153
|
}
|
|
43
154
|
|
|
44
155
|
/** A link-in-bio hub (Linktree, Beacons, bio.link): its links, classified like a website's. */
|
|
@@ -8,6 +8,28 @@ export const BROWSER_UA =
|
|
|
8
8
|
|
|
9
9
|
export const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
|
10
10
|
|
|
11
|
+
const HTML_ENT = { amp: "&", lt: "<", gt: ">", quot: '"', apos: "'", nbsp: " " };
|
|
12
|
+
|
|
13
|
+
/** Decodes the HTML entities a page's attributes or text may hold, named and numeric. */
|
|
14
|
+
export const decodeHtml = (s) => String(s).replace(/&(#x[0-9a-f]+|#\d+|[a-z]+);/gi, (m, e) => {
|
|
15
|
+
if (e[0] === "#") {
|
|
16
|
+
const n = e[1].toLowerCase() === "x" ? parseInt(e.slice(2), 16) : Number(e.slice(1));
|
|
17
|
+
return Number.isFinite(n) ? String.fromCodePoint(n) : m;
|
|
18
|
+
}
|
|
19
|
+
return HTML_ENT[e.toLowerCase()] ?? m;
|
|
20
|
+
});
|
|
21
|
+
|
|
22
|
+
// Instagram paths that are not a profile.
|
|
23
|
+
export const IG_RESERVED = new Set(["p", "reel", "reels", "explore", "accounts", "tv", "stories", "share", "direct", "about", "legal", "web", "developer"]);
|
|
24
|
+
|
|
25
|
+
/** The profile handle in an instagram.com URL, or "" for a post, reel or other non-profile path. */
|
|
26
|
+
export const instagramHandle = (url) => {
|
|
27
|
+
try {
|
|
28
|
+
const h = new URL(url).pathname.split("/")[1]?.toLowerCase();
|
|
29
|
+
return h && !IG_RESERVED.has(h) && /^[a-z0-9._]+$/.test(h) ? h : "";
|
|
30
|
+
} catch { return ""; }
|
|
31
|
+
};
|
|
32
|
+
|
|
11
33
|
/** Runs `fn(item, index)` over `items` with at most `size` in flight; results keep the order of `items`. */
|
|
12
34
|
export async function mapPool(items, size, fn) {
|
|
13
35
|
const results = new Array(items.length);
|
|
@@ -2,19 +2,9 @@
|
|
|
2
2
|
// TripAdvisor page when it lets us in) with plain fetch and regexes: no HTML
|
|
3
3
|
// parser dependency. Falls back to Playwright when a page renders client-side.
|
|
4
4
|
import { loadPlaywright } from "../../lib/playwright.mjs";
|
|
5
|
-
import { fetchText, sleep, BROWSER_UA } from "./util.mjs";
|
|
5
|
+
import { fetchText, sleep, BROWSER_UA, decodeHtml as decode, instagramHandle } from "./util.mjs";
|
|
6
6
|
import { fromOsm, fromSpec } from "./hours.mjs";
|
|
7
7
|
|
|
8
|
-
const ENT = { amp: "&", lt: "<", gt: ">", quot: '"', apos: "'", nbsp: " " };
|
|
9
|
-
const decode = (s) =>
|
|
10
|
-
s.replace(/&(#x[0-9a-f]+|#\d+|[a-z]+);/gi, (m, e) => {
|
|
11
|
-
if (e[0] === "#") {
|
|
12
|
-
const n = e[1].toLowerCase() === "x" ? parseInt(e.slice(2), 16) : Number(e.slice(1));
|
|
13
|
-
return Number.isFinite(n) ? String.fromCodePoint(n) : m;
|
|
14
|
-
}
|
|
15
|
-
return ENT[e.toLowerCase()] ?? m;
|
|
16
|
-
});
|
|
17
|
-
|
|
18
8
|
const attrs = (tag) => {
|
|
19
9
|
const out = {};
|
|
20
10
|
for (const m of tag.matchAll(/([a-zA-Z_:][-\w:.]*)\s*(?:=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'>]+)))?/g))
|
|
@@ -25,7 +15,6 @@ const abs = (u, base) => { try { return new URL(u, base).href; } catch { return
|
|
|
25
15
|
const strip = (s) => decode(s.replace(/<[^>]+>/g, " ")).replace(/\s+/g, " ").trim();
|
|
26
16
|
const host = (u) => { try { return new URL(u).hostname.replace(/^www\./, ""); } catch { return ""; } };
|
|
27
17
|
|
|
28
|
-
const IG_RESERVED = new Set(["p", "reel", "reels", "explore", "accounts", "tv", "stories", "share", "direct", "about", "legal", "web", "developer"]);
|
|
29
18
|
const RESERVE = /(opentable|resy\.com|thefork|eltenedor|exploretock|sevenrooms|covermanager|quandoo|tablein|mesa247|bookatable|tablecheck|resos\.com|reservandonos|agendapro|fudo\.)/i;
|
|
30
19
|
const DELIVERY = /(rappi|ubereats|pedidosya|doordash|grubhub|domicilios\.com|didi-food|glovoapp)/i;
|
|
31
20
|
const HUBS = /(linktr\.ee|beacons\.ai|bio\.link|lnk\.bio|linktree\.com|taplink|campsite\.bio|solo\.to|linkin\.bio|flow\.page)/i;
|
|
@@ -102,8 +91,8 @@ export function analyzeHtml(html, base, lines = visibleLines(html)) {
|
|
|
102
91
|
const num = (u.pathname.match(/^\/(\d{7,})/)?.[1]) ?? u.searchParams.get("phone");
|
|
103
92
|
push("whatsapp", num ? num.replace(/\D/g, "") : href);
|
|
104
93
|
} else if (/(^|\.)instagram\.com$/.test(h)) {
|
|
105
|
-
const handle =
|
|
106
|
-
if (handle
|
|
94
|
+
const handle = instagramHandle(href);
|
|
95
|
+
if (handle) ig.set(handle, (ig.get(handle) ?? 0) + 1);
|
|
107
96
|
} else if (/facebook\.com$|fb\.com$/.test(h)) push("facebook", href);
|
|
108
97
|
else if (/tiktok\.com$/.test(h)) push("tiktok", href);
|
|
109
98
|
else if (/tripadvisor\./.test(h)) push("tripadvisor", href);
|
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
// Usage: tablefacts research "Restaurant name" "City, Country" [options]
|
|
3
3
|
import { parseArgs } from "node:util";
|
|
4
4
|
import { research } from "./index.mjs";
|
|
5
|
+
import { summaryLines } from "./lib/report.mjs";
|
|
5
6
|
import { loadEnv } from "../lib/env.mjs";
|
|
6
7
|
import { cliMessage, exitCodeFor } from "../lib/errors.mjs";
|
|
7
8
|
import { consoleLog } from "../lib/log.mjs";
|
|
@@ -10,9 +11,11 @@ const HELP = `Research a restaurant from public sources and write a profile for
|
|
|
10
11
|
|
|
11
12
|
tablefacts research "<name>" "<city, country>" [options]
|
|
12
13
|
|
|
13
|
-
Sources: Google Maps (Places API), OpenStreetMap,
|
|
14
|
-
Instagram, TripAdvisor and link-in-bio pages (Linktree
|
|
15
|
-
each other: the website or Google leads to
|
|
14
|
+
Sources: Google Maps (Places API), OpenStreetMap, a key-free web search, the
|
|
15
|
+
restaurant's website, Instagram, TripAdvisor and link-in-bio pages (Linktree
|
|
16
|
+
and similar). They find each other: the search, website or Google leads to
|
|
17
|
+
Instagram, TripAdvisor and the hub. The web search runs when Google and
|
|
18
|
+
OpenStreetMap find nothing.
|
|
16
19
|
|
|
17
20
|
Options:
|
|
18
21
|
--country <ISO> country code, narrows the search (CO, MX, US...)
|
|
@@ -26,8 +29,9 @@ Options:
|
|
|
26
29
|
--out <dir> output folder (default .tablefacts/research/<slug>)
|
|
27
30
|
-h, --help this text
|
|
28
31
|
|
|
29
|
-
Writes profile.json, report.md and setup-answers.txt.
|
|
30
|
-
in
|
|
32
|
+
Writes profile.json, report.md and setup-answers.txt (and latest.json next to the
|
|
33
|
+
folder, in the default location). GOOGLE_PLACES_API_KEY goes in .env (see
|
|
34
|
+
.env.example); without it OpenStreetMap, the web search and the web pages
|
|
31
35
|
still work, with fewer facts.`;
|
|
32
36
|
|
|
33
37
|
const { values, positionals } = parseArgs({
|
|
@@ -52,9 +56,9 @@ try {
|
|
|
52
56
|
tripadvisor: values.tripadvisor, linktree: values.linktree, render: values.render,
|
|
53
57
|
google: !values["no-google"], photos: Number(values.photos ?? 0), out: values.out, log: consoleLog,
|
|
54
58
|
});
|
|
55
|
-
|
|
56
|
-
console.log(
|
|
57
|
-
|
|
59
|
+
// The lines are built in one place (summaryLines) so tests cover the exact wording the CLI prints.
|
|
60
|
+
console.log();
|
|
61
|
+
for (const line of summaryLines(profile)) console.log(line);
|
|
58
62
|
for (const w of profile.warnings) console.log(`Warning: ${w}`);
|
|
59
63
|
console.log(`
|
|
60
64
|
Wrote ${outDir}
|
package/types/lib/types.d.mts
CHANGED
|
@@ -253,7 +253,19 @@ export type ImportOptions = {
|
|
|
253
253
|
*/
|
|
254
254
|
json?: string;
|
|
255
255
|
/**
|
|
256
|
-
*
|
|
256
|
+
* This restaurant's table prefix (e.g. "makibar_"); empty for the unprefixed menu_* tables. Default "": importCluvi/importImageMenu fall back to the source config's.
|
|
257
|
+
*/
|
|
258
|
+
tablePrefix?: string;
|
|
259
|
+
/**
|
|
260
|
+
* Write the unprefixed menu_* tables even when other restaurants' prefixed tables exist. Only for a single-restaurant database.
|
|
261
|
+
*/
|
|
262
|
+
allowUnprefixed?: boolean;
|
|
263
|
+
/**
|
|
264
|
+
* Confirm a destructive `replaceAll`, which empties the target tables.
|
|
265
|
+
*/
|
|
266
|
+
yes?: boolean;
|
|
267
|
+
/**
|
|
268
|
+
* Replace the whole menu, not only the categories in this import. Needs `yes`.
|
|
257
269
|
*/
|
|
258
270
|
replaceAll?: boolean;
|
|
259
271
|
/**
|
|
@@ -284,6 +296,7 @@ export type ImportResult = {
|
|
|
284
296
|
*/
|
|
285
297
|
database: {
|
|
286
298
|
label: string;
|
|
299
|
+
tables: string[];
|
|
287
300
|
current: {
|
|
288
301
|
categories: number;
|
|
289
302
|
products: number;
|
|
@@ -297,6 +310,10 @@ export type ImportMenuOptions = ImportOptions & {
|
|
|
297
310
|
title?: string;
|
|
298
311
|
};
|
|
299
312
|
export type CluviConfig = {
|
|
313
|
+
/**
|
|
314
|
+
* This restaurant's table prefix in a shared database (e.g. "cannario_"); empty for the unprefixed menu_* tables.
|
|
315
|
+
*/
|
|
316
|
+
tablePrefix?: string;
|
|
300
317
|
/**
|
|
301
318
|
* Any page of the restaurant's Cluvi menu.
|
|
302
319
|
*/
|
|
@@ -315,6 +332,10 @@ export type CluviConfig = {
|
|
|
315
332
|
sections?: Record<string, string>;
|
|
316
333
|
};
|
|
317
334
|
export type RawConfig = {
|
|
335
|
+
/**
|
|
336
|
+
* This restaurant's table prefix in a shared database (e.g. "mombasa_"); empty for the unprefixed menu_* tables.
|
|
337
|
+
*/
|
|
338
|
+
tablePrefix?: string;
|
|
318
339
|
/**
|
|
319
340
|
* Page with the menu pictures, or a direct image URL.
|
|
320
341
|
*/
|
|
@@ -574,7 +595,10 @@ export type TablefactsErrorCode = 'EUSAGE' | 'ECONFIG' | 'EDEPENDENCY' | 'EFAILE
|
|
|
574
595
|
* @typedef {object} ImportOptions
|
|
575
596
|
* @property {boolean} [dryRun] Check and report, write nothing.
|
|
576
597
|
* @property {string} [json] Also save the extracted menu as JSON at this path (resolved against projectDir).
|
|
577
|
-
* @property {
|
|
598
|
+
* @property {string} [tablePrefix] This restaurant's table prefix (e.g. "makibar_"); empty for the unprefixed menu_* tables. Default "": importCluvi/importImageMenu fall back to the source config's.
|
|
599
|
+
* @property {boolean} [allowUnprefixed] Write the unprefixed menu_* tables even when other restaurants' prefixed tables exist. Only for a single-restaurant database.
|
|
600
|
+
* @property {boolean} [yes] Confirm a destructive `replaceAll`, which empties the target tables.
|
|
601
|
+
* @property {boolean} [replaceAll] Replace the whole menu, not only the categories in this import. Needs `yes`.
|
|
578
602
|
* @property {boolean} [force] Write even if the import has far fewer products than it replaces.
|
|
579
603
|
* @property {string} [databaseUrl] Default: env.SUPABASE_DB_URL.
|
|
580
604
|
* @property {Env} [env] Environment the database URL and keys are read from. Default process.env.
|
|
@@ -587,7 +611,7 @@ export type TablefactsErrorCode = 'EUSAGE' | 'ECONFIG' | 'EDEPENDENCY' | 'EFAILE
|
|
|
587
611
|
* @property {string[]} notes
|
|
588
612
|
* @property {boolean} written
|
|
589
613
|
* @property {boolean} dryRun
|
|
590
|
-
* @property {{ label: string, current: { categories: number, products: number, kept: string[] } } | null} database Null when the database was not reached.
|
|
614
|
+
* @property {{ label: string, tables: string[], current: { categories: number, products: number, kept: string[] } } | null} database Null when the database was not reached.
|
|
591
615
|
*/
|
|
592
616
|
/**
|
|
593
617
|
* @typedef {ImportOptions & { menu: Menu, notes?: string[], title?: string }} ImportMenuOptions
|
|
@@ -595,6 +619,7 @@ export type TablefactsErrorCode = 'EUSAGE' | 'ECONFIG' | 'EDEPENDENCY' | 'EFAILE
|
|
|
595
619
|
/**
|
|
596
620
|
* Restaurant-specific part of the Cluvi source (src/menu/cluvi/config.mjs).
|
|
597
621
|
* @typedef {object} CluviConfig
|
|
622
|
+
* @property {string} [tablePrefix] This restaurant's table prefix in a shared database (e.g. "cannario_"); empty for the unprefixed menu_* tables.
|
|
598
623
|
* @property {string} [url] Any page of the restaurant's Cluvi menu.
|
|
599
624
|
* @property {{ slug: string, name: string, from: string[] }[]} [categories] Cluvi main categories folded into each site category.
|
|
600
625
|
* @property {Record<string, string>} [sections] Cluvi subcategory to section name.
|
|
@@ -602,6 +627,7 @@ export type TablefactsErrorCode = 'EUSAGE' | 'ECONFIG' | 'EDEPENDENCY' | 'EFAILE
|
|
|
602
627
|
/**
|
|
603
628
|
* Restaurant-specific part of the picture-menu source (src/menu/raw/config.mjs).
|
|
604
629
|
* @typedef {object} RawConfig
|
|
630
|
+
* @property {string} [tablePrefix] This restaurant's table prefix in a shared database (e.g. "mombasa_"); empty for the unprefixed menu_* tables.
|
|
605
631
|
* @property {string} [url] Page with the menu pictures, or a direct image URL.
|
|
606
632
|
* @property {string} currency ISO code of the prices.
|
|
607
633
|
* @property {string} [thousands] Thousands separator the menu prints. Default ".".
|
|
@@ -4,6 +4,7 @@
|
|
|
4
4
|
*/
|
|
5
5
|
export declare function fetchCluviMenu({ url, service, lang, config }?: {
|
|
6
6
|
config?: {
|
|
7
|
+
tablePrefix: string;
|
|
7
8
|
url: string;
|
|
8
9
|
categories: {
|
|
9
10
|
slug: string;
|
|
@@ -26,6 +27,7 @@ export declare function fetchCluviMenu({ url, service, lang, config }?: {
|
|
|
26
27
|
}[];
|
|
27
28
|
notes: string[];
|
|
28
29
|
title: string;
|
|
30
|
+
tablePrefix: string;
|
|
29
31
|
}>;
|
|
30
32
|
/**
|
|
31
33
|
* Reads the menu from Cluvi and imports it. Takes importMenu's options too.
|
package/types/menu/lib/db.d.mts
CHANGED
|
@@ -3,19 +3,47 @@ export declare function connect(url: any, { env }?: {}): Promise<{
|
|
|
3
3
|
client: any;
|
|
4
4
|
label: string;
|
|
5
5
|
}>;
|
|
6
|
+
/**
|
|
7
|
+
* Refuses to touch another site's tables before anything is written.
|
|
8
|
+
*
|
|
9
|
+
* Checks the three target tables exist (pointing at the migration when they do not). Then, when
|
|
10
|
+
* no prefix is set, lists the other restaurants' `<prefix>_menu_categories` tables: any of them
|
|
11
|
+
* means this database is shared, so the import stops unless the caller confirmed it targets the
|
|
12
|
+
* one unprefixed set with `allowUnprefixed`. A whole-menu `replaceAll` is refused in that case
|
|
13
|
+
* whatever the flags say, because deleting every unprefixed row could destroy another site.
|
|
14
|
+
* Only reads; returns `{ prefix, tables, others }`.
|
|
15
|
+
* @param {any} client
|
|
16
|
+
* @param {{ tablePrefix?: string, allowUnprefixed?: boolean, replaceAll?: boolean }} [options]
|
|
17
|
+
*/
|
|
18
|
+
export declare function assertTarget(client: any, { tablePrefix, allowUnprefixed, replaceAll }?: {
|
|
19
|
+
tablePrefix?: string;
|
|
20
|
+
allowUnprefixed?: boolean;
|
|
21
|
+
replaceAll?: boolean;
|
|
22
|
+
}): Promise<{
|
|
23
|
+
prefix: string;
|
|
24
|
+
tables: {
|
|
25
|
+
categories: string;
|
|
26
|
+
sections: string;
|
|
27
|
+
products: string;
|
|
28
|
+
};
|
|
29
|
+
others: any;
|
|
30
|
+
}>;
|
|
6
31
|
/** What an import would replace: the categories it writes, or the whole menu. */
|
|
7
|
-
export declare function inspect(client: any, menu: any, { replaceAll }?: {
|
|
32
|
+
export declare function inspect(client: any, menu: any, { replaceAll, tablePrefix }?: {
|
|
8
33
|
replaceAll?: boolean | undefined;
|
|
34
|
+
tablePrefix?: string | undefined;
|
|
9
35
|
}): Promise<any>;
|
|
10
36
|
/**
|
|
11
37
|
* Replaces the menu in one transaction, so readers see the old menu or the new
|
|
12
38
|
* one and a failure changes nothing. By default only the categories in `menu`
|
|
13
39
|
* are replaced (deleting a category cascades to its sections and products);
|
|
14
40
|
* `replaceAll` empties the menu first. Ids are generated here so the rows can
|
|
15
|
-
* be inserted in bulk, a column at a time.
|
|
41
|
+
* be inserted in bulk, a column at a time. `tablePrefix` is the restaurant's
|
|
42
|
+
* own table set; the delete is scoped to it, never to another site's tables.
|
|
16
43
|
*/
|
|
17
|
-
export declare function replaceMenu(client: any, menu: any, { replaceAll }?: {
|
|
44
|
+
export declare function replaceMenu(client: any, menu: any, { replaceAll, tablePrefix }?: {
|
|
18
45
|
replaceAll?: boolean | undefined;
|
|
46
|
+
tablePrefix?: string | undefined;
|
|
19
47
|
}): Promise<{
|
|
20
48
|
categories: any;
|
|
21
49
|
sections: any;
|
|
@@ -8,4 +8,4 @@ export declare function templateHints(menu: any, { projectDir }?: {}): Promise<s
|
|
|
8
8
|
* @param {import('../../lib/types.mjs').ImportMenuOptions} options
|
|
9
9
|
* @returns {Promise<import('../../lib/types.mjs').ImportResult>}
|
|
10
10
|
*/
|
|
11
|
-
export declare function importMenu({ menu, notes, title, dryRun, json, replaceAll, force, databaseUrl, env, projectDir, log: logOption, }?: import('../../lib/types.mjs').ImportMenuOptions): Promise<import('../../lib/types.mjs').ImportResult>;
|
|
11
|
+
export declare function importMenu({ menu, notes, title, dryRun, json, replaceAll, force, tablePrefix, allowUnprefixed, yes, databaseUrl, env, projectDir, log: logOption, }?: import('../../lib/types.mjs').ImportMenuOptions): Promise<import('../../lib/types.mjs').ImportResult>;
|
package/types/menu/lib/run.d.mts
CHANGED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `tablePrefix` when it is empty or a safe `<name>_` prefix (an ECONFIG error otherwise).
|
|
3
|
+
* @param {string} [tablePrefix]
|
|
4
|
+
* @returns {string}
|
|
5
|
+
*/
|
|
6
|
+
export declare function validateTablePrefix(tablePrefix?: string): string;
|
|
7
|
+
/**
|
|
8
|
+
* The three `public` menu tables for a prefix, e.g.
|
|
9
|
+
* `{ categories: "public.makibar_menu_categories", sections: "public.makibar_menu_sections", products: "public.makibar_menu_products" }`.
|
|
10
|
+
* The default (empty prefix) is the unprefixed `public.menu_*` set.
|
|
11
|
+
* @param {string} [tablePrefix]
|
|
12
|
+
* @returns {{ categories: string, sections: string, products: string }}
|
|
13
|
+
*/
|
|
14
|
+
export declare function menuTables(tablePrefix?: string): {
|
|
15
|
+
categories: string;
|
|
16
|
+
sections: string;
|
|
17
|
+
products: string;
|
|
18
|
+
};
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
/** Every image found, and the numbered ones `only` selects. `config` defaults to config.mjs. */
|
|
2
2
|
export declare function findPages({ urls, only, minWidth, config }?: {
|
|
3
3
|
config?: {
|
|
4
|
+
tablePrefix: string;
|
|
4
5
|
url: string;
|
|
5
6
|
currency: string;
|
|
6
7
|
thousands: string;
|
|
@@ -32,6 +33,7 @@ export declare function listMenuImages(options?: import('../../lib/types.mjs').L
|
|
|
32
33
|
/** Reads the pages and normalizes them: `{ menu, notes, title }`, ready for importMenu. */
|
|
33
34
|
export declare function fetchImageMenu({ urls, only, provider, model, minWidth, refresh, apiKey, env, projectDir, config, log: logOption }?: {
|
|
34
35
|
config?: {
|
|
36
|
+
tablePrefix: string;
|
|
35
37
|
url: string;
|
|
36
38
|
currency: string;
|
|
37
39
|
thousands: string;
|
|
@@ -55,6 +57,7 @@ export declare function fetchImageMenu({ urls, only, provider, model, minWidth,
|
|
|
55
57
|
menu: import("../../lib/types.mjs").Menu;
|
|
56
58
|
notes: string[];
|
|
57
59
|
title: string;
|
|
60
|
+
tablePrefix: string;
|
|
58
61
|
}>;
|
|
59
62
|
/**
|
|
60
63
|
* Reads the menu from pictures with a vision model and imports it. Takes importMenu's options too.
|
|
@@ -1,9 +1,10 @@
|
|
|
1
|
-
export declare function buildProfile({ query, google, osm, site, hub, instagram, tripadvisor }: {
|
|
1
|
+
export declare function buildProfile({ query, google, osm, site, hub, instagram, tripadvisor, search }: {
|
|
2
2
|
google: any;
|
|
3
3
|
hub: any;
|
|
4
4
|
instagram: any;
|
|
5
5
|
osm: any;
|
|
6
6
|
query: any;
|
|
7
|
+
search: any;
|
|
7
8
|
site: any;
|
|
8
9
|
tripadvisor: any;
|
|
9
10
|
}): {
|
|
@@ -82,6 +83,8 @@ export declare function buildProfile({ query, google, osm, site, hub, instagram,
|
|
|
82
83
|
followers: any;
|
|
83
84
|
posts: any;
|
|
84
85
|
blocked: any;
|
|
86
|
+
partial: any;
|
|
87
|
+
surface: any;
|
|
85
88
|
} | null;
|
|
86
89
|
};
|
|
87
90
|
};
|
|
@@ -1,4 +1,17 @@
|
|
|
1
|
-
|
|
1
|
+
/**
|
|
2
|
+
* Splits the fields into what was discovered from a source, what was only echoed
|
|
3
|
+
* back from the user's own input, and low-confidence guesses, so a summary can
|
|
4
|
+
* say "N facts, M from your input" instead of counting all of them as found.
|
|
5
|
+
*/
|
|
6
|
+
export declare function fieldSummary(p: any): {
|
|
7
|
+
discovered: string[];
|
|
8
|
+
echoed: string[];
|
|
9
|
+
guessed: string[];
|
|
10
|
+
};
|
|
11
|
+
/** The CLI summary: what came from a source, what was only echoed, what was a low guess. */
|
|
12
|
+
export declare function summaryLines(p: any): string[];
|
|
13
|
+
export declare function renderReport(p: any, { notes, photos, kept }: {
|
|
14
|
+
kept?: never[] | undefined;
|
|
2
15
|
notes: any;
|
|
3
16
|
photos: any;
|
|
4
17
|
}): string;
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
/** Unwraps DuckDuckGo's `//duckduckgo.com/l/?uddg=<url>` redirect to the real result URL. */
|
|
2
|
+
export declare function resultUrl(href: any): string | null;
|
|
3
|
+
/** Pulls { url, title, snippet } out of DuckDuckGo's HTML result page (no DOM parser). */
|
|
4
|
+
export declare function parseSearchResults(html: any): {
|
|
5
|
+
url: string;
|
|
6
|
+
title: string;
|
|
7
|
+
snippet: string;
|
|
8
|
+
}[];
|
|
9
|
+
/**
|
|
10
|
+
* Sorts results into the links the pipeline can follow. `website` is the best
|
|
11
|
+
* non-directory result whose title/host looks like the restaurant's name; a
|
|
12
|
+
* result that only matches the place is kept as a weaker candidate.
|
|
13
|
+
*/
|
|
14
|
+
export declare function classifyResults(results: any, { name }?: {
|
|
15
|
+
name?: string | undefined;
|
|
16
|
+
}): {
|
|
17
|
+
website: any;
|
|
18
|
+
instagram: never;
|
|
19
|
+
tripadvisor: never;
|
|
20
|
+
facebook: never;
|
|
21
|
+
tiktok: never;
|
|
22
|
+
mapsUrl: never;
|
|
23
|
+
hubs: never[];
|
|
24
|
+
candidates: {
|
|
25
|
+
url: any;
|
|
26
|
+
title: any;
|
|
27
|
+
}[];
|
|
28
|
+
};
|
|
29
|
+
/**
|
|
30
|
+
* Runs the search for a restaurant and returns its candidate links. Tries every
|
|
31
|
+
* endpoint until one returns results; if all answer but none has results, the
|
|
32
|
+
* empty result is returned so the caller still says "no results" rather than
|
|
33
|
+
* "failed". Throws only when no endpoint answers.
|
|
34
|
+
*/
|
|
35
|
+
export declare function searchWeb({ name, location }: {
|
|
36
|
+
location?: string | undefined;
|
|
37
|
+
name: any;
|
|
38
|
+
}): Promise<{
|
|
39
|
+
website: any;
|
|
40
|
+
instagram: never;
|
|
41
|
+
tripadvisor: never;
|
|
42
|
+
facebook: never;
|
|
43
|
+
tiktok: never;
|
|
44
|
+
mapsUrl: never;
|
|
45
|
+
hubs: never[];
|
|
46
|
+
candidates: {
|
|
47
|
+
url: any;
|
|
48
|
+
title: any;
|
|
49
|
+
}[];
|
|
50
|
+
source: string;
|
|
51
|
+
url: string;
|
|
52
|
+
query: string;
|
|
53
|
+
results: {
|
|
54
|
+
url: string;
|
|
55
|
+
title: string;
|
|
56
|
+
snippet: string;
|
|
57
|
+
}[];
|
|
58
|
+
}>;
|