tablefacts 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +3 -1
- package/CHANGELOG.md +61 -0
- package/README.md +67 -19
- package/package.json +14 -3
- package/src/lib/types.mjs +16 -4
- package/src/menu/README.md +63 -15
- package/src/menu/cluvi/config.mjs +6 -0
- package/src/menu/cluvi/import.mjs +6 -2
- package/src/menu/lib/db.mjs +81 -17
- package/src/menu/lib/import.mjs +20 -3
- package/src/menu/lib/run.mjs +18 -0
- package/src/menu/lib/tables.mjs +44 -0
- package/src/menu/raw/config.mjs +12 -2
- package/src/menu/raw/extract.mjs +36 -15
- package/src/menu/raw/images.mjs +109 -0
- package/src/menu/raw/import.mjs +330 -51
- package/src/menu/raw/normalize.mjs +15 -4
- package/src/menu/raw/pdf.mjs +330 -0
- package/src/menu/raw/pdfjs.mjs +57 -0
- package/src/menu/raw/source.mjs +9 -2
- package/src/menu/raw/vision.mjs +124 -65
- package/src/research/README.md +23 -13
- package/src/research/index.mjs +52 -17
- package/src/research/lib/merge.mjs +11 -10
- package/src/research/lib/report.mjs +72 -19
- package/src/research/lib/search.mjs +127 -0
- package/src/research/lib/social.mjs +137 -26
- package/src/research/lib/util.mjs +22 -0
- package/src/research/lib/website.mjs +3 -14
- package/src/research/research.mjs +12 -8
- package/types/lib/types.d.mts +70 -6
- package/types/menu/cluvi/config.d.mts +1 -0
- package/types/menu/cluvi/import.d.mts +2 -0
- package/types/menu/lib/db.d.mts +31 -3
- package/types/menu/lib/import.d.mts +1 -1
- package/types/menu/lib/run.d.mts +6 -0
- package/types/menu/lib/tables.d.mts +18 -0
- package/types/menu/raw/config.d.mts +2 -0
- package/types/menu/raw/images.d.mts +27 -0
- package/types/menu/raw/import.d.mts +36 -7
- package/types/menu/raw/normalize.d.mts +7 -1
- package/types/menu/raw/pdf.d.mts +98 -0
- package/types/menu/raw/pdfjs.d.mts +12 -0
- package/types/menu/raw/source.d.mts +2 -0
- package/types/menu/raw/vision.d.mts +11 -1
- package/types/research/lib/merge.d.mts +4 -1
- package/types/research/lib/report.d.mts +14 -1
- package/types/research/lib/search.d.mts +58 -0
- package/types/research/lib/social.d.mts +32 -31
- package/types/research/lib/util.d.mts +5 -0
- package/types/research/lib/website.d.mts +20 -20
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
/** Unwraps DuckDuckGo's `//duckduckgo.com/l/?uddg=<url>` redirect to the real result URL. */
|
|
2
|
+
export declare function resultUrl(href: any): string | null;
|
|
3
|
+
/** Pulls { url, title, snippet } out of DuckDuckGo's HTML result page (no DOM parser). */
|
|
4
|
+
export declare function parseSearchResults(html: any): {
|
|
5
|
+
url: string;
|
|
6
|
+
title: string;
|
|
7
|
+
snippet: string;
|
|
8
|
+
}[];
|
|
9
|
+
/**
|
|
10
|
+
* Sorts results into the links the pipeline can follow. `website` is the best
|
|
11
|
+
* non-directory result whose title/host looks like the restaurant's name; a
|
|
12
|
+
* result that only matches the place is kept as a weaker candidate.
|
|
13
|
+
*/
|
|
14
|
+
export declare function classifyResults(results: any, { name }?: {
|
|
15
|
+
name?: string | undefined;
|
|
16
|
+
}): {
|
|
17
|
+
website: any;
|
|
18
|
+
instagram: never;
|
|
19
|
+
tripadvisor: never;
|
|
20
|
+
facebook: never;
|
|
21
|
+
tiktok: never;
|
|
22
|
+
mapsUrl: never;
|
|
23
|
+
hubs: never[];
|
|
24
|
+
candidates: {
|
|
25
|
+
url: any;
|
|
26
|
+
title: any;
|
|
27
|
+
}[];
|
|
28
|
+
};
|
|
29
|
+
/**
|
|
30
|
+
* Runs the search for a restaurant and returns its candidate links. Tries every
|
|
31
|
+
* endpoint until one returns results; if all answer but none has results, the
|
|
32
|
+
* empty result is returned so the caller still says "no results" rather than
|
|
33
|
+
* "failed". Throws only when no endpoint answers.
|
|
34
|
+
*/
|
|
35
|
+
export declare function searchWeb({ name, location }: {
|
|
36
|
+
location?: string | undefined;
|
|
37
|
+
name: any;
|
|
38
|
+
}): Promise<{
|
|
39
|
+
website: any;
|
|
40
|
+
instagram: never;
|
|
41
|
+
tripadvisor: never;
|
|
42
|
+
facebook: never;
|
|
43
|
+
tiktok: never;
|
|
44
|
+
mapsUrl: never;
|
|
45
|
+
hubs: never[];
|
|
46
|
+
candidates: {
|
|
47
|
+
url: any;
|
|
48
|
+
title: any;
|
|
49
|
+
}[];
|
|
50
|
+
source: string;
|
|
51
|
+
url: string;
|
|
52
|
+
query: string;
|
|
53
|
+
results: {
|
|
54
|
+
url: string;
|
|
55
|
+
title: string;
|
|
56
|
+
snippet: string;
|
|
57
|
+
}[];
|
|
58
|
+
}>;
|
|
@@ -1,34 +1,31 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Public Instagram profile metadata
|
|
3
|
-
* share card
|
|
4
|
-
*
|
|
2
|
+
* Public Instagram profile metadata, trying the public surfaces in turn: the
|
|
3
|
+
* share card, the web profile API, oEmbed, and finally the bio a search engine
|
|
4
|
+
* indexed (`snippet`). Only honest requests are made; a walled profile is
|
|
5
|
+
* reported as blocked, never worked around. Returns `blocked` only when none of
|
|
6
|
+
* them had anything.
|
|
5
7
|
*/
|
|
6
|
-
export declare function readInstagram(handle: any
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
displayName: any;
|
|
22
|
-
followers: any;
|
|
23
|
-
posts: any;
|
|
24
|
-
bio: any;
|
|
25
|
-
phones: any[];
|
|
26
|
-
links: any;
|
|
27
|
-
image: any;
|
|
28
|
-
}>;
|
|
29
|
-
/** TripAdvisor: schema.org JSON-LD when the page loads; DataDome often blocks it. */
|
|
8
|
+
export declare function readInstagram(handle: any, { snippet }?: {
|
|
9
|
+
snippet?: string | undefined;
|
|
10
|
+
}): Promise<any>;
|
|
11
|
+
/**
|
|
12
|
+
* Name and location out of a TripAdvisor restaurant URL slug, e.g.
|
|
13
|
+
* "...-Reviews-Maki_Bar_Medellin-Medellin_Antioquia_Department.html" gives
|
|
14
|
+
* { name: "Maki Bar Medellin", location: "Medellin Antioquia Department" }. The
|
|
15
|
+
* city/region boundary is ambiguous for multi-word cities, so the whole location
|
|
16
|
+
* is kept and the caller treats it as a hint, never as a confirmed locality.
|
|
17
|
+
*/
|
|
18
|
+
export declare function parseTripadvisorUrl(raw: any): {
|
|
19
|
+
name: string;
|
|
20
|
+
location: string;
|
|
21
|
+
} | null;
|
|
22
|
+
/** TripAdvisor: schema.org JSON-LD when the page loads; DataDome often blocks it. The URL slug is always kept. */
|
|
30
23
|
export declare function readTripadvisor(url: any, { browser }?: {}): Promise<{
|
|
31
24
|
url: any;
|
|
25
|
+
slug: {
|
|
26
|
+
name: string;
|
|
27
|
+
location: string;
|
|
28
|
+
} | null;
|
|
32
29
|
blocked: string;
|
|
33
30
|
jsonld?: undefined;
|
|
34
31
|
description?: undefined;
|
|
@@ -37,6 +34,10 @@ export declare function readTripadvisor(url: any, { browser }?: {}): Promise<{
|
|
|
37
34
|
} | {
|
|
38
35
|
blocked?: undefined;
|
|
39
36
|
url: any;
|
|
37
|
+
slug: {
|
|
38
|
+
name: string;
|
|
39
|
+
location: string;
|
|
40
|
+
} | null;
|
|
40
41
|
jsonld: {
|
|
41
42
|
name: any;
|
|
42
43
|
telephone: any;
|
|
@@ -68,23 +69,23 @@ export declare function readTripadvisor(url: any, { browser }?: {}): Promise<{
|
|
|
68
69
|
} | null;
|
|
69
70
|
} | null;
|
|
70
71
|
description: any;
|
|
71
|
-
hoursText:
|
|
72
|
+
hoursText: string[];
|
|
72
73
|
images: any[];
|
|
73
74
|
}>;
|
|
74
75
|
/** A link-in-bio hub (Linktree, Beacons, bio.link): its links, classified like a website's. */
|
|
75
76
|
export declare function readHub(url: any, { browser }?: {}): Promise<{
|
|
76
|
-
links?: undefined;
|
|
77
77
|
images?: undefined;
|
|
78
78
|
url: any;
|
|
79
79
|
blocked: string;
|
|
80
80
|
title?: undefined;
|
|
81
81
|
meta?: undefined;
|
|
82
|
+
links?: undefined;
|
|
82
83
|
anchors?: undefined;
|
|
83
84
|
logos?: undefined;
|
|
84
85
|
} | {
|
|
85
86
|
blocked?: undefined;
|
|
86
87
|
url: any;
|
|
87
|
-
title:
|
|
88
|
+
title: string;
|
|
88
89
|
meta: {
|
|
89
90
|
siteName: any;
|
|
90
91
|
title: any;
|
|
@@ -108,7 +109,7 @@ export declare function readHub(url: any, { browser }?: {}): Promise<{
|
|
|
108
109
|
};
|
|
109
110
|
anchors: {
|
|
110
111
|
href: string | null;
|
|
111
|
-
text:
|
|
112
|
+
text: string;
|
|
112
113
|
}[];
|
|
113
114
|
logos: {
|
|
114
115
|
url: string | null;
|
|
@@ -2,6 +2,11 @@ import { fold } from "../../lib/text.mjs";
|
|
|
2
2
|
export { fold };
|
|
3
3
|
export declare const BROWSER_UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/130.0 Safari/537.36 cannario-research";
|
|
4
4
|
export declare const sleep: (ms: any) => Promise<any>;
|
|
5
|
+
/** Decodes the HTML entities a page's attributes or text may hold, named and numeric. */
|
|
6
|
+
export declare const decodeHtml: (s: any) => string;
|
|
7
|
+
export declare const IG_RESERVED: Set<string>;
|
|
8
|
+
/** The profile handle in an instagram.com URL, or "" for a post, reel or other non-profile path. */
|
|
9
|
+
export declare const instagramHandle: (url: any) => string;
|
|
5
10
|
/** Runs `fn(item, index)` over `items` with at most `size` in flight; results keep the order of `items`. */
|
|
6
11
|
export declare function mapPool(items: any, size: any, fn: any): Promise<any[]>;
|
|
7
12
|
export declare const MAX_BODY_BYTES: number;
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/** Pulls every fact a page holds. `base` resolves relative links. */
|
|
2
|
-
export declare function analyzeHtml(html: any, base: any, lines?:
|
|
2
|
+
export declare function analyzeHtml(html: any, base: any, lines?: string[]): {
|
|
3
3
|
url: any;
|
|
4
|
-
title:
|
|
4
|
+
title: string;
|
|
5
5
|
lang: any;
|
|
6
6
|
meta: {
|
|
7
7
|
siteName: any;
|
|
@@ -59,13 +59,13 @@ export declare function analyzeHtml(html: any, base: any, lines?: any): {
|
|
|
59
59
|
url: string | null;
|
|
60
60
|
kind: any;
|
|
61
61
|
}[];
|
|
62
|
-
textLength:
|
|
63
|
-
headings:
|
|
64
|
-
paragraphs:
|
|
65
|
-
hoursText:
|
|
62
|
+
textLength: number;
|
|
63
|
+
headings: string[];
|
|
64
|
+
paragraphs: string[];
|
|
65
|
+
hoursText: string[];
|
|
66
66
|
anchors: {
|
|
67
67
|
href: string | null;
|
|
68
|
-
text:
|
|
68
|
+
text: string;
|
|
69
69
|
}[];
|
|
70
70
|
};
|
|
71
71
|
/**
|
|
@@ -86,7 +86,7 @@ export declare function readPage(url: any, { renderJs, browser }?: {
|
|
|
86
86
|
blocked?: undefined;
|
|
87
87
|
page: {
|
|
88
88
|
url: any;
|
|
89
|
-
title:
|
|
89
|
+
title: string;
|
|
90
90
|
lang: any;
|
|
91
91
|
meta: {
|
|
92
92
|
siteName: any;
|
|
@@ -144,13 +144,13 @@ export declare function readPage(url: any, { renderJs, browser }?: {
|
|
|
144
144
|
url: string | null;
|
|
145
145
|
kind: any;
|
|
146
146
|
}[];
|
|
147
|
-
textLength:
|
|
148
|
-
headings:
|
|
149
|
-
paragraphs:
|
|
150
|
-
hoursText:
|
|
147
|
+
textLength: number;
|
|
148
|
+
headings: string[];
|
|
149
|
+
paragraphs: string[];
|
|
150
|
+
hoursText: string[];
|
|
151
151
|
anchors: {
|
|
152
152
|
href: string | null;
|
|
153
|
-
text:
|
|
153
|
+
text: string;
|
|
154
154
|
}[];
|
|
155
155
|
rendered: boolean;
|
|
156
156
|
};
|
|
@@ -158,7 +158,7 @@ export declare function readPage(url: any, { renderJs, browser }?: {
|
|
|
158
158
|
blocked?: undefined;
|
|
159
159
|
page: {
|
|
160
160
|
url: any;
|
|
161
|
-
title:
|
|
161
|
+
title: string;
|
|
162
162
|
lang: any;
|
|
163
163
|
meta: {
|
|
164
164
|
siteName: any;
|
|
@@ -216,13 +216,13 @@ export declare function readPage(url: any, { renderJs, browser }?: {
|
|
|
216
216
|
url: string | null;
|
|
217
217
|
kind: any;
|
|
218
218
|
}[];
|
|
219
|
-
textLength:
|
|
220
|
-
headings:
|
|
221
|
-
paragraphs:
|
|
222
|
-
hoursText:
|
|
219
|
+
textLength: number;
|
|
220
|
+
headings: string[];
|
|
221
|
+
paragraphs: string[];
|
|
222
|
+
hoursText: string[];
|
|
223
223
|
anchors: {
|
|
224
224
|
href: string | null;
|
|
225
|
-
text:
|
|
225
|
+
text: string;
|
|
226
226
|
}[];
|
|
227
227
|
};
|
|
228
228
|
}>;
|
|
@@ -234,7 +234,7 @@ export declare function scrapeSite(startUrl: any, { renderJs, maxPages, browser
|
|
|
234
234
|
url: any;
|
|
235
235
|
pages: {
|
|
236
236
|
url: any;
|
|
237
|
-
title:
|
|
237
|
+
title: string;
|
|
238
238
|
rendered: boolean;
|
|
239
239
|
}[];
|
|
240
240
|
lang: any;
|