tablefacts 0.1.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/.env.example +3 -1
  2. package/CHANGELOG.md +61 -0
  3. package/README.md +67 -19
  4. package/package.json +14 -3
  5. package/src/lib/types.mjs +16 -4
  6. package/src/menu/README.md +63 -15
  7. package/src/menu/cluvi/config.mjs +6 -0
  8. package/src/menu/cluvi/import.mjs +6 -2
  9. package/src/menu/lib/db.mjs +81 -17
  10. package/src/menu/lib/import.mjs +20 -3
  11. package/src/menu/lib/run.mjs +18 -0
  12. package/src/menu/lib/tables.mjs +44 -0
  13. package/src/menu/raw/config.mjs +12 -2
  14. package/src/menu/raw/extract.mjs +36 -15
  15. package/src/menu/raw/images.mjs +109 -0
  16. package/src/menu/raw/import.mjs +330 -51
  17. package/src/menu/raw/normalize.mjs +15 -4
  18. package/src/menu/raw/pdf.mjs +330 -0
  19. package/src/menu/raw/pdfjs.mjs +57 -0
  20. package/src/menu/raw/source.mjs +9 -2
  21. package/src/menu/raw/vision.mjs +124 -65
  22. package/src/research/README.md +23 -13
  23. package/src/research/index.mjs +52 -17
  24. package/src/research/lib/merge.mjs +11 -10
  25. package/src/research/lib/report.mjs +72 -19
  26. package/src/research/lib/search.mjs +127 -0
  27. package/src/research/lib/social.mjs +137 -26
  28. package/src/research/lib/util.mjs +22 -0
  29. package/src/research/lib/website.mjs +3 -14
  30. package/src/research/research.mjs +12 -8
  31. package/types/lib/types.d.mts +70 -6
  32. package/types/menu/cluvi/config.d.mts +1 -0
  33. package/types/menu/cluvi/import.d.mts +2 -0
  34. package/types/menu/lib/db.d.mts +31 -3
  35. package/types/menu/lib/import.d.mts +1 -1
  36. package/types/menu/lib/run.d.mts +6 -0
  37. package/types/menu/lib/tables.d.mts +18 -0
  38. package/types/menu/raw/config.d.mts +2 -0
  39. package/types/menu/raw/images.d.mts +27 -0
  40. package/types/menu/raw/import.d.mts +36 -7
  41. package/types/menu/raw/normalize.d.mts +7 -1
  42. package/types/menu/raw/pdf.d.mts +98 -0
  43. package/types/menu/raw/pdfjs.d.mts +12 -0
  44. package/types/menu/raw/source.d.mts +2 -0
  45. package/types/menu/raw/vision.d.mts +11 -1
  46. package/types/research/lib/merge.d.mts +4 -1
  47. package/types/research/lib/report.d.mts +14 -1
  48. package/types/research/lib/search.d.mts +58 -0
  49. package/types/research/lib/social.d.mts +32 -31
  50. package/types/research/lib/util.d.mts +5 -0
  51. package/types/research/lib/website.d.mts +20 -20
@@ -0,0 +1,58 @@
1
+ /** Unwraps DuckDuckGo's `//duckduckgo.com/l/?uddg=<url>` redirect to the real result URL. */
2
+ export declare function resultUrl(href: any): string | null;
3
+ /** Pulls { url, title, snippet } out of DuckDuckGo's HTML result page (no DOM parser). */
4
+ export declare function parseSearchResults(html: any): {
5
+ url: string;
6
+ title: string;
7
+ snippet: string;
8
+ }[];
9
+ /**
10
+ * Sorts results into the links the pipeline can follow. `website` is the best
11
+ * non-directory result whose title/host looks like the restaurant's name; a
12
+ * result that only matches the place is kept as a weaker candidate.
13
+ */
14
+ export declare function classifyResults(results: any, { name }?: {
15
+ name?: string | undefined;
16
+ }): {
17
+ website: any;
18
+ instagram: never;
19
+ tripadvisor: never;
20
+ facebook: never;
21
+ tiktok: never;
22
+ mapsUrl: never;
23
+ hubs: never[];
24
+ candidates: {
25
+ url: any;
26
+ title: any;
27
+ }[];
28
+ };
29
+ /**
30
+ * Runs the search for a restaurant and returns its candidate links. Tries every
31
+ * endpoint until one returns results; if all answer but none has results, the
32
+ * empty result is returned so the caller still says "no results" rather than
33
+ * "failed". Throws only when no endpoint answers.
34
+ */
35
+ export declare function searchWeb({ name, location }: {
36
+ location?: string | undefined;
37
+ name: any;
38
+ }): Promise<{
39
+ website: any;
40
+ instagram: never;
41
+ tripadvisor: never;
42
+ facebook: never;
43
+ tiktok: never;
44
+ mapsUrl: never;
45
+ hubs: never[];
46
+ candidates: {
47
+ url: any;
48
+ title: any;
49
+ }[];
50
+ source: string;
51
+ url: string;
52
+ query: string;
53
+ results: {
54
+ url: string;
55
+ title: string;
56
+ snippet: string;
57
+ }[];
58
+ }>;
@@ -1,34 +1,31 @@
1
1
  /**
2
- * Public Instagram profile metadata. Without a login the page only exposes the
3
- * share card: "1,234 Followers, 56 Following, 789 Posts - Name (@handle) on
4
- * Instagram: "bio"". The bio often carries the phone, the area and a hub link.
2
+ * Public Instagram profile metadata, trying the public surfaces in turn: the
3
+ * share card, the web profile API, oEmbed, and finally the bio a search engine
4
+ * indexed (`snippet`). Only honest requests are made; a walled profile is
5
+ * reported as blocked, never worked around. Returns `blocked` only when none of
6
+ * them had anything.
5
7
  */
6
- export declare function readInstagram(handle: any): Promise<{
7
- handle: any;
8
- url: string;
9
- blocked: any;
10
- displayName?: undefined;
11
- followers?: undefined;
12
- posts?: undefined;
13
- bio?: undefined;
14
- phones?: undefined;
15
- links?: undefined;
16
- image?: undefined;
17
- } | {
18
- blocked?: undefined;
19
- handle: any;
20
- url: string;
21
- displayName: any;
22
- followers: any;
23
- posts: any;
24
- bio: any;
25
- phones: any[];
26
- links: any;
27
- image: any;
28
- }>;
29
- /** TripAdvisor: schema.org JSON-LD when the page loads; DataDome often blocks it. */
8
+ export declare function readInstagram(handle: any, { snippet }?: {
9
+ snippet?: string | undefined;
10
+ }): Promise<any>;
11
+ /**
12
+ * Name and location out of a TripAdvisor restaurant URL slug, e.g.
13
+ * "...-Reviews-Maki_Bar_Medellin-Medellin_Antioquia_Department.html" gives
14
+ * { name: "Maki Bar Medellin", location: "Medellin Antioquia Department" }. The
15
+ * city/region boundary is ambiguous for multi-word cities, so the whole location
16
+ * is kept and the caller treats it as a hint, never as a confirmed locality.
17
+ */
18
+ export declare function parseTripadvisorUrl(raw: any): {
19
+ name: string;
20
+ location: string;
21
+ } | null;
22
+ /** TripAdvisor: schema.org JSON-LD when the page loads; DataDome often blocks it. The URL slug is always kept. */
30
23
  export declare function readTripadvisor(url: any, { browser }?: {}): Promise<{
31
24
  url: any;
25
+ slug: {
26
+ name: string;
27
+ location: string;
28
+ } | null;
32
29
  blocked: string;
33
30
  jsonld?: undefined;
34
31
  description?: undefined;
@@ -37,6 +34,10 @@ export declare function readTripadvisor(url: any, { browser }?: {}): Promise<{
37
34
  } | {
38
35
  blocked?: undefined;
39
36
  url: any;
37
+ slug: {
38
+ name: string;
39
+ location: string;
40
+ } | null;
40
41
  jsonld: {
41
42
  name: any;
42
43
  telephone: any;
@@ -68,23 +69,23 @@ export declare function readTripadvisor(url: any, { browser }?: {}): Promise<{
68
69
  } | null;
69
70
  } | null;
70
71
  description: any;
71
- hoursText: any;
72
+ hoursText: string[];
72
73
  images: any[];
73
74
  }>;
74
75
  /** A link-in-bio hub (Linktree, Beacons, bio.link): its links, classified like a website's. */
75
76
  export declare function readHub(url: any, { browser }?: {}): Promise<{
76
- links?: undefined;
77
77
  images?: undefined;
78
78
  url: any;
79
79
  blocked: string;
80
80
  title?: undefined;
81
81
  meta?: undefined;
82
+ links?: undefined;
82
83
  anchors?: undefined;
83
84
  logos?: undefined;
84
85
  } | {
85
86
  blocked?: undefined;
86
87
  url: any;
87
- title: any;
88
+ title: string;
88
89
  meta: {
89
90
  siteName: any;
90
91
  title: any;
@@ -108,7 +109,7 @@ export declare function readHub(url: any, { browser }?: {}): Promise<{
108
109
  };
109
110
  anchors: {
110
111
  href: string | null;
111
- text: any;
112
+ text: string;
112
113
  }[];
113
114
  logos: {
114
115
  url: string | null;
@@ -2,6 +2,11 @@ import { fold } from "../../lib/text.mjs";
2
2
  export { fold };
3
3
  export declare const BROWSER_UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/130.0 Safari/537.36 cannario-research";
4
4
  export declare const sleep: (ms: any) => Promise<any>;
5
+ /** Decodes the HTML entities a page's attributes or text may hold, named and numeric. */
6
+ export declare const decodeHtml: (s: any) => string;
7
+ export declare const IG_RESERVED: Set<string>;
8
+ /** The profile handle in an instagram.com URL, or "" for a post, reel or other non-profile path. */
9
+ export declare const instagramHandle: (url: any) => string;
5
10
  /** Runs `fn(item, index)` over `items` with at most `size` in flight; results keep the order of `items`. */
6
11
  export declare function mapPool(items: any, size: any, fn: any): Promise<any[]>;
7
12
  export declare const MAX_BODY_BYTES: number;
@@ -1,7 +1,7 @@
1
1
  /** Pulls every fact a page holds. `base` resolves relative links. */
2
- export declare function analyzeHtml(html: any, base: any, lines?: any): {
2
+ export declare function analyzeHtml(html: any, base: any, lines?: string[]): {
3
3
  url: any;
4
- title: any;
4
+ title: string;
5
5
  lang: any;
6
6
  meta: {
7
7
  siteName: any;
@@ -59,13 +59,13 @@ export declare function analyzeHtml(html: any, base: any, lines?: any): {
59
59
  url: string | null;
60
60
  kind: any;
61
61
  }[];
62
- textLength: any;
63
- headings: any[];
64
- paragraphs: any;
65
- hoursText: any;
62
+ textLength: number;
63
+ headings: string[];
64
+ paragraphs: string[];
65
+ hoursText: string[];
66
66
  anchors: {
67
67
  href: string | null;
68
- text: any;
68
+ text: string;
69
69
  }[];
70
70
  };
71
71
  /**
@@ -86,7 +86,7 @@ export declare function readPage(url: any, { renderJs, browser }?: {
86
86
  blocked?: undefined;
87
87
  page: {
88
88
  url: any;
89
- title: any;
89
+ title: string;
90
90
  lang: any;
91
91
  meta: {
92
92
  siteName: any;
@@ -144,13 +144,13 @@ export declare function readPage(url: any, { renderJs, browser }?: {
144
144
  url: string | null;
145
145
  kind: any;
146
146
  }[];
147
- textLength: any;
148
- headings: any[];
149
- paragraphs: any;
150
- hoursText: any;
147
+ textLength: number;
148
+ headings: string[];
149
+ paragraphs: string[];
150
+ hoursText: string[];
151
151
  anchors: {
152
152
  href: string | null;
153
- text: any;
153
+ text: string;
154
154
  }[];
155
155
  rendered: boolean;
156
156
  };
@@ -158,7 +158,7 @@ export declare function readPage(url: any, { renderJs, browser }?: {
158
158
  blocked?: undefined;
159
159
  page: {
160
160
  url: any;
161
- title: any;
161
+ title: string;
162
162
  lang: any;
163
163
  meta: {
164
164
  siteName: any;
@@ -216,13 +216,13 @@ export declare function readPage(url: any, { renderJs, browser }?: {
216
216
  url: string | null;
217
217
  kind: any;
218
218
  }[];
219
- textLength: any;
220
- headings: any[];
221
- paragraphs: any;
222
- hoursText: any;
219
+ textLength: number;
220
+ headings: string[];
221
+ paragraphs: string[];
222
+ hoursText: string[];
223
223
  anchors: {
224
224
  href: string | null;
225
- text: any;
225
+ text: string;
226
226
  }[];
227
227
  };
228
228
  }>;
@@ -234,7 +234,7 @@ export declare function scrapeSite(startUrl: any, { renderJs, maxPages, browser
234
234
  url: any;
235
235
  pages: {
236
236
  url: any;
237
- title: any;
237
+ title: string;
238
238
  rendered: boolean;
239
239
  }[];
240
240
  lang: any;