tablefacts 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. package/.env.example +2 -0
  2. package/CHANGELOG.md +39 -0
  3. package/README.md +37 -9
  4. package/package.json +2 -2
  5. package/src/lib/types.mjs +7 -2
  6. package/src/menu/README.md +37 -11
  7. package/src/menu/cluvi/config.mjs +6 -0
  8. package/src/menu/cluvi/import.mjs +6 -2
  9. package/src/menu/lib/db.mjs +81 -17
  10. package/src/menu/lib/import.mjs +20 -3
  11. package/src/menu/lib/run.mjs +15 -0
  12. package/src/menu/lib/tables.mjs +44 -0
  13. package/src/menu/raw/config.mjs +6 -0
  14. package/src/menu/raw/import.mjs +5 -2
  15. package/src/research/README.md +23 -13
  16. package/src/research/index.mjs +52 -17
  17. package/src/research/lib/merge.mjs +11 -10
  18. package/src/research/lib/report.mjs +72 -19
  19. package/src/research/lib/search.mjs +127 -0
  20. package/src/research/lib/social.mjs +137 -26
  21. package/src/research/lib/util.mjs +22 -0
  22. package/src/research/lib/website.mjs +3 -14
  23. package/src/research/research.mjs +12 -8
  24. package/types/lib/types.d.mts +29 -3
  25. package/types/menu/cluvi/config.d.mts +1 -0
  26. package/types/menu/cluvi/import.d.mts +2 -0
  27. package/types/menu/lib/db.d.mts +31 -3
  28. package/types/menu/lib/import.d.mts +1 -1
  29. package/types/menu/lib/run.d.mts +3 -0
  30. package/types/menu/lib/tables.d.mts +18 -0
  31. package/types/menu/raw/config.d.mts +1 -0
  32. package/types/menu/raw/import.d.mts +3 -0
  33. package/types/research/lib/merge.d.mts +4 -1
  34. package/types/research/lib/report.d.mts +14 -1
  35. package/types/research/lib/search.d.mts +58 -0
  36. package/types/research/lib/social.d.mts +32 -31
  37. package/types/research/lib/util.d.mts +5 -0
  38. package/types/research/lib/website.d.mts +20 -20
@@ -1,34 +1,31 @@
1
1
  /**
2
- * Public Instagram profile metadata. Without a login the page only exposes the
3
- * share card: "1,234 Followers, 56 Following, 789 Posts - Name (@handle) on
4
- * Instagram: "bio"". The bio often carries the phone, the area and a hub link.
2
+ * Public Instagram profile metadata, trying the public surfaces in turn: the
3
+ * share card, the web profile API, oEmbed, and finally the bio a search engine
4
+ * indexed (`snippet`). Only honest requests are made; a walled profile is
5
+ * reported as blocked, never worked around. Returns `blocked` only when none of
6
+ * them had anything.
5
7
  */
6
- export declare function readInstagram(handle: any): Promise<{
7
- handle: any;
8
- url: string;
9
- blocked: any;
10
- displayName?: undefined;
11
- followers?: undefined;
12
- posts?: undefined;
13
- bio?: undefined;
14
- phones?: undefined;
15
- links?: undefined;
16
- image?: undefined;
17
- } | {
18
- blocked?: undefined;
19
- handle: any;
20
- url: string;
21
- displayName: any;
22
- followers: any;
23
- posts: any;
24
- bio: any;
25
- phones: any[];
26
- links: any;
27
- image: any;
28
- }>;
29
- /** TripAdvisor: schema.org JSON-LD when the page loads; DataDome often blocks it. */
8
+ export declare function readInstagram(handle: any, { snippet }?: {
9
+ snippet?: string | undefined;
10
+ }): Promise<any>;
11
+ /**
12
+ * Name and location out of a TripAdvisor restaurant URL slug, e.g.
13
+ * "...-Reviews-Maki_Bar_Medellin-Medellin_Antioquia_Department.html" gives
14
+ * { name: "Maki Bar Medellin", location: "Medellin Antioquia Department" }. The
15
+ * city/region boundary is ambiguous for multi-word cities, so the whole location
16
+ * is kept and the caller treats it as a hint, never as a confirmed locality.
17
+ */
18
+ export declare function parseTripadvisorUrl(raw: any): {
19
+ name: string;
20
+ location: string;
21
+ } | null;
22
+ /** TripAdvisor: schema.org JSON-LD when the page loads; DataDome often blocks it. The URL slug is always kept. */
30
23
  export declare function readTripadvisor(url: any, { browser }?: {}): Promise<{
31
24
  url: any;
25
+ slug: {
26
+ name: string;
27
+ location: string;
28
+ } | null;
32
29
  blocked: string;
33
30
  jsonld?: undefined;
34
31
  description?: undefined;
@@ -37,6 +34,10 @@ export declare function readTripadvisor(url: any, { browser }?: {}): Promise<{
37
34
  } | {
38
35
  blocked?: undefined;
39
36
  url: any;
37
+ slug: {
38
+ name: string;
39
+ location: string;
40
+ } | null;
40
41
  jsonld: {
41
42
  name: any;
42
43
  telephone: any;
@@ -68,23 +69,23 @@ export declare function readTripadvisor(url: any, { browser }?: {}): Promise<{
68
69
  } | null;
69
70
  } | null;
70
71
  description: any;
71
- hoursText: any;
72
+ hoursText: string[];
72
73
  images: any[];
73
74
  }>;
74
75
  /** A link-in-bio hub (Linktree, Beacons, bio.link): its links, classified like a website's. */
75
76
  export declare function readHub(url: any, { browser }?: {}): Promise<{
76
- links?: undefined;
77
77
  images?: undefined;
78
78
  url: any;
79
79
  blocked: string;
80
80
  title?: undefined;
81
81
  meta?: undefined;
82
+ links?: undefined;
82
83
  anchors?: undefined;
83
84
  logos?: undefined;
84
85
  } | {
85
86
  blocked?: undefined;
86
87
  url: any;
87
- title: any;
88
+ title: string;
88
89
  meta: {
89
90
  siteName: any;
90
91
  title: any;
@@ -108,7 +109,7 @@ export declare function readHub(url: any, { browser }?: {}): Promise<{
108
109
  };
109
110
  anchors: {
110
111
  href: string | null;
111
- text: any;
112
+ text: string;
112
113
  }[];
113
114
  logos: {
114
115
  url: string | null;
@@ -2,6 +2,11 @@ import { fold } from "../../lib/text.mjs";
2
2
  export { fold };
3
3
  export declare const BROWSER_UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/130.0 Safari/537.36 cannario-research";
4
4
  export declare const sleep: (ms: any) => Promise<any>;
5
+ /** Decodes the HTML entities a page's attributes or text may hold, named and numeric. */
6
+ export declare const decodeHtml: (s: any) => string;
7
+ export declare const IG_RESERVED: Set<string>;
8
+ /** The profile handle in an instagram.com URL, or "" for a post, reel or other non-profile path. */
9
+ export declare const instagramHandle: (url: any) => string;
5
10
  /** Runs `fn(item, index)` over `items` with at most `size` in flight; results keep the order of `items`. */
6
11
  export declare function mapPool(items: any, size: any, fn: any): Promise<any[]>;
7
12
  export declare const MAX_BODY_BYTES: number;
@@ -1,7 +1,7 @@
1
1
  /** Pulls every fact a page holds. `base` resolves relative links. */
2
- export declare function analyzeHtml(html: any, base: any, lines?: any): {
2
+ export declare function analyzeHtml(html: any, base: any, lines?: string[]): {
3
3
  url: any;
4
- title: any;
4
+ title: string;
5
5
  lang: any;
6
6
  meta: {
7
7
  siteName: any;
@@ -59,13 +59,13 @@ export declare function analyzeHtml(html: any, base: any, lines?: any): {
59
59
  url: string | null;
60
60
  kind: any;
61
61
  }[];
62
- textLength: any;
63
- headings: any[];
64
- paragraphs: any;
65
- hoursText: any;
62
+ textLength: number;
63
+ headings: string[];
64
+ paragraphs: string[];
65
+ hoursText: string[];
66
66
  anchors: {
67
67
  href: string | null;
68
- text: any;
68
+ text: string;
69
69
  }[];
70
70
  };
71
71
  /**
@@ -86,7 +86,7 @@ export declare function readPage(url: any, { renderJs, browser }?: {
86
86
  blocked?: undefined;
87
87
  page: {
88
88
  url: any;
89
- title: any;
89
+ title: string;
90
90
  lang: any;
91
91
  meta: {
92
92
  siteName: any;
@@ -144,13 +144,13 @@ export declare function readPage(url: any, { renderJs, browser }?: {
144
144
  url: string | null;
145
145
  kind: any;
146
146
  }[];
147
- textLength: any;
148
- headings: any[];
149
- paragraphs: any;
150
- hoursText: any;
147
+ textLength: number;
148
+ headings: string[];
149
+ paragraphs: string[];
150
+ hoursText: string[];
151
151
  anchors: {
152
152
  href: string | null;
153
- text: any;
153
+ text: string;
154
154
  }[];
155
155
  rendered: boolean;
156
156
  };
@@ -158,7 +158,7 @@ export declare function readPage(url: any, { renderJs, browser }?: {
158
158
  blocked?: undefined;
159
159
  page: {
160
160
  url: any;
161
- title: any;
161
+ title: string;
162
162
  lang: any;
163
163
  meta: {
164
164
  siteName: any;
@@ -216,13 +216,13 @@ export declare function readPage(url: any, { renderJs, browser }?: {
216
216
  url: string | null;
217
217
  kind: any;
218
218
  }[];
219
- textLength: any;
220
- headings: any[];
221
- paragraphs: any;
222
- hoursText: any;
219
+ textLength: number;
220
+ headings: string[];
221
+ paragraphs: string[];
222
+ hoursText: string[];
223
223
  anchors: {
224
224
  href: string | null;
225
- text: any;
225
+ text: string;
226
226
  }[];
227
227
  };
228
228
  }>;
@@ -234,7 +234,7 @@ export declare function scrapeSite(startUrl: any, { renderJs, maxPages, browser
234
234
  url: any;
235
235
  pages: {
236
236
  url: any;
237
- title: any;
237
+ title: string;
238
238
  rendered: boolean;
239
239
  }[];
240
240
  lang: any;