extract-webpage 1.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/README.md +212 -0
  2. package/dist/config/env.d.ts +6 -0
  3. package/dist/config/index.d.ts +23 -0
  4. package/dist/config/serverRegistry.d.ts +7 -0
  5. package/dist/config/types.d.ts +4 -0
  6. package/dist/extract-webpage.cjs.js +2 -0
  7. package/dist/extract-webpage.cjs.js.map +1 -0
  8. package/dist/extract-webpage.es.js +5 -0
  9. package/dist/extract-webpage.es.js.map +1 -0
  10. package/dist/html-to-cite/extract-author.d.ts +11 -0
  11. package/dist/html-to-cite/extract-cite.d.ts +33 -0
  12. package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
  13. package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
  14. package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
  15. package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
  16. package/dist/html-to-cite/extract-source.d.ts +7 -0
  17. package/dist/html-to-cite/extract-title.d.ts +11 -0
  18. package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
  19. package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
  20. package/dist/html-to-cite/url-to-domain.d.ts +20 -0
  21. package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
  22. package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
  23. package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
  24. package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
  25. package/dist/html-to-content/html-to-content.d.ts +51 -0
  26. package/dist/html-to-content/html-utils.d.ts +76 -0
  27. package/dist/index.d.ts +26 -0
  28. package/dist/search/index.d.ts +14 -0
  29. package/dist/search/meta-search-agent-reexport.d.ts +8 -0
  30. package/dist/search/public-searxng.d.ts +47 -0
  31. package/dist/search/search-web.d.ts +33 -0
  32. package/dist/search/tavily.d.ts +20 -0
  33. package/dist/search/url-to-html.d.ts +62 -0
  34. package/dist/seektopic/fold-keyphrases.d.ts +28 -0
  35. package/dist/seektopic/ngrams.d.ts +27 -0
  36. package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
  37. package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
  38. package/dist/seektopic/types.d.ts +86 -0
  39. package/dist/seektopic/vector-search.d.ts +89 -0
  40. package/dist/seektopic/weight-keyphrases.d.ts +22 -0
  41. package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
  42. package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
  43. package/dist/tokenize/suggest-complete-word.d.ts +48 -0
  44. package/dist/tokenize/text-to-chunks.d.ts +48 -0
  45. package/dist/tokenize/text-to-sentences.d.ts +35 -0
  46. package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
  47. package/dist/tokenize/word-is-ignored.d.ts +12 -0
  48. package/dist/tokenize/word-to-root-stem.d.ts +16 -0
  49. package/dist/url-to-content/docx-to-content.d.ts +22 -0
  50. package/dist/url-to-content/is-url-adult.d.ts +26 -0
  51. package/dist/url-to-content/url-to-content.d.ts +127 -0
  52. package/dist/url-to-content/url-to-html.d.ts +60 -0
  53. package/dist/url-to-content/youtube-helpers.d.ts +23 -0
  54. package/dist/url-to-content/youtube-to-text.d.ts +70 -0
  55. package/dist/utils/documents.d.ts +4 -0
  56. package/dist/utils/grab.d.ts +18 -0
  57. package/package.json +109 -0
  58. package/src/config/env.ts +8 -0
  59. package/src/config/index.ts +233 -0
  60. package/src/config/serverRegistry.ts +24 -0
  61. package/src/config/types.ts +17 -0
  62. package/src/fs-mock.js +22 -0
  63. package/src/global.d.ts +8 -0
  64. package/src/html-to-cite/extract-author.ts +125 -0
  65. package/src/html-to-cite/extract-cite.ts +97 -0
  66. package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
  67. package/src/html-to-cite/extract-date/date-validators.ts +191 -0
  68. package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
  69. package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
  70. package/src/html-to-cite/extract-source.ts +30 -0
  71. package/src/html-to-cite/extract-title.ts +78 -0
  72. package/src/html-to-cite/human-names-92k.json +1 -0
  73. package/src/html-to-cite/human-names-recognize.ts +396 -0
  74. package/src/html-to-cite/metadata-to-cite.ts +73 -0
  75. package/src/html-to-cite/url-to-domain.ts +50 -0
  76. package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
  77. package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
  78. package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
  79. package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
  80. package/src/html-to-content/html-to-basic-html.ts +282 -0
  81. package/src/html-to-content/html-to-content.ts +97 -0
  82. package/src/html-to-content/html-utils.ts +398 -0
  83. package/src/index.ts +29 -0
  84. package/src/search/__tests__/public-searxng.test.ts +529 -0
  85. package/src/search/index.ts +43 -0
  86. package/src/search/meta-search-agent-reexport.ts +38 -0
  87. package/src/search/public-searxng.ts +470 -0
  88. package/src/search/search-web.ts +668 -0
  89. package/src/search/tavily.ts +106 -0
  90. package/src/search/url-to-html.ts +278 -0
  91. package/src/seektopic/fold-keyphrases.ts +87 -0
  92. package/src/seektopic/ngrams.ts +64 -0
  93. package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
  94. package/src/seektopic/seektopic-keyphrases.ts +279 -0
  95. package/src/seektopic/types.ts +92 -0
  96. package/src/seektopic/vector-search.ts +232 -0
  97. package/src/seektopic/weight-keyphrases.ts +59 -0
  98. package/src/suggest-next-words/autocomplete-ai.ts +38 -0
  99. package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
  100. package/src/tokenize/suggest-complete-word.ts +137 -0
  101. package/src/tokenize/text-to-chunks.ts +150 -0
  102. package/src/tokenize/text-to-sentences.ts +614 -0
  103. package/src/tokenize/text-to-topic-tokens.ts +175 -0
  104. package/src/tokenize/word-is-ignored.ts +53 -0
  105. package/src/tokenize/word-to-root-stem.ts +151 -0
  106. package/src/types.d.ts +130 -0
  107. package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
  108. package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
  109. package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
  110. package/src/url-to-content/docx-to-content.ts +702 -0
  111. package/src/url-to-content/is-url-adult.ts +318 -0
  112. package/src/url-to-content/url-to-content.ts +367 -0
  113. package/src/url-to-content/url-to-html.ts +436 -0
  114. package/src/url-to-content/youtube-helpers.ts +64 -0
  115. package/src/url-to-content/youtube-to-text.ts +468 -0
  116. package/src/utils/documents.ts +71 -0
  117. package/src/utils/grab.ts +51 -0
@@ -0,0 +1,396 @@
1
+ /**
2
+ * @fileoverview Advanced human name recognition and normalization utility.
3
+ * Distinguishes between person names and organizations using a 92k-entry name database.
4
+ */
5
+ import dataHumanNames from "./human-names-92k.json";
6
+
7
+ // Common organization terms for detection - extended list from provided code
8
+ const TERMS_ORG =
9
+ "abc,ag,ap,academy,advisors,agency,airbnb,amazon,america,american,apple,associated,association,atlantic," +
10
+ "attorneys,authority,axel,bank,baptiste,bbc,bertelsmann,blackrock,bloomberg,bmw,boston,broadcasting,bureau," +
11
+ "business,buzzfeed,cambridge,capital,cbs,center,chase,chicago,china,church,citigroup,clinic,club,cnn,coca-cola," +
12
+ "college,commission,communications,cond\u00e9,consulting,corp,corps,costco,daily,department,der,deutsche,division,dow," +
13
+ "economist,enterprises,eu,european,fabrication,facebook,fargo,ferrari,financial,ford,forbes,fox,france,fund," +
14
+ "gannett,general,global,globe,gm,gmbh,goldman,google,group,guardian,harvard,hearst,herald,hill,holdings,home," +
15
+ "honda,hospital,huffington,ibm,inc,industries,institute,intel,international,investments,jazeera,japan,jones," +
16
+ "jpmorgan,lancet,laboratories,law,legal,linkedin,llc,los,ltd,manufacturing,macy's,mcdonald's,media,medical," +
17
+ "mercedes-benz,meta,microsoft,ministry,mit,morgan,mosque,msnbc,nast,national,nato,nbc,netflix,news,newsweek," +
18
+ "new,nike,nordstrom,npr,ny,organization,oxford,pa\u00eds,partners,pbs,pentagon,plc,politico,porsche,post,press," +
19
+ "productions,r&d,regiment,retail,reuters,research,rt,sachs,school,science,scientific,securities,services,silicon," +
20
+ "society,solutions,south,spacex,spiegel,springer,stanford,stanley,starbucks,straits,studios,sydney,synagogue," +
21
+ "systems,target,team,tech,techcrunch,temple,tesla,the,thomson,times,toronto,toyota,trust,twitter,uber,union," +
22
+ "united,university,usa,valley,vanguard,vice,volkswagen,volvo,vox,wall,walmart,welle,wells,who,white,wired," +
23
+ "worldwide,works,world,wsj,york,yorker";
24
+
25
+ // Qualification terms that might suggest the name belongs to a person rather than an organization
26
+ const TERMS_QUALIFICATIONS =
27
+ "is,senior,associate,professor,fellow,assistant,lecturer,ceo,staff,strategist,specialist,worked,directed," +
28
+ "correspondent,president,author,director,prof,asst,editor,analyst,degree,administrator,served,member," +
29
+ "institute,economist,reporter,head,heads,newspaper,deputy,advocate,colonel,officer,founder,founded,visiting," +
30
+ "journalist,former,retired,expert,executive,manager,doctoral,candidate,chief,contributor,student,blogger," +
31
+ "chair,chairman,major,general,ambassador,phd,secretary,physicist,engineer,research,office,school,department," +
32
+ "writer,teacher,advisor,award,center,commentator,rand,brookings,heritage,cato,un,aei,forbes,nyt,cbo";
33
+
34
+ /**
35
+ * Parses a full name into its component parts:
36
+ * Title, Firstname, Prefix, Middle, Lastname, Honorific, Alias
37
+ * https://en.wikipedia.org/wiki/List_of_family_name_affixes
38
+ * @param {string} input - The full name to parse.
39
+ * @returns {Object}
40
+ */
41
+ const extractHumanNameParts = (input) => {
42
+ // Initialize the result object
43
+ const result = {
44
+ prefix: "", //van der von de
45
+ firstname: "",
46
+ middle: "",
47
+ lastname: "",
48
+ honorific: "", //Jr Phd II
49
+ };
50
+
51
+ // Input validation
52
+ if (!input || typeof input !== "string") {
53
+ return result;
54
+ }
55
+
56
+ // Trim input and determine case fixing mode
57
+ input = input.trim();
58
+ const shouldFixCase =
59
+ input === input.toUpperCase() || input === input.toLowerCase();
60
+
61
+ // Define lists for parsing
62
+ const lists = {
63
+ honorific: ["esq", "esquire", "jr", "sr", "ii", "iii", "iv", "phd",
64
+ "md", "ms", "mrs", "mr", "miss", "dr"],
65
+ prefix: ["de", "van", "von", "der", "den", "vel", "le", "la", "da"],
66
+ title: ["mr", "mrs", "ms", "miss", "dr", "rev", "prof"],
67
+ };
68
+
69
+ // Extract alias
70
+ const aliasRegex =
71
+ /\s(['']([^'']+)['']|[""]([^""]+)[""]|\[([^\]]+)\]|\(([^\)]+)\)),?\s/g;
72
+ const aliasMatch = input.match(aliasRegex);
73
+ if (aliasMatch) {
74
+ input = input.replace(aliasRegex, " ");
75
+ }
76
+
77
+ // Split the name into parts
78
+ let parts = input.split(/\s+/);
79
+
80
+ // Extract honorific
81
+ const honorificIndex = parts.findIndex((part) =>
82
+ lists.honorific.includes(part.toLowerCase().replace(/\.$/, ""))
83
+ );
84
+ if (honorificIndex !== -1) {
85
+ result.honorific = parts.splice(honorificIndex).join(", ");
86
+ }
87
+
88
+ // Extract title
89
+ const titleIndex = parts.findIndex((part) =>
90
+ lists.title.includes(part.toLowerCase().replace(/\.$/, ""))
91
+ );
92
+ if (titleIndex !== -1) {
93
+ result.prefix = parts.splice(titleIndex, 1)[0];
94
+ }
95
+
96
+ // Join prefixes to following name parts
97
+ for (let i = parts.length - 2; i >= 0; i--) {
98
+ if (lists.prefix.includes(parts[i]?.toLowerCase())) {
99
+ parts[i] += " " + parts[i + 1];
100
+ parts.splice(i + 1, 1);
101
+ }
102
+ }
103
+
104
+ // Extract lastname name (if comma present)
105
+ const commaIndex = parts.findIndex((part) => part.endsWith(","));
106
+ if (commaIndex !== -1) {
107
+ result.lastname = parts
108
+ .splice(0, commaIndex + 1)
109
+ .join(" ")
110
+ .replace(/,$/, "");
111
+ } else {
112
+ result.lastname = parts.pop();
113
+ }
114
+
115
+ // Assign remaining parts to firstname and middle names
116
+ if (parts.length > 0) {
117
+ result.firstname = parts.shift();
118
+ if (parts.length > 0) {
119
+ result.middle = parts.join(" ");
120
+ }
121
+ }
122
+
123
+ // Fix case if needed
124
+ if (shouldFixCase) {
125
+ Object.keys(result).forEach((key) => {
126
+ if (result[key]) {
127
+ result[key] = result[key]
128
+ .split(" ")
129
+ .map(
130
+ (word) => word.charAt(0).toUpperCase() + word.slice(1)?.toLowerCase()
131
+ )
132
+ .join(" ");
133
+ }
134
+ });
135
+ }
136
+
137
+ return result;
138
+ };
139
+
140
+ export interface ExtractHumanNameOptions {
141
+ formatCiteShortenAuthor?: boolean;
142
+ maxAuthorsBeforeEtAl?: number;
143
+ }
144
+
145
+ /**
146
+ * Validates and formats author names properly handling multiple authors and multi-word names
147
+ *
148
+ * @param {string} author - The author name string(s) to be processed
149
+ * @param {ExtractHumanNameOptions} [options={}] - Configuration options
150
+ * @returns {object} Formatted author information for citation
151
+ */
152
+ export function extractHumanName(
153
+ author: string,
154
+ options: ExtractHumanNameOptions = {}
155
+ ) {
156
+ const { formatCiteShortenAuthor = false, maxAuthorsBeforeEtAl = 2 } = options;
157
+
158
+ if (!author || !author.split) {
159
+ return { author_cite: "", author_short: "", author_type: 4 };
160
+ }
161
+
162
+ // Clean up the input string
163
+ author = author.trim()
164
+ .replace(/^by:?\s*/i, "") // Remove "by:" prefix
165
+ .replace(/\s{2,}/g, " "); // Normalize spaces
166
+
167
+ // Split into multiple authors if present
168
+ const authorNames = splitMultipleAuthors(author);
169
+
170
+ if (authorNames.length === 0) {
171
+ return { author_cite: "", author_short: "", author_type: 4 };
172
+ }
173
+
174
+ // Process each author
175
+ const processedAuthors = authorNames.map(authorName => {
176
+ // Check if name is likely an organization
177
+ const isOrg = isOrganization(authorName);
178
+
179
+ // Extract name parts using the provided function
180
+ const nameParts = extractHumanNameParts(authorName);
181
+
182
+ return {
183
+ original: authorName,
184
+ nameParts,
185
+ isOrg
186
+ };
187
+ });
188
+
189
+ // Determine overall author type
190
+ let authorType = 4; // Default to unknown
191
+
192
+ if (processedAuthors.length === 1) {
193
+ authorType = processedAuthors[0].isOrg ? 3 : 0; // 3 = org, 0 = single
194
+ } else if (processedAuthors.length === 2) {
195
+ authorType = 1; // two-author
196
+ } else if (processedAuthors.length > 2) {
197
+ authorType = 2; // more-than-two
198
+ }
199
+
200
+ // Format authors for citation
201
+ const formattedAuthors = processedAuthors.map(author => {
202
+ if (author.isOrg) {
203
+ // Don't reverse organization names
204
+ const maxOrgNameLength = 60;
205
+ let orgName = author.original;
206
+ if (orgName.length > maxOrgNameLength) {
207
+ orgName = orgName.substring(0, orgName.slice(0, maxOrgNameLength).lastIndexOf(" "));
208
+ }
209
+ return {
210
+ cite: orgName,
211
+ short: orgName
212
+ };
213
+ } else {
214
+ // Format human name using name parts
215
+ const nameParts = author.nameParts;
216
+
217
+ // Handle empty or malformed name parts
218
+ if (!nameParts || !nameParts.lastname) {
219
+ return { cite: author.original, short: author.original };
220
+ }
221
+
222
+ let formattedFirstName = nameParts.firstname;
223
+
224
+ // Include middle name with first name if present
225
+ if (nameParts.middle) {
226
+ formattedFirstName += " " + nameParts.middle;
227
+ }
228
+
229
+ // Add prefix to last name if present
230
+ let lastName = nameParts.lastname;
231
+ if (nameParts.prefix) {
232
+ lastName = `${nameParts.prefix} ${lastName}`;
233
+ }
234
+
235
+ // Shorten first name if option is set
236
+ if (formatCiteShortenAuthor && formattedFirstName) {
237
+ formattedFirstName = formattedFirstName
238
+ .split(/\s+/)
239
+ .map(part => part[0] + ".")
240
+ .join(" ");
241
+ }
242
+
243
+ // Add honorific if present
244
+ let cite = `${lastName}, ${formattedFirstName}`.trim();
245
+ if (nameParts.honorific) {
246
+ cite += `, ${nameParts.honorific}`;
247
+ }
248
+
249
+ return {
250
+ cite: cite.replace(/,\s*$/, ""),
251
+ short: lastName
252
+ };
253
+ }
254
+ });
255
+
256
+ // Generate citation strings
257
+ let authorCite, authorShort;
258
+
259
+ if (authorType === 0 || authorType === 3) {
260
+ // Single author or organization
261
+ authorCite = formattedAuthors[0].cite;
262
+ authorShort = formattedAuthors[0].short;
263
+ } else if (authorType === 1) {
264
+ // Two authors
265
+ authorCite = `${formattedAuthors[0].cite} & ${formattedAuthors[1].cite}`;
266
+ authorShort = `${formattedAuthors[0].short} & ${formattedAuthors[1].short}`;
267
+ } else if (authorType === 2) {
268
+ // More than two authors
269
+ if (processedAuthors.length <= maxAuthorsBeforeEtAl) {
270
+ // List all authors with commas and "and" before the last one
271
+ const lastAuthor = formattedAuthors.pop();
272
+ authorCite = formattedAuthors.map(a => a.cite).join(", ");
273
+ if (lastAuthor) {
274
+ authorCite += ` & ${lastAuthor.cite}`;
275
+ }
276
+ authorShort = `${formattedAuthors[0].short} et al.`;
277
+ } else {
278
+ // Use et al. format
279
+ authorCite = `${formattedAuthors[0].cite} et al.`;
280
+ authorShort = `${formattedAuthors[0].short} et al.`;
281
+ }
282
+ } else {
283
+ // Unknown/error case
284
+ authorCite = author;
285
+ authorShort = author;
286
+ }
287
+
288
+ return {
289
+ author_cite: authorCite,
290
+ author_short: authorShort,
291
+ author_type: authorType
292
+ };
293
+ }
294
+
295
+ /**
296
+ * Splits a string containing multiple authors into individual author names
297
+ *
298
+ * @param {string} authorString - String potentially containing multiple authors
299
+ * @returns {string[]} Array of individual author names
300
+ */
301
+ function splitMultipleAuthors(authorString) {
302
+ if (!authorString) return [];
303
+
304
+ // Remove "et al." since we're parsing actual authors
305
+ authorString = authorString.replace(/\s+et\s+al\.?/gi, "");
306
+
307
+ // Handle common formatting patterns for multiple authors
308
+
309
+ // Pattern 1: Last, First & Last, First
310
+ if (/\w+,\s*\w+\s*&\s*\w+,\s*\w+/.test(authorString)) {
311
+ return authorString.split(/\s*&\s*/);
312
+ }
313
+
314
+ // Pattern 2: Last, First, Last, First, and Last, First
315
+ if (/\w+,\s*\w+,\s*\w+,\s*\w+/.test(authorString)) {
316
+ // Replace the last comma+and with a standard separator
317
+ authorString = authorString.replace(/,\s*(and|&)\s*(?=[^,]*$)/, " & ");
318
+ return authorString.split(/\s*,\s*(?=[^,]*(?:,|$))/).map(s => s.trim());
319
+ }
320
+
321
+ // Pattern 3: First Last, First Last, and First Last
322
+ if (/\w+\s\w+,\s\w+\s\w+/.test(authorString)) {
323
+ // Replace the last comma+and with a standard separator
324
+ authorString = authorString.replace(/,\s*(and|&)\s*(?=[^,]*$)/, " & ");
325
+ return authorString.split(/\s*,\s*/).map(s => s.trim());
326
+ }
327
+
328
+ // Pattern 4: First Last and First Last
329
+ if (/\w+\s\w+\s(and|&)\s\w+\s\w+/.test(authorString)) {
330
+ return authorString.split(/\s+(and|&)\s+/).map(s => s.trim());
331
+ }
332
+
333
+ // Default pattern - try to split by various separators
334
+ // Replace common author separators with a standard one for easier processing
335
+ authorString = authorString
336
+ .replace(/\s+and\s+/gi, " & ")
337
+ .replace(/\s*[,;]\s*(?!(?:[^(]*\)))/g, " & "); // Replace commas/semicolons outside parentheses
338
+
339
+ // Split by the standard separator
340
+ return authorString.split(/\s*&\s*/).filter(author => author.trim().length > 0);
341
+ }
342
+
343
+ /**
344
+ * Determines if a name string represents an organization rather than a person
345
+ *
346
+ * @param {string} nameString - The name to analyze
347
+ * @returns {boolean} True if the name appears to be an organization
348
+ */
349
+ function isOrganization(nameString) {
350
+ if (!nameString) return false;
351
+
352
+ // Convert organization terms to array
353
+ const orgTerms = TERMS_ORG.split(",");
354
+ const qualTerms = TERMS_QUALIFICATIONS.split(",");
355
+
356
+ // Clean and normalize the name string
357
+ const nameLower = nameString.toLowerCase().replace(/[^\w\s]/g, " ");
358
+ const words = nameLower.split(/\s+/);
359
+
360
+ // Check for organization terms
361
+ for (const word of words) {
362
+ if (orgTerms.includes(word)) {
363
+ return true;
364
+ }
365
+ }
366
+
367
+ // Check for qualification terms that suggest it's a person
368
+ for (const word of words) {
369
+ if (qualTerms.includes(word)) {
370
+ return false;
371
+ }
372
+ }
373
+
374
+ // Look for name patterns that suggest it's a person
375
+ if (/,\s*\w+/.test(nameString)) { // Has comma format like "Smith, John"
376
+ return false;
377
+ }
378
+
379
+ // If there are more than 4 words and no commas, it's likely an organization
380
+ if (words.length > 4 && !nameString.includes(",")) {
381
+ return true;
382
+ }
383
+
384
+ // Check if the name contains any human name parts according to our database
385
+ let hasHumanNamePart = false;
386
+ for (const word of words) {
387
+ const nameTitle = word[0]?.toUpperCase() + word.slice(1)?.toLowerCase();
388
+ if (dataHumanNames[nameTitle] === 1 || dataHumanNames[nameTitle] === 2) {
389
+ hasHumanNamePart = true;
390
+ break;
391
+ }
392
+ }
393
+
394
+ // If no human name parts found and more than 2 words, likely an organization
395
+ return !hasHumanNamePart && words.length > 2;
396
+ }
@@ -0,0 +1,73 @@
1
+
2
+ export interface CiteMetadata {
3
+ author?: string;
4
+ date?: string;
5
+ title?: string;
6
+ source?: string;
7
+ }
8
+
9
+ /**
10
+ * Extract cite info from common property names in webpage's metadata
11
+ * @param {Document} doc dom object of document
12
+ * @returns {CiteMetadata} author, date, title, source
13
+ */
14
+ export function extractCiteFromMetadata(doc: Document): CiteMetadata {
15
+ if (!doc) return null;
16
+
17
+ const commonCiteMetaTags = {
18
+ source: ["application-name", "og:site_name", "twitter:site", "dc.title"],
19
+ title: ["title", "og:title", "twitter:title", "parsely-title"],
20
+ author: [
21
+ "author",
22
+ "creator",
23
+ "og:creator",
24
+ "article:author",
25
+ "dc.creator",
26
+ "parsely-author",
27
+ ],
28
+ date: [
29
+ "article:published_time",
30
+ "article:modified_time",
31
+ "og:updated_time",
32
+ "dc.date",
33
+ "dc.date.issued",
34
+ "dc.date.created",
35
+ "dc:created",
36
+ "dcterms.date",
37
+ "datepublished",
38
+ "datemodified",
39
+ "updated_time",
40
+ "modified_time",
41
+ "published_time",
42
+ "release_date",
43
+ "date",
44
+ "parsely-pub-date",
45
+ "article:published",
46
+ "article:published_time",
47
+ "og:pubdate",
48
+ "pubdate",
49
+ "date",
50
+ "dateCreated",
51
+ "pdate",
52
+ "sailthru.date",
53
+ "dcterms.created",
54
+ ],
55
+ };
56
+
57
+ const result = {};
58
+
59
+ Array.from(doc.getElementsByTagName("meta")).forEach((metaElem) => {
60
+ const property =
61
+ metaElem.getAttribute("property") || metaElem.getAttribute("itemprop");
62
+ const name = metaElem.getAttribute("name");
63
+
64
+ for (const [key, attrs] of Object.entries(commonCiteMetaTags))
65
+ if (
66
+ metaElem.getAttribute("content") &&
67
+ (attrs.includes(property?.toLowerCase()) ||
68
+ attrs.includes(name?.toLowerCase()))
69
+ )
70
+ result[key] = metaElem.getAttribute("content");
71
+ });
72
+ return result;
73
+ }
@@ -0,0 +1,50 @@
1
+
2
+ /**
3
+ * @fileoverview Utility for extracting and normalizing domain names from URLs.
4
+ * Handles subdomains and TLD cleaning for source attribution.
5
+ */
6
+ /**
7
+ * Extract TLD and hostname from domain in Regex. There's [two or more part
8
+ * TLDs](https://en.wikipedia.org/wiki/List_of_Internet_top-level_domains)
9
+ * so it is hard to tell if host.secondTLD.tld or host.tld is correct way
10
+ * to get root domain (e.g. abc.go.jp, abc.co.uk)
11
+ * @param {string} domain
12
+ * @returns {string} rootDomain
13
+ */
14
+ export function convertURLToDomain(domain) {
15
+ var tldRegExp = new RegExp(
16
+ "(?=[^^]).(fr|de|cz|at|com|wiki|co|edu|gov|info|mil|id|" +
17
+ "gv|tv|int|name|net|org|pro|ac|me|ltd|parliament)(.|$).*$"
18
+ );
19
+ var match =
20
+ domain.match(tldRegExp) ||
21
+ domain.match(/(?=[^^])\.[^a-z]{1,2}\.[^\.]{2,4}$/) ||
22
+ domain.match(/\.[^\.]{2,}$/);
23
+ var tld = match && match.index;
24
+ var domainWithoutSuffix = domain.substring(0, tld);
25
+
26
+ // Get the main domain part, handling subdomains
27
+ if (domainWithoutSuffix.includes(".")) {
28
+ // Split by dots and get the last two parts for domains like en.wikipedia.org
29
+ const parts = domainWithoutSuffix.split(".");
30
+ if (parts.length >= 2) {
31
+ domainWithoutSuffix = parts.slice(-2).join(".");
32
+ } else {
33
+ domainWithoutSuffix = parts[parts.length - 1];
34
+ }
35
+ }
36
+ return domainWithoutSuffix;
37
+ }
38
+
39
+
40
+ /**
41
+ * Checks if a string is a valid URL.
42
+ * @param {string} string
43
+ * @returns {boolean} true if the string is a valid URL
44
+ * @private
45
+ */
46
+ export function isURLValid(string) {
47
+ return /^(https?:\/\/)?([\da-z\.-]+)\.([a-z\.]{2,6})([\/\w \.-]*)*\/?$/
48
+ .test(string);
49
+ }
50
+