extract-webpage 1.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +212 -0
- package/dist/config/env.d.ts +6 -0
- package/dist/config/index.d.ts +23 -0
- package/dist/config/serverRegistry.d.ts +7 -0
- package/dist/config/types.d.ts +4 -0
- package/dist/extract-webpage.cjs.js +2 -0
- package/dist/extract-webpage.cjs.js.map +1 -0
- package/dist/extract-webpage.es.js +5 -0
- package/dist/extract-webpage.es.js.map +1 -0
- package/dist/html-to-cite/extract-author.d.ts +11 -0
- package/dist/html-to-cite/extract-cite.d.ts +33 -0
- package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
- package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
- package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
- package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
- package/dist/html-to-cite/extract-source.d.ts +7 -0
- package/dist/html-to-cite/extract-title.d.ts +11 -0
- package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
- package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
- package/dist/html-to-cite/url-to-domain.d.ts +20 -0
- package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
- package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
- package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
- package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
- package/dist/html-to-content/html-to-content.d.ts +51 -0
- package/dist/html-to-content/html-utils.d.ts +76 -0
- package/dist/index.d.ts +26 -0
- package/dist/search/index.d.ts +14 -0
- package/dist/search/meta-search-agent-reexport.d.ts +8 -0
- package/dist/search/public-searxng.d.ts +47 -0
- package/dist/search/search-web.d.ts +33 -0
- package/dist/search/tavily.d.ts +20 -0
- package/dist/search/url-to-html.d.ts +62 -0
- package/dist/seektopic/fold-keyphrases.d.ts +28 -0
- package/dist/seektopic/ngrams.d.ts +27 -0
- package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
- package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
- package/dist/seektopic/types.d.ts +86 -0
- package/dist/seektopic/vector-search.d.ts +89 -0
- package/dist/seektopic/weight-keyphrases.d.ts +22 -0
- package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
- package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
- package/dist/tokenize/suggest-complete-word.d.ts +48 -0
- package/dist/tokenize/text-to-chunks.d.ts +48 -0
- package/dist/tokenize/text-to-sentences.d.ts +35 -0
- package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
- package/dist/tokenize/word-is-ignored.d.ts +12 -0
- package/dist/tokenize/word-to-root-stem.d.ts +16 -0
- package/dist/url-to-content/docx-to-content.d.ts +22 -0
- package/dist/url-to-content/is-url-adult.d.ts +26 -0
- package/dist/url-to-content/url-to-content.d.ts +127 -0
- package/dist/url-to-content/url-to-html.d.ts +60 -0
- package/dist/url-to-content/youtube-helpers.d.ts +23 -0
- package/dist/url-to-content/youtube-to-text.d.ts +70 -0
- package/dist/utils/documents.d.ts +4 -0
- package/dist/utils/grab.d.ts +18 -0
- package/package.json +109 -0
- package/src/config/env.ts +8 -0
- package/src/config/index.ts +233 -0
- package/src/config/serverRegistry.ts +24 -0
- package/src/config/types.ts +17 -0
- package/src/fs-mock.js +22 -0
- package/src/global.d.ts +8 -0
- package/src/html-to-cite/extract-author.ts +125 -0
- package/src/html-to-cite/extract-cite.ts +97 -0
- package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
- package/src/html-to-cite/extract-date/date-validators.ts +191 -0
- package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
- package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
- package/src/html-to-cite/extract-source.ts +30 -0
- package/src/html-to-cite/extract-title.ts +78 -0
- package/src/html-to-cite/human-names-92k.json +1 -0
- package/src/html-to-cite/human-names-recognize.ts +396 -0
- package/src/html-to-cite/metadata-to-cite.ts +73 -0
- package/src/html-to-cite/url-to-domain.ts +50 -0
- package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
- package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
- package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
- package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
- package/src/html-to-content/html-to-basic-html.ts +282 -0
- package/src/html-to-content/html-to-content.ts +97 -0
- package/src/html-to-content/html-utils.ts +398 -0
- package/src/index.ts +29 -0
- package/src/search/__tests__/public-searxng.test.ts +529 -0
- package/src/search/index.ts +43 -0
- package/src/search/meta-search-agent-reexport.ts +38 -0
- package/src/search/public-searxng.ts +470 -0
- package/src/search/search-web.ts +668 -0
- package/src/search/tavily.ts +106 -0
- package/src/search/url-to-html.ts +278 -0
- package/src/seektopic/fold-keyphrases.ts +87 -0
- package/src/seektopic/ngrams.ts +64 -0
- package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
- package/src/seektopic/seektopic-keyphrases.ts +279 -0
- package/src/seektopic/types.ts +92 -0
- package/src/seektopic/vector-search.ts +232 -0
- package/src/seektopic/weight-keyphrases.ts +59 -0
- package/src/suggest-next-words/autocomplete-ai.ts +38 -0
- package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
- package/src/tokenize/suggest-complete-word.ts +137 -0
- package/src/tokenize/text-to-chunks.ts +150 -0
- package/src/tokenize/text-to-sentences.ts +614 -0
- package/src/tokenize/text-to-topic-tokens.ts +175 -0
- package/src/tokenize/word-is-ignored.ts +53 -0
- package/src/tokenize/word-to-root-stem.ts +151 -0
- package/src/types.d.ts +130 -0
- package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
- package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
- package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
- package/src/url-to-content/docx-to-content.ts +702 -0
- package/src/url-to-content/is-url-adult.ts +318 -0
- package/src/url-to-content/url-to-content.ts +367 -0
- package/src/url-to-content/url-to-html.ts +436 -0
- package/src/url-to-content/youtube-helpers.ts +64 -0
- package/src/url-to-content/youtube-to-text.ts +468 -0
- package/src/utils/documents.ts +71 -0
- package/src/utils/grab.ts +51 -0
|
@@ -0,0 +1,396 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Advanced human name recognition and normalization utility.
|
|
3
|
+
* Distinguishes between person names and organizations using a 92k-entry name database.
|
|
4
|
+
*/
|
|
5
|
+
import dataHumanNames from "./human-names-92k.json";
|
|
6
|
+
|
|
7
|
+
// Common organization terms for detection - extended list from provided code
|
|
8
|
+
const TERMS_ORG =
|
|
9
|
+
"abc,ag,ap,academy,advisors,agency,airbnb,amazon,america,american,apple,associated,association,atlantic," +
|
|
10
|
+
"attorneys,authority,axel,bank,baptiste,bbc,bertelsmann,blackrock,bloomberg,bmw,boston,broadcasting,bureau," +
|
|
11
|
+
"business,buzzfeed,cambridge,capital,cbs,center,chase,chicago,china,church,citigroup,clinic,club,cnn,coca-cola," +
|
|
12
|
+
"college,commission,communications,cond\u00e9,consulting,corp,corps,costco,daily,department,der,deutsche,division,dow," +
|
|
13
|
+
"economist,enterprises,eu,european,fabrication,facebook,fargo,ferrari,financial,ford,forbes,fox,france,fund," +
|
|
14
|
+
"gannett,general,global,globe,gm,gmbh,goldman,google,group,guardian,harvard,hearst,herald,hill,holdings,home," +
|
|
15
|
+
"honda,hospital,huffington,ibm,inc,industries,institute,intel,international,investments,jazeera,japan,jones," +
|
|
16
|
+
"jpmorgan,lancet,laboratories,law,legal,linkedin,llc,los,ltd,manufacturing,macy's,mcdonald's,media,medical," +
|
|
17
|
+
"mercedes-benz,meta,microsoft,ministry,mit,morgan,mosque,msnbc,nast,national,nato,nbc,netflix,news,newsweek," +
|
|
18
|
+
"new,nike,nordstrom,npr,ny,organization,oxford,pa\u00eds,partners,pbs,pentagon,plc,politico,porsche,post,press," +
|
|
19
|
+
"productions,r&d,regiment,retail,reuters,research,rt,sachs,school,science,scientific,securities,services,silicon," +
|
|
20
|
+
"society,solutions,south,spacex,spiegel,springer,stanford,stanley,starbucks,straits,studios,sydney,synagogue," +
|
|
21
|
+
"systems,target,team,tech,techcrunch,temple,tesla,the,thomson,times,toronto,toyota,trust,twitter,uber,union," +
|
|
22
|
+
"united,university,usa,valley,vanguard,vice,volkswagen,volvo,vox,wall,walmart,welle,wells,who,white,wired," +
|
|
23
|
+
"worldwide,works,world,wsj,york,yorker";
|
|
24
|
+
|
|
25
|
+
// Qualification terms that might suggest the name belongs to a person rather than an organization
|
|
26
|
+
const TERMS_QUALIFICATIONS =
|
|
27
|
+
"is,senior,associate,professor,fellow,assistant,lecturer,ceo,staff,strategist,specialist,worked,directed," +
|
|
28
|
+
"correspondent,president,author,director,prof,asst,editor,analyst,degree,administrator,served,member," +
|
|
29
|
+
"institute,economist,reporter,head,heads,newspaper,deputy,advocate,colonel,officer,founder,founded,visiting," +
|
|
30
|
+
"journalist,former,retired,expert,executive,manager,doctoral,candidate,chief,contributor,student,blogger," +
|
|
31
|
+
"chair,chairman,major,general,ambassador,phd,secretary,physicist,engineer,research,office,school,department," +
|
|
32
|
+
"writer,teacher,advisor,award,center,commentator,rand,brookings,heritage,cato,un,aei,forbes,nyt,cbo";
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* Parses a full name into its component parts:
|
|
36
|
+
* Title, Firstname, Prefix, Middle, Lastname, Honorific, Alias
|
|
37
|
+
* https://en.wikipedia.org/wiki/List_of_family_name_affixes
|
|
38
|
+
* @param {string} input - The full name to parse.
|
|
39
|
+
* @returns {Object}
|
|
40
|
+
*/
|
|
41
|
+
const extractHumanNameParts = (input) => {
|
|
42
|
+
// Initialize the result object
|
|
43
|
+
const result = {
|
|
44
|
+
prefix: "", //van der von de
|
|
45
|
+
firstname: "",
|
|
46
|
+
middle: "",
|
|
47
|
+
lastname: "",
|
|
48
|
+
honorific: "", //Jr Phd II
|
|
49
|
+
};
|
|
50
|
+
|
|
51
|
+
// Input validation
|
|
52
|
+
if (!input || typeof input !== "string") {
|
|
53
|
+
return result;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
// Trim input and determine case fixing mode
|
|
57
|
+
input = input.trim();
|
|
58
|
+
const shouldFixCase =
|
|
59
|
+
input === input.toUpperCase() || input === input.toLowerCase();
|
|
60
|
+
|
|
61
|
+
// Define lists for parsing
|
|
62
|
+
const lists = {
|
|
63
|
+
honorific: ["esq", "esquire", "jr", "sr", "ii", "iii", "iv", "phd",
|
|
64
|
+
"md", "ms", "mrs", "mr", "miss", "dr"],
|
|
65
|
+
prefix: ["de", "van", "von", "der", "den", "vel", "le", "la", "da"],
|
|
66
|
+
title: ["mr", "mrs", "ms", "miss", "dr", "rev", "prof"],
|
|
67
|
+
};
|
|
68
|
+
|
|
69
|
+
// Extract alias
|
|
70
|
+
const aliasRegex =
|
|
71
|
+
/\s(['']([^'']+)['']|[""]([^""]+)[""]|\[([^\]]+)\]|\(([^\)]+)\)),?\s/g;
|
|
72
|
+
const aliasMatch = input.match(aliasRegex);
|
|
73
|
+
if (aliasMatch) {
|
|
74
|
+
input = input.replace(aliasRegex, " ");
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
// Split the name into parts
|
|
78
|
+
let parts = input.split(/\s+/);
|
|
79
|
+
|
|
80
|
+
// Extract honorific
|
|
81
|
+
const honorificIndex = parts.findIndex((part) =>
|
|
82
|
+
lists.honorific.includes(part.toLowerCase().replace(/\.$/, ""))
|
|
83
|
+
);
|
|
84
|
+
if (honorificIndex !== -1) {
|
|
85
|
+
result.honorific = parts.splice(honorificIndex).join(", ");
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
// Extract title
|
|
89
|
+
const titleIndex = parts.findIndex((part) =>
|
|
90
|
+
lists.title.includes(part.toLowerCase().replace(/\.$/, ""))
|
|
91
|
+
);
|
|
92
|
+
if (titleIndex !== -1) {
|
|
93
|
+
result.prefix = parts.splice(titleIndex, 1)[0];
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
// Join prefixes to following name parts
|
|
97
|
+
for (let i = parts.length - 2; i >= 0; i--) {
|
|
98
|
+
if (lists.prefix.includes(parts[i]?.toLowerCase())) {
|
|
99
|
+
parts[i] += " " + parts[i + 1];
|
|
100
|
+
parts.splice(i + 1, 1);
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
// Extract lastname name (if comma present)
|
|
105
|
+
const commaIndex = parts.findIndex((part) => part.endsWith(","));
|
|
106
|
+
if (commaIndex !== -1) {
|
|
107
|
+
result.lastname = parts
|
|
108
|
+
.splice(0, commaIndex + 1)
|
|
109
|
+
.join(" ")
|
|
110
|
+
.replace(/,$/, "");
|
|
111
|
+
} else {
|
|
112
|
+
result.lastname = parts.pop();
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
// Assign remaining parts to firstname and middle names
|
|
116
|
+
if (parts.length > 0) {
|
|
117
|
+
result.firstname = parts.shift();
|
|
118
|
+
if (parts.length > 0) {
|
|
119
|
+
result.middle = parts.join(" ");
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
// Fix case if needed
|
|
124
|
+
if (shouldFixCase) {
|
|
125
|
+
Object.keys(result).forEach((key) => {
|
|
126
|
+
if (result[key]) {
|
|
127
|
+
result[key] = result[key]
|
|
128
|
+
.split(" ")
|
|
129
|
+
.map(
|
|
130
|
+
(word) => word.charAt(0).toUpperCase() + word.slice(1)?.toLowerCase()
|
|
131
|
+
)
|
|
132
|
+
.join(" ");
|
|
133
|
+
}
|
|
134
|
+
});
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
return result;
|
|
138
|
+
};
|
|
139
|
+
|
|
140
|
+
export interface ExtractHumanNameOptions {
|
|
141
|
+
formatCiteShortenAuthor?: boolean;
|
|
142
|
+
maxAuthorsBeforeEtAl?: number;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/**
|
|
146
|
+
* Validates and formats author names properly handling multiple authors and multi-word names
|
|
147
|
+
*
|
|
148
|
+
* @param {string} author - The author name string(s) to be processed
|
|
149
|
+
* @param {ExtractHumanNameOptions} [options={}] - Configuration options
|
|
150
|
+
* @returns {object} Formatted author information for citation
|
|
151
|
+
*/
|
|
152
|
+
export function extractHumanName(
|
|
153
|
+
author: string,
|
|
154
|
+
options: ExtractHumanNameOptions = {}
|
|
155
|
+
) {
|
|
156
|
+
const { formatCiteShortenAuthor = false, maxAuthorsBeforeEtAl = 2 } = options;
|
|
157
|
+
|
|
158
|
+
if (!author || !author.split) {
|
|
159
|
+
return { author_cite: "", author_short: "", author_type: 4 };
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
// Clean up the input string
|
|
163
|
+
author = author.trim()
|
|
164
|
+
.replace(/^by:?\s*/i, "") // Remove "by:" prefix
|
|
165
|
+
.replace(/\s{2,}/g, " "); // Normalize spaces
|
|
166
|
+
|
|
167
|
+
// Split into multiple authors if present
|
|
168
|
+
const authorNames = splitMultipleAuthors(author);
|
|
169
|
+
|
|
170
|
+
if (authorNames.length === 0) {
|
|
171
|
+
return { author_cite: "", author_short: "", author_type: 4 };
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
// Process each author
|
|
175
|
+
const processedAuthors = authorNames.map(authorName => {
|
|
176
|
+
// Check if name is likely an organization
|
|
177
|
+
const isOrg = isOrganization(authorName);
|
|
178
|
+
|
|
179
|
+
// Extract name parts using the provided function
|
|
180
|
+
const nameParts = extractHumanNameParts(authorName);
|
|
181
|
+
|
|
182
|
+
return {
|
|
183
|
+
original: authorName,
|
|
184
|
+
nameParts,
|
|
185
|
+
isOrg
|
|
186
|
+
};
|
|
187
|
+
});
|
|
188
|
+
|
|
189
|
+
// Determine overall author type
|
|
190
|
+
let authorType = 4; // Default to unknown
|
|
191
|
+
|
|
192
|
+
if (processedAuthors.length === 1) {
|
|
193
|
+
authorType = processedAuthors[0].isOrg ? 3 : 0; // 3 = org, 0 = single
|
|
194
|
+
} else if (processedAuthors.length === 2) {
|
|
195
|
+
authorType = 1; // two-author
|
|
196
|
+
} else if (processedAuthors.length > 2) {
|
|
197
|
+
authorType = 2; // more-than-two
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
// Format authors for citation
|
|
201
|
+
const formattedAuthors = processedAuthors.map(author => {
|
|
202
|
+
if (author.isOrg) {
|
|
203
|
+
// Don't reverse organization names
|
|
204
|
+
const maxOrgNameLength = 60;
|
|
205
|
+
let orgName = author.original;
|
|
206
|
+
if (orgName.length > maxOrgNameLength) {
|
|
207
|
+
orgName = orgName.substring(0, orgName.slice(0, maxOrgNameLength).lastIndexOf(" "));
|
|
208
|
+
}
|
|
209
|
+
return {
|
|
210
|
+
cite: orgName,
|
|
211
|
+
short: orgName
|
|
212
|
+
};
|
|
213
|
+
} else {
|
|
214
|
+
// Format human name using name parts
|
|
215
|
+
const nameParts = author.nameParts;
|
|
216
|
+
|
|
217
|
+
// Handle empty or malformed name parts
|
|
218
|
+
if (!nameParts || !nameParts.lastname) {
|
|
219
|
+
return { cite: author.original, short: author.original };
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
let formattedFirstName = nameParts.firstname;
|
|
223
|
+
|
|
224
|
+
// Include middle name with first name if present
|
|
225
|
+
if (nameParts.middle) {
|
|
226
|
+
formattedFirstName += " " + nameParts.middle;
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
// Add prefix to last name if present
|
|
230
|
+
let lastName = nameParts.lastname;
|
|
231
|
+
if (nameParts.prefix) {
|
|
232
|
+
lastName = `${nameParts.prefix} ${lastName}`;
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
// Shorten first name if option is set
|
|
236
|
+
if (formatCiteShortenAuthor && formattedFirstName) {
|
|
237
|
+
formattedFirstName = formattedFirstName
|
|
238
|
+
.split(/\s+/)
|
|
239
|
+
.map(part => part[0] + ".")
|
|
240
|
+
.join(" ");
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
// Add honorific if present
|
|
244
|
+
let cite = `${lastName}, ${formattedFirstName}`.trim();
|
|
245
|
+
if (nameParts.honorific) {
|
|
246
|
+
cite += `, ${nameParts.honorific}`;
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
return {
|
|
250
|
+
cite: cite.replace(/,\s*$/, ""),
|
|
251
|
+
short: lastName
|
|
252
|
+
};
|
|
253
|
+
}
|
|
254
|
+
});
|
|
255
|
+
|
|
256
|
+
// Generate citation strings
|
|
257
|
+
let authorCite, authorShort;
|
|
258
|
+
|
|
259
|
+
if (authorType === 0 || authorType === 3) {
|
|
260
|
+
// Single author or organization
|
|
261
|
+
authorCite = formattedAuthors[0].cite;
|
|
262
|
+
authorShort = formattedAuthors[0].short;
|
|
263
|
+
} else if (authorType === 1) {
|
|
264
|
+
// Two authors
|
|
265
|
+
authorCite = `${formattedAuthors[0].cite} & ${formattedAuthors[1].cite}`;
|
|
266
|
+
authorShort = `${formattedAuthors[0].short} & ${formattedAuthors[1].short}`;
|
|
267
|
+
} else if (authorType === 2) {
|
|
268
|
+
// More than two authors
|
|
269
|
+
if (processedAuthors.length <= maxAuthorsBeforeEtAl) {
|
|
270
|
+
// List all authors with commas and "and" before the last one
|
|
271
|
+
const lastAuthor = formattedAuthors.pop();
|
|
272
|
+
authorCite = formattedAuthors.map(a => a.cite).join(", ");
|
|
273
|
+
if (lastAuthor) {
|
|
274
|
+
authorCite += ` & ${lastAuthor.cite}`;
|
|
275
|
+
}
|
|
276
|
+
authorShort = `${formattedAuthors[0].short} et al.`;
|
|
277
|
+
} else {
|
|
278
|
+
// Use et al. format
|
|
279
|
+
authorCite = `${formattedAuthors[0].cite} et al.`;
|
|
280
|
+
authorShort = `${formattedAuthors[0].short} et al.`;
|
|
281
|
+
}
|
|
282
|
+
} else {
|
|
283
|
+
// Unknown/error case
|
|
284
|
+
authorCite = author;
|
|
285
|
+
authorShort = author;
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
return {
|
|
289
|
+
author_cite: authorCite,
|
|
290
|
+
author_short: authorShort,
|
|
291
|
+
author_type: authorType
|
|
292
|
+
};
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
/**
|
|
296
|
+
* Splits a string containing multiple authors into individual author names
|
|
297
|
+
*
|
|
298
|
+
* @param {string} authorString - String potentially containing multiple authors
|
|
299
|
+
* @returns {string[]} Array of individual author names
|
|
300
|
+
*/
|
|
301
|
+
function splitMultipleAuthors(authorString) {
|
|
302
|
+
if (!authorString) return [];
|
|
303
|
+
|
|
304
|
+
// Remove "et al." since we're parsing actual authors
|
|
305
|
+
authorString = authorString.replace(/\s+et\s+al\.?/gi, "");
|
|
306
|
+
|
|
307
|
+
// Handle common formatting patterns for multiple authors
|
|
308
|
+
|
|
309
|
+
// Pattern 1: Last, First & Last, First
|
|
310
|
+
if (/\w+,\s*\w+\s*&\s*\w+,\s*\w+/.test(authorString)) {
|
|
311
|
+
return authorString.split(/\s*&\s*/);
|
|
312
|
+
}
|
|
313
|
+
|
|
314
|
+
// Pattern 2: Last, First, Last, First, and Last, First
|
|
315
|
+
if (/\w+,\s*\w+,\s*\w+,\s*\w+/.test(authorString)) {
|
|
316
|
+
// Replace the last comma+and with a standard separator
|
|
317
|
+
authorString = authorString.replace(/,\s*(and|&)\s*(?=[^,]*$)/, " & ");
|
|
318
|
+
return authorString.split(/\s*,\s*(?=[^,]*(?:,|$))/).map(s => s.trim());
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
// Pattern 3: First Last, First Last, and First Last
|
|
322
|
+
if (/\w+\s\w+,\s\w+\s\w+/.test(authorString)) {
|
|
323
|
+
// Replace the last comma+and with a standard separator
|
|
324
|
+
authorString = authorString.replace(/,\s*(and|&)\s*(?=[^,]*$)/, " & ");
|
|
325
|
+
return authorString.split(/\s*,\s*/).map(s => s.trim());
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
// Pattern 4: First Last and First Last
|
|
329
|
+
if (/\w+\s\w+\s(and|&)\s\w+\s\w+/.test(authorString)) {
|
|
330
|
+
return authorString.split(/\s+(and|&)\s+/).map(s => s.trim());
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
// Default pattern - try to split by various separators
|
|
334
|
+
// Replace common author separators with a standard one for easier processing
|
|
335
|
+
authorString = authorString
|
|
336
|
+
.replace(/\s+and\s+/gi, " & ")
|
|
337
|
+
.replace(/\s*[,;]\s*(?!(?:[^(]*\)))/g, " & "); // Replace commas/semicolons outside parentheses
|
|
338
|
+
|
|
339
|
+
// Split by the standard separator
|
|
340
|
+
return authorString.split(/\s*&\s*/).filter(author => author.trim().length > 0);
|
|
341
|
+
}
|
|
342
|
+
|
|
343
|
+
/**
|
|
344
|
+
* Determines if a name string represents an organization rather than a person
|
|
345
|
+
*
|
|
346
|
+
* @param {string} nameString - The name to analyze
|
|
347
|
+
* @returns {boolean} True if the name appears to be an organization
|
|
348
|
+
*/
|
|
349
|
+
function isOrganization(nameString) {
|
|
350
|
+
if (!nameString) return false;
|
|
351
|
+
|
|
352
|
+
// Convert organization terms to array
|
|
353
|
+
const orgTerms = TERMS_ORG.split(",");
|
|
354
|
+
const qualTerms = TERMS_QUALIFICATIONS.split(",");
|
|
355
|
+
|
|
356
|
+
// Clean and normalize the name string
|
|
357
|
+
const nameLower = nameString.toLowerCase().replace(/[^\w\s]/g, " ");
|
|
358
|
+
const words = nameLower.split(/\s+/);
|
|
359
|
+
|
|
360
|
+
// Check for organization terms
|
|
361
|
+
for (const word of words) {
|
|
362
|
+
if (orgTerms.includes(word)) {
|
|
363
|
+
return true;
|
|
364
|
+
}
|
|
365
|
+
}
|
|
366
|
+
|
|
367
|
+
// Check for qualification terms that suggest it's a person
|
|
368
|
+
for (const word of words) {
|
|
369
|
+
if (qualTerms.includes(word)) {
|
|
370
|
+
return false;
|
|
371
|
+
}
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
// Look for name patterns that suggest it's a person
|
|
375
|
+
if (/,\s*\w+/.test(nameString)) { // Has comma format like "Smith, John"
|
|
376
|
+
return false;
|
|
377
|
+
}
|
|
378
|
+
|
|
379
|
+
// If there are more than 4 words and no commas, it's likely an organization
|
|
380
|
+
if (words.length > 4 && !nameString.includes(",")) {
|
|
381
|
+
return true;
|
|
382
|
+
}
|
|
383
|
+
|
|
384
|
+
// Check if the name contains any human name parts according to our database
|
|
385
|
+
let hasHumanNamePart = false;
|
|
386
|
+
for (const word of words) {
|
|
387
|
+
const nameTitle = word[0]?.toUpperCase() + word.slice(1)?.toLowerCase();
|
|
388
|
+
if (dataHumanNames[nameTitle] === 1 || dataHumanNames[nameTitle] === 2) {
|
|
389
|
+
hasHumanNamePart = true;
|
|
390
|
+
break;
|
|
391
|
+
}
|
|
392
|
+
}
|
|
393
|
+
|
|
394
|
+
// If no human name parts found and more than 2 words, likely an organization
|
|
395
|
+
return !hasHumanNamePart && words.length > 2;
|
|
396
|
+
}
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
|
|
2
|
+
export interface CiteMetadata {
|
|
3
|
+
author?: string;
|
|
4
|
+
date?: string;
|
|
5
|
+
title?: string;
|
|
6
|
+
source?: string;
|
|
7
|
+
}
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* Extract cite info from common property names in webpage's metadata
|
|
11
|
+
* @param {Document} doc dom object of document
|
|
12
|
+
* @returns {CiteMetadata} author, date, title, source
|
|
13
|
+
*/
|
|
14
|
+
export function extractCiteFromMetadata(doc: Document): CiteMetadata {
|
|
15
|
+
if (!doc) return null;
|
|
16
|
+
|
|
17
|
+
const commonCiteMetaTags = {
|
|
18
|
+
source: ["application-name", "og:site_name", "twitter:site", "dc.title"],
|
|
19
|
+
title: ["title", "og:title", "twitter:title", "parsely-title"],
|
|
20
|
+
author: [
|
|
21
|
+
"author",
|
|
22
|
+
"creator",
|
|
23
|
+
"og:creator",
|
|
24
|
+
"article:author",
|
|
25
|
+
"dc.creator",
|
|
26
|
+
"parsely-author",
|
|
27
|
+
],
|
|
28
|
+
date: [
|
|
29
|
+
"article:published_time",
|
|
30
|
+
"article:modified_time",
|
|
31
|
+
"og:updated_time",
|
|
32
|
+
"dc.date",
|
|
33
|
+
"dc.date.issued",
|
|
34
|
+
"dc.date.created",
|
|
35
|
+
"dc:created",
|
|
36
|
+
"dcterms.date",
|
|
37
|
+
"datepublished",
|
|
38
|
+
"datemodified",
|
|
39
|
+
"updated_time",
|
|
40
|
+
"modified_time",
|
|
41
|
+
"published_time",
|
|
42
|
+
"release_date",
|
|
43
|
+
"date",
|
|
44
|
+
"parsely-pub-date",
|
|
45
|
+
"article:published",
|
|
46
|
+
"article:published_time",
|
|
47
|
+
"og:pubdate",
|
|
48
|
+
"pubdate",
|
|
49
|
+
"date",
|
|
50
|
+
"dateCreated",
|
|
51
|
+
"pdate",
|
|
52
|
+
"sailthru.date",
|
|
53
|
+
"dcterms.created",
|
|
54
|
+
],
|
|
55
|
+
};
|
|
56
|
+
|
|
57
|
+
const result = {};
|
|
58
|
+
|
|
59
|
+
Array.from(doc.getElementsByTagName("meta")).forEach((metaElem) => {
|
|
60
|
+
const property =
|
|
61
|
+
metaElem.getAttribute("property") || metaElem.getAttribute("itemprop");
|
|
62
|
+
const name = metaElem.getAttribute("name");
|
|
63
|
+
|
|
64
|
+
for (const [key, attrs] of Object.entries(commonCiteMetaTags))
|
|
65
|
+
if (
|
|
66
|
+
metaElem.getAttribute("content") &&
|
|
67
|
+
(attrs.includes(property?.toLowerCase()) ||
|
|
68
|
+
attrs.includes(name?.toLowerCase()))
|
|
69
|
+
)
|
|
70
|
+
result[key] = metaElem.getAttribute("content");
|
|
71
|
+
});
|
|
72
|
+
return result;
|
|
73
|
+
}
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
|
|
2
|
+
/**
|
|
3
|
+
* @fileoverview Utility for extracting and normalizing domain names from URLs.
|
|
4
|
+
* Handles subdomains and TLD cleaning for source attribution.
|
|
5
|
+
*/
|
|
6
|
+
/**
|
|
7
|
+
* Extract TLD and hostname from domain in Regex. There's [two or more part
|
|
8
|
+
* TLDs](https://en.wikipedia.org/wiki/List_of_Internet_top-level_domains)
|
|
9
|
+
* so it is hard to tell if host.secondTLD.tld or host.tld is correct way
|
|
10
|
+
* to get root domain (e.g. abc.go.jp, abc.co.uk)
|
|
11
|
+
* @param {string} domain
|
|
12
|
+
* @returns {string} rootDomain
|
|
13
|
+
*/
|
|
14
|
+
export function convertURLToDomain(domain) {
|
|
15
|
+
var tldRegExp = new RegExp(
|
|
16
|
+
"(?=[^^]).(fr|de|cz|at|com|wiki|co|edu|gov|info|mil|id|" +
|
|
17
|
+
"gv|tv|int|name|net|org|pro|ac|me|ltd|parliament)(.|$).*$"
|
|
18
|
+
);
|
|
19
|
+
var match =
|
|
20
|
+
domain.match(tldRegExp) ||
|
|
21
|
+
domain.match(/(?=[^^])\.[^a-z]{1,2}\.[^\.]{2,4}$/) ||
|
|
22
|
+
domain.match(/\.[^\.]{2,}$/);
|
|
23
|
+
var tld = match && match.index;
|
|
24
|
+
var domainWithoutSuffix = domain.substring(0, tld);
|
|
25
|
+
|
|
26
|
+
// Get the main domain part, handling subdomains
|
|
27
|
+
if (domainWithoutSuffix.includes(".")) {
|
|
28
|
+
// Split by dots and get the last two parts for domains like en.wikipedia.org
|
|
29
|
+
const parts = domainWithoutSuffix.split(".");
|
|
30
|
+
if (parts.length >= 2) {
|
|
31
|
+
domainWithoutSuffix = parts.slice(-2).join(".");
|
|
32
|
+
} else {
|
|
33
|
+
domainWithoutSuffix = parts[parts.length - 1];
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
return domainWithoutSuffix;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
/**
|
|
41
|
+
* Checks if a string is a valid URL.
|
|
42
|
+
* @param {string} string
|
|
43
|
+
* @returns {boolean} true if the string is a valid URL
|
|
44
|
+
* @private
|
|
45
|
+
*/
|
|
46
|
+
export function isURLValid(string) {
|
|
47
|
+
return /^(https?:\/\/)?([\da-z\.-]+)\.([a-z\.]{2,6})([\/\w \.-]*)*\/?$/
|
|
48
|
+
.test(string);
|
|
49
|
+
}
|
|
50
|
+
|