extract-webpage 1.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/README.md +212 -0
  2. package/dist/config/env.d.ts +6 -0
  3. package/dist/config/index.d.ts +23 -0
  4. package/dist/config/serverRegistry.d.ts +7 -0
  5. package/dist/config/types.d.ts +4 -0
  6. package/dist/extract-webpage.cjs.js +2 -0
  7. package/dist/extract-webpage.cjs.js.map +1 -0
  8. package/dist/extract-webpage.es.js +5 -0
  9. package/dist/extract-webpage.es.js.map +1 -0
  10. package/dist/html-to-cite/extract-author.d.ts +11 -0
  11. package/dist/html-to-cite/extract-cite.d.ts +33 -0
  12. package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
  13. package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
  14. package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
  15. package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
  16. package/dist/html-to-cite/extract-source.d.ts +7 -0
  17. package/dist/html-to-cite/extract-title.d.ts +11 -0
  18. package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
  19. package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
  20. package/dist/html-to-cite/url-to-domain.d.ts +20 -0
  21. package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
  22. package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
  23. package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
  24. package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
  25. package/dist/html-to-content/html-to-content.d.ts +51 -0
  26. package/dist/html-to-content/html-utils.d.ts +76 -0
  27. package/dist/index.d.ts +26 -0
  28. package/dist/search/index.d.ts +14 -0
  29. package/dist/search/meta-search-agent-reexport.d.ts +8 -0
  30. package/dist/search/public-searxng.d.ts +47 -0
  31. package/dist/search/search-web.d.ts +33 -0
  32. package/dist/search/tavily.d.ts +20 -0
  33. package/dist/search/url-to-html.d.ts +62 -0
  34. package/dist/seektopic/fold-keyphrases.d.ts +28 -0
  35. package/dist/seektopic/ngrams.d.ts +27 -0
  36. package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
  37. package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
  38. package/dist/seektopic/types.d.ts +86 -0
  39. package/dist/seektopic/vector-search.d.ts +89 -0
  40. package/dist/seektopic/weight-keyphrases.d.ts +22 -0
  41. package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
  42. package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
  43. package/dist/tokenize/suggest-complete-word.d.ts +48 -0
  44. package/dist/tokenize/text-to-chunks.d.ts +48 -0
  45. package/dist/tokenize/text-to-sentences.d.ts +35 -0
  46. package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
  47. package/dist/tokenize/word-is-ignored.d.ts +12 -0
  48. package/dist/tokenize/word-to-root-stem.d.ts +16 -0
  49. package/dist/url-to-content/docx-to-content.d.ts +22 -0
  50. package/dist/url-to-content/is-url-adult.d.ts +26 -0
  51. package/dist/url-to-content/url-to-content.d.ts +127 -0
  52. package/dist/url-to-content/url-to-html.d.ts +60 -0
  53. package/dist/url-to-content/youtube-helpers.d.ts +23 -0
  54. package/dist/url-to-content/youtube-to-text.d.ts +70 -0
  55. package/dist/utils/documents.d.ts +4 -0
  56. package/dist/utils/grab.d.ts +18 -0
  57. package/package.json +109 -0
  58. package/src/config/env.ts +8 -0
  59. package/src/config/index.ts +233 -0
  60. package/src/config/serverRegistry.ts +24 -0
  61. package/src/config/types.ts +17 -0
  62. package/src/fs-mock.js +22 -0
  63. package/src/global.d.ts +8 -0
  64. package/src/html-to-cite/extract-author.ts +125 -0
  65. package/src/html-to-cite/extract-cite.ts +97 -0
  66. package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
  67. package/src/html-to-cite/extract-date/date-validators.ts +191 -0
  68. package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
  69. package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
  70. package/src/html-to-cite/extract-source.ts +30 -0
  71. package/src/html-to-cite/extract-title.ts +78 -0
  72. package/src/html-to-cite/human-names-92k.json +1 -0
  73. package/src/html-to-cite/human-names-recognize.ts +396 -0
  74. package/src/html-to-cite/metadata-to-cite.ts +73 -0
  75. package/src/html-to-cite/url-to-domain.ts +50 -0
  76. package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
  77. package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
  78. package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
  79. package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
  80. package/src/html-to-content/html-to-basic-html.ts +282 -0
  81. package/src/html-to-content/html-to-content.ts +97 -0
  82. package/src/html-to-content/html-utils.ts +398 -0
  83. package/src/index.ts +29 -0
  84. package/src/search/__tests__/public-searxng.test.ts +529 -0
  85. package/src/search/index.ts +43 -0
  86. package/src/search/meta-search-agent-reexport.ts +38 -0
  87. package/src/search/public-searxng.ts +470 -0
  88. package/src/search/search-web.ts +668 -0
  89. package/src/search/tavily.ts +106 -0
  90. package/src/search/url-to-html.ts +278 -0
  91. package/src/seektopic/fold-keyphrases.ts +87 -0
  92. package/src/seektopic/ngrams.ts +64 -0
  93. package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
  94. package/src/seektopic/seektopic-keyphrases.ts +279 -0
  95. package/src/seektopic/types.ts +92 -0
  96. package/src/seektopic/vector-search.ts +232 -0
  97. package/src/seektopic/weight-keyphrases.ts +59 -0
  98. package/src/suggest-next-words/autocomplete-ai.ts +38 -0
  99. package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
  100. package/src/tokenize/suggest-complete-word.ts +137 -0
  101. package/src/tokenize/text-to-chunks.ts +150 -0
  102. package/src/tokenize/text-to-sentences.ts +614 -0
  103. package/src/tokenize/text-to-topic-tokens.ts +175 -0
  104. package/src/tokenize/word-is-ignored.ts +53 -0
  105. package/src/tokenize/word-to-root-stem.ts +151 -0
  106. package/src/types.d.ts +130 -0
  107. package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
  108. package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
  109. package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
  110. package/src/url-to-content/docx-to-content.ts +702 -0
  111. package/src/url-to-content/is-url-adult.ts +318 -0
  112. package/src/url-to-content/url-to-content.ts +367 -0
  113. package/src/url-to-content/url-to-html.ts +436 -0
  114. package/src/url-to-content/youtube-helpers.ts +64 -0
  115. package/src/url-to-content/youtube-to-text.ts +468 -0
  116. package/src/utils/documents.ts +71 -0
  117. package/src/utils/grab.ts +51 -0
@@ -0,0 +1,233 @@
1
+ /**
2
+ * @fileoverview Simplified ConfigManager for the research-agent package.
3
+ * Manages model providers and search config in memory.
4
+ */
5
+ import type { ConfigModelProvider, MCPServerConfig, Config, UIConfigSections } from "./types";
6
+ import { getEnv } from "./env";
7
+
8
+ const hashObj = (obj: { [key: string]: any }) => {
9
+ const str = JSON.stringify(obj, Object.keys(obj).sort());
10
+ let hash = 0;
11
+ for (let i = 0; i < str.length; i++) {
12
+ hash = (hash << 5) - hash + str.charCodeAt(i);
13
+ hash |= 0;
14
+ }
15
+ return String(Math.abs(hash).toString(36));
16
+ };
17
+
18
+ class ConfigManager {
19
+ configVersion = 1;
20
+ currentConfig: Config = {
21
+ version: this.configVersion,
22
+ setupComplete: getEnv("SETUP_COMPLETE") === "true" || false,
23
+ preferences: {},
24
+ personalization: {},
25
+ modelProviders: [],
26
+ mcpServers: [],
27
+ search: {
28
+ searxngURL: "",
29
+ tavilyApiKey: "",
30
+ sourceScrapeCount: 3,
31
+ sourceScrapeTimeout: 5,
32
+ },
33
+ };
34
+ uiConfigSections: UIConfigSections = {
35
+ preferences: [],
36
+ personalization: [],
37
+ modelProviders: [],
38
+ mcpServers: [],
39
+ search: [
40
+ {
41
+ name: "SearXNG URL",
42
+ key: "searxngURL",
43
+ type: "string",
44
+ required: false,
45
+ description: "The URL of your SearXNG instance",
46
+ placeholder: "http://localhost:4000",
47
+ default: "",
48
+ scope: "server",
49
+ env: "SEARXNG_API_URL",
50
+ },
51
+ {
52
+ name: "Tavily API Key",
53
+ key: "tavilyApiKey",
54
+ type: "password",
55
+ required: false,
56
+ description:
57
+ "Optional. Enter your own Tavily API key to override the site default.",
58
+ placeholder: "tvly-...",
59
+ default: "",
60
+ scope: "server",
61
+ env: "TAVILY_API_KEY",
62
+ },
63
+ {
64
+ name: "Source pages to scrape",
65
+ key: "sourceScrapeCount",
66
+ type: "select",
67
+ options: [
68
+ { name: "Disabled (snippet only)", value: "0" },
69
+ { name: "1 page", value: "1" },
70
+ { name: "2 pages", value: "2" },
71
+ { name: "3 pages (default)", value: "3" },
72
+ { name: "5 pages", value: "5" },
73
+ ],
74
+ required: false,
75
+ description: "Number of top search result URLs to fully scrape.",
76
+ default: "3",
77
+ scope: "server",
78
+ },
79
+ {
80
+ name: "Scrape timeout (seconds)",
81
+ key: "sourceScrapeTimeout",
82
+ type: "select",
83
+ options: [
84
+ { name: "3 seconds", value: "3" },
85
+ { name: "5 seconds (default)", value: "5" },
86
+ { name: "10 seconds", value: "10" },
87
+ { name: "15 seconds", value: "15" },
88
+ { name: "20 seconds", value: "20" },
89
+ ],
90
+ required: false,
91
+ description: "Maximum time to wait when scraping each source URL.",
92
+ default: "5",
93
+ scope: "server",
94
+ },
95
+ ],
96
+ };
97
+
98
+ private initialized = false;
99
+
100
+ constructor() {
101
+ // Don't initialize in constructor to avoid circular dependency
102
+ // Initialize lazily when config is first accessed
103
+ }
104
+
105
+ private ensureInitialized() {
106
+ if (this.initialized) return;
107
+ this.initialize();
108
+ this.initialized = true;
109
+ }
110
+
111
+ private initialize() {
112
+ // Search config from env
113
+ this.uiConfigSections.search.forEach((f) => {
114
+ if (f.env && !this.currentConfig.search[f.key]) {
115
+ this.currentConfig.search[f.key] =
116
+ getEnv(f.env) ?? f.default ?? "";
117
+ }
118
+ });
119
+ }
120
+
121
+ public getConfig(key: string, defaultValue?: any): any {
122
+ this.ensureInitialized();
123
+ const nested = key.split(".");
124
+ let obj: any = this.currentConfig;
125
+
126
+ for (let i = 0; i < nested.length; i++) {
127
+ const part = nested[i];
128
+ if (obj == null) return defaultValue;
129
+ obj = obj[part];
130
+ }
131
+
132
+ return obj === undefined ? defaultValue : obj;
133
+ }
134
+
135
+ public updateConfig(key: string, val: any) {
136
+ const parts = key.split(".");
137
+ if (parts.length === 0) return;
138
+
139
+ let target: any = this.currentConfig;
140
+ for (let i = 0; i < parts.length - 1; i++) {
141
+ const part = parts[i];
142
+ if (target[part] === null || typeof target[part] !== "object") {
143
+ target[part] = {};
144
+ }
145
+ target = target[part];
146
+ }
147
+
148
+ const finalKey = parts[parts.length - 1];
149
+ target[finalKey] = val;
150
+ }
151
+
152
+ public addModelProvider(type: string, config: any) {
153
+ this.ensureInitialized();
154
+ const hash = hashObj(config);
155
+ const section = this.uiConfigSections.modelProviders.find(s => s.key === type);
156
+ const name = section?.name || type;
157
+
158
+ const newModelProvider: ConfigModelProvider = {
159
+ id: hash,
160
+ name,
161
+ type,
162
+ config,
163
+ chatModels: [],
164
+ hash: hash,
165
+ };
166
+
167
+ this.currentConfig.modelProviders.push(newModelProvider);
168
+ return newModelProvider;
169
+ }
170
+
171
+ public removeModelProvider(id: string) {
172
+ this.currentConfig.modelProviders =
173
+ this.currentConfig.modelProviders.filter((p) => p.id !== id);
174
+ }
175
+
176
+ public async updateModelProvider(id: string, config: any) {
177
+ const provider = this.currentConfig.modelProviders.find(
178
+ (p) => p.id === id,
179
+ );
180
+ if (!provider) throw new Error("Provider not found");
181
+
182
+ provider.config = config;
183
+ return provider;
184
+ }
185
+
186
+ public addProviderModel(providerId: string, type: "chat", model: any) {
187
+ const provider = this.currentConfig.modelProviders.find(
188
+ (p) => p.id === providerId,
189
+ );
190
+ if (!provider) throw new Error("Invalid provider id");
191
+
192
+ delete model.type;
193
+ provider.chatModels.push(model);
194
+ return model;
195
+ }
196
+
197
+ public removeProviderModel(
198
+ providerId: string,
199
+ type: "chat",
200
+ modelKey: string,
201
+ ) {
202
+ const provider = this.currentConfig.modelProviders.find(
203
+ (p) => p.id === providerId,
204
+ );
205
+ if (!provider) throw new Error("Invalid provider id");
206
+
207
+ provider.chatModels = provider.chatModels.filter((m) => m.key !== modelKey);
208
+ }
209
+
210
+ public isSetupComplete() {
211
+ return this.currentConfig.setupComplete;
212
+ }
213
+
214
+ public markSetupComplete() {
215
+ if (!this.currentConfig.setupComplete) {
216
+ this.currentConfig.setupComplete = true;
217
+ }
218
+ }
219
+
220
+ public getUIConfigSections(): UIConfigSections {
221
+ this.ensureInitialized();
222
+ return this.uiConfigSections;
223
+ }
224
+
225
+ public getCurrentConfig(): Config {
226
+ this.ensureInitialized();
227
+ return JSON.parse(JSON.stringify(this.currentConfig));
228
+ }
229
+ }
230
+
231
+ const configManager = new ConfigManager();
232
+
233
+ export default configManager;
@@ -0,0 +1,24 @@
1
+ import configManager from "./index";
2
+ import type { ConfigModelProvider } from "./types";
3
+
4
+ export const getConfiguredModelProviders = (): ConfigModelProvider[] => {
5
+ return configManager.getConfig("modelProviders", []);
6
+ };
7
+
8
+ export const getConfiguredModelProviderById = (
9
+ id: string,
10
+ ): ConfigModelProvider | undefined => {
11
+ return getConfiguredModelProviders().find((p) => p.id === id) ?? undefined;
12
+ };
13
+
14
+ export const getSearxngURL = () =>
15
+ configManager.getConfig("search.searxngURL", "");
16
+
17
+ export const getTavilyApiKey = () =>
18
+ configManager.getConfig("search.tavilyApiKey", "");
19
+
20
+ export const getSourceScrapeCount = (): number =>
21
+ parseInt(configManager.getConfig("search.sourceScrapeCount", "3"), 10) || 3;
22
+
23
+ export const getSourceScrapeTimeout = (): number =>
24
+ parseInt(configManager.getConfig("search.sourceScrapeTimeout", "5"), 10) || 5;
@@ -0,0 +1,17 @@
1
+ /**
2
+ * Re-export config types from the package's central type definitions.
3
+ */
4
+ export type {
5
+ UIConfigField,
6
+ Config,
7
+ EnvMap,
8
+ UIConfigSections,
9
+ SelectUIConfigField,
10
+ StringUIConfigField,
11
+ ModelProviderUISection,
12
+ MCPServerUISection,
13
+ ConfigModelProvider,
14
+ MCPServerConfig,
15
+ TextareaUIConfigField,
16
+ SwitchUIConfigField,
17
+ } from "../types";
package/src/fs-mock.js ADDED
@@ -0,0 +1,22 @@
1
+ export const promises = {
2
+ mkdir: async () => {},
3
+ readFile: async () => "",
4
+ writeFile: async () => {},
5
+ appendFile: async () => {},
6
+ stat: async () => ({ isDirectory: () => false }),
7
+ };
8
+
9
+ export const mkdir = promises.mkdir;
10
+ export const readFile = promises.readFile;
11
+ export const writeFile = promises.writeFile;
12
+ export const appendFile = promises.appendFile;
13
+ export const stat = promises.stat;
14
+
15
+ export default {
16
+ promises,
17
+ mkdir,
18
+ readFile,
19
+ writeFile,
20
+ appendFile,
21
+ stat,
22
+ };
@@ -0,0 +1,8 @@
1
+ declare module "@huggingface/transformers" {
2
+ export function pipeline(task: string, model: string, options?: any): Promise<any>;
3
+ }
4
+
5
+ declare module "*.csv?raw" {
6
+ const content: string;
7
+ export default content;
8
+ }
@@ -0,0 +1,125 @@
1
+ /**
2
+ * @fileoverview Utility for identifying and extracting author names from HTML metadata and content.
3
+ * Handles individual authors, multiple authors, and organizational authors.
4
+ */
5
+ import { extractHumanName } from "./human-names-recognize";
6
+
7
+ // https://www.scribbr.com/citation/generator/folders/2rx21jyIjZIKRrcwLk2oXE/lists/4OeJQ4euxzyTk9BiPTRjwn/
8
+ const AUTHOR_META_TAGS = [
9
+ "byl",
10
+ "clmst",
11
+ "dc.author",
12
+ "dcsext.author",
13
+ "dc.creator",
14
+ "rbauthors",
15
+ "authors",
16
+ "sailthru.author",
17
+ "article:author",
18
+ "parsely-author",
19
+ ];
20
+
21
+ const AUTHOR_SELECTORS = [
22
+ ".entry .entry-author",
23
+ ".author.vcard .fn",
24
+ ".author .vcard .fn",
25
+ ".byline.vcard .fn",
26
+ ".byline .vcard .fn",
27
+ ".byline .by .author",
28
+ ".byline .by",
29
+ ".byline .author",
30
+ ".post-author.vcard",
31
+ ".post-author .vcard",
32
+ "a[rel=author]",
33
+ "#by_author",
34
+ ".by_author",
35
+ "#entryAuthor",
36
+ ".entryAuthor",
37
+ ".byline a[href*=author]",
38
+ "#author .authorname",
39
+ ".author .authorname",
40
+ "#author",
41
+ ".author",
42
+ ".articleauthor",
43
+ ".ArticleAuthor",
44
+ ".byline",
45
+ ];
46
+
47
+ const AUTHOR_MAX_LENGTH = 300;
48
+ const BYLINE_REGEX = /^[\n\s]*By\s*:?\s*/i;
49
+ const CLEAN_AUTHOR_RE = /^\s*(posted |written )?by\s*:?\s*(.*)/i;
50
+
51
+ /**
52
+ * Extracts the author from the document and validates it as a human name
53
+ *
54
+ * @param {Document} document
55
+ * @returns {object|null} author_cite, author_short, author_type - or null if no valid author found
56
+ */
57
+ export function extractAuthor(document) {
58
+
59
+
60
+ // 1. Check meta tags
61
+ for (const tag of AUTHOR_META_TAGS) {
62
+ const metaElement = document.querySelector(
63
+ `meta[name="${tag}"], meta[property="${tag}"]`
64
+ );
65
+ if (metaElement) {
66
+ const author = extractAndValidateHumanName(
67
+ metaElement.getAttribute("content")
68
+ );
69
+ if (author) return author;
70
+ }
71
+ }
72
+
73
+ // 2. Check selectors
74
+ for (const selector of AUTHOR_SELECTORS) {
75
+ const element = document.querySelector(selector);
76
+ if (element) {
77
+ const author = extractAndValidateHumanName(element.textContent);
78
+ if (author) return author;
79
+ }
80
+ }
81
+
82
+ // 4. Broader search in divs and spans
83
+ const elements = [
84
+ ...document.getElementsByTagName("div"),
85
+ ...document.getElementsByTagName("span"),
86
+ ];
87
+ for (const element of elements) {
88
+ if (
89
+ element.id.match(/author|byline/i) ||
90
+ element?.className?.match(/author|byline/i)
91
+ ) {
92
+ const author = extractAndValidateHumanName(element.innerText);
93
+ if (author) return author;
94
+ }
95
+ }
96
+
97
+
98
+ return null;
99
+ }
100
+
101
+ // Helper function to clean and validate author string
102
+ const validateAuthor = (author) => {
103
+ if (!author) return null;
104
+
105
+ if (author.startsWith("@")) author = author.slice(1);
106
+ if (author.startsWith("http")) return null;
107
+
108
+ author = author?.replace(CLEAN_AUTHOR_RE, "$2").trim();
109
+ return author.length > 0 && author.length < AUTHOR_MAX_LENGTH
110
+ ? author
111
+ : null;
112
+ };
113
+
114
+ // Function to extract and validate human name
115
+ const extractAndValidateHumanName = (author) => {
116
+ const validatedAuthor = validateAuthor(author);
117
+ if (validatedAuthor) {
118
+ return extractHumanName(validatedAuthor, {
119
+ formatCiteShortenAuthor: false
120
+ });
121
+ }
122
+ return null;
123
+ };
124
+
125
+
@@ -0,0 +1,97 @@
1
+ /**
2
+ * @fileoverview Orchestrator for extracting high-quality citations (author, date, title, source) from HTML.
3
+ * Validates names and sources against known patterns and benchmarks.
4
+ */
5
+ import { parseHTML } from "linkedom";
6
+ import { extractAuthor } from "./extract-author";
7
+ import { extractDateQuick } from "./extract-date/extract-date-quick";
8
+ import { extractDate } from "./extract-date/extract-date";
9
+ import { extractSource } from "./extract-source";
10
+ import { extractTitle } from "./extract-title";
11
+ import { extractCiteFromMetadata } from "./metadata-to-cite";
12
+ import { extractHumanName } from "./human-names-recognize";
13
+ import { parseDate } from "chrono-node";
14
+
15
+ export interface ExtractCiteResult {
16
+ author?: string;
17
+ author_cite?: string;
18
+ date?: string;
19
+ title?: string;
20
+ source?: string;
21
+ }
22
+
23
+ export interface ExtractCiteOptions {
24
+ url?: string;
25
+ }
26
+
27
+ /**
28
+ * ### \u1f4da\u1f48e Extract Expert Excerpt
29
+ * <img width="350px" src="https://i.imgur.com/4GOOM9s.jpeg" />
30
+ *
31
+ * Extract author, date, source, and title from HTML using meta tags
32
+ * and common class names. Validates human name from author string to check
33
+ * against common list of 90k first names, last names,and organizations to infer
34
+ * if it should be reversed starting by author last name (accounting for affixes/titles),
35
+ * since organizations are not reversed.
36
+ * [Article Extraction Benchmark](https://github.com/scrapinghub/article-extraction-benchmark?tab=readme-ov-file#results)
37
+ * @param {Document | string} document dom object or html string with article content
38
+ * @param {ExtractCiteOptions} [options={}]
39
+ * @returns {ExtractCiteResult | null} An object containing extracted citation information.
40
+ * @category Extract
41
+ * @author [vtempest (2025)](https://github.com/vtempest)
42
+ */
43
+ export function extractCite(
44
+ document: Document | string,
45
+ options: ExtractCiteOptions = {}
46
+ ) {
47
+ const { url = "" } = options;
48
+
49
+ if (!document) return null;
50
+
51
+ //if passing in html string, convert to dom object
52
+ if (typeof document === "string") document = parseHTML(document)?.document;
53
+
54
+ var { author, date, title, source } = extractCiteFromMetadata(document);
55
+
56
+ if (author?.length < 3 || author?.length > 100) author = null;
57
+
58
+ var { author_cite, author_short, author_type } =
59
+ extractAuthor(document) || extractHumanName(author);
60
+
61
+ date = extractDateQuick(document, url) || date;
62
+
63
+ //extract from ways of writing dates in natural language into standard format
64
+ date = parseDate(date)?.toISOString().split("T")[0] || null;
65
+
66
+ if (!date)
67
+ try {
68
+ date = extractDate(
69
+ document,
70
+ true,
71
+ true,
72
+ "%Y-%m-%d",
73
+ url,
74
+ false,
75
+ new Date("2002-01-01")
76
+ );
77
+ } catch (e) {
78
+ console.log(e);
79
+ }
80
+
81
+ date = parseDate(date)?.toISOString().split("T")[0] || null;
82
+
83
+ title = extractTitle(document) || title;
84
+ source = extractSource(document) || source;
85
+
86
+ // URL TO SOURCE
87
+ if (!source && url?.length > 20)
88
+ source = url
89
+ .split("//")[1]
90
+ .split("/")[0]
91
+ ?.replace("www.", "")
92
+ .split(" ")
93
+ .map((word) => word[0].toUpperCase() + word.slice(1))
94
+ .join(" ");
95
+
96
+ return { author, author_cite, date, title, source };
97
+ }