extract-webpage 1.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +212 -0
- package/dist/config/env.d.ts +6 -0
- package/dist/config/index.d.ts +23 -0
- package/dist/config/serverRegistry.d.ts +7 -0
- package/dist/config/types.d.ts +4 -0
- package/dist/extract-webpage.cjs.js +2 -0
- package/dist/extract-webpage.cjs.js.map +1 -0
- package/dist/extract-webpage.es.js +5 -0
- package/dist/extract-webpage.es.js.map +1 -0
- package/dist/html-to-cite/extract-author.d.ts +11 -0
- package/dist/html-to-cite/extract-cite.d.ts +33 -0
- package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
- package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
- package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
- package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
- package/dist/html-to-cite/extract-source.d.ts +7 -0
- package/dist/html-to-cite/extract-title.d.ts +11 -0
- package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
- package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
- package/dist/html-to-cite/url-to-domain.d.ts +20 -0
- package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
- package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
- package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
- package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
- package/dist/html-to-content/html-to-content.d.ts +51 -0
- package/dist/html-to-content/html-utils.d.ts +76 -0
- package/dist/index.d.ts +26 -0
- package/dist/search/index.d.ts +14 -0
- package/dist/search/meta-search-agent-reexport.d.ts +8 -0
- package/dist/search/public-searxng.d.ts +47 -0
- package/dist/search/search-web.d.ts +33 -0
- package/dist/search/tavily.d.ts +20 -0
- package/dist/search/url-to-html.d.ts +62 -0
- package/dist/seektopic/fold-keyphrases.d.ts +28 -0
- package/dist/seektopic/ngrams.d.ts +27 -0
- package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
- package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
- package/dist/seektopic/types.d.ts +86 -0
- package/dist/seektopic/vector-search.d.ts +89 -0
- package/dist/seektopic/weight-keyphrases.d.ts +22 -0
- package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
- package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
- package/dist/tokenize/suggest-complete-word.d.ts +48 -0
- package/dist/tokenize/text-to-chunks.d.ts +48 -0
- package/dist/tokenize/text-to-sentences.d.ts +35 -0
- package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
- package/dist/tokenize/word-is-ignored.d.ts +12 -0
- package/dist/tokenize/word-to-root-stem.d.ts +16 -0
- package/dist/url-to-content/docx-to-content.d.ts +22 -0
- package/dist/url-to-content/is-url-adult.d.ts +26 -0
- package/dist/url-to-content/url-to-content.d.ts +127 -0
- package/dist/url-to-content/url-to-html.d.ts +60 -0
- package/dist/url-to-content/youtube-helpers.d.ts +23 -0
- package/dist/url-to-content/youtube-to-text.d.ts +70 -0
- package/dist/utils/documents.d.ts +4 -0
- package/dist/utils/grab.d.ts +18 -0
- package/package.json +109 -0
- package/src/config/env.ts +8 -0
- package/src/config/index.ts +233 -0
- package/src/config/serverRegistry.ts +24 -0
- package/src/config/types.ts +17 -0
- package/src/fs-mock.js +22 -0
- package/src/global.d.ts +8 -0
- package/src/html-to-cite/extract-author.ts +125 -0
- package/src/html-to-cite/extract-cite.ts +97 -0
- package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
- package/src/html-to-cite/extract-date/date-validators.ts +191 -0
- package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
- package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
- package/src/html-to-cite/extract-source.ts +30 -0
- package/src/html-to-cite/extract-title.ts +78 -0
- package/src/html-to-cite/human-names-92k.json +1 -0
- package/src/html-to-cite/human-names-recognize.ts +396 -0
- package/src/html-to-cite/metadata-to-cite.ts +73 -0
- package/src/html-to-cite/url-to-domain.ts +50 -0
- package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
- package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
- package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
- package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
- package/src/html-to-content/html-to-basic-html.ts +282 -0
- package/src/html-to-content/html-to-content.ts +97 -0
- package/src/html-to-content/html-utils.ts +398 -0
- package/src/index.ts +29 -0
- package/src/search/__tests__/public-searxng.test.ts +529 -0
- package/src/search/index.ts +43 -0
- package/src/search/meta-search-agent-reexport.ts +38 -0
- package/src/search/public-searxng.ts +470 -0
- package/src/search/search-web.ts +668 -0
- package/src/search/tavily.ts +106 -0
- package/src/search/url-to-html.ts +278 -0
- package/src/seektopic/fold-keyphrases.ts +87 -0
- package/src/seektopic/ngrams.ts +64 -0
- package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
- package/src/seektopic/seektopic-keyphrases.ts +279 -0
- package/src/seektopic/types.ts +92 -0
- package/src/seektopic/vector-search.ts +232 -0
- package/src/seektopic/weight-keyphrases.ts +59 -0
- package/src/suggest-next-words/autocomplete-ai.ts +38 -0
- package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
- package/src/tokenize/suggest-complete-word.ts +137 -0
- package/src/tokenize/text-to-chunks.ts +150 -0
- package/src/tokenize/text-to-sentences.ts +614 -0
- package/src/tokenize/text-to-topic-tokens.ts +175 -0
- package/src/tokenize/word-is-ignored.ts +53 -0
- package/src/tokenize/word-to-root-stem.ts +151 -0
- package/src/types.d.ts +130 -0
- package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
- package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
- package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
- package/src/url-to-content/docx-to-content.ts +702 -0
- package/src/url-to-content/is-url-adult.ts +318 -0
- package/src/url-to-content/url-to-content.ts +367 -0
- package/src/url-to-content/url-to-html.ts +436 -0
- package/src/url-to-content/youtube-helpers.ts +64 -0
- package/src/url-to-content/youtube-to-text.ts +468 -0
- package/src/utils/documents.ts +71 -0
- package/src/utils/grab.ts +51 -0
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
|
|
2
|
+
/**
|
|
3
|
+
* Extract source from document using common class names
|
|
4
|
+
*
|
|
5
|
+
* @param {document} document document or dom object with article content
|
|
6
|
+
* @returns {object} source
|
|
7
|
+
*/
|
|
8
|
+
export function extractSource(document) {
|
|
9
|
+
var source, arrSources;
|
|
10
|
+
|
|
11
|
+
if (typeof source == "undefined") {
|
|
12
|
+
arrSources = document.getElementsByClassName("og:site_name");
|
|
13
|
+
if (arrSources.length <= 0) {
|
|
14
|
+
arrSources = document.getElementsByClassName("cre");
|
|
15
|
+
}
|
|
16
|
+
if (arrSources.length <= 0) {
|
|
17
|
+
var arrMeta = document.getElementsByTagName("meta");
|
|
18
|
+
|
|
19
|
+
for (var i = 0; i < arrMeta.length; i++) {
|
|
20
|
+
if (arrMeta[i].getAttribute("property") == "og:site_name") {
|
|
21
|
+
source = arrMeta[i].content;
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
if (arrSources.length > 0) {
|
|
26
|
+
source = arrSources[0].content;
|
|
27
|
+
}
|
|
28
|
+
}
|
|
29
|
+
return source;
|
|
30
|
+
}
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Utility for identifying, extracting, and normalizing document titles from HTML.
|
|
3
|
+
* Handles metadata, selectors, and breadcrumb cleaning.
|
|
4
|
+
*/
|
|
5
|
+
/**
|
|
6
|
+
* Extract and clean title from document
|
|
7
|
+
*
|
|
8
|
+
* @param {Document} document - DOM object with article content
|
|
9
|
+
* @returns {string} Extracted and cleaned title
|
|
10
|
+
*/
|
|
11
|
+
export function extractTitle(document) {
|
|
12
|
+
const META_TAGS = [
|
|
13
|
+
'tweetmeme-title', 'dc.title', 'rbtitle', 'headline', 'title', 'og:title'
|
|
14
|
+
];
|
|
15
|
+
|
|
16
|
+
const SELECTORS = [
|
|
17
|
+
'.hentry .entry-title', 'h1#articleHeader', 'h1.articleHeader', 'h1.article',
|
|
18
|
+
'.instapaper_title', '#meebo-title', 'article h1', '#entry-title', '.entry-title',
|
|
19
|
+
'#entryTitle', '#entrytitle', '.entryTitle', '.entrytitle', '#articleTitle',
|
|
20
|
+
'.articleTitle', 'post post-title', 'h1.title', 'h2.article', 'h1',
|
|
21
|
+
'html head title', 'title'
|
|
22
|
+
];
|
|
23
|
+
|
|
24
|
+
let title = '';
|
|
25
|
+
|
|
26
|
+
// Check meta tags
|
|
27
|
+
for (const tag of META_TAGS) {
|
|
28
|
+
const metaTag = document.querySelector(`meta[name="${tag}"], meta[property="${tag}"]`);
|
|
29
|
+
if (metaTag) {
|
|
30
|
+
title = metaTag.getAttribute('content');
|
|
31
|
+
break;
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
// Check selectors if no title found in meta tags
|
|
36
|
+
if (!title) {
|
|
37
|
+
for (const selector of SELECTORS) {
|
|
38
|
+
const element = document.querySelector(selector);
|
|
39
|
+
if (element) {
|
|
40
|
+
title = element.textContent.trim();
|
|
41
|
+
break;
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
// Fall back to document.title if nothing else worked
|
|
47
|
+
if (!title) {
|
|
48
|
+
title = document.title;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
// Clean and normalize the title
|
|
52
|
+
const TITLE_SPLITTERS_RE = /( [|\-\/:\u00bb] )|( - )|(\|)/;
|
|
53
|
+
const DOMAIN_ENDINGS_RE = /\.(com|net|org|io|gov|edu|co\.uk)$/i;
|
|
54
|
+
|
|
55
|
+
// Handle split titles
|
|
56
|
+
if (TITLE_SPLITTERS_RE.test(title)) {
|
|
57
|
+
const splitTitle = title.split(TITLE_SPLITTERS_RE);
|
|
58
|
+
|
|
59
|
+
// Handle breadcrumbed titles
|
|
60
|
+
if (splitTitle.length >= 2) {
|
|
61
|
+
const longestPart = splitTitle.reduce((acc, part) => part?.length > acc?.length ? part : acc, '');
|
|
62
|
+
if (longestPart.length > 10) {
|
|
63
|
+
title = longestPart;
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
// Truncate title if it's too long
|
|
72
|
+
if (title.length > 150) {
|
|
73
|
+
title = title.substring(0, 150);
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
// Strip any remaining HTML tags and normalize spaces
|
|
77
|
+
return title?.replace(/<\/?[^>]+(>|$)/g, '').replace(/\s+/g, ' ').trim();
|
|
78
|
+
}
|