extract-webpage 1.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/README.md +212 -0
  2. package/dist/config/env.d.ts +6 -0
  3. package/dist/config/index.d.ts +23 -0
  4. package/dist/config/serverRegistry.d.ts +7 -0
  5. package/dist/config/types.d.ts +4 -0
  6. package/dist/extract-webpage.cjs.js +2 -0
  7. package/dist/extract-webpage.cjs.js.map +1 -0
  8. package/dist/extract-webpage.es.js +5 -0
  9. package/dist/extract-webpage.es.js.map +1 -0
  10. package/dist/html-to-cite/extract-author.d.ts +11 -0
  11. package/dist/html-to-cite/extract-cite.d.ts +33 -0
  12. package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
  13. package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
  14. package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
  15. package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
  16. package/dist/html-to-cite/extract-source.d.ts +7 -0
  17. package/dist/html-to-cite/extract-title.d.ts +11 -0
  18. package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
  19. package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
  20. package/dist/html-to-cite/url-to-domain.d.ts +20 -0
  21. package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
  22. package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
  23. package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
  24. package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
  25. package/dist/html-to-content/html-to-content.d.ts +51 -0
  26. package/dist/html-to-content/html-utils.d.ts +76 -0
  27. package/dist/index.d.ts +26 -0
  28. package/dist/search/index.d.ts +14 -0
  29. package/dist/search/meta-search-agent-reexport.d.ts +8 -0
  30. package/dist/search/public-searxng.d.ts +47 -0
  31. package/dist/search/search-web.d.ts +33 -0
  32. package/dist/search/tavily.d.ts +20 -0
  33. package/dist/search/url-to-html.d.ts +62 -0
  34. package/dist/seektopic/fold-keyphrases.d.ts +28 -0
  35. package/dist/seektopic/ngrams.d.ts +27 -0
  36. package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
  37. package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
  38. package/dist/seektopic/types.d.ts +86 -0
  39. package/dist/seektopic/vector-search.d.ts +89 -0
  40. package/dist/seektopic/weight-keyphrases.d.ts +22 -0
  41. package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
  42. package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
  43. package/dist/tokenize/suggest-complete-word.d.ts +48 -0
  44. package/dist/tokenize/text-to-chunks.d.ts +48 -0
  45. package/dist/tokenize/text-to-sentences.d.ts +35 -0
  46. package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
  47. package/dist/tokenize/word-is-ignored.d.ts +12 -0
  48. package/dist/tokenize/word-to-root-stem.d.ts +16 -0
  49. package/dist/url-to-content/docx-to-content.d.ts +22 -0
  50. package/dist/url-to-content/is-url-adult.d.ts +26 -0
  51. package/dist/url-to-content/url-to-content.d.ts +127 -0
  52. package/dist/url-to-content/url-to-html.d.ts +60 -0
  53. package/dist/url-to-content/youtube-helpers.d.ts +23 -0
  54. package/dist/url-to-content/youtube-to-text.d.ts +70 -0
  55. package/dist/utils/documents.d.ts +4 -0
  56. package/dist/utils/grab.d.ts +18 -0
  57. package/package.json +109 -0
  58. package/src/config/env.ts +8 -0
  59. package/src/config/index.ts +233 -0
  60. package/src/config/serverRegistry.ts +24 -0
  61. package/src/config/types.ts +17 -0
  62. package/src/fs-mock.js +22 -0
  63. package/src/global.d.ts +8 -0
  64. package/src/html-to-cite/extract-author.ts +125 -0
  65. package/src/html-to-cite/extract-cite.ts +97 -0
  66. package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
  67. package/src/html-to-cite/extract-date/date-validators.ts +191 -0
  68. package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
  69. package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
  70. package/src/html-to-cite/extract-source.ts +30 -0
  71. package/src/html-to-cite/extract-title.ts +78 -0
  72. package/src/html-to-cite/human-names-92k.json +1 -0
  73. package/src/html-to-cite/human-names-recognize.ts +396 -0
  74. package/src/html-to-cite/metadata-to-cite.ts +73 -0
  75. package/src/html-to-cite/url-to-domain.ts +50 -0
  76. package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
  77. package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
  78. package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
  79. package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
  80. package/src/html-to-content/html-to-basic-html.ts +282 -0
  81. package/src/html-to-content/html-to-content.ts +97 -0
  82. package/src/html-to-content/html-utils.ts +398 -0
  83. package/src/index.ts +29 -0
  84. package/src/search/__tests__/public-searxng.test.ts +529 -0
  85. package/src/search/index.ts +43 -0
  86. package/src/search/meta-search-agent-reexport.ts +38 -0
  87. package/src/search/public-searxng.ts +470 -0
  88. package/src/search/search-web.ts +668 -0
  89. package/src/search/tavily.ts +106 -0
  90. package/src/search/url-to-html.ts +278 -0
  91. package/src/seektopic/fold-keyphrases.ts +87 -0
  92. package/src/seektopic/ngrams.ts +64 -0
  93. package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
  94. package/src/seektopic/seektopic-keyphrases.ts +279 -0
  95. package/src/seektopic/types.ts +92 -0
  96. package/src/seektopic/vector-search.ts +232 -0
  97. package/src/seektopic/weight-keyphrases.ts +59 -0
  98. package/src/suggest-next-words/autocomplete-ai.ts +38 -0
  99. package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
  100. package/src/tokenize/suggest-complete-word.ts +137 -0
  101. package/src/tokenize/text-to-chunks.ts +150 -0
  102. package/src/tokenize/text-to-sentences.ts +614 -0
  103. package/src/tokenize/text-to-topic-tokens.ts +175 -0
  104. package/src/tokenize/word-is-ignored.ts +53 -0
  105. package/src/tokenize/word-to-root-stem.ts +151 -0
  106. package/src/types.d.ts +130 -0
  107. package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
  108. package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
  109. package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
  110. package/src/url-to-content/docx-to-content.ts +702 -0
  111. package/src/url-to-content/is-url-adult.ts +318 -0
  112. package/src/url-to-content/url-to-content.ts +367 -0
  113. package/src/url-to-content/url-to-html.ts +436 -0
  114. package/src/url-to-content/youtube-helpers.ts +64 -0
  115. package/src/url-to-content/youtube-to-text.ts +468 -0
  116. package/src/utils/documents.ts +71 -0
  117. package/src/utils/grab.ts +51 -0
@@ -0,0 +1,30 @@
1
+
2
+ /**
3
+ * Extract source from document using common class names
4
+ *
5
+ * @param {document} document document or dom object with article content
6
+ * @returns {object} source
7
+ */
8
+ export function extractSource(document) {
9
+ var source, arrSources;
10
+
11
+ if (typeof source == "undefined") {
12
+ arrSources = document.getElementsByClassName("og:site_name");
13
+ if (arrSources.length <= 0) {
14
+ arrSources = document.getElementsByClassName("cre");
15
+ }
16
+ if (arrSources.length <= 0) {
17
+ var arrMeta = document.getElementsByTagName("meta");
18
+
19
+ for (var i = 0; i < arrMeta.length; i++) {
20
+ if (arrMeta[i].getAttribute("property") == "og:site_name") {
21
+ source = arrMeta[i].content;
22
+ }
23
+ }
24
+ }
25
+ if (arrSources.length > 0) {
26
+ source = arrSources[0].content;
27
+ }
28
+ }
29
+ return source;
30
+ }
@@ -0,0 +1,78 @@
1
+ /**
2
+ * @fileoverview Utility for identifying, extracting, and normalizing document titles from HTML.
3
+ * Handles metadata, selectors, and breadcrumb cleaning.
4
+ */
5
+ /**
6
+ * Extract and clean title from document
7
+ *
8
+ * @param {Document} document - DOM object with article content
9
+ * @returns {string} Extracted and cleaned title
10
+ */
11
+ export function extractTitle(document) {
12
+ const META_TAGS = [
13
+ 'tweetmeme-title', 'dc.title', 'rbtitle', 'headline', 'title', 'og:title'
14
+ ];
15
+
16
+ const SELECTORS = [
17
+ '.hentry .entry-title', 'h1#articleHeader', 'h1.articleHeader', 'h1.article',
18
+ '.instapaper_title', '#meebo-title', 'article h1', '#entry-title', '.entry-title',
19
+ '#entryTitle', '#entrytitle', '.entryTitle', '.entrytitle', '#articleTitle',
20
+ '.articleTitle', 'post post-title', 'h1.title', 'h2.article', 'h1',
21
+ 'html head title', 'title'
22
+ ];
23
+
24
+ let title = '';
25
+
26
+ // Check meta tags
27
+ for (const tag of META_TAGS) {
28
+ const metaTag = document.querySelector(`meta[name="${tag}"], meta[property="${tag}"]`);
29
+ if (metaTag) {
30
+ title = metaTag.getAttribute('content');
31
+ break;
32
+ }
33
+ }
34
+
35
+ // Check selectors if no title found in meta tags
36
+ if (!title) {
37
+ for (const selector of SELECTORS) {
38
+ const element = document.querySelector(selector);
39
+ if (element) {
40
+ title = element.textContent.trim();
41
+ break;
42
+ }
43
+ }
44
+ }
45
+
46
+ // Fall back to document.title if nothing else worked
47
+ if (!title) {
48
+ title = document.title;
49
+ }
50
+
51
+ // Clean and normalize the title
52
+ const TITLE_SPLITTERS_RE = /( [|\-\/:\u00bb] )|( - )|(\|)/;
53
+ const DOMAIN_ENDINGS_RE = /\.(com|net|org|io|gov|edu|co\.uk)$/i;
54
+
55
+ // Handle split titles
56
+ if (TITLE_SPLITTERS_RE.test(title)) {
57
+ const splitTitle = title.split(TITLE_SPLITTERS_RE);
58
+
59
+ // Handle breadcrumbed titles
60
+ if (splitTitle.length >= 2) {
61
+ const longestPart = splitTitle.reduce((acc, part) => part?.length > acc?.length ? part : acc, '');
62
+ if (longestPart.length > 10) {
63
+ title = longestPart;
64
+ }
65
+ }
66
+
67
+ }
68
+
69
+
70
+
71
+ // Truncate title if it's too long
72
+ if (title.length > 150) {
73
+ title = title.substring(0, 150);
74
+ }
75
+
76
+ // Strip any remaining HTML tags and normalize spaces
77
+ return title?.replace(/<\/?[^>]+(>|$)/g, '').replace(/\s+/g, ' ').trim();
78
+ }