extract-webpage 1.2.119 → 1.2.121

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,3 +1,7 @@
1
+ /**
2
+ * @fileoverview Utility for extracting the publication/source name from a
3
+ * document's `og:site_name` meta tag or class-based markers.
4
+ */
1
5
  /**
2
6
  * Extract source from document using common class names
3
7
  *
@@ -1,3 +1,7 @@
1
+ /**
2
+ * @fileoverview Extracts citation metadata (author, date, title, source) from
3
+ * a document's `<meta>` tags using a list of commonly used property/name attributes.
4
+ */
1
5
  export interface CiteMetadata {
2
6
  author?: string;
3
7
  date?: string;
@@ -0,0 +1,4 @@
1
+ /**
2
+ * @fileoverview Local small-model next-word prediction utility (DistilGPT2 via
3
+ * HuggingFace Transformers). Currently disabled/commented out; kept for reference.
4
+ */
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "extract-webpage",
3
- "version": "1.2.119",
3
+ "version": "1.2.121",
4
4
  "module": "./dist/extract-webpage.es.js",
5
5
  "description": "Search, extract, cite, and outline the web for a topic with AI Research Agent.",
6
6
  "author": "vtempest <grokthiscontact@gmail.com>",
@@ -41,6 +41,7 @@
41
41
  "test": "vitest",
42
42
  "ship": "npm run build && npx standard-version --release-as patch; rm CHANGELOG.md; npm publish",
43
43
  "test-ui": "vitest --ui --watch",
44
+ "test:coverage": "vitest run --coverage",
44
45
  "make": "rm -rf dist/*; NODE_OPTIONS=--max-old-space-size=15192 BUN_JSC_forceRAMSize=15192 vite build "
45
46
  },
46
47
  "peerDependencies": {
@@ -54,6 +55,7 @@
54
55
  "devDependencies": {
55
56
  "@tsconfig/svelte": "^5.0.8",
56
57
  "@types/node": "^22.0.0",
58
+ "@vitest/coverage-v8": "^4.0.18",
57
59
  "@vitest/ui": "^4.0.18",
58
60
  "axios": "^1.13.6",
59
61
  "clsx": "^2.1.1",
@@ -72,10 +74,10 @@
72
74
  "dependencies": {
73
75
  "@huggingface/transformers": "^3.8.1",
74
76
  "ai": "^5.0.0",
75
- "chat-agent-toolkit": "^1.2.119",
77
+ "chat-agent-toolkit": "^1.2.121",
76
78
  "chrono-node": "^2.9.0",
77
79
  "drizzle-orm": "^0.45.1",
78
- "extract-pdf": "^0.1.106",
80
+ "extract-pdf": "^0.1.108",
79
81
  "extract-youtube": "^1.0.103",
80
82
  "html-entities": "^2.6.0",
81
83
  "js-yaml": "^4.1.1",
@@ -1,3 +1,7 @@
1
+ /**
2
+ * @fileoverview Convenience accessors for reading server-side config values
3
+ * (model providers, search backends, scrape limits) from the shared ConfigManager.
4
+ */
1
5
  import configManager from "./index";
2
6
  import type { ConfigModelProvider } from "./types";
3
7
 
package/src/fs-mock.js CHANGED
@@ -1,3 +1,7 @@
1
+ /**
2
+ * @fileoverview No-op `fs` module mock used to stub out Node's filesystem API
3
+ * in bundled/browser builds where real filesystem access is unavailable.
4
+ */
1
5
  export const promises = {
2
6
  mkdir: async () => {},
3
7
  readFile: async () => "",
@@ -1,4 +1,9 @@
1
1
 
2
+ /**
3
+ * @fileoverview Utility for extracting the publication/source name from a
4
+ * document's `og:site_name` meta tag or class-based markers.
5
+ */
6
+
2
7
  /**
3
8
  * Extract source from document using common class names
4
9
  *
@@ -1,3 +1,7 @@
1
+ /**
2
+ * @fileoverview Extracts citation metadata (author, date, title, source) from
3
+ * a document's `<meta>` tags using a list of commonly used property/name attributes.
4
+ */
1
5
 
2
6
  export interface CiteMetadata {
3
7
  author?: string;
@@ -1,3 +1,8 @@
1
+ /**
2
+ * @fileoverview Text embedding and semantic similarity utilities using
3
+ * HuggingFace Transformers (local MiniLM model) or the remote Inference API.
4
+ * Provides cosine-similarity-based document reranking against one or more queries.
5
+ */
1
6
  import grab from "../utils/grab";
2
7
 
3
8
  /**
@@ -1,3 +1,7 @@
1
+ /**
2
+ * @fileoverview Local small-model next-word prediction utility (DistilGPT2 via
3
+ * HuggingFace Transformers). Currently disabled/commented out; kept for reference.
4
+ */
1
5
  // import { pipeline, type TextGenerationSingle } from "@huggingface/transformers";
2
6
 
3
7
  // /**
@@ -1,4 +1,8 @@
1
1
  // @ts-nocheck
2
+ /**
3
+ * @fileoverview Web crawler utility ("Tardigrade") that fetches a URL's raw HTML
4
+ * with bot-detection handling and Cloudflare/JINA fallbacks, plus robots.txt checking.
5
+ */
2
6
  import { convertHTMLToBasicHTML } from "../html-to-content/html-to-basic-html";
3
7
  import { convertMarkdownToFormattedHTML } from "../html-to-content/html-utils";
4
8
  import grab from "../utils/grab";
@@ -1,3 +1,7 @@
1
+ /**
2
+ * @fileoverview Fetches web pages from a list of links, strips them to plain
3
+ * text, and splits them into `Document` chunks for downstream use.
4
+ */
1
5
  import axios from 'axios';
2
6
  import { splitTextIntoChunks, type Document } from 'chat-agent-toolkit';
3
7