@lmcc-dev/mult-fetch-mcp-server 1.3.1 → 1.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (109) hide show
  1. package/README.md +70 -15
  2. package/README.zh.md +31 -0
  3. package/dist/src/client.js +118 -91
  4. package/dist/src/index.js +0 -0
  5. package/dist/src/lib/fetchers/browser/BrowserFetcher.js +27 -10
  6. package/dist/src/lib/fetchers/browser/BrowserInstance.js +49 -43
  7. package/dist/src/lib/fetchers/common/BaseFetcher.js +106 -10
  8. package/dist/src/lib/fetchers/common/types.js +10 -1
  9. package/dist/src/lib/fetchers/common/utils.js +1 -1
  10. package/dist/src/lib/fetchers/node/HttpClient.js +47 -21
  11. package/dist/src/lib/fetchers/node/NodeFetcher.js +24 -8
  12. package/dist/src/lib/i18n/index.js +2 -2
  13. package/dist/src/lib/i18n/keys/client.js +1 -0
  14. package/dist/src/lib/i18n/keys/extractor.js +26 -0
  15. package/dist/src/lib/i18n/keys/fetcher.js +12 -1
  16. package/dist/src/lib/i18n/keys/index.js +1 -0
  17. package/dist/src/lib/i18n/keys/node.js +2 -0
  18. package/dist/src/lib/i18n/locales/en/client.js +1 -0
  19. package/dist/src/lib/i18n/locales/en/extractor.js +24 -0
  20. package/dist/src/lib/i18n/locales/en/fetcher.js +3 -1
  21. package/dist/src/lib/i18n/locales/en/index.js +2 -0
  22. package/dist/src/lib/i18n/locales/en/node.js +2 -0
  23. package/dist/src/lib/i18n/locales/zh/client.js +1 -0
  24. package/dist/src/lib/i18n/locales/zh/extractor.js +22 -0
  25. package/dist/src/lib/i18n/locales/zh/fetcher.js +13 -3
  26. package/dist/src/lib/i18n/locales/zh/index.js +2 -0
  27. package/dist/src/lib/i18n/locales/zh/node.js +2 -0
  28. package/dist/src/lib/i18n/logger.js +2 -2
  29. package/dist/src/lib/logger.js +38 -17
  30. package/dist/src/lib/server/browser.js +2 -2
  31. package/dist/src/lib/server/fetcher.js +0 -3
  32. package/dist/src/lib/server/index.js +2 -2
  33. package/dist/src/lib/server/prompts.js +4 -4
  34. package/dist/src/lib/server/tools.js +127 -354
  35. package/dist/src/lib/utils/ChunkManager.js +2 -2
  36. package/dist/src/lib/utils/ContentExtractor.js +141 -0
  37. package/dist/src/lib/utils/ContentProcessor.js +5 -11
  38. package/dist/src/lib/utils/ContentSizeManager.js +2 -2
  39. package/dist/src/lib/utils/ErrorHandler.js +1 -0
  40. package/dist/src/lib/utils/TemplateUtils.js +6 -2
  41. package/dist/src/lib/utils/UrlValidator.js +205 -0
  42. package/dist/src/mcp-server.js +0 -0
  43. package/dist/tests/client.test.js +1 -1
  44. package/dist/tests/fetch.test.js +0 -0
  45. package/dist/tests/fetchers/node/HttpClient.test.js +50 -0
  46. package/dist/tests/i18n-missing-keys.js +0 -0
  47. package/dist/tests/i18n-unused-keys.js +0 -0
  48. package/dist/tests/i18n.test.js +0 -0
  49. package/dist/tests/logger.test.js +0 -0
  50. package/dist/tests/mcp-server.test.js +0 -0
  51. package/dist/tests/setup.js +0 -0
  52. package/dist/tests/test-direct-client.js +0 -0
  53. package/dist/tests/test-extract-single.js +389 -0
  54. package/dist/tests/test-i18n.js +0 -0
  55. package/dist/tests/test-mcp-methods.js +0 -0
  56. package/dist/tests/test-mcp.js +0 -0
  57. package/dist/tests/test-mini4k.js +0 -0
  58. package/dist/tests/types.test.js +0 -0
  59. package/dist/tests/utils/ContentExtractor.test.js +173 -0
  60. package/dist/tests/utils/ContentProcessor.test.js +136 -0
  61. package/dist/tests/utils/TemplateUtils.test.js +118 -0
  62. package/dist/tests/utils/UrlValidator.test.js +58 -0
  63. package/package.json +33 -25
  64. package/dist/i18n-test-report.json +0 -8
  65. package/dist/i18n-unused-keys-report.json +0 -8
  66. package/dist/src/lib/BrowserFetcher.js +0 -787
  67. package/dist/src/lib/NodeFetcher.js +0 -492
  68. package/dist/src/lib/i18n/keys.js +0 -529
  69. package/dist/src/test-i18n.js +0 -139
  70. package/dist/tests/BrowserFetcher.test.js +0 -951
  71. package/dist/tests/NodeFetcher.test.js +0 -263
  72. package/dist/tests/i18n-remove-unused-keys.js +0 -236
  73. package/dist/tests/i18n-test-report.json +0 -2004
  74. package/dist/tests/src/lib/i18n/index.js +0 -108
  75. package/dist/tests/src/lib/i18n/keys/base.js +0 -47
  76. package/dist/tests/src/lib/i18n/keys/browser.js +0 -93
  77. package/dist/tests/src/lib/i18n/keys/client.js +0 -70
  78. package/dist/tests/src/lib/i18n/keys/errors.js +0 -34
  79. package/dist/tests/src/lib/i18n/keys/fetcher.js +0 -84
  80. package/dist/tests/src/lib/i18n/keys/index.js +0 -31
  81. package/dist/tests/src/lib/i18n/keys/node.js +0 -56
  82. package/dist/tests/src/lib/i18n/keys/prompts.js +0 -82
  83. package/dist/tests/src/lib/i18n/keys/resources.js +0 -50
  84. package/dist/tests/src/lib/i18n/keys/server.js +0 -64
  85. package/dist/tests/src/lib/i18n/keys/tools.js +0 -34
  86. package/dist/tests/src/lib/i18n/locales/en/browser.js +0 -88
  87. package/dist/tests/src/lib/i18n/locales/en/client.js +0 -66
  88. package/dist/tests/src/lib/i18n/locales/en/errors.js +0 -28
  89. package/dist/tests/src/lib/i18n/locales/en/fetcher.js +0 -71
  90. package/dist/tests/src/lib/i18n/locales/en/index.js +0 -29
  91. package/dist/tests/src/lib/i18n/locales/en/node.js +0 -51
  92. package/dist/tests/src/lib/i18n/locales/en/prompts.js +0 -52
  93. package/dist/tests/src/lib/i18n/locales/en/resources.js +0 -50
  94. package/dist/tests/src/lib/i18n/locales/en/server.js +0 -56
  95. package/dist/tests/src/lib/i18n/locales/en/tools.js +0 -28
  96. package/dist/tests/src/lib/i18n/locales/zh/browser.js +0 -87
  97. package/dist/tests/src/lib/i18n/locales/zh/client.js +0 -66
  98. package/dist/tests/src/lib/i18n/locales/zh/errors.js +0 -28
  99. package/dist/tests/src/lib/i18n/locales/zh/fetcher.js +0 -71
  100. package/dist/tests/src/lib/i18n/locales/zh/index.js +0 -29
  101. package/dist/tests/src/lib/i18n/locales/zh/node.js +0 -51
  102. package/dist/tests/src/lib/i18n/locales/zh/prompts.js +0 -53
  103. package/dist/tests/src/lib/i18n/locales/zh/resources.js +0 -50
  104. package/dist/tests/src/lib/i18n/locales/zh/server.js +0 -57
  105. package/dist/tests/src/lib/i18n/locales/zh/tools.js +0 -28
  106. package/dist/tests/src/lib/i18n/logger.js +0 -114
  107. package/dist/tests/src/lib/logger.js +0 -181
  108. package/dist/tests/tests/test-i18n.js +0 -588
  109. package/dist/vitest.config.js +0 -29
@@ -0,0 +1,141 @@
1
+ /*
2
+ * Author: Martin <lmccc.dev@gmail.com>
3
+ * Co-Author: AI Assistant (Claude)
4
+ * Description: This code was collaboratively developed by Martin and AI Assistant.
5
+ */
6
+ import { Readability } from '@mozilla/readability';
7
+ import { JSDOM } from 'jsdom';
8
+ import { log, COMPONENTS } from '../logger.js';
9
+ /**
10
+ * 内容提取器类 (Content extractor class)
11
+ * 提供智能内容提取功能,基于Mozilla的Readability库 (Provides intelligent content extraction based on Mozilla's Readability library)
12
+ */
13
+ export class ContentExtractor {
14
+ /**
15
+ * 提取HTML中的主要内容 (Extract main content from HTML)
16
+ * @param html HTML内容 (HTML content)
17
+ * @param url 页面URL,用于处理相对路径 (Page URL, used for handling relative paths)
18
+ * @param debug 是否开启调试 (Whether to enable debugging)
19
+ * @returns 提取的内容对象,包括标题、内容、文本等 (Extracted content object, including title, content, text, etc.)
20
+ */
21
+ static extractContent(html, url, debug = false) {
22
+ log('extractor.creating_jsdom', debug, { url }, COMPONENTS.EXTRACTOR);
23
+ try {
24
+ // 创建JSDOM文档 (Create JSDOM document)
25
+ const doc = new JSDOM(html, { url });
26
+ // 检查页面是否适合进行可读性提取 (Check if the page is suitable for readability extraction)
27
+ const isReaderable = this.isProbablyReaderable(doc.window.document, debug);
28
+ // 如果页面不适合提取,返回空结果 (If the page is not suitable for extraction, return empty result)
29
+ if (!isReaderable && debug) {
30
+ log('extractor.page_not_readerable', true, { url }, COMPONENTS.EXTRACTOR);
31
+ }
32
+ // 即使页面可能不适合提取,我们也尝试进行提取 (Even if the page may not be suitable for extraction, we try to extract anyway)
33
+ // 创建Readability解析器 (Create Readability parser)
34
+ log('extractor.creating_reader', debug, {}, COMPONENTS.EXTRACTOR);
35
+ const reader = new Readability(doc.window.document);
36
+ // 解析内容 (Parse content)
37
+ log('extractor.parsing_content', debug, {}, COMPONENTS.EXTRACTOR);
38
+ const article = reader.parse();
39
+ // 如果解析失败,返回空结果 (If parsing failed, return empty result)
40
+ if (!article) {
41
+ log('extractor.parsing_failed', debug, { url }, COMPONENTS.EXTRACTOR);
42
+ return {
43
+ title: null,
44
+ content: null,
45
+ textContent: null,
46
+ excerpt: null,
47
+ byline: null,
48
+ siteName: null,
49
+ length: 0,
50
+ isReaderable
51
+ };
52
+ }
53
+ log('extractor.parsing_success', debug, {
54
+ title: article.title,
55
+ contentLength: article.content?.length || 0,
56
+ textLength: article.textContent?.length || 0
57
+ }, COMPONENTS.EXTRACTOR);
58
+ // 返回解析结果 (Return parsing result)
59
+ return {
60
+ title: article.title,
61
+ content: article.content,
62
+ textContent: article.textContent,
63
+ excerpt: article.excerpt,
64
+ byline: article.byline,
65
+ siteName: article.siteName,
66
+ length: article.length,
67
+ isReaderable
68
+ };
69
+ }
70
+ catch (error) {
71
+ // 处理错误 (Handle error)
72
+ log('extractor.extraction_error', true, {
73
+ error: error instanceof Error ? error.message : String(error),
74
+ url
75
+ }, COMPONENTS.EXTRACTOR);
76
+ return {
77
+ title: null,
78
+ content: null,
79
+ textContent: null,
80
+ excerpt: null,
81
+ byline: null,
82
+ siteName: null,
83
+ length: 0,
84
+ isReaderable: false
85
+ };
86
+ }
87
+ }
88
+ /**
89
+ * 检查页面是否适合进行可读性提取 (Check if the page is suitable for readability extraction)
90
+ * @param document DOM文档 (DOM document)
91
+ * @param debug 是否开启调试 (Whether to enable debugging)
92
+ * @returns 是否适合提取 (Whether it is suitable for extraction)
93
+ */
94
+ static isProbablyReaderable(document, debug = false) {
95
+ // 这些是常见的无法提取内容的页面类型 (These are common page types that cannot extract content)
96
+ const unlikelyPageTypes = [
97
+ /login/i, /signup/i, /register/i, /404/i, /403/i, /error/i, /captcha/i,
98
+ /password/i, /forgot/i, /reset/i, /signin/i, /signout/i, /logout/i,
99
+ /search/i, /contact/i, /about/i, /faq/i, /help/i, /support/i,
100
+ /dashboard/i, /admin/i, /profile/i, /account/i, /settings/i,
101
+ /cart/i, /checkout/i, /basket/i, /purchase/i, /payment/i,
102
+ /calculator/i, /converter/i, /translator/i
103
+ ];
104
+ // 检查URL和标题是否包含不太可能包含文章的关键词 (Check if URL and title contain keywords that are unlikely to contain articles)
105
+ const url = document.location?.href || '';
106
+ const title = document.title || '';
107
+ for (const pattern of unlikelyPageTypes) {
108
+ if (pattern.test(url) || pattern.test(title)) {
109
+ if (debug) {
110
+ log('extractor.unlikely_page_type', debug, { pattern: pattern.toString(), url, title }, COMPONENTS.EXTRACTOR);
111
+ }
112
+ return false;
113
+ }
114
+ }
115
+ // 检查是否有文章相关元素 (Check if there are article-related elements)
116
+ const hasArticleElements = !!document.querySelector('article') ||
117
+ !!document.querySelector('[role="article"]') ||
118
+ !!document.querySelector('[itemprop="articleBody"]') ||
119
+ !!document.querySelector('.post-content') ||
120
+ !!document.querySelector('.article-content') ||
121
+ !!document.querySelector('.entry-content');
122
+ if (hasArticleElements) {
123
+ if (debug) {
124
+ log('extractor.has_article_elements', debug, {}, COMPONENTS.EXTRACTOR);
125
+ }
126
+ return true;
127
+ }
128
+ // 检查是否有足够的段落 (Check if there are enough paragraphs)
129
+ const paragraphs = document.querySelectorAll('p');
130
+ if (paragraphs.length >= 5) {
131
+ if (debug) {
132
+ log('extractor.has_enough_paragraphs', debug, { count: paragraphs.length }, COMPONENTS.EXTRACTOR);
133
+ }
134
+ return true;
135
+ }
136
+ if (debug) {
137
+ log('extractor.not_enough_content', debug, { paragraphs: paragraphs.length }, COMPONENTS.EXTRACTOR);
138
+ }
139
+ return false;
140
+ }
141
+ }
@@ -16,10 +16,9 @@ export class ContentProcessor {
16
16
  * 将HTML转换为Markdown (Convert HTML to Markdown)
17
17
  * @param html HTML内容 (HTML content)
18
18
  * @param debug 是否开启调试 (Whether to enable debugging)
19
- * @param component 组件名称 (Component name for logging) - 仅用于向下兼容,实际会使用PROCESSOR组件
20
19
  * @returns Markdown内容 (Markdown content)
21
20
  */
22
- static htmlToMarkdown(html, debug, component) {
21
+ static htmlToMarkdown(html, debug) {
23
22
  // 始终使用COMPONENTS.PROCESSOR作为组件标识符 (Always use COMPONENTS.PROCESSOR as component identifier)
24
23
  log('processor.creatingTurndown', debug, {}, COMPONENTS.PROCESSOR);
25
24
  const turndownService = new TurndownService({
@@ -45,10 +44,9 @@ export class ContentProcessor {
45
44
  * 将HTML转换为纯文本 (Convert HTML to plain text)
46
45
  * @param html HTML内容 (HTML content)
47
46
  * @param debug 是否开启调试 (Whether to enable debugging)
48
- * @param component 组件名称 (Component name for logging) - 仅用于向下兼容,实际会使用PROCESSOR组件
49
47
  * @returns 纯文本内容 (Plain text content)
50
48
  */
51
- static htmlToText(html, debug, component) {
49
+ static htmlToText(html, debug) {
52
50
  // 始终使用COMPONENTS.PROCESSOR作为组件标识符 (Always use COMPONENTS.PROCESSOR as component identifier)
53
51
  log('processor.creatingHtmlToText', debug, {}, COMPONENTS.PROCESSOR);
54
52
  // 配置html-to-text选项 (Configure html-to-text options)
@@ -70,10 +68,9 @@ export class ContentProcessor {
70
68
  * 解析JSON字符串 (Parse JSON string)
71
69
  * @param text JSON字符串 (JSON string)
72
70
  * @param debug 是否开启调试 (Whether to enable debugging)
73
- * @param component 组件名称 (Component name for logging) - 仅用于向下兼容,实际会使用PROCESSOR组件
74
71
  * @returns 解析结果 (Parse result - success or error with message)
75
72
  */
76
- static parseJson(text, debug, component) {
73
+ static parseJson(text, debug) {
77
74
  log('processor.parsingJson', debug, {}, COMPONENTS.PROCESSOR);
78
75
  try {
79
76
  const parsed = JSON.parse(text);
@@ -95,10 +92,9 @@ export class ContentProcessor {
95
92
  * 处理文本内容,确保是UTF-8编码 (Process text content, ensure it's UTF-8 encoded)
96
93
  * @param text 文本内容 (Text content)
97
94
  * @param debug 是否开启调试 (Whether to enable debugging)
98
- * @param component 组件名称 (Component name for logging) - 仅用于向下兼容,实际会使用PROCESSOR组件
99
95
  * @returns 处理后的文本 (Processed text)
100
96
  */
101
- static processTextContent(text, debug, component) {
97
+ static processTextContent(text, debug) {
102
98
  log('processor.processingText', debug, { length: text.length }, COMPONENTS.PROCESSOR);
103
99
  // 这里可以添加文本处理逻辑,如编码转换、去除特殊字符等
104
100
  // (Add text processing logic here, such as encoding conversion, removing special characters, etc.)
@@ -109,10 +105,8 @@ export class ContentProcessor {
109
105
  * @param content 内容 (Content)
110
106
  * @param contentSizeLimit 内容大小限制 (Content size limit)
111
107
  * @param debug 是否开启调试 (Whether to enable debugging)
112
- * @param component 组件名称 (Component name for logging) - 仅用于向下兼容,实际会使用PROCESSOR组件
113
- * @throws {ToolError} 如果内容太大且不允许分割 (If content is too large and splitting is not allowed)
114
108
  */
115
- static validateContentSize(content, contentSizeLimit, debug, component) {
109
+ static validateContentSize(content, contentSizeLimit, debug) {
116
110
  const contentSize = content.length;
117
111
  if (contentSize > contentSizeLimit) {
118
112
  log('processor.contentTooLarge', debug, {
@@ -89,10 +89,10 @@ export class ContentSizeManager {
89
89
  * @param content 原始内容 (Original content)
90
90
  * @param sizeLimit 每个片段的大小限制,单位为字节 (Size limit for each chunk in bytes)
91
91
  * @param debug 是否启用调试模式 (Whether debug mode is enabled)
92
- * @param offset 当前偏移量,用于确定是首次请求还是后续请求 (Current offset, used to determine if it's initial or subsequent request)
92
+ * @param _offset 当前偏移量,用于确定是首次请求还是后续请求 (Current offset, used to determine if it's initial or subsequent request)
93
93
  * @returns 内容片段数组 (Array of content chunks)
94
94
  */
95
- static splitContentIntoChunks(content, sizeLimit = this.DEFAULT_SIZE_LIMIT, debug = false, offset = 0) {
95
+ static splitContentIntoChunks(content, sizeLimit = this.DEFAULT_SIZE_LIMIT, debug = false, _offset = 0) {
96
96
  // 计算内容总字节大小 (Calculate total size of content in bytes)
97
97
  const totalBytes = Buffer.byteLength(content, 'utf8');
98
98
  // 使用TemplateUtils中的常量和方法生成示例模板以计算大小
@@ -5,6 +5,7 @@
5
5
  */
6
6
  import { isAccessDeniedError, isNetworkError } from './errorDetection.js';
7
7
  import { log, COMPONENTS } from '../logger.js';
8
+ // import { ERROR_KEYS } from '../i18n/keys/errors.js';
8
9
  /**
9
10
  * 错误类型枚举
10
11
  * (Error type enumeration)
@@ -44,6 +44,11 @@ export class TemplateUtils {
44
44
  * @returns 格式化的提示文本 (Formatted prompt text)
45
45
  */
46
46
  static generateSizeBasedChunkPrompt(fetchedBytes, totalBytes, chunkId, remainingBytes, estimatedRequests, currentSizeLimit, isFirstRequest = true) {
47
+ // 如果剩余字节数为0,使用完成提示而不是继续获取的提示
48
+ // (If remaining bytes is 0, use completion prompt instead of continuation prompt)
49
+ if (remainingBytes <= 0) {
50
+ return this.generateSizeBasedLastChunkPrompt(fetchedBytes, totalBytes, isFirstRequest);
51
+ }
47
52
  const prefix = isFirstRequest ? 'Content is too long and has been split. ' : '';
48
53
  const fetchedPercent = Math.round((fetchedBytes / totalBytes) * 100);
49
54
  return `\n\n${TemplateUtils.SYSTEM_NOTE.START}\n${prefix}You've retrieved ${fetchedBytes.toLocaleString()} bytes (${fetchedPercent}% of total ${totalBytes.toLocaleString()} bytes). ${remainingBytes.toLocaleString()} bytes remaining. With current contentSizeLimit=${currentSizeLimit.toLocaleString()}, approximately ${estimatedRequests} more requests needed to retrieve all content. To continue, use the same tool function with parameters chunkId="${chunkId}" and startCursor=${fetchedBytes}.\n${TemplateUtils.SYSTEM_NOTE.END}`;
@@ -57,8 +62,7 @@ export class TemplateUtils {
57
62
  */
58
63
  static generateSizeBasedLastChunkPrompt(fetchedBytes, totalBytes, isFirstRequest = true) {
59
64
  const prefix = isFirstRequest ? 'Content is too long and has been split. ' : '';
60
- const fetchedPercent = Math.round((fetchedBytes / totalBytes) * 100);
61
- return `\n\n${TemplateUtils.SYSTEM_NOTE.START}\n${prefix}You've retrieved ${fetchedBytes.toLocaleString()} bytes (100% of total ${totalBytes.toLocaleString()} bytes).\nThis is the last part of the content.\n${TemplateUtils.SYSTEM_NOTE.END}`;
65
+ return `\n\n${TemplateUtils.SYSTEM_NOTE.START}\n${prefix}You've retrieved ${fetchedBytes.toLocaleString()} bytes (100% of total ${totalBytes.toLocaleString()} bytes).\nThis is the last part of the content. No further requests needed.\n${TemplateUtils.SYSTEM_NOTE.END}`;
62
66
  }
63
67
  /**
64
68
  * 检查内容是否已包含系统提示 (Check if content already contains system prompt)
@@ -0,0 +1,205 @@
1
+ /**
2
+ * Author: Martin <lmccc.dev@gmail.com>
3
+ * Co-Author: AI Assistant (Claude)
4
+ * Description: SSRF protection for outbound fetch requests.
5
+ */
6
+ import { isIP } from 'net';
7
+ import { lookup } from 'dns/promises';
8
+ const ALLOWED_SCHEMES = new Set(['http:', 'https:']);
9
+ const BLOCKED_HOSTNAMES = new Set([
10
+ 'localhost',
11
+ 'metadata',
12
+ 'metadata.google.internal',
13
+ 'metadata.goog',
14
+ ]);
15
+ export class UrlValidationError extends Error {
16
+ code = 'EBLOCKEDURL';
17
+ constructor(message) {
18
+ super(message);
19
+ this.name = 'UrlValidationError';
20
+ }
21
+ }
22
+ function ipv4ToInt(ip) {
23
+ const parts = ip.split('.');
24
+ if (parts.length !== 4) {
25
+ return -1;
26
+ }
27
+ let value = 0;
28
+ for (const part of parts) {
29
+ const octet = Number(part);
30
+ if (!Number.isInteger(octet) || octet < 0 || octet > 255) {
31
+ return -1;
32
+ }
33
+ value = (value << 8) + octet;
34
+ }
35
+ return value >>> 0;
36
+ }
37
+ function isBlockedIPv4(ip) {
38
+ const value = ipv4ToInt(ip);
39
+ if (value < 0) {
40
+ return true;
41
+ }
42
+ const firstOctet = value >>> 24;
43
+ const secondOctet = (value >>> 16) & 0xff;
44
+ if (firstOctet === 0)
45
+ return true; // 0.0.0.0/8
46
+ if (firstOctet === 10)
47
+ return true; // 10.0.0.0/8
48
+ if (firstOctet === 127)
49
+ return true; // 127.0.0.0/8
50
+ if (firstOctet === 100 && (value & 0xc0000000) === 0x64400000)
51
+ return true; // 100.64.0.0/10
52
+ if (firstOctet === 169 && secondOctet === 254)
53
+ return true; // 169.254.0.0/16
54
+ if (firstOctet === 172 && secondOctet >= 16 && secondOctet <= 31)
55
+ return true; // 172.16.0.0/12
56
+ if (firstOctet === 192 && secondOctet === 168)
57
+ return true; // 192.168.0.0/16
58
+ if (firstOctet === 192 && secondOctet === 0)
59
+ return true; // 192.0.0.0/24
60
+ if (firstOctet === 198 && (secondOctet === 18 || secondOctet === 19))
61
+ return true; // 198.18.0.0/15
62
+ if (firstOctet >= 224)
63
+ return true; // multicast and reserved ranges
64
+ return false;
65
+ }
66
+ function expandIPv6(ip) {
67
+ const lower = ip.toLowerCase();
68
+ const [head = '', tail = ''] = lower.split('::');
69
+ const headParts = head ? head.split(':') : [];
70
+ const tailParts = tail ? tail.split(':') : [];
71
+ if (headParts.length + tailParts.length > 7) {
72
+ return null;
73
+ }
74
+ const missing = 8 - headParts.length - tailParts.length;
75
+ const parts = [
76
+ ...headParts,
77
+ ...Array.from({ length: missing }, () => '0'),
78
+ ...tailParts,
79
+ ];
80
+ if (parts.length !== 8) {
81
+ return null;
82
+ }
83
+ const groups = [];
84
+ for (const part of parts) {
85
+ if (!part) {
86
+ return null;
87
+ }
88
+ const value = Number.parseInt(part, 16);
89
+ if (!Number.isFinite(value) || value < 0 || value > 0xffff) {
90
+ return null;
91
+ }
92
+ groups.push(value);
93
+ }
94
+ return groups;
95
+ }
96
+ function isBlockedIPv6(ip) {
97
+ const normalized = ip.toLowerCase();
98
+ if (normalized === '::' || normalized === '::1') {
99
+ return true;
100
+ }
101
+ const ipv4MappedMatch = normalized.match(/^::ffff:(\d{1,3}(?:\.\d{1,3}){3})$/);
102
+ if (ipv4MappedMatch) {
103
+ return isBlockedIPv4(ipv4MappedMatch[1]);
104
+ }
105
+ const groups = expandIPv6(normalized);
106
+ if (!groups) {
107
+ return true;
108
+ }
109
+ if (groups[0] === 0 &&
110
+ groups[1] === 0 &&
111
+ groups[2] === 0 &&
112
+ groups[3] === 0 &&
113
+ groups[4] === 0 &&
114
+ groups[5] === 0xffff) {
115
+ const mappedValue = (groups[6] << 16) | groups[7];
116
+ const mappedIpv4 = [
117
+ (mappedValue >>> 24) & 0xff,
118
+ (mappedValue >>> 16) & 0xff,
119
+ (mappedValue >>> 8) & 0xff,
120
+ mappedValue & 0xff,
121
+ ].join('.');
122
+ return isBlockedIPv4(mappedIpv4);
123
+ }
124
+ const first = groups[0];
125
+ if ((first & 0xffc0) === 0xfe80)
126
+ return true; // fe80::/10
127
+ if ((first & 0xfe00) === 0xfc00)
128
+ return true; // fc00::/7
129
+ if (first === 0x2001 && groups[1] === 0xdb8)
130
+ return true; // documentation range 2001:db8::/32
131
+ return false;
132
+ }
133
+ function isBlockedIpAddress(ip) {
134
+ const version = isIP(ip);
135
+ if (version === 4) {
136
+ return isBlockedIPv4(ip);
137
+ }
138
+ if (version === 6) {
139
+ return isBlockedIPv6(ip);
140
+ }
141
+ return true;
142
+ }
143
+ function isBlockedHostname(hostname) {
144
+ const normalized = hostname.toLowerCase().replace(/\.$/, '');
145
+ if (BLOCKED_HOSTNAMES.has(normalized)) {
146
+ return true;
147
+ }
148
+ return normalized.endsWith('.localhost');
149
+ }
150
+ function parseFetchUrl(urlString) {
151
+ try {
152
+ return new URL(urlString);
153
+ }
154
+ catch {
155
+ throw new UrlValidationError(`Invalid URL: ${urlString}`);
156
+ }
157
+ }
158
+ function validateScheme(url) {
159
+ if (!ALLOWED_SCHEMES.has(url.protocol)) {
160
+ throw new UrlValidationError(`Blocked URL scheme: ${url.protocol}`);
161
+ }
162
+ }
163
+ function normalizeHostname(hostname) {
164
+ if (hostname.startsWith('[') && hostname.endsWith(']')) {
165
+ return hostname.slice(1, -1);
166
+ }
167
+ return hostname;
168
+ }
169
+ function validateHostnameOrIp(hostname) {
170
+ const normalizedHostname = normalizeHostname(hostname);
171
+ const ipVersion = isIP(normalizedHostname);
172
+ if (ipVersion) {
173
+ if (isBlockedIpAddress(normalizedHostname)) {
174
+ throw new UrlValidationError(`Blocked IP address: ${normalizedHostname}`);
175
+ }
176
+ return;
177
+ }
178
+ if (isBlockedHostname(normalizedHostname)) {
179
+ throw new UrlValidationError(`Blocked hostname: ${normalizedHostname}`);
180
+ }
181
+ }
182
+ async function validateResolvedAddresses(hostname) {
183
+ const normalizedHostname = normalizeHostname(hostname);
184
+ const addresses = await lookup(normalizedHostname, { all: true, verbatim: true });
185
+ if (addresses.length === 0) {
186
+ throw new UrlValidationError(`Unable to resolve hostname: ${normalizedHostname}`);
187
+ }
188
+ for (const address of addresses) {
189
+ if (isBlockedIpAddress(address.address)) {
190
+ throw new UrlValidationError(`Blocked IP address resolved for ${normalizedHostname}: ${address.address}`);
191
+ }
192
+ }
193
+ }
194
+ /**
195
+ * Validate a URL before making an outbound fetch request.
196
+ * Blocks private, loopback, link-local, and metadata targets.
197
+ */
198
+ export async function validateFetchUrl(urlString) {
199
+ const url = parseFetchUrl(urlString);
200
+ validateScheme(url);
201
+ validateHostnameOrIp(url.hostname);
202
+ if (!isIP(normalizeHostname(url.hostname))) {
203
+ await validateResolvedAddresses(url.hostname);
204
+ }
205
+ }
File without changes
@@ -94,7 +94,7 @@ const mockClientModule = {
94
94
  return;
95
95
  }
96
96
  const method = process.argv[2];
97
- let paramsJson = process.argv[3];
97
+ const paramsJson = process.argv[3];
98
98
  try {
99
99
  JSON.parse(paramsJson);
100
100
  }
File without changes
@@ -0,0 +1,50 @@
1
+ /**
2
+ * Author: Martin <lmccc.dev@gmail.com>
3
+ * Co-Author: AI Assistant (Claude)
4
+ * Description: Tests for HttpClient SSRF protections.
5
+ */
6
+ import { describe, expect, test, vi, beforeEach } from 'vitest';
7
+ import fetch from 'node-fetch';
8
+ import { lookup } from 'dns/promises';
9
+ import { HttpClient } from '../../../src/lib/fetchers/node/HttpClient.js';
10
+ import { UrlValidationError } from '../../../src/lib/utils/UrlValidator.js';
11
+ vi.mock('node-fetch', () => ({
12
+ default: vi.fn(),
13
+ }));
14
+ vi.mock('dns/promises', () => ({
15
+ lookup: vi.fn(),
16
+ }));
17
+ vi.mock('../../../src/lib/fetchers/common/utils.js', () => ({
18
+ getRandomUserAgent: vi.fn(() => 'test-agent'),
19
+ randomDelay: vi.fn(async () => undefined),
20
+ getSystemProxy: vi.fn(() => undefined),
21
+ }));
22
+ describe('HttpClient SSRF protection', () => {
23
+ beforeEach(() => {
24
+ vi.clearAllMocks();
25
+ vi.mocked(lookup).mockResolvedValue([{ address: '93.184.216.34', family: 4 }]);
26
+ });
27
+ test('blocks direct loopback requests before fetch is called', async () => {
28
+ await expect(HttpClient.fetchWithRedirects({
29
+ url: 'http://127.0.0.1:8080/latest/meta-data/',
30
+ useSystemProxy: false,
31
+ })).rejects.toThrow(UrlValidationError);
32
+ expect(fetch).not.toHaveBeenCalled();
33
+ });
34
+ test('blocks redirect targets that point to private addresses', async () => {
35
+ vi.mocked(fetch).mockResolvedValueOnce({
36
+ status: 302,
37
+ statusText: 'Found',
38
+ headers: {
39
+ has: (name) => name.toLowerCase() === 'location',
40
+ get: () => 'http://127.0.0.1:8080/internal',
41
+ },
42
+ text: async () => '',
43
+ });
44
+ await expect(HttpClient.fetchWithRedirects({
45
+ url: 'https://example.com/redirect',
46
+ useSystemProxy: false,
47
+ })).rejects.toThrow(/Blocked IP address/);
48
+ expect(fetch).toHaveBeenCalledTimes(1);
49
+ });
50
+ });
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes