@lmcc-dev/mult-fetch-mcp-server 1.3.1 → 1.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (109) hide show
  1. package/README.md +70 -15
  2. package/README.zh.md +31 -0
  3. package/dist/src/client.js +118 -91
  4. package/dist/src/index.js +0 -0
  5. package/dist/src/lib/fetchers/browser/BrowserFetcher.js +27 -10
  6. package/dist/src/lib/fetchers/browser/BrowserInstance.js +49 -43
  7. package/dist/src/lib/fetchers/common/BaseFetcher.js +106 -10
  8. package/dist/src/lib/fetchers/common/types.js +10 -1
  9. package/dist/src/lib/fetchers/common/utils.js +1 -1
  10. package/dist/src/lib/fetchers/node/HttpClient.js +47 -21
  11. package/dist/src/lib/fetchers/node/NodeFetcher.js +24 -8
  12. package/dist/src/lib/i18n/index.js +2 -2
  13. package/dist/src/lib/i18n/keys/client.js +1 -0
  14. package/dist/src/lib/i18n/keys/extractor.js +26 -0
  15. package/dist/src/lib/i18n/keys/fetcher.js +12 -1
  16. package/dist/src/lib/i18n/keys/index.js +1 -0
  17. package/dist/src/lib/i18n/keys/node.js +2 -0
  18. package/dist/src/lib/i18n/locales/en/client.js +1 -0
  19. package/dist/src/lib/i18n/locales/en/extractor.js +24 -0
  20. package/dist/src/lib/i18n/locales/en/fetcher.js +3 -1
  21. package/dist/src/lib/i18n/locales/en/index.js +2 -0
  22. package/dist/src/lib/i18n/locales/en/node.js +2 -0
  23. package/dist/src/lib/i18n/locales/zh/client.js +1 -0
  24. package/dist/src/lib/i18n/locales/zh/extractor.js +22 -0
  25. package/dist/src/lib/i18n/locales/zh/fetcher.js +13 -3
  26. package/dist/src/lib/i18n/locales/zh/index.js +2 -0
  27. package/dist/src/lib/i18n/locales/zh/node.js +2 -0
  28. package/dist/src/lib/i18n/logger.js +2 -2
  29. package/dist/src/lib/logger.js +38 -17
  30. package/dist/src/lib/server/browser.js +2 -2
  31. package/dist/src/lib/server/fetcher.js +0 -3
  32. package/dist/src/lib/server/index.js +2 -2
  33. package/dist/src/lib/server/prompts.js +4 -4
  34. package/dist/src/lib/server/tools.js +127 -354
  35. package/dist/src/lib/utils/ChunkManager.js +2 -2
  36. package/dist/src/lib/utils/ContentExtractor.js +141 -0
  37. package/dist/src/lib/utils/ContentProcessor.js +5 -11
  38. package/dist/src/lib/utils/ContentSizeManager.js +2 -2
  39. package/dist/src/lib/utils/ErrorHandler.js +1 -0
  40. package/dist/src/lib/utils/TemplateUtils.js +6 -2
  41. package/dist/src/lib/utils/UrlValidator.js +205 -0
  42. package/dist/src/mcp-server.js +0 -0
  43. package/dist/tests/client.test.js +1 -1
  44. package/dist/tests/fetch.test.js +0 -0
  45. package/dist/tests/fetchers/node/HttpClient.test.js +50 -0
  46. package/dist/tests/i18n-missing-keys.js +0 -0
  47. package/dist/tests/i18n-unused-keys.js +0 -0
  48. package/dist/tests/i18n.test.js +0 -0
  49. package/dist/tests/logger.test.js +0 -0
  50. package/dist/tests/mcp-server.test.js +0 -0
  51. package/dist/tests/setup.js +0 -0
  52. package/dist/tests/test-direct-client.js +0 -0
  53. package/dist/tests/test-extract-single.js +389 -0
  54. package/dist/tests/test-i18n.js +0 -0
  55. package/dist/tests/test-mcp-methods.js +0 -0
  56. package/dist/tests/test-mcp.js +0 -0
  57. package/dist/tests/test-mini4k.js +0 -0
  58. package/dist/tests/types.test.js +0 -0
  59. package/dist/tests/utils/ContentExtractor.test.js +173 -0
  60. package/dist/tests/utils/ContentProcessor.test.js +136 -0
  61. package/dist/tests/utils/TemplateUtils.test.js +118 -0
  62. package/dist/tests/utils/UrlValidator.test.js +58 -0
  63. package/package.json +33 -25
  64. package/dist/i18n-test-report.json +0 -8
  65. package/dist/i18n-unused-keys-report.json +0 -8
  66. package/dist/src/lib/BrowserFetcher.js +0 -787
  67. package/dist/src/lib/NodeFetcher.js +0 -492
  68. package/dist/src/lib/i18n/keys.js +0 -529
  69. package/dist/src/test-i18n.js +0 -139
  70. package/dist/tests/BrowserFetcher.test.js +0 -951
  71. package/dist/tests/NodeFetcher.test.js +0 -263
  72. package/dist/tests/i18n-remove-unused-keys.js +0 -236
  73. package/dist/tests/i18n-test-report.json +0 -2004
  74. package/dist/tests/src/lib/i18n/index.js +0 -108
  75. package/dist/tests/src/lib/i18n/keys/base.js +0 -47
  76. package/dist/tests/src/lib/i18n/keys/browser.js +0 -93
  77. package/dist/tests/src/lib/i18n/keys/client.js +0 -70
  78. package/dist/tests/src/lib/i18n/keys/errors.js +0 -34
  79. package/dist/tests/src/lib/i18n/keys/fetcher.js +0 -84
  80. package/dist/tests/src/lib/i18n/keys/index.js +0 -31
  81. package/dist/tests/src/lib/i18n/keys/node.js +0 -56
  82. package/dist/tests/src/lib/i18n/keys/prompts.js +0 -82
  83. package/dist/tests/src/lib/i18n/keys/resources.js +0 -50
  84. package/dist/tests/src/lib/i18n/keys/server.js +0 -64
  85. package/dist/tests/src/lib/i18n/keys/tools.js +0 -34
  86. package/dist/tests/src/lib/i18n/locales/en/browser.js +0 -88
  87. package/dist/tests/src/lib/i18n/locales/en/client.js +0 -66
  88. package/dist/tests/src/lib/i18n/locales/en/errors.js +0 -28
  89. package/dist/tests/src/lib/i18n/locales/en/fetcher.js +0 -71
  90. package/dist/tests/src/lib/i18n/locales/en/index.js +0 -29
  91. package/dist/tests/src/lib/i18n/locales/en/node.js +0 -51
  92. package/dist/tests/src/lib/i18n/locales/en/prompts.js +0 -52
  93. package/dist/tests/src/lib/i18n/locales/en/resources.js +0 -50
  94. package/dist/tests/src/lib/i18n/locales/en/server.js +0 -56
  95. package/dist/tests/src/lib/i18n/locales/en/tools.js +0 -28
  96. package/dist/tests/src/lib/i18n/locales/zh/browser.js +0 -87
  97. package/dist/tests/src/lib/i18n/locales/zh/client.js +0 -66
  98. package/dist/tests/src/lib/i18n/locales/zh/errors.js +0 -28
  99. package/dist/tests/src/lib/i18n/locales/zh/fetcher.js +0 -71
  100. package/dist/tests/src/lib/i18n/locales/zh/index.js +0 -29
  101. package/dist/tests/src/lib/i18n/locales/zh/node.js +0 -51
  102. package/dist/tests/src/lib/i18n/locales/zh/prompts.js +0 -53
  103. package/dist/tests/src/lib/i18n/locales/zh/resources.js +0 -50
  104. package/dist/tests/src/lib/i18n/locales/zh/server.js +0 -57
  105. package/dist/tests/src/lib/i18n/locales/zh/tools.js +0 -28
  106. package/dist/tests/src/lib/i18n/logger.js +0 -114
  107. package/dist/tests/src/lib/logger.js +0 -181
  108. package/dist/tests/tests/test-i18n.js +0 -588
  109. package/dist/vitest.config.js +0 -29
@@ -11,6 +11,7 @@ import { getRandomUserAgent, getSystemProxy } from '../common/utils.js';
11
11
  import { ContentSizeManager } from '../../utils/ContentSizeManager.js';
12
12
  import { BaseFetcher } from '../common/BaseFetcher.js';
13
13
  import { ContentProcessor } from '../../utils/ContentProcessor.js';
14
+ import { validateFetchUrl } from '../../utils/UrlValidator.js';
14
15
  /**
15
16
  * 浏览器模式获取器类 (Browser mode fetcher class)
16
17
  * 使用Puppeteer实现浏览器模式的网页获取 (Implements webpage fetching in browser mode using Puppeteer)
@@ -75,6 +76,7 @@ export class BrowserFetcher extends BaseFetcher {
75
76
  if (effectiveProxy) {
76
77
  log('browser.usingProxy', debug, { proxy: effectiveProxy }, COMPONENTS.BROWSER_FETCH);
77
78
  }
79
+ await validateFetchUrl(url);
78
80
  // 导航到URL
79
81
  log('browser.navigating', debug, { url }, COMPONENTS.BROWSER_FETCH);
80
82
  await page.goto(url, {
@@ -144,7 +146,7 @@ export class BrowserFetcher extends BaseFetcher {
144
146
  * @returns 获取结果 (Fetch result)
145
147
  */
146
148
  async html(requestPayload) {
147
- const { url, debug = false, closeBrowser = false, contentSizeLimit = ContentSizeManager.getDefaultSizeLimit(), enableContentSplitting = true, chunkId, startCursor = 0 } = requestPayload;
149
+ const { url, debug = false, closeBrowser: _closeBrowser = false, contentSizeLimit = ContentSizeManager.getDefaultSizeLimit(), enableContentSplitting = true, chunkId, startCursor = 0, extractContent = false, includeMetadata = true, fallbackToOriginal = true } = requestPayload;
148
150
  // 如果提供了分段ID和起始游标,则从缓存中获取分段内容 (If chunk ID and startCursor are provided, get chunk content from cache)
149
151
  if (chunkId && startCursor !== undefined) {
150
152
  return this.getChunkContent(chunkId, startCursor, contentSizeLimit, debug, COMPONENTS.BROWSER_FETCH);
@@ -159,13 +161,28 @@ export class BrowserFetcher extends BaseFetcher {
159
161
  }
160
162
  // 获取HTML内容 (Get HTML content)
161
163
  const html = result.content[0].text;
164
+ // 处理内容提取 (Process content extraction)
165
+ const { content: processedContent, metadata } = this.processWithContentExtraction(html, url, {
166
+ extractContent,
167
+ includeMetadata,
168
+ fallbackToOriginal
169
+ }, debug, COMPONENTS.BROWSER_FETCH);
162
170
  // 检查内容大小并处理 (Check content size and process)
163
- const chunkingResult = this.handleContentChunking(html, contentSizeLimit, enableContentSplitting, debug, COMPONENTS.BROWSER_FETCH);
171
+ const chunkingResult = this.handleContentChunking(processedContent, contentSizeLimit, enableContentSplitting, debug, COMPONENTS.BROWSER_FETCH);
164
172
  if (chunkingResult) {
173
+ // 如果有元数据,添加到分块结果中 (If there is metadata, add it to the chunk result)
174
+ if (metadata) {
175
+ chunkingResult.metadata = metadata;
176
+ }
165
177
  return chunkingResult;
166
178
  }
167
- // 返回HTML内容 (Return HTML content)
168
- return result;
179
+ // 创建新的结果(使用处理后的内容) (Create new result with processed content)
180
+ const processedResult = BaseFetcher.createSuccessResponse(processedContent);
181
+ // 如果有元数据,添加到结果中 (If there is metadata, add it to the result)
182
+ if (metadata) {
183
+ processedResult.metadata = metadata;
184
+ }
185
+ return processedResult;
169
186
  }
170
187
  catch (error) {
171
188
  // 处理错误 (Handle error)
@@ -180,7 +197,7 @@ export class BrowserFetcher extends BaseFetcher {
180
197
  * @returns 获取结果 (Fetch result)
181
198
  */
182
199
  async json(requestPayload) {
183
- const { url, debug = false, closeBrowser = false, contentSizeLimit = ContentSizeManager.getDefaultSizeLimit(), enableContentSplitting = true, chunkId, startCursor = 0 } = requestPayload;
200
+ const { url, debug = false, closeBrowser: _closeBrowser = false, contentSizeLimit = ContentSizeManager.getDefaultSizeLimit(), enableContentSplitting = true, chunkId, startCursor = 0 } = requestPayload;
184
201
  // 如果提供了分段ID和起始游标,则从缓存中获取分段内容 (If chunk ID and startCursor are provided, get chunk content from cache)
185
202
  if (chunkId && startCursor !== undefined) {
186
203
  return this.getChunkContent(chunkId, startCursor, contentSizeLimit, debug, COMPONENTS.BROWSER_FETCH);
@@ -226,7 +243,7 @@ export class BrowserFetcher extends BaseFetcher {
226
243
  * @returns 获取结果 (Fetch result)
227
244
  */
228
245
  async txt(requestPayload) {
229
- const { url, debug = false, closeBrowser = false, contentSizeLimit = ContentSizeManager.getDefaultSizeLimit(), enableContentSplitting = true, chunkId, startCursor = 0 } = requestPayload;
246
+ const { url, debug = false, closeBrowser: _closeBrowser = false, contentSizeLimit = ContentSizeManager.getDefaultSizeLimit(), enableContentSplitting = true, chunkId, startCursor = 0 } = requestPayload;
230
247
  // 如果提供了分段ID和起始游标,则从缓存中获取分段内容 (If chunk ID and startCursor are provided, get chunk content from cache)
231
248
  if (chunkId && startCursor !== undefined) {
232
249
  return this.getChunkContent(chunkId, startCursor, contentSizeLimit, debug, COMPONENTS.BROWSER_FETCH);
@@ -266,7 +283,7 @@ export class BrowserFetcher extends BaseFetcher {
266
283
  * @returns 纯文本响应 (Plain text response)
267
284
  */
268
285
  async plainText(requestPayload) {
269
- const { url, debug = false, closeBrowser = false, contentSizeLimit = ContentSizeManager.getDefaultSizeLimit(), enableContentSplitting = true, chunkId, startCursor = 0 } = requestPayload;
286
+ const { url: _url, debug = false, closeBrowser: _closeBrowser = false, contentSizeLimit = ContentSizeManager.getDefaultSizeLimit(), enableContentSplitting = true, chunkId, startCursor = 0 } = requestPayload;
270
287
  // 如果提供了分段ID和起始游标,则从缓存中获取分段内容 (If chunk ID and startCursor are provided, get chunk content from cache)
271
288
  if (chunkId && startCursor !== undefined) {
272
289
  return this.getChunkContent(chunkId, startCursor, contentSizeLimit, debug, COMPONENTS.BROWSER_FETCH);
@@ -280,7 +297,7 @@ export class BrowserFetcher extends BaseFetcher {
280
297
  // 获取HTML内容 (Get HTML content)
281
298
  const html = result.content[0].text;
282
299
  // 使用ContentProcessor将HTML转换为纯文本 (Use ContentProcessor to convert HTML to plain text)
283
- const plainText = ContentProcessor.htmlToText(html, debug, COMPONENTS.BROWSER_FETCH);
300
+ const plainText = ContentProcessor.htmlToText(html, debug);
284
301
  // 检查内容大小并处理 (Check content size and process)
285
302
  const chunkingResult = this.handleContentChunking(plainText, contentSizeLimit, enableContentSplitting, debug, COMPONENTS.BROWSER_FETCH);
286
303
  if (chunkingResult) {
@@ -295,7 +312,7 @@ export class BrowserFetcher extends BaseFetcher {
295
312
  * @returns Markdown内容 (Markdown content)
296
313
  */
297
314
  async markdown(requestPayload) {
298
- const { url, debug = false, closeBrowser = false, contentSizeLimit = ContentSizeManager.getDefaultSizeLimit(), enableContentSplitting = true, chunkId, startCursor = 0 } = requestPayload;
315
+ const { url: _url, debug = false, closeBrowser: _closeBrowser = false, contentSizeLimit = ContentSizeManager.getDefaultSizeLimit(), enableContentSplitting = true, chunkId, startCursor = 0 } = requestPayload;
299
316
  // 如果提供了分段ID和起始游标,则从缓存中获取分段内容 (If chunk ID and startCursor are provided, get chunk content from cache)
300
317
  if (chunkId && startCursor !== undefined) {
301
318
  return this.getChunkContent(chunkId, startCursor, contentSizeLimit, debug, COMPONENTS.BROWSER_FETCH);
@@ -309,7 +326,7 @@ export class BrowserFetcher extends BaseFetcher {
309
326
  // 获取HTML内容 (Get HTML content)
310
327
  const html = result.content[0].text;
311
328
  // 使用ContentProcessor将HTML转换为Markdown (Use ContentProcessor to convert HTML to Markdown)
312
- const markdown = ContentProcessor.htmlToMarkdown(html, debug, COMPONENTS.BROWSER_FETCH);
329
+ const markdown = ContentProcessor.htmlToMarkdown(html, debug);
313
330
  // 检查内容大小并处理 (Check content size and process)
314
331
  const chunkingResult = this.handleContentChunking(markdown, contentSizeLimit, enableContentSplitting, debug, COMPONENTS.BROWSER_FETCH);
315
332
  if (chunkingResult) {
@@ -88,13 +88,13 @@ export class BrowserInstance {
88
88
  // 标记浏览器正在启动 (Mark browser as starting)
89
89
  this.browserStarting = true;
90
90
  // 创建启动Promise (Create startup Promise)
91
- this.browserStartPromise = new Promise(async (resolve, reject) => {
92
- try {
93
- log('browser.startingBrowser', debug, {}, COMPONENTS.BROWSER_FETCH);
94
- // 准备浏览器启动参数 (Prepare browser startup parameters)
95
- const launchOptions = {
96
- headless: 'new',
97
- args: [
91
+ this.browserStartPromise = new Promise((resolve, reject) => {
92
+ // 内部定义并立即调用异步函数
93
+ (async () => {
94
+ try {
95
+ log('browser.startingBrowser', debug, {}, COMPONENTS.BROWSER_FETCH);
96
+ // 准备浏览器启动参数 (Prepare browser startup parameters)
97
+ const browserArgs = [
98
98
  '--no-sandbox',
99
99
  '--disable-setuid-sandbox',
100
100
  '--disable-dev-shm-usage',
@@ -112,51 +112,57 @@ export class BrowserInstance {
112
112
  '--disable-features=IsolateOrigins,site-per-process',
113
113
  '--disable-web-security',
114
114
  '--disable-site-isolation-trials',
115
- '--disable-features=BlockInsecurePrivateNetworkRequests',
116
115
  '--disable-features=IsolateOrigins',
117
116
  '--disable-features=site-per-process',
118
117
  '--disable-blink-features=AutomationControlled',
119
118
  '--user-agent=' + getRandomUserAgent()
120
- ],
121
- ignoreHTTPSErrors: true,
122
- defaultViewport: {
123
- width: 1920,
124
- height: 1080
119
+ ];
120
+ if (process.env.ALLOW_PRIVATE_NETWORK_REQUESTS === 'true') {
121
+ browserArgs.push('--disable-features=BlockInsecurePrivateNetworkRequests');
125
122
  }
126
- };
127
- // 检查是否有环境变量指定的Chrome路径 (Check if there is a Chrome path specified by environment variables)
128
- const executablePath = process.env.PUPPETEER_EXECUTABLE_PATH;
129
- if (executablePath) {
130
- log('browser.usingCustomChromePath', debug, { path: executablePath }, COMPONENTS.BROWSER_FETCH);
131
- launchOptions.executablePath = executablePath;
132
- launchOptions.channel = undefined;
123
+ const launchOptions = {
124
+ headless: 'new',
125
+ args: browserArgs,
126
+ ignoreHTTPSErrors: true,
127
+ defaultViewport: {
128
+ width: 1920,
129
+ height: 1080
130
+ }
131
+ };
132
+ // 检查是否有环境变量指定的Chrome路径 (Check if there is a Chrome path specified by environment variables)
133
+ const executablePath = process.env.PUPPETEER_EXECUTABLE_PATH;
134
+ if (executablePath) {
135
+ log('browser.usingCustomChromePath', debug, { path: executablePath }, COMPONENTS.BROWSER_FETCH);
136
+ launchOptions.executablePath = executablePath;
137
+ launchOptions.channel = undefined;
138
+ }
139
+ // 启动浏览器 (Launch browser)
140
+ // @ts-ignore - 忽略puppeteer-extra的类型错误
141
+ this.browser = await puppeteerExtra.launch(launchOptions);
142
+ log('browser.browserStarted', debug, {}, COMPONENTS.BROWSER_FETCH);
143
+ // 设置浏览器关闭事件处理 (Set browser close event handling)
144
+ this.browser.on('disconnected', () => {
145
+ log('browser.browserDisconnected', debug, {}, COMPONENTS.BROWSER_FETCH);
146
+ this.browser = null;
147
+ this.browserStarting = false;
148
+ this.browserStartPromise = null;
149
+ });
150
+ // 重置标志 (Reset flags)
151
+ this.browserStarting = false;
152
+ // 解析Promise (Resolve Promise)
153
+ resolve(this.browser);
133
154
  }
134
- // 启动浏览器 (Launch browser)
135
- // @ts-ignore - 忽略puppeteer-extra的类型错误
136
- this.browser = await puppeteerExtra.launch(launchOptions);
137
- log('browser.browserStarted', debug, {}, COMPONENTS.BROWSER_FETCH);
138
- // 设置浏览器关闭事件处理 (Set browser close event handling)
139
- this.browser.on('disconnected', () => {
140
- log('browser.browserDisconnected', debug, {}, COMPONENTS.BROWSER_FETCH);
155
+ catch (error) {
156
+ // 记录错误 (Log error)
157
+ log('browser.browserStartError', true, { error: String(error) }, COMPONENTS.BROWSER_FETCH);
158
+ // 重置标志 (Reset flags)
141
159
  this.browser = null;
142
160
  this.browserStarting = false;
143
161
  this.browserStartPromise = null;
144
- });
145
- // 重置标志 (Reset flags)
146
- this.browserStarting = false;
147
- // 解析Promise (Resolve Promise)
148
- resolve(this.browser);
149
- }
150
- catch (error) {
151
- // 记录错误 (Log error)
152
- log('browser.browserStartError', true, { error: String(error) }, COMPONENTS.BROWSER_FETCH);
153
- // 重置标志 (Reset flags)
154
- this.browser = null;
155
- this.browserStarting = false;
156
- this.browserStartPromise = null;
157
- // 拒绝Promise (Reject Promise)
158
- reject(error);
159
- }
162
+ // 拒绝Promise (Reject Promise)
163
+ reject(error);
164
+ }
165
+ })();
160
166
  });
161
167
  return this.browserStartPromise;
162
168
  }
@@ -3,10 +3,11 @@
3
3
  * Co-Author: AI Assistant (Claude)
4
4
  * Description: This code was collaboratively developed by Martin and AI Assistant.
5
5
  */
6
- import { log } from "../../logger.js";
7
- import { ContentSizeManager } from "../../utils/ContentSizeManager.js";
8
- import { ChunkManager } from "../../utils/ChunkManager.js";
9
- import { TemplateUtils } from "../../utils/TemplateUtils.js";
6
+ import { log } from '../../logger.js';
7
+ import { ChunkManager } from '../../utils/ChunkManager.js';
8
+ import { ContentSizeManager } from '../../utils/ContentSizeManager.js';
9
+ import { TemplateUtils } from '../../utils/TemplateUtils.js';
10
+ import { ContentExtractor } from '../../utils/ContentExtractor.js';
10
11
  /**
11
12
  * 基础获取器类 (Base fetcher class)
12
13
  * 提供所有获取器共用的基础功能 (Provides basic functionality used by all fetchers)
@@ -80,7 +81,14 @@ export class BaseFetcher {
80
81
  const estimatedRequests = Math.ceil(remainingBytes / contentSizeLimit);
81
82
  // 添加分段提示 (Add chunk prompt)
82
83
  const isFirstRequest = true;
83
- const chunkWithPrompt = firstChunk + TemplateUtils.generateSizeBasedChunkPrompt(fetchedBytes, totalSize, chunkId, remainingBytes, estimatedRequests, contentSizeLimit, isFirstRequest);
84
+ let chunkWithPrompt;
85
+ // 检查是否已经检索了全部内容
86
+ if (remainingBytes <= 0) {
87
+ chunkWithPrompt = firstChunk + TemplateUtils.generateSizeBasedLastChunkPrompt(fetchedBytes, totalSize, isFirstRequest);
88
+ }
89
+ else {
90
+ chunkWithPrompt = firstChunk + TemplateUtils.generateSizeBasedChunkPrompt(fetchedBytes, totalSize, chunkId, remainingBytes, estimatedRequests, contentSizeLimit, isFirstRequest);
91
+ }
84
92
  // 创建响应 (Create response)
85
93
  return {
86
94
  isError: false,
@@ -136,12 +144,26 @@ export class BaseFetcher {
136
144
  }
137
145
  // 获取相关信息 (Get related information)
138
146
  const { content, fetchedBytes, remainingBytes, isLastChunk, totalBytes } = chunkResult;
147
+ // 记录是否为最后一个分块的详细信息
148
+ log('fetcher.chunkInfo', debug, {
149
+ chunkId,
150
+ fetchedBytes,
151
+ totalBytes,
152
+ remainingBytes,
153
+ isLastChunk,
154
+ percentage: `${Math.round((fetchedBytes / totalBytes) * 100)}%`
155
+ }, component);
139
156
  // 计算预计还需要的请求次数 (Calculate estimated number of requests needed)
140
157
  const estimatedRequests = Math.ceil(remainingBytes / sizeLimit);
141
158
  // 添加分段提示 (Add chunk prompt)
142
159
  let contentWithPrompt;
143
- if (isLastChunk) {
144
- // 如果是最后一个分段 (If it's the last chunk)
160
+ // 如果是最后一个分段或没有剩余内容 (If it's the last chunk or there's no remaining content)
161
+ if (isLastChunk || remainingBytes <= 0) {
162
+ log('fetcher.lastChunkDetected', debug, {
163
+ fetchedBytes,
164
+ totalBytes,
165
+ remainingBytes
166
+ }, component);
145
167
  contentWithPrompt = content + TemplateUtils.generateSizeBasedLastChunkPrompt(fetchedBytes, totalBytes, startCursor === 0 // 如果startCursor为0,则为首次请求 (If startCursor is 0, it's the first request)
146
168
  );
147
169
  }
@@ -164,7 +186,8 @@ export class BaseFetcher {
164
186
  totalBytes,
165
187
  fetchedBytes,
166
188
  remainingBytes,
167
- hasMoreChunks: !isLastChunk
189
+ hasMoreChunks: remainingBytes > 0,
190
+ isLastChunk // 明确添加isLastChunk属性以便客户端使用
168
191
  };
169
192
  }
170
193
  /**
@@ -184,8 +207,18 @@ export class BaseFetcher {
184
207
  return response;
185
208
  }
186
209
  // 使用基于字节的提示 (Use byte-based prompt)
187
- const promptText = TemplateUtils.generateSizeBasedChunkPrompt(response.fetchedBytes, response.totalBytes, chunkId, response.remainingBytes, estimatedRequests, currentSizeLimit, false // 设置为false表示这不是首次请求 (Set to false indicating this is not the first request)
188
- );
210
+ let promptText;
211
+ // 检查是否还有剩余字节 (Check if there are remaining bytes)
212
+ if (response.remainingBytes <= 0) {
213
+ // 没有剩余字节,使用最后一块的提示 (No remaining bytes, use last chunk prompt)
214
+ promptText = TemplateUtils.generateSizeBasedLastChunkPrompt(response.fetchedBytes, response.totalBytes, false // 设置为false表示这不是首次请求 (Set to false indicating this is not the first request)
215
+ );
216
+ }
217
+ else {
218
+ // 仍有剩余字节,使用普通分块提示 (Still has remaining bytes, use regular chunk prompt)
219
+ promptText = TemplateUtils.generateSizeBasedChunkPrompt(response.fetchedBytes, response.totalBytes, chunkId, response.remainingBytes, estimatedRequests, currentSizeLimit, false // 设置为false表示这不是首次请求 (Set to false indicating this is not the first request)
220
+ );
221
+ }
189
222
  // 创建新的响应对象,避免修改原始对象 (Create new response object to avoid modifying the original)
190
223
  return {
191
224
  ...response,
@@ -200,4 +233,67 @@ export class BaseFetcher {
200
233
  // 如果不需要添加提示,返回原始响应 (If no prompt needed, return original response)
201
234
  return response;
202
235
  }
236
+ /**
237
+ * 处理内容提取 (Process content extraction)
238
+ * 根据参数决定是否提取内容,返回处理后的内容和元数据 (Decide whether to extract content based on parameters, return processed content and metadata)
239
+ * @param html 原始HTML内容 (Original HTML content)
240
+ * @param url 页面URL (Page URL)
241
+ * @param options 内容提取选项 (Content extraction options)
242
+ * @param debug 是否启用调试 (Whether to enable debugging)
243
+ * @param component 组件名称,用于日志 (Component name for logging)
244
+ * @returns 处理后的内容和元数据 (Processed content and metadata)
245
+ */
246
+ processWithContentExtraction(html, url, options, debug = false, component) {
247
+ const { extractContent = false, includeMetadata = true, fallbackToOriginal = true } = options;
248
+ // 如果不需要提取内容,直接返回原始内容 (If content extraction is not needed, return original content)
249
+ if (!extractContent) {
250
+ return { content: html, metadata: null };
251
+ }
252
+ log('fetcher.extractingContent', debug, { url }, component);
253
+ try {
254
+ // 提取内容 (Extract content)
255
+ const extractResult = ContentExtractor.extractContent(html, url, debug);
256
+ // 如果提取成功且有内容 (If extraction is successful and has content)
257
+ if (extractResult.content) {
258
+ // 准备元数据 (Prepare metadata)
259
+ const metadata = includeMetadata ? {
260
+ title: extractResult.title,
261
+ byline: extractResult.byline,
262
+ siteName: extractResult.siteName,
263
+ excerpt: extractResult.excerpt,
264
+ isReaderable: extractResult.isReaderable,
265
+ length: extractResult.length
266
+ } : null;
267
+ log('fetcher.extractionSuccess', debug, {
268
+ contentLength: extractResult.content.length,
269
+ title: extractResult.title
270
+ }, component);
271
+ // 直接返回处理后的内容 (Directly return the processed content)
272
+ return { content: extractResult.content, metadata };
273
+ }
274
+ else {
275
+ log('fetcher.extractionFailed', debug, { url }, component);
276
+ // 如果提取失败且不允许回退到原始内容 (If extraction failed and fallback is not allowed)
277
+ if (!fallbackToOriginal) {
278
+ log('fetcher.noFallback', debug, {}, component);
279
+ throw new Error("Content extraction failed, and fallback is disabled");
280
+ }
281
+ log('fetcher.usingOriginalContent', debug, {}, component);
282
+ return { content: html, metadata: null };
283
+ }
284
+ }
285
+ catch (error) {
286
+ log('fetcher.extractionError', debug, {
287
+ error: error instanceof Error ? error.message : String(error),
288
+ url
289
+ }, component);
290
+ // 如果不允许回退到原始内容,则抛出错误 (If fallback is not allowed, throw error)
291
+ if (!fallbackToOriginal) {
292
+ throw error;
293
+ }
294
+ // 回退到原始内容 (Fallback to original content)
295
+ log('fetcher.fallbackToOriginal', debug, {}, component);
296
+ return { content: html, metadata: null };
297
+ }
298
+ }
203
299
  }
@@ -17,6 +17,15 @@ export const BrowserParamsSchema = z.object({
17
17
  autoDetectMode: z.boolean().optional(),
18
18
  closeBrowser: z.boolean().optional(),
19
19
  });
20
+ /**
21
+ * 通用提取选项 (Common extraction options)
22
+ * 用于配置内容提取行为 (Used to configure content extraction behavior)
23
+ */
24
+ export const ContentExtractionOptionsSchema = z.object({
25
+ extractContent: z.boolean().optional(),
26
+ includeMetadata: z.boolean().optional(),
27
+ fallbackToOriginal: z.boolean().optional()
28
+ });
20
29
  /**
21
30
  * 请求参数模式 (Request parameters schema)
22
31
  */
@@ -42,4 +51,4 @@ export const RequestPayloadSchema = z.object({
42
51
  enableContentSplitting: z.boolean().optional(),
43
52
  chunkId: z.string().optional(),
44
53
  startCursor: z.number().optional(),
45
- }).merge(BrowserParamsSchema);
54
+ }).merge(BrowserParamsSchema).merge(ContentExtractionOptionsSchema);
@@ -51,7 +51,7 @@ export function getDomain(url) {
51
51
  const urlObj = new URL(url);
52
52
  return urlObj.hostname;
53
53
  }
54
- catch (error) {
54
+ catch (_error) {
55
55
  return '';
56
56
  }
57
57
  }
@@ -8,6 +8,7 @@ import { HttpProxyAgent } from "http-proxy-agent";
8
8
  import { HttpsProxyAgent } from "https-proxy-agent";
9
9
  import { log, COMPONENTS } from '../../logger.js';
10
10
  import { getRandomUserAgent, randomDelay, getSystemProxy } from '../common/utils.js';
11
+ import { validateFetchUrl, UrlValidationError } from '../../utils/UrlValidator.js';
11
12
  /**
12
13
  * HTTP客户端类 (HTTP client class)
13
14
  * 处理HTTP请求、重定向和错误 (Handle HTTP requests, redirects and errors)
@@ -58,6 +59,7 @@ export class HttpClient {
58
59
  */
59
60
  static async handleDelay(noDelay, isRedirect, debug) {
60
61
  if (!noDelay && isRedirect) {
62
+ log('node.addingRandomDelay', debug, {}, COMPONENTS.NODE_FETCH);
61
63
  await randomDelay();
62
64
  }
63
65
  }
@@ -163,6 +165,7 @@ export class HttpClient {
163
165
  // 初始化重定向计数器 (Initialize redirect counter)
164
166
  let redirectCount = 0;
165
167
  let currentUrl = url;
168
+ await validateFetchUrl(currentUrl);
166
169
  // 记录请求开始时间 (Record request start time)
167
170
  const fetchStart = Date.now();
168
171
  // 创建AbortController用于超时控制 (Create AbortController for timeout control)
@@ -202,38 +205,61 @@ export class HttpClient {
202
205
  log('node.responseStatus', debug, {
203
206
  status: response.status,
204
207
  statusText: response.statusText,
205
- headers: Object.fromEntries(response.headers.entries()),
208
+ duration: Date.now() - fetchStart,
206
209
  }, COMPONENTS.NODE_FETCH);
207
- // 处理重定向 (Handle redirects)
210
+ // 处理错误响应 (Handle error response)
211
+ if (response.status >= 400) {
212
+ await this.handleErrorResponse(response, debug);
213
+ }
214
+ // 检查是否需要重定向 (Check if redirect is needed)
208
215
  if (response.status >= 300 && response.status < 400 && response.headers.has('location')) {
209
- redirectCount++;
210
216
  // 获取重定向URL (Get redirect URL)
211
217
  const location = response.headers.get('location');
212
- log('node.redirectingTo', debug, { location }, COMPONENTS.NODE_FETCH);
213
- // 构建完整的重定向URL (Build complete redirect URL)
214
218
  currentUrl = this.buildRedirectUrl(location, currentUrl, debug);
219
+ await validateFetchUrl(currentUrl);
220
+ // 增加重定向计数 (Increase redirect count)
221
+ redirectCount++;
222
+ // 记录重定向 (Log redirect)
223
+ log('node.redirecting', debug, {
224
+ to: currentUrl,
225
+ redirectCount,
226
+ }, COMPONENTS.NODE_FETCH);
227
+ // 关闭body读取器 (Close body reader)
228
+ await response.text();
215
229
  continue;
216
230
  }
217
- // 如果响应成功,返回响应 (If response is successful, return response)
218
- if (response.ok) {
219
- log('node.requestSuccess', debug, {}, COMPONENTS.NODE_FETCH);
220
- return response;
221
- }
222
- else {
223
- // 处理错误响应 (Handle error response)
224
- return await this.handleErrorResponse(response, debug);
225
- }
231
+ // 返回成功响应 (Return successful response)
232
+ clearTimeout(timeoutId);
233
+ return response;
226
234
  }
227
- // 如果达到最大重定向次数,抛出错误 (If maximum redirects reached, throw error)
228
- throw new Error(`Too many redirects (${maxRedirects})`);
235
+ // 重定向次数超过限制 (Redirect count exceeds limit)
236
+ const error = new Error(`Too many redirects (${maxRedirects})`);
237
+ error.code = 'EMAXREDIRECTS';
238
+ throw error;
229
239
  }
230
240
  catch (error) {
231
- // 处理请求错误 (Handle request error)
232
- this.handleRequestError(error, fetchStart, timeout, redirectCount, maxRedirects, debug);
233
- }
234
- finally {
235
- // 清除超时定时器 (Clear timeout timer)
236
241
  clearTimeout(timeoutId);
242
+ if (error instanceof UrlValidationError) {
243
+ throw error;
244
+ }
245
+ return this.handleRequestError(error, fetchStart, timeout, redirectCount, maxRedirects, debug);
237
246
  }
238
247
  }
248
+ /**
249
+ * 计算代理通道等待时间 (Calculate proxy channel wait time)
250
+ * @param proxyType 代理类型 (Proxy type)
251
+ * @param proxy 代理服务器 (Proxy server)
252
+ * @param _debug 调试模式 (Debug mode)
253
+ * @returns 等待时间(毫秒) (Wait time in milliseconds)
254
+ */
255
+ static calculateProxyDelay(proxyType, proxy, _debug) {
256
+ // 根据代理类型和代理服务器地址计算合适的延迟时间
257
+ if (!proxy)
258
+ return 0;
259
+ // 基础延迟根据代理类型不同
260
+ const baseDelay = proxyType === 'http' ? 2000 : 3000;
261
+ // 计算延迟变量 (0-1000ms)
262
+ const variableDelay = Math.floor(Math.random() * 1000);
263
+ return baseDelay + variableDelay;
264
+ }
239
265
  }
@@ -9,6 +9,7 @@ import { ErrorHandler } from '../../utils/ErrorHandler.js';
9
9
  import { ContentSizeManager } from '../../utils/ContentSizeManager.js';
10
10
  import { BaseFetcher } from "../common/BaseFetcher.js";
11
11
  import { ContentProcessor } from "../../utils/ContentProcessor.js";
12
+ // import { ContentExtractor } from '../../utils/ContentExtractor.js';
12
13
  /**
13
14
  * Node模式获取器类 (Node mode fetcher class)
14
15
  * 使用node-fetch实现标准模式的网页获取 (Implements webpage fetching in standard mode using node-fetch)
@@ -20,7 +21,7 @@ export class NodeFetcher extends BaseFetcher {
20
21
  * @returns HTML内容 (HTML content)
21
22
  */
22
23
  async html(requestPayload) {
23
- const { debug = false, contentSizeLimit = ContentSizeManager.getDefaultSizeLimit(), enableContentSplitting = true, chunkId, startCursor = 0 } = requestPayload;
24
+ const { debug = false, contentSizeLimit = ContentSizeManager.getDefaultSizeLimit(), enableContentSplitting = true, chunkId, startCursor = 0, extractContent = false, includeMetadata = true, fallbackToOriginal = true } = requestPayload;
24
25
  // 如果提供了分段ID和起始游标,则从缓存中获取分段内容 (If chunk ID and startCursor are provided, get chunk content from cache)
25
26
  if (chunkId && startCursor !== undefined) {
26
27
  return this.getChunkContent(chunkId, startCursor, contentSizeLimit, debug, COMPONENTS.NODE_FETCH);
@@ -33,13 +34,28 @@ export class NodeFetcher extends BaseFetcher {
33
34
  log('node.readingText', debug, {}, COMPONENTS.NODE_FETCH);
34
35
  const html = await response.text();
35
36
  log('node.htmlContentLength', debug, { length: html.length }, COMPONENTS.NODE_FETCH);
37
+ // 处理内容提取 (Process content extraction)
38
+ const { content: processedContent, metadata } = this.processWithContentExtraction(html, requestPayload.url, {
39
+ extractContent,
40
+ includeMetadata,
41
+ fallbackToOriginal
42
+ }, debug, COMPONENTS.NODE_FETCH);
36
43
  // 检查内容大小并处理 (Check content size and process)
37
- const chunkingResult = this.handleContentChunking(html, contentSizeLimit, enableContentSplitting, debug, COMPONENTS.NODE_FETCH);
44
+ const chunkingResult = this.handleContentChunking(processedContent, contentSizeLimit, enableContentSplitting, debug, COMPONENTS.NODE_FETCH);
38
45
  if (chunkingResult) {
46
+ // 如果有元数据,添加到分块结果中 (If there is metadata, add it to the chunk result)
47
+ if (metadata) {
48
+ chunkingResult.metadata = metadata;
49
+ }
39
50
  return chunkingResult;
40
51
  }
41
- // 返回HTML内容 (Return HTML content)
42
- return BaseFetcher.createSuccessResponse(html);
52
+ // 返回内容 (Return content)
53
+ const result = BaseFetcher.createSuccessResponse(processedContent);
54
+ // 如果有元数据,添加到结果中 (If there is metadata, add it to the result)
55
+ if (metadata) {
56
+ result.metadata = metadata;
57
+ }
58
+ return result;
43
59
  }
44
60
  catch (error) {
45
61
  // 使用ErrorHandler处理错误 (Use ErrorHandler to handle error)
@@ -151,8 +167,8 @@ export class NodeFetcher extends BaseFetcher {
151
167
  const response = await HttpClient.fetchWithRedirects(requestPayload);
152
168
  // 读取响应文本 (Read response text)
153
169
  const html = await response.text();
154
- // 使用ContentProcessor将HTML转换为纯文本 (Use ContentProcessor to convert HTML to plain text)
155
- const plainText = ContentProcessor.htmlToText(html, debug, COMPONENTS.NODE_FETCH);
170
+ // 将HTML转换为纯文本 (Convert HTML to plain text)
171
+ const plainText = ContentProcessor.htmlToText(html, debug);
156
172
  // 检查内容大小并处理 (Check content size and process)
157
173
  const chunkingResult = this.handleContentChunking(plainText, contentSizeLimit, enableContentSplitting, debug, COMPONENTS.NODE_FETCH);
158
174
  if (chunkingResult) {
@@ -185,8 +201,8 @@ export class NodeFetcher extends BaseFetcher {
185
201
  const response = await HttpClient.fetchWithRedirects(requestPayload);
186
202
  // 读取响应文本 (Read response text)
187
203
  const html = await response.text();
188
- // 使用ContentProcessor将HTML转换为Markdown (Use ContentProcessor to convert HTML to Markdown)
189
- const markdown = ContentProcessor.htmlToMarkdown(html, debug, COMPONENTS.NODE_FETCH);
204
+ // 将HTML转换为Markdown (Convert HTML to Markdown)
205
+ const markdown = ContentProcessor.htmlToMarkdown(html, debug);
190
206
  // 检查内容大小并处理 (Check content size and process)
191
207
  const chunkingResult = this.handleContentChunking(markdown, contentSizeLimit, enableContentSplitting, debug, COMPONENTS.NODE_FETCH);
192
208
  if (chunkingResult) {
@@ -63,7 +63,7 @@ const initializeI18n = () => {
63
63
  saveMissing: false, // 不保存缺失的翻译键 (Don't save missing translation keys)
64
64
  keySeparator: false, // 禁用键分隔符,使用完整的键路径 (Disable key separator, use full key path)
65
65
  nsSeparator: false, // 禁用命名空间分隔符 (Disable namespace separator)
66
- missingKeyHandler: (lng, ns, key, fallbackValue) => {
66
+ missingKeyHandler: (_lng, _ns, _key, _fallbackValue) => {
67
67
  // 不输出任何日志,由上层调用决定是否输出 (Don't output any logs, let the upper-level call decide whether to output)
68
68
  }
69
69
  });
@@ -89,7 +89,7 @@ export const t = (key, options) => {
89
89
  const result = i18n.t(key, options);
90
90
  return result;
91
91
  }
92
- catch (error) {
92
+ catch (_error) {
93
93
  return key; // 出错时返回原始键 (Return the original key when an error occurs)
94
94
  }
95
95
  };
@@ -54,6 +54,7 @@ export const CLIENT_KEYS = (() => {
54
54
  responseStructure: keyGen('responseStructure'),
55
55
  parsedByteChunkInfo: keyGen('parsedByteChunkInfo'),
56
56
  parsedChunkInfo: keyGen('parsedChunkInfo'),
57
+ chunkInfoParsed: keyGen('chunkInfoParsed'),
57
58
  chunkLimitNotice: keyGen('chunkLimitNotice'),
58
59
  chunkLimitHint: keyGen('chunkLimitHint'),
59
60
  fetchingChunkProgress: keyGen('fetchingChunkProgress'),