@lmcc-dev/mult-fetch-mcp-server 1.2.0 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. package/README.md +80 -29
  2. package/README.zh.md +80 -29
  3. package/dist/i18n-test-report.json +2 -2
  4. package/dist/i18n-unused-keys-report.json +4 -4
  5. package/dist/src/client.js +528 -119
  6. package/dist/src/index.js +0 -0
  7. package/dist/src/lib/fetch.js +8 -0
  8. package/dist/src/lib/fetchers/browser/BrowserFetcher.js +139 -199
  9. package/dist/src/lib/fetchers/common/BaseFetcher.js +203 -0
  10. package/dist/src/lib/fetchers/common/types.js +4 -0
  11. package/dist/src/lib/fetchers/common/utils.js +1 -1
  12. package/dist/src/lib/fetchers/index.js +11 -0
  13. package/dist/src/lib/fetchers/node/NodeFetcher.js +91 -66
  14. package/dist/src/lib/i18n/keys/browser.js +40 -2
  15. package/dist/src/lib/i18n/keys/chunkManager.js +43 -0
  16. package/dist/src/lib/i18n/keys/client.js +53 -1
  17. package/dist/src/lib/i18n/keys/contentSize.js +20 -0
  18. package/dist/src/lib/i18n/keys/fetcher.js +88 -79
  19. package/dist/src/lib/i18n/keys/index.js +4 -0
  20. package/dist/src/lib/i18n/keys/node.js +7 -1
  21. package/dist/src/lib/i18n/keys/processor.js +30 -0
  22. package/dist/src/lib/i18n/keys/tools.js +5 -1
  23. package/dist/src/lib/i18n/keys/url.js +45 -0
  24. package/dist/src/lib/i18n/keys.js +25 -0
  25. package/dist/src/lib/i18n/locales/en/browser.js +52 -14
  26. package/dist/src/lib/i18n/locales/en/chunkManager.js +41 -0
  27. package/dist/src/lib/i18n/locales/en/client.js +62 -9
  28. package/dist/src/lib/i18n/locales/en/contentSize.js +17 -0
  29. package/dist/src/lib/i18n/locales/en/fetcher.js +50 -30
  30. package/dist/src/lib/i18n/locales/en/index.js +9 -1
  31. package/dist/src/lib/i18n/locales/en/node.js +26 -20
  32. package/dist/src/lib/i18n/locales/en/processor.js +28 -0
  33. package/dist/src/lib/i18n/locales/en/tools.js +5 -1
  34. package/dist/src/lib/i18n/locales/en/url.js +40 -0
  35. package/dist/src/lib/i18n/locales/en.js +51 -0
  36. package/dist/src/lib/i18n/locales/zh/browser.js +85 -47
  37. package/dist/src/lib/i18n/locales/zh/chunkManager.js +41 -0
  38. package/dist/src/lib/i18n/locales/zh/client.js +53 -1
  39. package/dist/src/lib/i18n/locales/zh/contentSize.js +17 -0
  40. package/dist/src/lib/i18n/locales/zh/fetcher.js +67 -47
  41. package/dist/src/lib/i18n/locales/zh/index.js +9 -1
  42. package/dist/src/lib/i18n/locales/zh/node.js +50 -44
  43. package/dist/src/lib/i18n/locales/zh/processor.js +28 -0
  44. package/dist/src/lib/i18n/locales/zh/tools.js +5 -1
  45. package/dist/src/lib/i18n/locales/zh/url.js +40 -0
  46. package/dist/src/lib/i18n/locales/zh.js +51 -0
  47. package/dist/src/lib/logger.js +11 -2
  48. package/dist/src/lib/server/browser.js +40 -8
  49. package/dist/src/lib/server/fetcher.js +96 -15
  50. package/dist/src/lib/server/tools.js +202 -52
  51. package/dist/src/lib/types.js +5 -0
  52. package/dist/src/lib/utils/ChunkManager.js +177 -0
  53. package/dist/src/lib/utils/ContentProcessor.js +134 -0
  54. package/dist/src/lib/utils/ContentSizeManager.js +123 -0
  55. package/dist/src/lib/utils/TemplateUtils.js +71 -0
  56. package/dist/src/lib/utils/errors.js +2 -8
  57. package/dist/src/mcp-server.js +0 -0
  58. package/dist/src/test-i18n.js +0 -0
  59. package/dist/tests/BrowserFetcher.test.js +0 -0
  60. package/dist/tests/NodeFetcher.test.js +0 -0
  61. package/dist/tests/client.test.js +0 -0
  62. package/dist/tests/fetch.test.js +0 -0
  63. package/dist/tests/fetchers/plain-text.test.js +146 -0
  64. package/dist/tests/i18n-missing-keys.js +0 -0
  65. package/dist/tests/i18n-remove-unused-keys.js +0 -0
  66. package/dist/tests/i18n-unused-keys.js +0 -0
  67. package/dist/tests/i18n.test.js +0 -0
  68. package/dist/tests/logger.test.js +0 -0
  69. package/dist/tests/mcp-server.test.js +0 -0
  70. package/dist/tests/server/fetcher.test.js +69 -15
  71. package/dist/tests/server/tools.test.js +165 -16
  72. package/dist/tests/setup.js +0 -0
  73. package/dist/tests/test-direct-client.js +113 -5
  74. package/dist/tests/test-i18n.js +0 -0
  75. package/dist/tests/test-mcp-methods.js +67 -1
  76. package/dist/tests/test-mcp.js +83 -3
  77. package/dist/tests/test-mini4k.js +90 -3
  78. package/dist/tests/types.test.js +0 -0
  79. package/dist/tests/utils/ChunkManager.test.js +163 -0
  80. package/dist/tests/utils/ContentSizeManager.test.js +170 -0
  81. package/dist/vitest.config.js +0 -0
  82. package/package.json +27 -23
@@ -6,6 +6,7 @@
6
6
  import { Fetcher } from '../fetch.js';
7
7
  import { log, COMPONENTS } from '../logger.js';
8
8
  import { initializeBrowser, closeBrowserInstance, shouldSwitchToBrowser } from './browser.js';
9
+ import { BaseFetcher } from '../fetchers/common/BaseFetcher.js';
9
10
  /**
10
11
  * 辅助函数:根据类型和参数自动选择合适的获取方法 (Helper function: automatically select appropriate fetching method based on type and parameters)
11
12
  * 支持自动在标准模式和浏览器模式之间切换 (Supports automatic switching between standard mode and browser mode)
@@ -38,6 +39,10 @@ export async function fetchWithAutoDetect(params, type) {
38
39
  return await Fetcher.txt(params);
39
40
  case 'markdown':
40
41
  return await Fetcher.markdown(params);
42
+ case 'plaintext':
43
+ return await Fetcher.plainText(params);
44
+ default:
45
+ throw new Error(`Unsupported content type: ${type}`);
41
46
  }
42
47
  }
43
48
  else {
@@ -47,43 +52,122 @@ export async function fetchWithAutoDetect(params, type) {
47
52
  }
48
53
  try {
49
54
  // 根据类型选择合适的标准获取方法 (Choose appropriate standard fetching method based on type)
55
+ let result;
50
56
  switch (type) {
51
57
  case 'html': {
52
- return await Fetcher.html(params);
58
+ result = await Fetcher.html(params);
59
+ break;
53
60
  }
54
61
  case 'json': {
55
- return await Fetcher.json(params);
62
+ result = await Fetcher.json(params);
63
+ break;
56
64
  }
57
65
  case 'txt': {
58
- return await Fetcher.txt(params);
66
+ result = await Fetcher.txt(params);
67
+ break;
59
68
  }
60
69
  case 'markdown': {
61
- return await Fetcher.markdown(params);
70
+ result = await Fetcher.markdown(params);
71
+ break;
72
+ }
73
+ case 'plaintext': {
74
+ result = await Fetcher.plainText(params);
75
+ break;
76
+ }
77
+ default:
78
+ throw new Error(`Unsupported content type: ${type}`);
79
+ }
80
+ // 检查结果是否为错误,并且是否包含403或forbidden关键词
81
+ if (result.isError && autoDetectMode) {
82
+ const errorText = result.content && result.content[0] && result.content[0].text
83
+ ? String(result.content[0].text)
84
+ : '';
85
+ const is403Error = errorText.includes('403') ||
86
+ errorText.toLowerCase().includes('forbidden');
87
+ if (is403Error || shouldSwitchToBrowser(errorText)) {
88
+ if (debug) {
89
+ log('server.switchingToBrowserMode', debug, {
90
+ url: params.url,
91
+ error: errorText,
92
+ reason: is403Error ? '403 Forbidden' : 'Other error requiring browser'
93
+ }, COMPONENTS.SERVER);
94
+ }
95
+ // 确保浏览器已初始化
96
+ await initializeBrowser(debug);
97
+ // 设置浏览器模式参数
98
+ const browserParams = {
99
+ ...params,
100
+ useBrowser: true,
101
+ debug: debug
102
+ };
103
+ // 根据类型选择合适的浏览器获取方法
104
+ let browserResult;
105
+ switch (type) {
106
+ case 'html':
107
+ browserResult = await Fetcher.html(browserParams);
108
+ break;
109
+ case 'json':
110
+ browserResult = await Fetcher.json(browserParams);
111
+ break;
112
+ case 'txt':
113
+ browserResult = await Fetcher.txt(browserParams);
114
+ break;
115
+ case 'markdown':
116
+ browserResult = await Fetcher.markdown(browserParams);
117
+ break;
118
+ case 'plaintext':
119
+ browserResult = await Fetcher.plainText(browserParams);
120
+ break;
121
+ default:
122
+ throw new Error(`Unsupported content type: ${type}`);
123
+ }
124
+ return browserResult;
62
125
  }
63
126
  }
127
+ return result;
64
128
  }
65
129
  catch (error) {
66
130
  // 如果标准模式失败且启用了自动检测
67
- if (autoDetectMode && shouldSwitchToBrowser(error)) {
131
+ const errorMessage = error instanceof Error ? error.message : String(error);
132
+ // 检查是否应该切换到浏览器模式
133
+ // 特别处理 HTTP 403 Forbidden 错误,这是最常见的需要切换到浏览器模式的情况
134
+ const shouldSwitch = autoDetectMode && (shouldSwitchToBrowser(error) ||
135
+ errorMessage.includes('403') ||
136
+ errorMessage.toLowerCase().includes('forbidden'));
137
+ if (shouldSwitch) {
68
138
  if (debug) {
69
139
  log('server.switchingToBrowserMode', debug, { url: params.url }, COMPONENTS.SERVER);
70
140
  }
71
141
  // 确保浏览器已初始化
72
142
  await initializeBrowser(debug);
73
143
  // 设置浏览器模式参数
74
- params.useBrowser = true;
75
- params.debug = debug;
144
+ const browserParams = {
145
+ ...params,
146
+ useBrowser: true,
147
+ debug: debug
148
+ };
76
149
  // 根据类型选择合适的浏览器获取方法
150
+ let browserResult;
77
151
  switch (type) {
78
152
  case 'html':
79
- return await Fetcher.html(params);
153
+ browserResult = await Fetcher.html(browserParams);
154
+ break;
80
155
  case 'json':
81
- return await Fetcher.json(params);
156
+ browserResult = await Fetcher.json(browserParams);
157
+ break;
82
158
  case 'txt':
83
- return await Fetcher.txt(params);
159
+ browserResult = await Fetcher.txt(browserParams);
160
+ break;
84
161
  case 'markdown':
85
- return await Fetcher.markdown(params);
162
+ browserResult = await Fetcher.markdown(browserParams);
163
+ break;
164
+ case 'plaintext':
165
+ browserResult = await Fetcher.plainText(browserParams);
166
+ break;
167
+ default:
168
+ throw new Error(`Unsupported content type: ${type}`);
86
169
  }
170
+ return browserResult;
87
171
  }
88
172
  // 如果不需要切换到浏览器模式,则抛出原始错误
89
173
  throw error;
@@ -102,8 +186,5 @@ export async function fetchWithAutoDetect(params, type) {
102
186
  throw error;
103
187
  }
104
188
  // 如果没有匹配的类型,返回错误
105
- return {
106
- content: [{ type: 'text', text: `Unsupported content type: ${type}` }],
107
- isError: true
108
- };
189
+ return BaseFetcher.createErrorResponse(`Unsupported content type: ${type}`);
109
190
  }
@@ -7,6 +7,8 @@ import { CallToolRequestSchema, ListToolsRequestSchema } from '@modelcontextprot
7
7
  import { log, COMPONENTS } from '../logger.js';
8
8
  import { fetchWithAutoDetect } from './fetcher.js';
9
9
  import { closeBrowserInstance } from './browser.js';
10
+ import { BaseFetcher } from '../fetchers/common/BaseFetcher.js';
11
+ import { TemplateUtils } from '../utils/TemplateUtils.js';
10
12
  /**
11
13
  * 获取HTML获取工具定义 (Get HTML fetch tool definition)
12
14
  * @returns HTML获取工具定义 (HTML fetch tool definition)
@@ -14,7 +16,7 @@ import { closeBrowserInstance } from './browser.js';
14
16
  function getHtmlFetchToolDefinition() {
15
17
  return {
16
18
  name: "fetch_html",
17
- description: "Fetch a website and return the content as HTML",
19
+ description: "Fetch a website and return the content as HTML. Best practices: 1) Always set startCursor=0 for initial requests, and use the fetchedBytes value from previous response for subsequent requests to ensure content continuity. 2) Set contentSizeLimit between 20000-50000 for large pages. 3) When handling large content, use the chunking system by following the startCursor instructions in the system notes rather than increasing contentSizeLimit. 4) If content retrieval fails, you can retry using the same chunkId and startCursor, or adjust startCursor as needed but you must handle any resulting data duplication or gaps yourself. 5) Always explain to users when content is chunked and ask if they want to continue retrieving subsequent parts.",
18
20
  inputSchema: {
19
21
  type: "object",
20
22
  properties: {
@@ -22,6 +24,10 @@ function getHtmlFetchToolDefinition() {
22
24
  type: "string",
23
25
  description: "URL of the website to fetch",
24
26
  },
27
+ startCursor: {
28
+ type: "number",
29
+ description: "Starting cursor position in bytes. Set to 0 for initial requests, and use the value from previous responses for subsequent requests to resume content retrieval.",
30
+ },
25
31
  headers: {
26
32
  type: "object",
27
33
  description: "Optional headers to include in the request",
@@ -56,7 +62,7 @@ function getHtmlFetchToolDefinition() {
56
62
  },
57
63
  autoDetectMode: {
58
64
  type: "boolean",
59
- description: "Optional flag to automatically switch to browser mode if standard fetch fails (default: true)",
65
+ description: "Optional flag to automatically switch to browser mode if standard fetch fails (default: true). Set to false to strictly use the specified mode without automatic switching.",
60
66
  },
61
67
  waitForSelector: {
62
68
  type: "string",
@@ -78,8 +84,20 @@ function getHtmlFetchToolDefinition() {
78
84
  type: "boolean",
79
85
  description: "Optional flag to close the browser after fetching (default: false)",
80
86
  },
87
+ contentSizeLimit: {
88
+ type: "number",
89
+ description: "Optional maximum content size in bytes before splitting into chunks (default: 50KB). Set between 20KB-50KB for optimal results. For large content, prefer smaller values (20KB-30KB) to avoid truncation.",
90
+ },
91
+ enableContentSplitting: {
92
+ type: "boolean",
93
+ description: "Optional flag to enable content splitting for large responses (default: true)",
94
+ },
95
+ chunkId: {
96
+ type: "string",
97
+ description: `Optional chunk ID for retrieving a specific chunk of content from a previous request. The system adds prompts in the format ${TemplateUtils.SYSTEM_NOTE.START} ... ${TemplateUtils.SYSTEM_NOTE.END} which AI models should ignore when processing the content.`,
98
+ },
81
99
  },
82
- required: ["url"],
100
+ required: ["url", "startCursor"],
83
101
  },
84
102
  };
85
103
  }
@@ -90,7 +108,7 @@ function getHtmlFetchToolDefinition() {
90
108
  function getJsonFetchToolDefinition() {
91
109
  return {
92
110
  name: "fetch_json",
93
- description: "Fetch a JSON file from a URL",
111
+ description: "Fetch a JSON file from a URL. Best practices: 1) Always set startCursor=0 for initial requests, and use the fetchedBytes value from previous response for subsequent requests to ensure content continuity. 2) Set contentSizeLimit between 20000-50000 for large files. 3) When handling large content, use the chunking system by following the startCursor instructions in the system notes rather than increasing contentSizeLimit. 4) If content retrieval fails, you can retry using the same chunkId and startCursor, or adjust startCursor as needed but you must handle any resulting data duplication or gaps yourself. 5) Always explain to users when content is chunked and ask if they want to continue retrieving subsequent parts.",
94
112
  inputSchema: {
95
113
  type: "object",
96
114
  properties: {
@@ -98,6 +116,10 @@ function getJsonFetchToolDefinition() {
98
116
  type: "string",
99
117
  description: "URL of the JSON to fetch",
100
118
  },
119
+ startCursor: {
120
+ type: "number",
121
+ description: "Starting cursor position in bytes. Set to 0 for initial requests, and use the value from previous responses for subsequent requests to resume content retrieval.",
122
+ },
101
123
  headers: {
102
124
  type: "object",
103
125
  description: "Optional headers to include in the request",
@@ -142,8 +164,20 @@ function getJsonFetchToolDefinition() {
142
164
  type: "boolean",
143
165
  description: "Optional flag to close the browser after fetching (default: false)",
144
166
  },
167
+ contentSizeLimit: {
168
+ type: "number",
169
+ description: "Optional maximum content size in bytes before splitting into chunks (default: 50KB). Set between 20KB-50KB for optimal results. For large content, prefer smaller values (20KB-30KB) to avoid truncation.",
170
+ },
171
+ enableContentSplitting: {
172
+ type: "boolean",
173
+ description: "Optional flag to enable content splitting for large responses (default: true)",
174
+ },
175
+ chunkId: {
176
+ type: "string",
177
+ description: `Optional chunk ID for retrieving a specific chunk of content from a previous request. The system adds prompts in the format ${TemplateUtils.SYSTEM_NOTE.START} ... ${TemplateUtils.SYSTEM_NOTE.END} which AI models should ignore when processing the content.`,
178
+ },
145
179
  },
146
- required: ["url"],
180
+ required: ["url", "startCursor"],
147
181
  },
148
182
  };
149
183
  }
@@ -154,7 +188,7 @@ function getJsonFetchToolDefinition() {
154
188
  function getTextFetchToolDefinition() {
155
189
  return {
156
190
  name: "fetch_txt",
157
- description: "Fetch a website, return the content as plain text (no HTML)",
191
+ description: "Fetch a website, return the content as plain text (no HTML). Best practices: 1) Always set startCursor=0 for initial requests, and use the fetchedBytes value from previous response for subsequent requests to ensure content continuity. 2) Set contentSizeLimit between 20000-50000 for large pages. 3) When handling large content, use the chunking system by following the startCursor instructions in the system notes rather than increasing contentSizeLimit. 4) If content retrieval fails, you can retry using the same chunkId and startCursor, or adjust startCursor as needed but you must handle any resulting data duplication or gaps yourself. 5) Always explain to users when content is chunked and ask if they want to continue retrieving subsequent parts.",
158
192
  inputSchema: {
159
193
  type: "object",
160
194
  properties: {
@@ -162,6 +196,10 @@ function getTextFetchToolDefinition() {
162
196
  type: "string",
163
197
  description: "URL of the website to fetch",
164
198
  },
199
+ startCursor: {
200
+ type: "number",
201
+ description: "Starting cursor position in bytes. Set to 0 for initial requests, and use the value from previous responses for subsequent requests to resume content retrieval.",
202
+ },
165
203
  headers: {
166
204
  type: "object",
167
205
  description: "Optional headers to include in the request",
@@ -206,8 +244,20 @@ function getTextFetchToolDefinition() {
206
244
  type: "boolean",
207
245
  description: "Optional flag to scroll to bottom of page in browser mode (default: false)",
208
246
  },
247
+ contentSizeLimit: {
248
+ type: "number",
249
+ description: "Optional maximum content size in bytes before splitting into chunks (default: 50KB). Set between 20KB-50KB for optimal results. For large content, prefer smaller values (20KB-30KB) to avoid truncation.",
250
+ },
251
+ enableContentSplitting: {
252
+ type: "boolean",
253
+ description: "Optional flag to enable content splitting for large responses (default: true)",
254
+ },
255
+ chunkId: {
256
+ type: "string",
257
+ description: `Optional chunk ID for retrieving a specific chunk of content from a previous request. The system adds prompts in the format ${TemplateUtils.SYSTEM_NOTE.START} ... ${TemplateUtils.SYSTEM_NOTE.END} which AI models should ignore when processing the content.`,
258
+ },
209
259
  },
210
- required: ["url"],
260
+ required: ["url", "startCursor"],
211
261
  },
212
262
  };
213
263
  }
@@ -218,7 +268,7 @@ function getTextFetchToolDefinition() {
218
268
  function getMarkdownFetchToolDefinition() {
219
269
  return {
220
270
  name: "fetch_markdown",
221
- description: "Fetch a website and return the content as Markdown",
271
+ description: "Fetch a website and return the content as Markdown. Best practices: 1) Always set startCursor=0 for initial requests, and use the fetchedBytes value from previous response for subsequent requests to ensure content continuity. 2) Set contentSizeLimit between 20000-50000 for large pages. 3) When handling large content, use the chunking system by following the startCursor instructions in the system notes rather than increasing contentSizeLimit. 4) If content retrieval fails, you can retry using the same chunkId and startCursor, or adjust startCursor as needed but you must handle any resulting data duplication or gaps yourself. 5) Always explain to users when content is chunked and ask if they want to continue retrieving subsequent parts.",
222
272
  inputSchema: {
223
273
  type: "object",
224
274
  properties: {
@@ -226,6 +276,10 @@ function getMarkdownFetchToolDefinition() {
226
276
  type: "string",
227
277
  description: "URL of the website to fetch",
228
278
  },
279
+ startCursor: {
280
+ type: "number",
281
+ description: "Starting cursor position in bytes. Set to 0 for initial requests, and use the value from previous responses for subsequent requests to resume content retrieval.",
282
+ },
229
283
  headers: {
230
284
  type: "object",
231
285
  description: "Optional headers to include in the request",
@@ -270,8 +324,112 @@ function getMarkdownFetchToolDefinition() {
270
324
  type: "boolean",
271
325
  description: "Optional flag to scroll to bottom of page in browser mode (default: false)",
272
326
  },
327
+ contentSizeLimit: {
328
+ type: "number",
329
+ description: "Optional maximum content size in bytes before splitting into chunks (default: 50KB). Set between 20KB-50KB for optimal results. For large content, prefer smaller values (20KB-30KB) to avoid truncation.",
330
+ },
331
+ enableContentSplitting: {
332
+ type: "boolean",
333
+ description: "Optional flag to enable content splitting for large responses (default: true)",
334
+ },
335
+ chunkId: {
336
+ type: "string",
337
+ description: `Optional chunk ID for retrieving a specific chunk of content from a previous request. The system adds prompts in the format ${TemplateUtils.SYSTEM_NOTE.START} ... ${TemplateUtils.SYSTEM_NOTE.END} which AI models should ignore when processing the content.`,
338
+ },
339
+ },
340
+ required: ["url", "startCursor"],
341
+ },
342
+ };
343
+ }
344
+ /**
345
+ * 获取纯文本获取工具定义 (Get plain text fetch tool definition)
346
+ * @returns 纯文本获取工具定义 (Plain text fetch tool definition)
347
+ */
348
+ function getPlainTextFetchToolDefinition() {
349
+ return {
350
+ name: "fetch_plaintext",
351
+ description: "Fetch a website and return the content as plain text with HTML tags removed. Best practices: 1) Always set startCursor=0 for initial requests, and use the fetchedBytes value from previous response for subsequent requests to ensure content continuity. 2) Set contentSizeLimit between 20000-50000 for large pages. 3) When handling large content, use the chunking system by following the startCursor instructions in the system notes rather than increasing contentSizeLimit. 4) If content retrieval fails, you can retry using the same chunkId and startCursor, or adjust startCursor as needed but you must handle any resulting data duplication or gaps yourself. 5) Always explain to users when content is chunked and ask if they want to continue retrieving subsequent parts.",
352
+ inputSchema: {
353
+ type: "object",
354
+ properties: {
355
+ url: {
356
+ type: "string",
357
+ description: "URL of the website to fetch",
358
+ },
359
+ startCursor: {
360
+ type: "number",
361
+ description: "Starting cursor position in bytes. Set to 0 for initial requests, and use the value from previous responses for subsequent requests to resume content retrieval.",
362
+ },
363
+ headers: {
364
+ type: "object",
365
+ description: "Optional headers to include in the request",
366
+ },
367
+ proxy: {
368
+ type: "string",
369
+ description: "Optional proxy server to use (format: http://host:port or https://host:port)",
370
+ },
371
+ noDelay: {
372
+ type: "boolean",
373
+ description: "Optional flag to disable random delay between requests (default: false)",
374
+ },
375
+ timeout: {
376
+ type: "number",
377
+ description: "Optional timeout in milliseconds (default: 30000)",
378
+ },
379
+ maxRedirects: {
380
+ type: "number",
381
+ description: "Optional maximum number of redirects to follow (default: 10)",
382
+ },
383
+ useSystemProxy: {
384
+ type: "boolean",
385
+ description: "Optional flag to use system proxy environment variables (default: true)",
386
+ },
387
+ debug: {
388
+ type: "boolean",
389
+ description: "Optional flag to enable detailed debug logging (default: false)",
390
+ },
391
+ useBrowser: {
392
+ type: "boolean",
393
+ description: "Optional flag to use headless browser for fetching (default: false)",
394
+ },
395
+ autoDetectMode: {
396
+ type: "boolean",
397
+ description: "Optional flag to automatically switch to browser mode if standard fetch fails (default: true). Set to false to strictly use the specified mode without automatic switching.",
398
+ },
399
+ waitForSelector: {
400
+ type: "string",
401
+ description: "Optional CSS selector to wait for when using browser mode",
402
+ },
403
+ waitForTimeout: {
404
+ type: "number",
405
+ description: "Optional timeout to wait after page load in browser mode (default: 5000)",
406
+ },
407
+ scrollToBottom: {
408
+ type: "boolean",
409
+ description: "Optional flag to scroll to bottom of page in browser mode (default: false)",
410
+ },
411
+ saveCookies: {
412
+ type: "boolean",
413
+ description: "Optional flag to save cookies for future requests to the same domain (default: true)",
414
+ },
415
+ closeBrowser: {
416
+ type: "boolean",
417
+ description: "Optional flag to close the browser after fetching (default: false)",
418
+ },
419
+ contentSizeLimit: {
420
+ type: "number",
421
+ description: "Optional maximum content size in bytes before splitting into chunks (default: 50KB). Set between 20KB-50KB for optimal results. For large content, prefer smaller values (20KB-30KB) to avoid truncation.",
422
+ },
423
+ enableContentSplitting: {
424
+ type: "boolean",
425
+ description: "Optional flag to enable content splitting for large responses (default: true)",
426
+ },
427
+ chunkId: {
428
+ type: "string",
429
+ description: `Optional chunk ID for retrieving a specific chunk of content from a previous request. The system adds prompts in the format ${TemplateUtils.SYSTEM_NOTE.START} ... ${TemplateUtils.SYSTEM_NOTE.END} which AI models should ignore when processing the content.`,
430
+ },
273
431
  },
274
- required: ["url"],
432
+ required: ["url", "startCursor"],
275
433
  },
276
434
  };
277
435
  }
@@ -284,7 +442,8 @@ function getAllToolDefinitions() {
284
442
  getHtmlFetchToolDefinition(),
285
443
  getJsonFetchToolDefinition(),
286
444
  getTextFetchToolDefinition(),
287
- getMarkdownFetchToolDefinition()
445
+ getMarkdownFetchToolDefinition(),
446
+ getPlainTextFetchToolDefinition()
288
447
  ];
289
448
  }
290
449
  /**
@@ -310,7 +469,7 @@ function registerToolCallHandler(server) {
310
469
  log('tools.callReceived', debug, { name, args: JSON.stringify(args) }, COMPONENTS.SERVER);
311
470
  try {
312
471
  // 处理不同类型的获取请求 (Handle different types of fetch requests)
313
- if (name === 'fetch_html' || name === 'fetch_json' || name === 'fetch_txt' || name === 'fetch_markdown') {
472
+ if (name === 'fetch_html' || name === 'fetch_json' || name === 'fetch_txt' || name === 'fetch_markdown' || name === 'fetch_plaintext') {
314
473
  const result = await handleFetchRequest(name, args, debug);
315
474
  // 直接返回符合 MCP SDK 要求的格式
316
475
  return result;
@@ -318,29 +477,13 @@ function registerToolCallHandler(server) {
318
477
  else {
319
478
  // 未知工具 (Unknown tool)
320
479
  log('tools.unknownTool', debug, { name }, COMPONENTS.SERVER);
321
- return {
322
- content: [
323
- {
324
- type: "text",
325
- text: `Unknown tool: ${name}`
326
- }
327
- ],
328
- isError: true
329
- };
480
+ return BaseFetcher.createErrorResponse(`Unknown tool: ${name}`);
330
481
  }
331
482
  }
332
483
  catch (error) {
333
484
  // 处理错误 (Handle error)
334
485
  log('tools.callError', debug, { name, error: error.message }, COMPONENTS.SERVER);
335
- return {
336
- content: [
337
- {
338
- type: "text",
339
- text: error.message
340
- }
341
- ],
342
- isError: true
343
- };
486
+ return BaseFetcher.createErrorResponse(`Error fetching ${args.url || `chunk ${args.chunkId}`}: ${error.message}`);
344
487
  }
345
488
  finally {
346
489
  // 如果请求参数中指定了关闭浏览器,则关闭浏览器 (If request parameters specify to close browser, close browser)
@@ -359,20 +502,22 @@ function registerToolCallHandler(server) {
359
502
  */
360
503
  async function handleFetchRequest(name, args, debug) {
361
504
  // 验证URL参数 (Validate URL parameter)
362
- if (!args.url) {
363
- log('tools.missingUrl', debug, {}, COMPONENTS.SERVER);
364
- return {
365
- content: [
366
- {
367
- type: "text",
368
- text: "URL parameter is required"
369
- }
370
- ],
371
- isError: true
372
- };
505
+ if (!args.url && !args.chunkId) {
506
+ log('tools.missingUrlOrChunkId', debug, {}, COMPONENTS.SERVER);
507
+ return BaseFetcher.createErrorResponse("Either URL or chunkId parameter is required");
508
+ }
509
+ // 验证startCursor参数 (Validate startCursor parameter)
510
+ if (args.startCursor === undefined) {
511
+ log('tools.missingStartCursor', debug, {}, COMPONENTS.SERVER);
512
+ return BaseFetcher.createErrorResponse("startCursor parameter is required. Use 0 for initial requests.");
373
513
  }
374
514
  // 记录请求 (Log request)
375
- log('tools.fetchRequest', debug, { url: args.url, type: name }, COMPONENTS.SERVER);
515
+ if (args.chunkId) {
516
+ log('tools.fetchChunkRequest', debug, { chunkId: args.chunkId, startCursor: args.startCursor, type: name }, COMPONENTS.SERVER);
517
+ }
518
+ else {
519
+ log('tools.fetchRequest', debug, { url: args.url, startCursor: args.startCursor, type: name }, COMPONENTS.SERVER);
520
+ }
376
521
  // 执行获取请求 (Execute fetch request)
377
522
  try {
378
523
  const params = {
@@ -381,7 +526,7 @@ async function handleFetchRequest(name, args, debug) {
381
526
  };
382
527
  // 根据工具名称确定内容类型 (Determine content type based on tool name)
383
528
  const type = name.replace('fetch_', '');
384
- const result = await fetchWithAutoDetect(params, type);
529
+ let result = await fetchWithAutoDetect(params, type);
385
530
  // 确保返回标准结构体 (Ensure returning standard structure)
386
531
  if (result.isError) {
387
532
  // 确保错误内容有正确的类型 (Ensure error content has correct type)
@@ -407,6 +552,19 @@ async function handleFetchRequest(name, args, debug) {
407
552
  if (!result.content[0].type) {
408
553
  result.content[0].type = "text";
409
554
  }
555
+ // 如果内容是分段的,添加提示词 (If content is chunked, add prompt)
556
+ if (result.isChunked && result.hasMoreChunks) {
557
+ // 检查内容中是否已经包含系统提示,避免重复添加
558
+ // Check if content already contains system note to avoid duplicate
559
+ if (!TemplateUtils.hasSystemPrompt(result.content[0].text || '')) {
560
+ // 使用BaseFetcher的addChunkPrompt方法添加提示词 (Use BaseFetcher's addChunkPrompt method to add prompt)
561
+ result = BaseFetcher.addChunkPrompt(result);
562
+ }
563
+ }
564
+ }
565
+ // 如果没有匹配的类型,返回错误
566
+ if (!result.content[0].type) {
567
+ return BaseFetcher.createErrorResponse(`Unsupported content type: ${type}`);
410
568
  }
411
569
  // 将FetchResponse转换为符合MCP SDK要求的格式
412
570
  return {
@@ -417,16 +575,8 @@ async function handleFetchRequest(name, args, debug) {
417
575
  catch (error) {
418
576
  // 处理错误 (Handle error)
419
577
  log('tools.fetchError', debug, { url: args.url, error: error.message }, COMPONENTS.SERVER);
420
- // 确保返回标准结构体 (Ensure returning standard structure)
421
- return {
422
- content: [
423
- {
424
- type: "text",
425
- text: `Error fetching ${args.url}: ${error.message}`
426
- }
427
- ],
428
- isError: true
429
- };
578
+ // 返回错误信息 (Return error message)
579
+ return BaseFetcher.createErrorResponse(`Error fetching ${args.url || `chunk ${args.chunkId}`}: ${error.message}`);
430
580
  }
431
581
  }
432
582
  /**
@@ -32,4 +32,9 @@ export const RequestPayloadSchema = z.object({
32
32
  waitForTimeout: z.number().optional(),
33
33
  scrollToBottom: z.boolean().optional(),
34
34
  closeBrowser: z.boolean().optional(),
35
+ chunkId: z.string().optional(),
36
+ chunkIndex: z.number().optional(),
37
+ contentSizeLimit: z.number().optional(),
38
+ startCursor: z.number().optional(),
39
+ enableContentSplitting: z.boolean().optional(),
35
40
  }).merge(BrowserParamsSchema);