@lmcc-dev/mult-fetch-mcp-server 1.2.0 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +80 -29
- package/README.zh.md +80 -29
- package/dist/i18n-test-report.json +2 -2
- package/dist/i18n-unused-keys-report.json +4 -4
- package/dist/src/client.js +528 -119
- package/dist/src/index.js +0 -0
- package/dist/src/lib/fetch.js +8 -0
- package/dist/src/lib/fetchers/browser/BrowserFetcher.js +139 -199
- package/dist/src/lib/fetchers/common/BaseFetcher.js +203 -0
- package/dist/src/lib/fetchers/common/types.js +4 -0
- package/dist/src/lib/fetchers/common/utils.js +1 -1
- package/dist/src/lib/fetchers/index.js +11 -0
- package/dist/src/lib/fetchers/node/NodeFetcher.js +91 -66
- package/dist/src/lib/i18n/keys/browser.js +40 -2
- package/dist/src/lib/i18n/keys/chunkManager.js +43 -0
- package/dist/src/lib/i18n/keys/client.js +53 -1
- package/dist/src/lib/i18n/keys/contentSize.js +20 -0
- package/dist/src/lib/i18n/keys/fetcher.js +88 -79
- package/dist/src/lib/i18n/keys/index.js +4 -0
- package/dist/src/lib/i18n/keys/node.js +7 -1
- package/dist/src/lib/i18n/keys/processor.js +30 -0
- package/dist/src/lib/i18n/keys/tools.js +5 -1
- package/dist/src/lib/i18n/keys/url.js +45 -0
- package/dist/src/lib/i18n/keys.js +25 -0
- package/dist/src/lib/i18n/locales/en/browser.js +52 -14
- package/dist/src/lib/i18n/locales/en/chunkManager.js +41 -0
- package/dist/src/lib/i18n/locales/en/client.js +62 -9
- package/dist/src/lib/i18n/locales/en/contentSize.js +17 -0
- package/dist/src/lib/i18n/locales/en/fetcher.js +50 -30
- package/dist/src/lib/i18n/locales/en/index.js +9 -1
- package/dist/src/lib/i18n/locales/en/node.js +26 -20
- package/dist/src/lib/i18n/locales/en/processor.js +28 -0
- package/dist/src/lib/i18n/locales/en/tools.js +5 -1
- package/dist/src/lib/i18n/locales/en/url.js +40 -0
- package/dist/src/lib/i18n/locales/en.js +51 -0
- package/dist/src/lib/i18n/locales/zh/browser.js +85 -47
- package/dist/src/lib/i18n/locales/zh/chunkManager.js +41 -0
- package/dist/src/lib/i18n/locales/zh/client.js +53 -1
- package/dist/src/lib/i18n/locales/zh/contentSize.js +17 -0
- package/dist/src/lib/i18n/locales/zh/fetcher.js +67 -47
- package/dist/src/lib/i18n/locales/zh/index.js +9 -1
- package/dist/src/lib/i18n/locales/zh/node.js +50 -44
- package/dist/src/lib/i18n/locales/zh/processor.js +28 -0
- package/dist/src/lib/i18n/locales/zh/tools.js +5 -1
- package/dist/src/lib/i18n/locales/zh/url.js +40 -0
- package/dist/src/lib/i18n/locales/zh.js +51 -0
- package/dist/src/lib/logger.js +11 -2
- package/dist/src/lib/server/browser.js +40 -8
- package/dist/src/lib/server/fetcher.js +96 -15
- package/dist/src/lib/server/tools.js +202 -52
- package/dist/src/lib/types.js +5 -0
- package/dist/src/lib/utils/ChunkManager.js +177 -0
- package/dist/src/lib/utils/ContentProcessor.js +134 -0
- package/dist/src/lib/utils/ContentSizeManager.js +123 -0
- package/dist/src/lib/utils/TemplateUtils.js +71 -0
- package/dist/src/lib/utils/errors.js +2 -8
- package/dist/src/mcp-server.js +0 -0
- package/dist/src/test-i18n.js +0 -0
- package/dist/tests/BrowserFetcher.test.js +0 -0
- package/dist/tests/NodeFetcher.test.js +0 -0
- package/dist/tests/client.test.js +0 -0
- package/dist/tests/fetch.test.js +0 -0
- package/dist/tests/fetchers/plain-text.test.js +146 -0
- package/dist/tests/i18n-missing-keys.js +0 -0
- package/dist/tests/i18n-remove-unused-keys.js +0 -0
- package/dist/tests/i18n-unused-keys.js +0 -0
- package/dist/tests/i18n.test.js +0 -0
- package/dist/tests/logger.test.js +0 -0
- package/dist/tests/mcp-server.test.js +0 -0
- package/dist/tests/server/fetcher.test.js +69 -15
- package/dist/tests/server/tools.test.js +165 -16
- package/dist/tests/setup.js +0 -0
- package/dist/tests/test-direct-client.js +113 -5
- package/dist/tests/test-i18n.js +0 -0
- package/dist/tests/test-mcp-methods.js +67 -1
- package/dist/tests/test-mcp.js +83 -3
- package/dist/tests/test-mini4k.js +90 -3
- package/dist/tests/types.test.js +0 -0
- package/dist/tests/utils/ChunkManager.test.js +163 -0
- package/dist/tests/utils/ContentSizeManager.test.js +170 -0
- package/dist/vitest.config.js +0 -0
- package/package.json +27 -23
|
@@ -6,6 +6,7 @@
|
|
|
6
6
|
import { Fetcher } from '../fetch.js';
|
|
7
7
|
import { log, COMPONENTS } from '../logger.js';
|
|
8
8
|
import { initializeBrowser, closeBrowserInstance, shouldSwitchToBrowser } from './browser.js';
|
|
9
|
+
import { BaseFetcher } from '../fetchers/common/BaseFetcher.js';
|
|
9
10
|
/**
|
|
10
11
|
* 辅助函数:根据类型和参数自动选择合适的获取方法 (Helper function: automatically select appropriate fetching method based on type and parameters)
|
|
11
12
|
* 支持自动在标准模式和浏览器模式之间切换 (Supports automatic switching between standard mode and browser mode)
|
|
@@ -38,6 +39,10 @@ export async function fetchWithAutoDetect(params, type) {
|
|
|
38
39
|
return await Fetcher.txt(params);
|
|
39
40
|
case 'markdown':
|
|
40
41
|
return await Fetcher.markdown(params);
|
|
42
|
+
case 'plaintext':
|
|
43
|
+
return await Fetcher.plainText(params);
|
|
44
|
+
default:
|
|
45
|
+
throw new Error(`Unsupported content type: ${type}`);
|
|
41
46
|
}
|
|
42
47
|
}
|
|
43
48
|
else {
|
|
@@ -47,43 +52,122 @@ export async function fetchWithAutoDetect(params, type) {
|
|
|
47
52
|
}
|
|
48
53
|
try {
|
|
49
54
|
// 根据类型选择合适的标准获取方法 (Choose appropriate standard fetching method based on type)
|
|
55
|
+
let result;
|
|
50
56
|
switch (type) {
|
|
51
57
|
case 'html': {
|
|
52
|
-
|
|
58
|
+
result = await Fetcher.html(params);
|
|
59
|
+
break;
|
|
53
60
|
}
|
|
54
61
|
case 'json': {
|
|
55
|
-
|
|
62
|
+
result = await Fetcher.json(params);
|
|
63
|
+
break;
|
|
56
64
|
}
|
|
57
65
|
case 'txt': {
|
|
58
|
-
|
|
66
|
+
result = await Fetcher.txt(params);
|
|
67
|
+
break;
|
|
59
68
|
}
|
|
60
69
|
case 'markdown': {
|
|
61
|
-
|
|
70
|
+
result = await Fetcher.markdown(params);
|
|
71
|
+
break;
|
|
72
|
+
}
|
|
73
|
+
case 'plaintext': {
|
|
74
|
+
result = await Fetcher.plainText(params);
|
|
75
|
+
break;
|
|
76
|
+
}
|
|
77
|
+
default:
|
|
78
|
+
throw new Error(`Unsupported content type: ${type}`);
|
|
79
|
+
}
|
|
80
|
+
// 检查结果是否为错误,并且是否包含403或forbidden关键词
|
|
81
|
+
if (result.isError && autoDetectMode) {
|
|
82
|
+
const errorText = result.content && result.content[0] && result.content[0].text
|
|
83
|
+
? String(result.content[0].text)
|
|
84
|
+
: '';
|
|
85
|
+
const is403Error = errorText.includes('403') ||
|
|
86
|
+
errorText.toLowerCase().includes('forbidden');
|
|
87
|
+
if (is403Error || shouldSwitchToBrowser(errorText)) {
|
|
88
|
+
if (debug) {
|
|
89
|
+
log('server.switchingToBrowserMode', debug, {
|
|
90
|
+
url: params.url,
|
|
91
|
+
error: errorText,
|
|
92
|
+
reason: is403Error ? '403 Forbidden' : 'Other error requiring browser'
|
|
93
|
+
}, COMPONENTS.SERVER);
|
|
94
|
+
}
|
|
95
|
+
// 确保浏览器已初始化
|
|
96
|
+
await initializeBrowser(debug);
|
|
97
|
+
// 设置浏览器模式参数
|
|
98
|
+
const browserParams = {
|
|
99
|
+
...params,
|
|
100
|
+
useBrowser: true,
|
|
101
|
+
debug: debug
|
|
102
|
+
};
|
|
103
|
+
// 根据类型选择合适的浏览器获取方法
|
|
104
|
+
let browserResult;
|
|
105
|
+
switch (type) {
|
|
106
|
+
case 'html':
|
|
107
|
+
browserResult = await Fetcher.html(browserParams);
|
|
108
|
+
break;
|
|
109
|
+
case 'json':
|
|
110
|
+
browserResult = await Fetcher.json(browserParams);
|
|
111
|
+
break;
|
|
112
|
+
case 'txt':
|
|
113
|
+
browserResult = await Fetcher.txt(browserParams);
|
|
114
|
+
break;
|
|
115
|
+
case 'markdown':
|
|
116
|
+
browserResult = await Fetcher.markdown(browserParams);
|
|
117
|
+
break;
|
|
118
|
+
case 'plaintext':
|
|
119
|
+
browserResult = await Fetcher.plainText(browserParams);
|
|
120
|
+
break;
|
|
121
|
+
default:
|
|
122
|
+
throw new Error(`Unsupported content type: ${type}`);
|
|
123
|
+
}
|
|
124
|
+
return browserResult;
|
|
62
125
|
}
|
|
63
126
|
}
|
|
127
|
+
return result;
|
|
64
128
|
}
|
|
65
129
|
catch (error) {
|
|
66
130
|
// 如果标准模式失败且启用了自动检测
|
|
67
|
-
|
|
131
|
+
const errorMessage = error instanceof Error ? error.message : String(error);
|
|
132
|
+
// 检查是否应该切换到浏览器模式
|
|
133
|
+
// 特别处理 HTTP 403 Forbidden 错误,这是最常见的需要切换到浏览器模式的情况
|
|
134
|
+
const shouldSwitch = autoDetectMode && (shouldSwitchToBrowser(error) ||
|
|
135
|
+
errorMessage.includes('403') ||
|
|
136
|
+
errorMessage.toLowerCase().includes('forbidden'));
|
|
137
|
+
if (shouldSwitch) {
|
|
68
138
|
if (debug) {
|
|
69
139
|
log('server.switchingToBrowserMode', debug, { url: params.url }, COMPONENTS.SERVER);
|
|
70
140
|
}
|
|
71
141
|
// 确保浏览器已初始化
|
|
72
142
|
await initializeBrowser(debug);
|
|
73
143
|
// 设置浏览器模式参数
|
|
74
|
-
|
|
75
|
-
|
|
144
|
+
const browserParams = {
|
|
145
|
+
...params,
|
|
146
|
+
useBrowser: true,
|
|
147
|
+
debug: debug
|
|
148
|
+
};
|
|
76
149
|
// 根据类型选择合适的浏览器获取方法
|
|
150
|
+
let browserResult;
|
|
77
151
|
switch (type) {
|
|
78
152
|
case 'html':
|
|
79
|
-
|
|
153
|
+
browserResult = await Fetcher.html(browserParams);
|
|
154
|
+
break;
|
|
80
155
|
case 'json':
|
|
81
|
-
|
|
156
|
+
browserResult = await Fetcher.json(browserParams);
|
|
157
|
+
break;
|
|
82
158
|
case 'txt':
|
|
83
|
-
|
|
159
|
+
browserResult = await Fetcher.txt(browserParams);
|
|
160
|
+
break;
|
|
84
161
|
case 'markdown':
|
|
85
|
-
|
|
162
|
+
browserResult = await Fetcher.markdown(browserParams);
|
|
163
|
+
break;
|
|
164
|
+
case 'plaintext':
|
|
165
|
+
browserResult = await Fetcher.plainText(browserParams);
|
|
166
|
+
break;
|
|
167
|
+
default:
|
|
168
|
+
throw new Error(`Unsupported content type: ${type}`);
|
|
86
169
|
}
|
|
170
|
+
return browserResult;
|
|
87
171
|
}
|
|
88
172
|
// 如果不需要切换到浏览器模式,则抛出原始错误
|
|
89
173
|
throw error;
|
|
@@ -102,8 +186,5 @@ export async function fetchWithAutoDetect(params, type) {
|
|
|
102
186
|
throw error;
|
|
103
187
|
}
|
|
104
188
|
// 如果没有匹配的类型,返回错误
|
|
105
|
-
return {
|
|
106
|
-
content: [{ type: 'text', text: `Unsupported content type: ${type}` }],
|
|
107
|
-
isError: true
|
|
108
|
-
};
|
|
189
|
+
return BaseFetcher.createErrorResponse(`Unsupported content type: ${type}`);
|
|
109
190
|
}
|
|
@@ -7,6 +7,8 @@ import { CallToolRequestSchema, ListToolsRequestSchema } from '@modelcontextprot
|
|
|
7
7
|
import { log, COMPONENTS } from '../logger.js';
|
|
8
8
|
import { fetchWithAutoDetect } from './fetcher.js';
|
|
9
9
|
import { closeBrowserInstance } from './browser.js';
|
|
10
|
+
import { BaseFetcher } from '../fetchers/common/BaseFetcher.js';
|
|
11
|
+
import { TemplateUtils } from '../utils/TemplateUtils.js';
|
|
10
12
|
/**
|
|
11
13
|
* 获取HTML获取工具定义 (Get HTML fetch tool definition)
|
|
12
14
|
* @returns HTML获取工具定义 (HTML fetch tool definition)
|
|
@@ -14,7 +16,7 @@ import { closeBrowserInstance } from './browser.js';
|
|
|
14
16
|
function getHtmlFetchToolDefinition() {
|
|
15
17
|
return {
|
|
16
18
|
name: "fetch_html",
|
|
17
|
-
description: "Fetch a website and return the content as HTML",
|
|
19
|
+
description: "Fetch a website and return the content as HTML. Best practices: 1) Always set startCursor=0 for initial requests, and use the fetchedBytes value from previous response for subsequent requests to ensure content continuity. 2) Set contentSizeLimit between 20000-50000 for large pages. 3) When handling large content, use the chunking system by following the startCursor instructions in the system notes rather than increasing contentSizeLimit. 4) If content retrieval fails, you can retry using the same chunkId and startCursor, or adjust startCursor as needed but you must handle any resulting data duplication or gaps yourself. 5) Always explain to users when content is chunked and ask if they want to continue retrieving subsequent parts.",
|
|
18
20
|
inputSchema: {
|
|
19
21
|
type: "object",
|
|
20
22
|
properties: {
|
|
@@ -22,6 +24,10 @@ function getHtmlFetchToolDefinition() {
|
|
|
22
24
|
type: "string",
|
|
23
25
|
description: "URL of the website to fetch",
|
|
24
26
|
},
|
|
27
|
+
startCursor: {
|
|
28
|
+
type: "number",
|
|
29
|
+
description: "Starting cursor position in bytes. Set to 0 for initial requests, and use the value from previous responses for subsequent requests to resume content retrieval.",
|
|
30
|
+
},
|
|
25
31
|
headers: {
|
|
26
32
|
type: "object",
|
|
27
33
|
description: "Optional headers to include in the request",
|
|
@@ -56,7 +62,7 @@ function getHtmlFetchToolDefinition() {
|
|
|
56
62
|
},
|
|
57
63
|
autoDetectMode: {
|
|
58
64
|
type: "boolean",
|
|
59
|
-
description: "Optional flag to automatically switch to browser mode if standard fetch fails (default: true)",
|
|
65
|
+
description: "Optional flag to automatically switch to browser mode if standard fetch fails (default: true). Set to false to strictly use the specified mode without automatic switching.",
|
|
60
66
|
},
|
|
61
67
|
waitForSelector: {
|
|
62
68
|
type: "string",
|
|
@@ -78,8 +84,20 @@ function getHtmlFetchToolDefinition() {
|
|
|
78
84
|
type: "boolean",
|
|
79
85
|
description: "Optional flag to close the browser after fetching (default: false)",
|
|
80
86
|
},
|
|
87
|
+
contentSizeLimit: {
|
|
88
|
+
type: "number",
|
|
89
|
+
description: "Optional maximum content size in bytes before splitting into chunks (default: 50KB). Set between 20KB-50KB for optimal results. For large content, prefer smaller values (20KB-30KB) to avoid truncation.",
|
|
90
|
+
},
|
|
91
|
+
enableContentSplitting: {
|
|
92
|
+
type: "boolean",
|
|
93
|
+
description: "Optional flag to enable content splitting for large responses (default: true)",
|
|
94
|
+
},
|
|
95
|
+
chunkId: {
|
|
96
|
+
type: "string",
|
|
97
|
+
description: `Optional chunk ID for retrieving a specific chunk of content from a previous request. The system adds prompts in the format ${TemplateUtils.SYSTEM_NOTE.START} ... ${TemplateUtils.SYSTEM_NOTE.END} which AI models should ignore when processing the content.`,
|
|
98
|
+
},
|
|
81
99
|
},
|
|
82
|
-
required: ["url"],
|
|
100
|
+
required: ["url", "startCursor"],
|
|
83
101
|
},
|
|
84
102
|
};
|
|
85
103
|
}
|
|
@@ -90,7 +108,7 @@ function getHtmlFetchToolDefinition() {
|
|
|
90
108
|
function getJsonFetchToolDefinition() {
|
|
91
109
|
return {
|
|
92
110
|
name: "fetch_json",
|
|
93
|
-
description: "Fetch a JSON file from a URL",
|
|
111
|
+
description: "Fetch a JSON file from a URL. Best practices: 1) Always set startCursor=0 for initial requests, and use the fetchedBytes value from previous response for subsequent requests to ensure content continuity. 2) Set contentSizeLimit between 20000-50000 for large files. 3) When handling large content, use the chunking system by following the startCursor instructions in the system notes rather than increasing contentSizeLimit. 4) If content retrieval fails, you can retry using the same chunkId and startCursor, or adjust startCursor as needed but you must handle any resulting data duplication or gaps yourself. 5) Always explain to users when content is chunked and ask if they want to continue retrieving subsequent parts.",
|
|
94
112
|
inputSchema: {
|
|
95
113
|
type: "object",
|
|
96
114
|
properties: {
|
|
@@ -98,6 +116,10 @@ function getJsonFetchToolDefinition() {
|
|
|
98
116
|
type: "string",
|
|
99
117
|
description: "URL of the JSON to fetch",
|
|
100
118
|
},
|
|
119
|
+
startCursor: {
|
|
120
|
+
type: "number",
|
|
121
|
+
description: "Starting cursor position in bytes. Set to 0 for initial requests, and use the value from previous responses for subsequent requests to resume content retrieval.",
|
|
122
|
+
},
|
|
101
123
|
headers: {
|
|
102
124
|
type: "object",
|
|
103
125
|
description: "Optional headers to include in the request",
|
|
@@ -142,8 +164,20 @@ function getJsonFetchToolDefinition() {
|
|
|
142
164
|
type: "boolean",
|
|
143
165
|
description: "Optional flag to close the browser after fetching (default: false)",
|
|
144
166
|
},
|
|
167
|
+
contentSizeLimit: {
|
|
168
|
+
type: "number",
|
|
169
|
+
description: "Optional maximum content size in bytes before splitting into chunks (default: 50KB). Set between 20KB-50KB for optimal results. For large content, prefer smaller values (20KB-30KB) to avoid truncation.",
|
|
170
|
+
},
|
|
171
|
+
enableContentSplitting: {
|
|
172
|
+
type: "boolean",
|
|
173
|
+
description: "Optional flag to enable content splitting for large responses (default: true)",
|
|
174
|
+
},
|
|
175
|
+
chunkId: {
|
|
176
|
+
type: "string",
|
|
177
|
+
description: `Optional chunk ID for retrieving a specific chunk of content from a previous request. The system adds prompts in the format ${TemplateUtils.SYSTEM_NOTE.START} ... ${TemplateUtils.SYSTEM_NOTE.END} which AI models should ignore when processing the content.`,
|
|
178
|
+
},
|
|
145
179
|
},
|
|
146
|
-
required: ["url"],
|
|
180
|
+
required: ["url", "startCursor"],
|
|
147
181
|
},
|
|
148
182
|
};
|
|
149
183
|
}
|
|
@@ -154,7 +188,7 @@ function getJsonFetchToolDefinition() {
|
|
|
154
188
|
function getTextFetchToolDefinition() {
|
|
155
189
|
return {
|
|
156
190
|
name: "fetch_txt",
|
|
157
|
-
description: "Fetch a website, return the content as plain text (no HTML)",
|
|
191
|
+
description: "Fetch a website, return the content as plain text (no HTML). Best practices: 1) Always set startCursor=0 for initial requests, and use the fetchedBytes value from previous response for subsequent requests to ensure content continuity. 2) Set contentSizeLimit between 20000-50000 for large pages. 3) When handling large content, use the chunking system by following the startCursor instructions in the system notes rather than increasing contentSizeLimit. 4) If content retrieval fails, you can retry using the same chunkId and startCursor, or adjust startCursor as needed but you must handle any resulting data duplication or gaps yourself. 5) Always explain to users when content is chunked and ask if they want to continue retrieving subsequent parts.",
|
|
158
192
|
inputSchema: {
|
|
159
193
|
type: "object",
|
|
160
194
|
properties: {
|
|
@@ -162,6 +196,10 @@ function getTextFetchToolDefinition() {
|
|
|
162
196
|
type: "string",
|
|
163
197
|
description: "URL of the website to fetch",
|
|
164
198
|
},
|
|
199
|
+
startCursor: {
|
|
200
|
+
type: "number",
|
|
201
|
+
description: "Starting cursor position in bytes. Set to 0 for initial requests, and use the value from previous responses for subsequent requests to resume content retrieval.",
|
|
202
|
+
},
|
|
165
203
|
headers: {
|
|
166
204
|
type: "object",
|
|
167
205
|
description: "Optional headers to include in the request",
|
|
@@ -206,8 +244,20 @@ function getTextFetchToolDefinition() {
|
|
|
206
244
|
type: "boolean",
|
|
207
245
|
description: "Optional flag to scroll to bottom of page in browser mode (default: false)",
|
|
208
246
|
},
|
|
247
|
+
contentSizeLimit: {
|
|
248
|
+
type: "number",
|
|
249
|
+
description: "Optional maximum content size in bytes before splitting into chunks (default: 50KB). Set between 20KB-50KB for optimal results. For large content, prefer smaller values (20KB-30KB) to avoid truncation.",
|
|
250
|
+
},
|
|
251
|
+
enableContentSplitting: {
|
|
252
|
+
type: "boolean",
|
|
253
|
+
description: "Optional flag to enable content splitting for large responses (default: true)",
|
|
254
|
+
},
|
|
255
|
+
chunkId: {
|
|
256
|
+
type: "string",
|
|
257
|
+
description: `Optional chunk ID for retrieving a specific chunk of content from a previous request. The system adds prompts in the format ${TemplateUtils.SYSTEM_NOTE.START} ... ${TemplateUtils.SYSTEM_NOTE.END} which AI models should ignore when processing the content.`,
|
|
258
|
+
},
|
|
209
259
|
},
|
|
210
|
-
required: ["url"],
|
|
260
|
+
required: ["url", "startCursor"],
|
|
211
261
|
},
|
|
212
262
|
};
|
|
213
263
|
}
|
|
@@ -218,7 +268,7 @@ function getTextFetchToolDefinition() {
|
|
|
218
268
|
function getMarkdownFetchToolDefinition() {
|
|
219
269
|
return {
|
|
220
270
|
name: "fetch_markdown",
|
|
221
|
-
description: "Fetch a website and return the content as Markdown",
|
|
271
|
+
description: "Fetch a website and return the content as Markdown. Best practices: 1) Always set startCursor=0 for initial requests, and use the fetchedBytes value from previous response for subsequent requests to ensure content continuity. 2) Set contentSizeLimit between 20000-50000 for large pages. 3) When handling large content, use the chunking system by following the startCursor instructions in the system notes rather than increasing contentSizeLimit. 4) If content retrieval fails, you can retry using the same chunkId and startCursor, or adjust startCursor as needed but you must handle any resulting data duplication or gaps yourself. 5) Always explain to users when content is chunked and ask if they want to continue retrieving subsequent parts.",
|
|
222
272
|
inputSchema: {
|
|
223
273
|
type: "object",
|
|
224
274
|
properties: {
|
|
@@ -226,6 +276,10 @@ function getMarkdownFetchToolDefinition() {
|
|
|
226
276
|
type: "string",
|
|
227
277
|
description: "URL of the website to fetch",
|
|
228
278
|
},
|
|
279
|
+
startCursor: {
|
|
280
|
+
type: "number",
|
|
281
|
+
description: "Starting cursor position in bytes. Set to 0 for initial requests, and use the value from previous responses for subsequent requests to resume content retrieval.",
|
|
282
|
+
},
|
|
229
283
|
headers: {
|
|
230
284
|
type: "object",
|
|
231
285
|
description: "Optional headers to include in the request",
|
|
@@ -270,8 +324,112 @@ function getMarkdownFetchToolDefinition() {
|
|
|
270
324
|
type: "boolean",
|
|
271
325
|
description: "Optional flag to scroll to bottom of page in browser mode (default: false)",
|
|
272
326
|
},
|
|
327
|
+
contentSizeLimit: {
|
|
328
|
+
type: "number",
|
|
329
|
+
description: "Optional maximum content size in bytes before splitting into chunks (default: 50KB). Set between 20KB-50KB for optimal results. For large content, prefer smaller values (20KB-30KB) to avoid truncation.",
|
|
330
|
+
},
|
|
331
|
+
enableContentSplitting: {
|
|
332
|
+
type: "boolean",
|
|
333
|
+
description: "Optional flag to enable content splitting for large responses (default: true)",
|
|
334
|
+
},
|
|
335
|
+
chunkId: {
|
|
336
|
+
type: "string",
|
|
337
|
+
description: `Optional chunk ID for retrieving a specific chunk of content from a previous request. The system adds prompts in the format ${TemplateUtils.SYSTEM_NOTE.START} ... ${TemplateUtils.SYSTEM_NOTE.END} which AI models should ignore when processing the content.`,
|
|
338
|
+
},
|
|
339
|
+
},
|
|
340
|
+
required: ["url", "startCursor"],
|
|
341
|
+
},
|
|
342
|
+
};
|
|
343
|
+
}
|
|
344
|
+
/**
|
|
345
|
+
* 获取纯文本获取工具定义 (Get plain text fetch tool definition)
|
|
346
|
+
* @returns 纯文本获取工具定义 (Plain text fetch tool definition)
|
|
347
|
+
*/
|
|
348
|
+
function getPlainTextFetchToolDefinition() {
|
|
349
|
+
return {
|
|
350
|
+
name: "fetch_plaintext",
|
|
351
|
+
description: "Fetch a website and return the content as plain text with HTML tags removed. Best practices: 1) Always set startCursor=0 for initial requests, and use the fetchedBytes value from previous response for subsequent requests to ensure content continuity. 2) Set contentSizeLimit between 20000-50000 for large pages. 3) When handling large content, use the chunking system by following the startCursor instructions in the system notes rather than increasing contentSizeLimit. 4) If content retrieval fails, you can retry using the same chunkId and startCursor, or adjust startCursor as needed but you must handle any resulting data duplication or gaps yourself. 5) Always explain to users when content is chunked and ask if they want to continue retrieving subsequent parts.",
|
|
352
|
+
inputSchema: {
|
|
353
|
+
type: "object",
|
|
354
|
+
properties: {
|
|
355
|
+
url: {
|
|
356
|
+
type: "string",
|
|
357
|
+
description: "URL of the website to fetch",
|
|
358
|
+
},
|
|
359
|
+
startCursor: {
|
|
360
|
+
type: "number",
|
|
361
|
+
description: "Starting cursor position in bytes. Set to 0 for initial requests, and use the value from previous responses for subsequent requests to resume content retrieval.",
|
|
362
|
+
},
|
|
363
|
+
headers: {
|
|
364
|
+
type: "object",
|
|
365
|
+
description: "Optional headers to include in the request",
|
|
366
|
+
},
|
|
367
|
+
proxy: {
|
|
368
|
+
type: "string",
|
|
369
|
+
description: "Optional proxy server to use (format: http://host:port or https://host:port)",
|
|
370
|
+
},
|
|
371
|
+
noDelay: {
|
|
372
|
+
type: "boolean",
|
|
373
|
+
description: "Optional flag to disable random delay between requests (default: false)",
|
|
374
|
+
},
|
|
375
|
+
timeout: {
|
|
376
|
+
type: "number",
|
|
377
|
+
description: "Optional timeout in milliseconds (default: 30000)",
|
|
378
|
+
},
|
|
379
|
+
maxRedirects: {
|
|
380
|
+
type: "number",
|
|
381
|
+
description: "Optional maximum number of redirects to follow (default: 10)",
|
|
382
|
+
},
|
|
383
|
+
useSystemProxy: {
|
|
384
|
+
type: "boolean",
|
|
385
|
+
description: "Optional flag to use system proxy environment variables (default: true)",
|
|
386
|
+
},
|
|
387
|
+
debug: {
|
|
388
|
+
type: "boolean",
|
|
389
|
+
description: "Optional flag to enable detailed debug logging (default: false)",
|
|
390
|
+
},
|
|
391
|
+
useBrowser: {
|
|
392
|
+
type: "boolean",
|
|
393
|
+
description: "Optional flag to use headless browser for fetching (default: false)",
|
|
394
|
+
},
|
|
395
|
+
autoDetectMode: {
|
|
396
|
+
type: "boolean",
|
|
397
|
+
description: "Optional flag to automatically switch to browser mode if standard fetch fails (default: true). Set to false to strictly use the specified mode without automatic switching.",
|
|
398
|
+
},
|
|
399
|
+
waitForSelector: {
|
|
400
|
+
type: "string",
|
|
401
|
+
description: "Optional CSS selector to wait for when using browser mode",
|
|
402
|
+
},
|
|
403
|
+
waitForTimeout: {
|
|
404
|
+
type: "number",
|
|
405
|
+
description: "Optional timeout to wait after page load in browser mode (default: 5000)",
|
|
406
|
+
},
|
|
407
|
+
scrollToBottom: {
|
|
408
|
+
type: "boolean",
|
|
409
|
+
description: "Optional flag to scroll to bottom of page in browser mode (default: false)",
|
|
410
|
+
},
|
|
411
|
+
saveCookies: {
|
|
412
|
+
type: "boolean",
|
|
413
|
+
description: "Optional flag to save cookies for future requests to the same domain (default: true)",
|
|
414
|
+
},
|
|
415
|
+
closeBrowser: {
|
|
416
|
+
type: "boolean",
|
|
417
|
+
description: "Optional flag to close the browser after fetching (default: false)",
|
|
418
|
+
},
|
|
419
|
+
contentSizeLimit: {
|
|
420
|
+
type: "number",
|
|
421
|
+
description: "Optional maximum content size in bytes before splitting into chunks (default: 50KB). Set between 20KB-50KB for optimal results. For large content, prefer smaller values (20KB-30KB) to avoid truncation.",
|
|
422
|
+
},
|
|
423
|
+
enableContentSplitting: {
|
|
424
|
+
type: "boolean",
|
|
425
|
+
description: "Optional flag to enable content splitting for large responses (default: true)",
|
|
426
|
+
},
|
|
427
|
+
chunkId: {
|
|
428
|
+
type: "string",
|
|
429
|
+
description: `Optional chunk ID for retrieving a specific chunk of content from a previous request. The system adds prompts in the format ${TemplateUtils.SYSTEM_NOTE.START} ... ${TemplateUtils.SYSTEM_NOTE.END} which AI models should ignore when processing the content.`,
|
|
430
|
+
},
|
|
273
431
|
},
|
|
274
|
-
required: ["url"],
|
|
432
|
+
required: ["url", "startCursor"],
|
|
275
433
|
},
|
|
276
434
|
};
|
|
277
435
|
}
|
|
@@ -284,7 +442,8 @@ function getAllToolDefinitions() {
|
|
|
284
442
|
getHtmlFetchToolDefinition(),
|
|
285
443
|
getJsonFetchToolDefinition(),
|
|
286
444
|
getTextFetchToolDefinition(),
|
|
287
|
-
getMarkdownFetchToolDefinition()
|
|
445
|
+
getMarkdownFetchToolDefinition(),
|
|
446
|
+
getPlainTextFetchToolDefinition()
|
|
288
447
|
];
|
|
289
448
|
}
|
|
290
449
|
/**
|
|
@@ -310,7 +469,7 @@ function registerToolCallHandler(server) {
|
|
|
310
469
|
log('tools.callReceived', debug, { name, args: JSON.stringify(args) }, COMPONENTS.SERVER);
|
|
311
470
|
try {
|
|
312
471
|
// 处理不同类型的获取请求 (Handle different types of fetch requests)
|
|
313
|
-
if (name === 'fetch_html' || name === 'fetch_json' || name === 'fetch_txt' || name === 'fetch_markdown') {
|
|
472
|
+
if (name === 'fetch_html' || name === 'fetch_json' || name === 'fetch_txt' || name === 'fetch_markdown' || name === 'fetch_plaintext') {
|
|
314
473
|
const result = await handleFetchRequest(name, args, debug);
|
|
315
474
|
// 直接返回符合 MCP SDK 要求的格式
|
|
316
475
|
return result;
|
|
@@ -318,29 +477,13 @@ function registerToolCallHandler(server) {
|
|
|
318
477
|
else {
|
|
319
478
|
// 未知工具 (Unknown tool)
|
|
320
479
|
log('tools.unknownTool', debug, { name }, COMPONENTS.SERVER);
|
|
321
|
-
return {
|
|
322
|
-
content: [
|
|
323
|
-
{
|
|
324
|
-
type: "text",
|
|
325
|
-
text: `Unknown tool: ${name}`
|
|
326
|
-
}
|
|
327
|
-
],
|
|
328
|
-
isError: true
|
|
329
|
-
};
|
|
480
|
+
return BaseFetcher.createErrorResponse(`Unknown tool: ${name}`);
|
|
330
481
|
}
|
|
331
482
|
}
|
|
332
483
|
catch (error) {
|
|
333
484
|
// 处理错误 (Handle error)
|
|
334
485
|
log('tools.callError', debug, { name, error: error.message }, COMPONENTS.SERVER);
|
|
335
|
-
return {
|
|
336
|
-
content: [
|
|
337
|
-
{
|
|
338
|
-
type: "text",
|
|
339
|
-
text: error.message
|
|
340
|
-
}
|
|
341
|
-
],
|
|
342
|
-
isError: true
|
|
343
|
-
};
|
|
486
|
+
return BaseFetcher.createErrorResponse(`Error fetching ${args.url || `chunk ${args.chunkId}`}: ${error.message}`);
|
|
344
487
|
}
|
|
345
488
|
finally {
|
|
346
489
|
// 如果请求参数中指定了关闭浏览器,则关闭浏览器 (If request parameters specify to close browser, close browser)
|
|
@@ -359,20 +502,22 @@ function registerToolCallHandler(server) {
|
|
|
359
502
|
*/
|
|
360
503
|
async function handleFetchRequest(name, args, debug) {
|
|
361
504
|
// 验证URL参数 (Validate URL parameter)
|
|
362
|
-
if (!args.url) {
|
|
363
|
-
log('tools.
|
|
364
|
-
return
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
],
|
|
371
|
-
isError: true
|
|
372
|
-
};
|
|
505
|
+
if (!args.url && !args.chunkId) {
|
|
506
|
+
log('tools.missingUrlOrChunkId', debug, {}, COMPONENTS.SERVER);
|
|
507
|
+
return BaseFetcher.createErrorResponse("Either URL or chunkId parameter is required");
|
|
508
|
+
}
|
|
509
|
+
// 验证startCursor参数 (Validate startCursor parameter)
|
|
510
|
+
if (args.startCursor === undefined) {
|
|
511
|
+
log('tools.missingStartCursor', debug, {}, COMPONENTS.SERVER);
|
|
512
|
+
return BaseFetcher.createErrorResponse("startCursor parameter is required. Use 0 for initial requests.");
|
|
373
513
|
}
|
|
374
514
|
// 记录请求 (Log request)
|
|
375
|
-
|
|
515
|
+
if (args.chunkId) {
|
|
516
|
+
log('tools.fetchChunkRequest', debug, { chunkId: args.chunkId, startCursor: args.startCursor, type: name }, COMPONENTS.SERVER);
|
|
517
|
+
}
|
|
518
|
+
else {
|
|
519
|
+
log('tools.fetchRequest', debug, { url: args.url, startCursor: args.startCursor, type: name }, COMPONENTS.SERVER);
|
|
520
|
+
}
|
|
376
521
|
// 执行获取请求 (Execute fetch request)
|
|
377
522
|
try {
|
|
378
523
|
const params = {
|
|
@@ -381,7 +526,7 @@ async function handleFetchRequest(name, args, debug) {
|
|
|
381
526
|
};
|
|
382
527
|
// 根据工具名称确定内容类型 (Determine content type based on tool name)
|
|
383
528
|
const type = name.replace('fetch_', '');
|
|
384
|
-
|
|
529
|
+
let result = await fetchWithAutoDetect(params, type);
|
|
385
530
|
// 确保返回标准结构体 (Ensure returning standard structure)
|
|
386
531
|
if (result.isError) {
|
|
387
532
|
// 确保错误内容有正确的类型 (Ensure error content has correct type)
|
|
@@ -407,6 +552,19 @@ async function handleFetchRequest(name, args, debug) {
|
|
|
407
552
|
if (!result.content[0].type) {
|
|
408
553
|
result.content[0].type = "text";
|
|
409
554
|
}
|
|
555
|
+
// 如果内容是分段的,添加提示词 (If content is chunked, add prompt)
|
|
556
|
+
if (result.isChunked && result.hasMoreChunks) {
|
|
557
|
+
// 检查内容中是否已经包含系统提示,避免重复添加
|
|
558
|
+
// Check if content already contains system note to avoid duplicate
|
|
559
|
+
if (!TemplateUtils.hasSystemPrompt(result.content[0].text || '')) {
|
|
560
|
+
// 使用BaseFetcher的addChunkPrompt方法添加提示词 (Use BaseFetcher's addChunkPrompt method to add prompt)
|
|
561
|
+
result = BaseFetcher.addChunkPrompt(result);
|
|
562
|
+
}
|
|
563
|
+
}
|
|
564
|
+
}
|
|
565
|
+
// 如果没有匹配的类型,返回错误
|
|
566
|
+
if (!result.content[0].type) {
|
|
567
|
+
return BaseFetcher.createErrorResponse(`Unsupported content type: ${type}`);
|
|
410
568
|
}
|
|
411
569
|
// 将FetchResponse转换为符合MCP SDK要求的格式
|
|
412
570
|
return {
|
|
@@ -417,16 +575,8 @@ async function handleFetchRequest(name, args, debug) {
|
|
|
417
575
|
catch (error) {
|
|
418
576
|
// 处理错误 (Handle error)
|
|
419
577
|
log('tools.fetchError', debug, { url: args.url, error: error.message }, COMPONENTS.SERVER);
|
|
420
|
-
//
|
|
421
|
-
return {
|
|
422
|
-
content: [
|
|
423
|
-
{
|
|
424
|
-
type: "text",
|
|
425
|
-
text: `Error fetching ${args.url}: ${error.message}`
|
|
426
|
-
}
|
|
427
|
-
],
|
|
428
|
-
isError: true
|
|
429
|
-
};
|
|
578
|
+
// 返回错误信息 (Return error message)
|
|
579
|
+
return BaseFetcher.createErrorResponse(`Error fetching ${args.url || `chunk ${args.chunkId}`}: ${error.message}`);
|
|
430
580
|
}
|
|
431
581
|
}
|
|
432
582
|
/**
|
package/dist/src/lib/types.js
CHANGED
|
@@ -32,4 +32,9 @@ export const RequestPayloadSchema = z.object({
|
|
|
32
32
|
waitForTimeout: z.number().optional(),
|
|
33
33
|
scrollToBottom: z.boolean().optional(),
|
|
34
34
|
closeBrowser: z.boolean().optional(),
|
|
35
|
+
chunkId: z.string().optional(),
|
|
36
|
+
chunkIndex: z.number().optional(),
|
|
37
|
+
contentSizeLimit: z.number().optional(),
|
|
38
|
+
startCursor: z.number().optional(),
|
|
39
|
+
enableContentSplitting: z.boolean().optional(),
|
|
35
40
|
}).merge(BrowserParamsSchema);
|