crawlforge-mcp-server 5.10.0 → 6.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +7 -1
- package/package.json +7 -5
- package/server.js +210 -251
- package/src/cli/commands/login.js +176 -0
- package/src/cli/index.js +2 -0
- package/src/core/ActionExecutor.js +1 -1
- package/src/core/AuthManager.js +19 -6
- package/src/core/ChangeTracker.js +1 -1
- package/src/core/ElicitationHelper.js +192 -62
- package/src/core/SamplingClient.js +8 -2
- package/src/core/analysis/ContentAnalyzer.js +1 -1
- package/src/core/processing/BrowserProcessor.js +1 -1
- package/src/core/processing/ContentProcessor.js +1 -1
- package/src/core/processing/PDFProcessor.js +2 -2
- package/src/server/registerTool.js +1 -1
- package/src/server/requestContext.js +50 -0
- package/src/server/specHygiene.js +17 -22
- package/src/server/transports/stdio.js +2 -3
- package/src/server/transports/streamableHttp.js +142 -67
- package/src/server/withAuth.js +47 -9
- package/src/tools/advanced/batchScrape/index.js +29 -19
- package/src/tools/agent/agent.js +9 -4
- package/src/tools/crawl/crawlDeep.js +14 -8
- package/src/tools/extract/analyzeContent.js +1 -1
- package/src/tools/extract/extractContent.js +1 -1
- package/src/tools/extract/extractStructured.js +62 -43
- package/src/tools/extract/processDocument.js +1 -1
- package/src/tools/extract/summarizeContent.js +1 -1
- package/src/tools/llmstxt/generateLLMsTxt.js +2 -2
- package/src/tools/research/deepResearch.js +11 -5
- package/src/tools/tracking/trackChanges/schema.js +4 -4
- package/src/utils/HumanBehaviorSimulator.js +7 -7
- package/src/server/taskSupport.js +0 -233
- package/src/server/transports/http.js +0 -22
|
@@ -83,41 +83,51 @@ export class BatchScrapeTool extends EventEmitter {
|
|
|
83
83
|
this._elicitation = new ElicitationHelper({ mcpServer });
|
|
84
84
|
}
|
|
85
85
|
|
|
86
|
-
async execute(params) {
|
|
86
|
+
async execute(params, ctx) {
|
|
87
87
|
try {
|
|
88
88
|
const validated = BatchScrapeSchema.parse(params);
|
|
89
|
-
this.stats.totalBatches++;
|
|
90
89
|
const batchId = this._generateBatchId();
|
|
91
|
-
const startTime = Date.now();
|
|
92
|
-
|
|
93
|
-
this._log('info', `Starting batch scrape ${batchId} with ${validated.urls.length} URLs in ${validated.mode} mode`);
|
|
94
|
-
|
|
95
|
-
const urlConfigs = this._normalizeUrlConfigs(validated.urls, validated);
|
|
96
90
|
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
//
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
91
|
+
// D1.4: Elicitation — warn when batch is large in sync mode. An
|
|
92
|
+
// unanswered gate returns an input-required result the SDK answers by
|
|
93
|
+
// re-entering this handler from the top, so it runs before anything that
|
|
94
|
+
// leaves a trace: below it are the batch counter and the webhook
|
|
95
|
+
// registration, which a second entry would repeat (a second webhook
|
|
96
|
+
// registration for one batch). The count comes from validated.urls,
|
|
97
|
+
// which _normalizeUrlConfigs maps one-for-one.
|
|
98
|
+
if (validated.mode === 'sync' && validated.urls.length > 25) {
|
|
99
|
+
const gate = this._elicitation.confirm(
|
|
100
|
+
ctx,
|
|
101
|
+
'batch_scrape:large_sync',
|
|
102
|
+
`batch_scrape (sync mode) will fetch ${validated.urls.length} URLs synchronously. This may take a while and consume significant credits.`,
|
|
106
103
|
{
|
|
107
|
-
url_count:
|
|
104
|
+
url_count: validated.urls.length,
|
|
108
105
|
mode: 'sync',
|
|
109
106
|
suggestion: 'Consider using mode:"async" for large batches.',
|
|
110
107
|
}
|
|
111
108
|
);
|
|
112
|
-
if (
|
|
109
|
+
if (gate.status === 'ask') return gate.result;
|
|
110
|
+
if (gate.status === 'cancelled') {
|
|
113
111
|
return {
|
|
114
112
|
batchId, mode: 'sync', success: false,
|
|
115
113
|
error: 'Batch scrape cancelled by user (elicitation declined).',
|
|
116
|
-
totalUrls:
|
|
114
|
+
totalUrls: validated.urls.length,
|
|
117
115
|
};
|
|
118
116
|
}
|
|
119
117
|
}
|
|
120
118
|
|
|
119
|
+
this.stats.totalBatches++;
|
|
120
|
+
const startTime = Date.now();
|
|
121
|
+
|
|
122
|
+
this._log('info', `Starting batch scrape ${batchId} with ${validated.urls.length} URLs in ${validated.mode} mode`);
|
|
123
|
+
|
|
124
|
+
const urlConfigs = this._normalizeUrlConfigs(validated.urls, validated);
|
|
125
|
+
|
|
126
|
+
let webhookConfig = null;
|
|
127
|
+
if (validated.webhook && this.enableWebhookNotifications) {
|
|
128
|
+
webhookConfig = this._registerWebhook(validated.webhook, batchId);
|
|
129
|
+
}
|
|
130
|
+
|
|
121
131
|
if (validated.mode === 'sync') {
|
|
122
132
|
return await this._processBatchSync(batchId, urlConfigs, validated, webhookConfig, startTime);
|
|
123
133
|
} else {
|
package/src/tools/agent/agent.js
CHANGED
|
@@ -35,16 +35,21 @@ export class AgentTool {
|
|
|
35
35
|
this._elicitation = new ElicitationHelper({ mcpServer });
|
|
36
36
|
}
|
|
37
37
|
|
|
38
|
-
async execute(params) {
|
|
38
|
+
async execute(params, ctx) {
|
|
39
39
|
const validated = AgentInputSchema.parse(params);
|
|
40
40
|
|
|
41
|
-
// Request confirmation before a pro run (expensive)
|
|
41
|
+
// Request confirmation before a pro run (expensive). An unanswered gate
|
|
42
|
+
// returns an input-required result the SDK answers by re-entering this
|
|
43
|
+
// handler from the top, so nothing above it may fetch or leave a trace.
|
|
42
44
|
if (validated.model === 'pro') {
|
|
43
|
-
const
|
|
45
|
+
const gate = this._elicitation.confirm(
|
|
46
|
+
ctx,
|
|
47
|
+
'agent:pro_model',
|
|
44
48
|
'agent tool: pro model uses ResearchOrchestrator and may incur significant costs.',
|
|
45
49
|
{ model: 'pro', maxUrls: validated.maxUrls, note: 'External LLM API costs billed separately if keys are set.' }
|
|
46
50
|
);
|
|
47
|
-
if (
|
|
51
|
+
if (gate.status === 'ask') return gate.result;
|
|
52
|
+
if (gate.status === 'cancelled') {
|
|
48
53
|
return {
|
|
49
54
|
success: false,
|
|
50
55
|
cancelled: true,
|
|
@@ -22,7 +22,7 @@ const CrawlDeepSchema = z.object({
|
|
|
22
22
|
dampingFactor: z.number().min(0).max(1).optional().default(0.85),
|
|
23
23
|
maxIterations: z.number().min(1).max(1000).optional().default(100),
|
|
24
24
|
enableCaching: z.boolean().optional().default(true)
|
|
25
|
-
}).optional().
|
|
25
|
+
}).optional().prefault({}),
|
|
26
26
|
// New domain filtering options
|
|
27
27
|
domain_filter: z.object({
|
|
28
28
|
whitelist: z.array(z.union([
|
|
@@ -59,7 +59,7 @@ const CrawlDeepSchema = z.object({
|
|
|
59
59
|
timeout: z.number().optional(),
|
|
60
60
|
maxPages: z.number().optional(),
|
|
61
61
|
concurrency: z.number().optional()
|
|
62
|
-
})).optional().
|
|
62
|
+
})).optional().prefault({})
|
|
63
63
|
}).optional(),
|
|
64
64
|
import_filter_config: z.string().optional(), // JSON string of exported config
|
|
65
65
|
// Session reuse: when enabled, all page fetches share a cookie jar and
|
|
@@ -67,11 +67,11 @@ const CrawlDeepSchema = z.object({
|
|
|
67
67
|
session: z.object({
|
|
68
68
|
enabled: z.boolean(),
|
|
69
69
|
persistCookies: z.boolean().optional().default(true),
|
|
70
|
-
headers: z.record(z.string()).optional().
|
|
70
|
+
headers: z.record(z.string()).optional().prefault({}),
|
|
71
71
|
initialRequest: z.object({
|
|
72
72
|
url: z.string().url(),
|
|
73
73
|
method: z.string().optional().default('GET'),
|
|
74
|
-
headers: z.record(z.string()).optional().
|
|
74
|
+
headers: z.record(z.string()).optional().prefault({}),
|
|
75
75
|
body: z.string().optional()
|
|
76
76
|
}).optional()
|
|
77
77
|
}).optional()
|
|
@@ -117,7 +117,7 @@ export class CrawlDeepTool {
|
|
|
117
117
|
this._elicitation = new ElicitationHelper({ mcpServer });
|
|
118
118
|
}
|
|
119
119
|
|
|
120
|
-
async execute(params) {
|
|
120
|
+
async execute(params, ctx) {
|
|
121
121
|
try {
|
|
122
122
|
const validated = CrawlDeepSchema.parse(params);
|
|
123
123
|
|
|
@@ -147,9 +147,14 @@ export class CrawlDeepTool {
|
|
|
147
147
|
if (cached) return { ...cached, cached: true };
|
|
148
148
|
}
|
|
149
149
|
|
|
150
|
-
// D1.4: Elicitation — warn when max_pages is very high
|
|
150
|
+
// D1.4: Elicitation — warn when max_pages is very high. An unanswered
|
|
151
|
+
// gate returns an input-required result the SDK answers by re-entering
|
|
152
|
+
// this handler from the top; everything above is the clamp arithmetic and
|
|
153
|
+
// a cache read, so a second entry fetches nothing and leaves no trace.
|
|
151
154
|
if (effectiveMaxPages > 500) {
|
|
152
|
-
const
|
|
155
|
+
const gate = this._elicitation.confirm(
|
|
156
|
+
ctx,
|
|
157
|
+
'crawl_deep:max_pages',
|
|
153
158
|
`crawl_deep will crawl up to ${effectiveMaxPages} pages from ${validated.url}. Large crawls consume many credits.`,
|
|
154
159
|
{
|
|
155
160
|
url: validated.url,
|
|
@@ -157,7 +162,8 @@ export class CrawlDeepTool {
|
|
|
157
162
|
max_depth: effectiveMaxDepth,
|
|
158
163
|
}
|
|
159
164
|
);
|
|
160
|
-
if (
|
|
165
|
+
if (gate.status === 'ask') return gate.result;
|
|
166
|
+
if (gate.status === 'cancelled') {
|
|
161
167
|
return {
|
|
162
168
|
success: false,
|
|
163
169
|
error: 'Crawl cancelled by user (elicitation declined).',
|
|
@@ -26,7 +26,7 @@ const AnalyzeContentSchema = z.object({
|
|
|
26
26
|
includeAdvancedMetrics: z.boolean().default(false),
|
|
27
27
|
groupEntitiesByType: z.boolean().default(true),
|
|
28
28
|
rankByRelevance: z.boolean().default(true)
|
|
29
|
-
}).optional().
|
|
29
|
+
}).optional().prefault({})
|
|
30
30
|
});
|
|
31
31
|
|
|
32
32
|
const AnalyzeContentResult = z.object({
|
|
@@ -45,7 +45,7 @@ const ExtractContentSchema = z.object({
|
|
|
45
45
|
includeRawHTML: z.boolean().default(false),
|
|
46
46
|
includeCleanedHTML: z.boolean().default(false),
|
|
47
47
|
outputFormat: z.enum(['text', 'markdown', 'structured']).default('structured')
|
|
48
|
-
}).optional().
|
|
48
|
+
}).optional().prefault({})
|
|
49
49
|
});
|
|
50
50
|
|
|
51
51
|
const ExtractContentResult = z.object({
|
|
@@ -174,36 +174,78 @@ export class ExtractStructuredTool {
|
|
|
174
174
|
* @param {Object} params - Extraction parameters
|
|
175
175
|
* @returns {Promise<Object>} Extraction result
|
|
176
176
|
*/
|
|
177
|
-
async execute(params) {
|
|
177
|
+
async execute(params, ctx) {
|
|
178
178
|
const startTime = Date.now();
|
|
179
179
|
|
|
180
180
|
try {
|
|
181
181
|
const validated = ExtractStructuredSchema.parse(params);
|
|
182
182
|
const { url, schema, prompt, llmConfig, fallbackToSelectors, selectorHints, respect_robots, user_agent, verify_numbers } = validated;
|
|
183
183
|
|
|
184
|
-
// Step 1: Fetch and parse — shared helper strips scripts/styles/iframes/svgs
|
|
185
|
-
const { html, $, textContent, warnings } = await fetchAndParse(url, {
|
|
186
|
-
userAgent: user_agent || this.userAgent,
|
|
187
|
-
respectRobots: respect_robots,
|
|
188
|
-
tool: 'extract_structured'
|
|
189
|
-
});
|
|
190
|
-
|
|
191
|
-
// What the model reads — see shownText().
|
|
192
|
-
const shown = shownText($, html, url, textContent);
|
|
193
|
-
|
|
194
|
-
// Step 3: Try LLM extraction first
|
|
195
184
|
let extractionResult = null;
|
|
196
185
|
let extractionMethod = 'llm';
|
|
197
186
|
let llmErrorMessage = null;
|
|
198
187
|
let llmAvailable = false;
|
|
188
|
+
let llm = null;
|
|
199
189
|
|
|
190
|
+
// Step 0: LLM readiness, resolved before the fetch so the D1.4 gate below
|
|
191
|
+
// can ask before any network work — an unanswered gate returns an
|
|
192
|
+
// input-required result the SDK answers by re-entering this handler from
|
|
193
|
+
// the top, and everything above the gate runs a second time.
|
|
200
194
|
try {
|
|
201
|
-
|
|
195
|
+
llm = this._ensureLLMManager(llmConfig || {});
|
|
202
196
|
// ready() probes Ollama, which has no API key to gate on. isAvailable()
|
|
203
197
|
// alone reported false on any machine without a cloud key, so a running
|
|
204
198
|
// local Ollama was never used.
|
|
205
199
|
llmAvailable = await llm.ready();
|
|
206
|
-
|
|
200
|
+
} catch (llmError) {
|
|
201
|
+
// No usable LLM — this falls through to the CSS fallback. Keep the
|
|
202
|
+
// message so callers can tell "LLM broken" apart from "no LLM
|
|
203
|
+
// configured".
|
|
204
|
+
llmErrorMessage = llmError.message;
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
// D1.4: no LLM configured and the schema demands more than 3 required
|
|
208
|
+
// fields — confirm before running the lower-fidelity CSS fallback. With
|
|
209
|
+
// no LLM, step 3 extracts nothing, so landing on that fallback is already
|
|
210
|
+
// settled here, before the page is fetched.
|
|
211
|
+
const requiredCount = (schema.required || []).length;
|
|
212
|
+
if (fallbackToSelectors !== false && !llmAvailable && requiredCount > 3) {
|
|
213
|
+
const gate = this._elicitation.confirm(
|
|
214
|
+
ctx,
|
|
215
|
+
'extract_structured:no_llm_required_fields',
|
|
216
|
+
`No LLM provider is configured and the requested schema has ${requiredCount} required fields. ` +
|
|
217
|
+
`extract_structured will fall back to lower-fidelity CSS selector extraction, which may miss required fields.`,
|
|
218
|
+
{ url, required_fields: requiredCount }
|
|
219
|
+
);
|
|
220
|
+
if (gate.status === 'ask') return gate.result;
|
|
221
|
+
if (gate.status === 'cancelled') {
|
|
222
|
+
return {
|
|
223
|
+
success: false,
|
|
224
|
+
url,
|
|
225
|
+
data: {},
|
|
226
|
+
extraction_method: 'none',
|
|
227
|
+
confidence: 0,
|
|
228
|
+
schema_used: schema,
|
|
229
|
+
processingTime: Date.now() - startTime,
|
|
230
|
+
error: 'Extraction cancelled by user (elicitation declined).',
|
|
231
|
+
validation: { valid: false, errors: ['Extraction cancelled by user (elicitation declined).'] }
|
|
232
|
+
};
|
|
233
|
+
}
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
// Step 1: Fetch and parse — shared helper strips scripts/styles/iframes/svgs
|
|
237
|
+
const { html, $, textContent, warnings } = await fetchAndParse(url, {
|
|
238
|
+
userAgent: user_agent || this.userAgent,
|
|
239
|
+
respectRobots: respect_robots,
|
|
240
|
+
tool: 'extract_structured'
|
|
241
|
+
});
|
|
242
|
+
|
|
243
|
+
// What the model reads — see shownText().
|
|
244
|
+
const shown = shownText($, html, url, textContent);
|
|
245
|
+
|
|
246
|
+
// Step 3: Try LLM extraction first (readiness resolved in step 0)
|
|
247
|
+
if (llmAvailable) {
|
|
248
|
+
try {
|
|
207
249
|
const result = await llm.extractStructured(shown, schema, {
|
|
208
250
|
prompt: prompt || '',
|
|
209
251
|
maxContentLength: SHOWN_TEXT_BUDGET
|
|
@@ -218,12 +260,11 @@ export class ExtractStructuredTool {
|
|
|
218
260
|
} else {
|
|
219
261
|
llmErrorMessage = result?.error || 'LLM did not return usable JSON';
|
|
220
262
|
}
|
|
263
|
+
} catch (llmError) {
|
|
264
|
+
// LLM failed — will fall through to CSS fallback.
|
|
265
|
+
extractionResult = null;
|
|
266
|
+
llmErrorMessage = llmError.message;
|
|
221
267
|
}
|
|
222
|
-
} catch (llmError) {
|
|
223
|
-
// LLM failed — will fall through to CSS fallback. Keep the message so
|
|
224
|
-
// callers can tell "LLM broken" apart from "no LLM configured".
|
|
225
|
-
extractionResult = null;
|
|
226
|
-
llmErrorMessage = llmError.message;
|
|
227
268
|
}
|
|
228
269
|
|
|
229
270
|
// Step 3b (3.4): numeric provenance. Only the LLM path invents numbers —
|
|
@@ -318,31 +359,9 @@ export class ExtractStructuredTool {
|
|
|
318
359
|
if (guarded.checked.skipped) provenance.skipped = guarded.checked.skipped;
|
|
319
360
|
}
|
|
320
361
|
|
|
321
|
-
// Step 4: CSS selector fallback if LLM unavailable or failed
|
|
362
|
+
// Step 4: CSS selector fallback if LLM unavailable or failed (the D1.4
|
|
363
|
+
// confirmation for this path is gated above, before the fetch)
|
|
322
364
|
if (!extractionResult && fallbackToSelectors !== false) {
|
|
323
|
-
// D1.4: no LLM configured and the schema demands more than 3 required
|
|
324
|
-
// fields — confirm before running the lower-fidelity CSS fallback.
|
|
325
|
-
const requiredCount = (schema.required || []).length;
|
|
326
|
-
if (!llmAvailable && requiredCount > 3) {
|
|
327
|
-
const proceed = await this._elicitation.confirm(
|
|
328
|
-
`No LLM provider is configured and the requested schema has ${requiredCount} required fields. ` +
|
|
329
|
-
`extract_structured will fall back to lower-fidelity CSS selector extraction, which may miss required fields.`,
|
|
330
|
-
{ url, required_fields: requiredCount }
|
|
331
|
-
);
|
|
332
|
-
if (!proceed) {
|
|
333
|
-
return {
|
|
334
|
-
success: false,
|
|
335
|
-
url,
|
|
336
|
-
data: {},
|
|
337
|
-
extraction_method: 'none',
|
|
338
|
-
confidence: 0,
|
|
339
|
-
schema_used: schema,
|
|
340
|
-
processingTime: Date.now() - startTime,
|
|
341
|
-
error: 'Extraction cancelled by user (elicitation declined).',
|
|
342
|
-
validation: { valid: false, errors: ['Extraction cancelled by user (elicitation declined).'] }
|
|
343
|
-
};
|
|
344
|
-
}
|
|
345
|
-
}
|
|
346
365
|
extractionResult = this._cssExtraction($, schema, selectorHints || {});
|
|
347
366
|
extractionMethod = 'css_fallback';
|
|
348
367
|
}
|
|
@@ -49,7 +49,7 @@ const ProcessDocumentSchema = z.object({
|
|
|
49
49
|
// Content filtering
|
|
50
50
|
minContentLength: z.number().min(0).default(50),
|
|
51
51
|
removeBoilerplate: z.boolean().default(true)
|
|
52
|
-
}).optional().
|
|
52
|
+
}).optional().prefault({})
|
|
53
53
|
});
|
|
54
54
|
|
|
55
55
|
const ProcessDocumentResult = z.object({
|
|
@@ -28,7 +28,7 @@ const SummarizeContentSchema = z.object({
|
|
|
28
28
|
maxKeywords: z.number().min(1).max(20).default(10),
|
|
29
29
|
preserveStructure: z.boolean().default(false),
|
|
30
30
|
language: z.string().optional()
|
|
31
|
-
}).optional().
|
|
31
|
+
}).optional().prefault({})
|
|
32
32
|
});
|
|
33
33
|
|
|
34
34
|
const SummarizeContentResult = z.object({
|
|
@@ -21,7 +21,7 @@ const GenerateLLMsTxtSchema = z.object({
|
|
|
21
21
|
checkSecurity: z.boolean().optional().default(false).describe('Whether to probe security-sensitive paths (opt-in; sends requests to /admin, /login, etc.)'),
|
|
22
22
|
probeRateLimit: z.boolean().optional().default(false).describe('Whether to send repeated probe requests to estimate rate limits (opt-in; fires ~5 requests)'),
|
|
23
23
|
respectRobots: z.boolean().optional().default(true).describe('Whether to respect robots.txt')
|
|
24
|
-
}).optional().
|
|
24
|
+
}).optional().prefault({}),
|
|
25
25
|
|
|
26
26
|
outputOptions: z.object({
|
|
27
27
|
includeDetailed: z.boolean().optional().default(true).describe('Generate detailed LLMs-full.txt'),
|
|
@@ -31,7 +31,7 @@ const GenerateLLMsTxtSchema = z.object({
|
|
|
31
31
|
customGuidelines: z.array(z.string()).optional().describe('Additional custom guidelines'),
|
|
32
32
|
customRestrictions: z.array(z.string()).optional().describe('Additional restrictions'),
|
|
33
33
|
robotsStyle: z.boolean().optional().default(false).describe('Emit legacy robots.txt-style directives instead of spec-compliant llmstxt.org markdown')
|
|
34
|
-
}).optional().
|
|
34
|
+
}).optional().prefault({}),
|
|
35
35
|
|
|
36
36
|
complianceLevel: z.enum(['basic', 'standard', 'strict']).optional().default('standard').describe('Compliance level for generated guidelines'),
|
|
37
37
|
|
|
@@ -115,7 +115,7 @@ export class DeepResearchTool {
|
|
|
115
115
|
this._elicitation = new ElicitationHelper({ mcpServer });
|
|
116
116
|
}
|
|
117
117
|
|
|
118
|
-
async execute(params) {
|
|
118
|
+
async execute(params, ctx) {
|
|
119
119
|
try {
|
|
120
120
|
const validated = DeepResearchSchema.parse(params);
|
|
121
121
|
const sessionId = this.generateSessionId();
|
|
@@ -137,10 +137,15 @@ export class DeepResearchTool {
|
|
|
137
137
|
}
|
|
138
138
|
|
|
139
139
|
// D1.4: Elicitation — warn user if projected cost exceeds 50 credits
|
|
140
|
-
// deep_research costs approximately 1 credit per URL; maxUrls > 50 → confirm
|
|
140
|
+
// deep_research costs approximately 1 credit per URL; maxUrls > 50 → confirm.
|
|
141
|
+
// An unanswered gate returns an input-required result the SDK answers by
|
|
142
|
+
// re-entering this handler from the top, so it sits above the session
|
|
143
|
+
// registration below — a second entry would otherwise leak a session.
|
|
141
144
|
if (validated.maxUrls > 50) {
|
|
142
145
|
const projectedCredits = validated.maxUrls;
|
|
143
|
-
const
|
|
146
|
+
const gate = this._elicitation.confirm(
|
|
147
|
+
ctx,
|
|
148
|
+
'deep_research:max_urls',
|
|
144
149
|
`deep_research will scan up to ${validated.maxUrls} URLs, projecting ~${projectedCredits} credits.`,
|
|
145
150
|
{
|
|
146
151
|
topic: validated.topic,
|
|
@@ -148,7 +153,8 @@ export class DeepResearchTool {
|
|
|
148
153
|
max_urls: validated.maxUrls,
|
|
149
154
|
}
|
|
150
155
|
);
|
|
151
|
-
if (
|
|
156
|
+
if (gate.status === 'ask') return gate.result;
|
|
157
|
+
if (gate.status === 'cancelled') {
|
|
152
158
|
return {
|
|
153
159
|
success: false,
|
|
154
160
|
error: 'Research cancelled by user before starting (elicitation declined).',
|
|
@@ -238,7 +244,7 @@ export class DeepResearchTool {
|
|
|
238
244
|
return {
|
|
239
245
|
success: false,
|
|
240
246
|
error: 'Invalid parameters',
|
|
241
|
-
details: validationError.
|
|
247
|
+
details: validationError.issues.map(err => ({
|
|
242
248
|
field: err.path.join('.'),
|
|
243
249
|
message: err.message,
|
|
244
250
|
received: err.received
|
|
@@ -48,7 +48,7 @@ export const TrackChangesSchema = z.object({
|
|
|
48
48
|
moderate: z.number().min(0).max(1).default(0.3),
|
|
49
49
|
major: z.number().min(0).max(1).default(0.7)
|
|
50
50
|
}).optional()
|
|
51
|
-
}).optional().
|
|
51
|
+
}).optional().prefault({}),
|
|
52
52
|
|
|
53
53
|
monitoringOptions: z.object({
|
|
54
54
|
enabled: z.boolean().default(false),
|
|
@@ -59,7 +59,7 @@ export const TrackChangesSchema = z.object({
|
|
|
59
59
|
enableWebhook: z.boolean().default(false),
|
|
60
60
|
webhookUrl: z.string().url().optional(),
|
|
61
61
|
webhookSecret: z.string().optional()
|
|
62
|
-
}).optional().
|
|
62
|
+
}).optional().prefault({}),
|
|
63
63
|
|
|
64
64
|
storageOptions: z.object({
|
|
65
65
|
enableSnapshots: z.boolean().default(true),
|
|
@@ -67,7 +67,7 @@ export const TrackChangesSchema = z.object({
|
|
|
67
67
|
maxHistoryEntries: z.number().min(1).max(1000).default(100),
|
|
68
68
|
compressionEnabled: z.boolean().default(true),
|
|
69
69
|
deltaStorageEnabled: z.boolean().default(true)
|
|
70
|
-
}).optional().
|
|
70
|
+
}).optional().prefault({}),
|
|
71
71
|
|
|
72
72
|
queryOptions: z.object({
|
|
73
73
|
limit: z.number().min(1).max(500).default(50),
|
|
@@ -76,7 +76,7 @@ export const TrackChangesSchema = z.object({
|
|
|
76
76
|
endTime: z.number().optional(),
|
|
77
77
|
includeContent: z.boolean().default(false),
|
|
78
78
|
significanceFilter: z.enum(['all', 'minor', 'moderate', 'major', 'critical']).optional()
|
|
79
|
-
}).optional().
|
|
79
|
+
}).optional().prefault({}),
|
|
80
80
|
|
|
81
81
|
notificationOptions: z.object({
|
|
82
82
|
email: z.object({
|
|
@@ -12,7 +12,7 @@ const BehaviorConfigSchema = z.object({
|
|
|
12
12
|
accuracy: z.number().min(0.1).max(1.0).default(0.8), // 0.1 = very inaccurate, 1.0 = perfect
|
|
13
13
|
naturalCurves: z.boolean().default(true),
|
|
14
14
|
randomMicroMovements: z.boolean().default(true)
|
|
15
|
-
}).
|
|
15
|
+
}).prefault({}),
|
|
16
16
|
|
|
17
17
|
typing: z.object({
|
|
18
18
|
enabled: z.boolean().default(true),
|
|
@@ -22,30 +22,30 @@ const BehaviorConfigSchema = z.object({
|
|
|
22
22
|
enabled: z.boolean().default(true),
|
|
23
23
|
frequency: z.number().min(0).max(0.1).default(0.02), // 2% mistake rate
|
|
24
24
|
correctionDelay: z.number().min(100).max(2000).default(500)
|
|
25
|
-
}).
|
|
26
|
-
}).
|
|
25
|
+
}).prefault({})
|
|
26
|
+
}).prefault({}),
|
|
27
27
|
|
|
28
28
|
scrolling: z.object({
|
|
29
29
|
enabled: z.boolean().default(true),
|
|
30
30
|
naturalAcceleration: z.boolean().default(true),
|
|
31
31
|
randomPauses: z.boolean().default(true),
|
|
32
32
|
scrollBackProbability: z.number().min(0).max(1).default(0.1)
|
|
33
|
-
}).
|
|
33
|
+
}).prefault({}),
|
|
34
34
|
|
|
35
35
|
interactions: z.object({
|
|
36
36
|
hoverBeforeClick: z.boolean().default(true),
|
|
37
37
|
clickDelay: z.object({
|
|
38
38
|
min: z.number().default(100),
|
|
39
39
|
max: z.number().default(300)
|
|
40
|
-
}).
|
|
40
|
+
}).prefault({}),
|
|
41
41
|
focusBlurSimulation: z.boolean().default(true),
|
|
42
42
|
idlePeriods: z.object({
|
|
43
43
|
enabled: z.boolean().default(true),
|
|
44
44
|
frequency: z.number().min(0).max(1).default(0.1), // 10% chance
|
|
45
45
|
minDuration: z.number().default(1000),
|
|
46
46
|
maxDuration: z.number().default(5000)
|
|
47
|
-
}).
|
|
48
|
-
}).
|
|
47
|
+
}).prefault({})
|
|
48
|
+
}).prefault({})
|
|
49
49
|
});
|
|
50
50
|
|
|
51
51
|
export class HumanBehaviorSimulator {
|