crawlforge-mcp-server 6.5.0 → 6.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,579 @@
1
+ /**
2
+ * BrowserSessionTool — `browser_session`: one browser page the caller keeps
3
+ * across several tool calls, driven by an `operation` enum.
4
+ *
5
+ * `scrape_with_actions` is one-shot and blind: the agent has to name up to 20
6
+ * actions up front, guessing selectors for a page it has never seen, and the
7
+ * browser closes when the call returns. A session inverts that loop — open,
8
+ * look (`snapshot`), act on the refs the snapshot handed back, look again —
9
+ * and one login is paid for once instead of once per call.
10
+ *
11
+ * Everything here is assembled from parts that already exist, deliberately:
12
+ * - `ActionExecutor.initializePage()` is the `open` primitive, SSRF guard and
13
+ * robots gate included, and `executeActionsOnPage()` runs actions against a
14
+ * page it does not own (the `navigate` action re-gates every hop itself).
15
+ * - Element refs live in a page-scoped WeakMap in core/browser/snapshot.js, so
16
+ * a session that keeps its page keeps its refs across calls for free, and
17
+ * loses them exactly when it should — on navigation.
18
+ * - `BrowserSessionStore` holds the sessions, the two TTL clocks and the caps.
19
+ * - `ExtractContentTool` turns the live DOM into the requested formats.
20
+ *
21
+ * Ownership is a tenant boundary, not a nicety — see ownerId() for what that
22
+ * means for the hosted REST path, which is served only to a request carrying a
23
+ * per-user owner token and refused to any other.
24
+ */
25
+
26
+ import { z } from 'zod';
27
+ import { createHash } from 'node:crypto';
28
+
29
+ import ActionExecutor from '../../core/ActionExecutor.js';
30
+ import ExtractContentTool from '../extract/extractContent.js';
31
+ import BrowserSessionStore, {
32
+ TTL_MIN_MS,
33
+ TTL_MAX_MS,
34
+ ACTIVITY_TTL_MIN_MS,
35
+ ACTIVITY_TTL_MAX_MS
36
+ } from '../../core/browser/SessionStore.js';
37
+ import { captureSnapshot } from '../../core/browser/snapshot.js';
38
+ import authManager from '../../core/AuthManager.js';
39
+ import { isCreatorModeVerified } from '../../core/creatorMode.js';
40
+ import { internalOwnerToken, isInternalRequest } from '../../server/requestContext.js';
41
+ import { isRemoteTransport } from '../../utils/remoteMode.js';
42
+ import { htmlToMarkdown } from '../../utils/htmlToMarkdown.js';
43
+ import { stealthDocumentVerdict } from '../../utils/stealthVerdict.js';
44
+
45
+ const SECOND = 1000;
46
+
47
+ /** The prefix that marks an owner id derived from a hosted REST owner token. */
48
+ const REST_OWNER_PREFIX = 'rest:';
49
+
50
+ /**
51
+ * A hosted REST customer may hold ONE session at a time, where a stdio install
52
+ * keeps the store's default of three.
53
+ *
54
+ * Arithmetic, not caution. On Render MAX_BROWSER_CONTEXTS=6, so
55
+ * DEFAULT_MAX_SESSIONS_TOTAL is floor(6/2) = 3 for the WHOLE box while
56
+ * DEFAULT_MAX_SESSIONS_PER_OWNER is 3 — meaning one REST customer opening three
57
+ * sessions occupies the entire hosted session capacity, and every other paying
58
+ * customer is refused until those sessions age out, up to ten minutes later.
59
+ * Capping REST owners at one lets three distinct customers work at once.
60
+ *
61
+ * stdio and self-hosted installs keep the three: there the process IS the
62
+ * customer, the box is theirs, and there is nobody else to lock out.
63
+ */
64
+ const REST_MAX_SESSIONS_PER_OWNER = 1;
65
+
66
+ /**
67
+ * The action array is `scrape_with_actions`' own, passed through untouched:
68
+ * ActionExecutor validates each action against its own union as it runs it, and
69
+ * a fourth copy of that union here could only drift from the three that exist.
70
+ *
71
+ * The two fields below are not decoration. ActionExecutor parses each action but
72
+ * discards the parsed value, so an action that arrives without them keeps the
73
+ * undefined it came with — and `action.retries > 0` is the gate on error
74
+ * recovery, `action.continueOnError` the per-action failure policy. These are
75
+ * the same defaults ScrapeWithActionsTool's schema stamps.
76
+ */
77
+ const SessionActionSchema = z.object({
78
+ type: z.string(),
79
+ continueOnError: z.boolean().default(false),
80
+ retries: z.number().min(0).max(5).default(1),
81
+ // The third default ActionExecutor declares and then throws away. It is read
82
+ // only by executeJavaScript, where `action.returnResult ? result : undefined`
83
+ // decides whether the script's return value survives at all — so without it
84
+ // stamped here every executeJavaScript in a session succeeded and handed back
85
+ // nothing, while the same action through scrape_with_actions (which keeps its
86
+ // parsed value) returned the data. Harmless on the action types that ignore it.
87
+ returnResult: z.boolean().default(true)
88
+ }).passthrough();
89
+
90
+ const BrowserSessionSchema = z.object({
91
+ operation: z.enum(['open', 'snapshot', 'act', 'read', 'screenshot', 'close', 'list']),
92
+ session_id: z.string().optional(),
93
+
94
+ // open. ttl/activity_ttl are seconds, as Firecrawl's are, so anyone arriving
95
+ // from their docs reads the same numbers; the store works in milliseconds.
96
+ url: z.string().url().optional(),
97
+ stealth: z.boolean().default(false),
98
+ ttl: z.number().min(TTL_MIN_MS / SECOND).max(TTL_MAX_MS / SECOND).optional(),
99
+ activity_ttl: z.number().min(ACTIVITY_TTL_MIN_MS / SECOND).max(ACTIVITY_TTL_MAX_MS / SECOND).optional(),
100
+ viewport: z.object({
101
+ width: z.number().min(800).max(1920),
102
+ height: z.number().min(600).max(1080)
103
+ }).optional(),
104
+ timeout: z.number().min(10000).max(120000).default(30000),
105
+
106
+ // Applies to the call it is sent on: on `open` to the first load, on `act` to
107
+ // every navigate in that call. It is never remembered by the session, so an
108
+ // override has to be repeated as deliberately as it was made.
109
+ respect_robots: z.boolean().optional(),
110
+
111
+ // snapshot
112
+ interactive_only: z.boolean().default(true),
113
+ max_nodes: z.number().min(1).max(1000).optional(),
114
+
115
+ // act
116
+ actions: z.array(SessionActionSchema).min(1).max(20).optional(),
117
+ continue_on_error: z.boolean().default(false),
118
+
119
+ // read
120
+ formats: z.array(z.enum(['markdown', 'html', 'text', 'json'])).default(['markdown']),
121
+
122
+ // screenshot
123
+ full_page: z.boolean().default(false),
124
+ format: z.enum(['png', 'jpeg']).default('png'),
125
+ quality: z.number().min(0).max(100).default(80),
126
+ selector: z.string().optional()
127
+ });
128
+
129
+ /** A refusal a caller (and a test) can match on by code rather than by prose. */
130
+ function refuse(code, message) {
131
+ const error = new Error(message);
132
+ error.name = 'BrowserSessionRefusal';
133
+ error.code = code;
134
+ return error;
135
+ }
136
+
137
+ /**
138
+ * The session fields every operation echoes back, in the same shape a store
139
+ * list row carries. Read it after touch() so the idle clock is the fresh one.
140
+ */
141
+ function sessionInfo(session) {
142
+ return {
143
+ sessionId: session.id,
144
+ url: session.url,
145
+ stealth: session.stealth,
146
+ expiresAt: session.createdAt + session.ttlMs,
147
+ idleExpiresAt: session.lastUsedAt + session.activityTtlMs
148
+ };
149
+ }
150
+
151
+ /**
152
+ * The verdict fields a result carries when there is something to say about the
153
+ * document — the same names `scrape` and `scrape_with_actions` publish.
154
+ */
155
+ function verdictFields(verdict) {
156
+ return {
157
+ ...(Number.isInteger(verdict.status) ? { httpStatus: verdict.status } : {}),
158
+ ...(verdict.blocked ? { blocked: verdict.blocked } : {})
159
+ };
160
+ }
161
+
162
+ /**
163
+ * `scrape_with_actions` publishes an executeJavaScript action's return value as
164
+ * a flat `jsResult` beside the nested one (processActionResults); a session's
165
+ * `act` returned the nested shape alone, so the same action read differently
166
+ * depending on which tool ran it. Same hoist, same field name.
167
+ */
168
+ function withJsResult(result) {
169
+ if (result?.type !== 'executeJavaScript' || !result.result) return result;
170
+ return { ...result, jsResult: result.result.result };
171
+ }
172
+
173
+ export class BrowserSessionTool {
174
+ constructor(options = {}) {
175
+ const {
176
+ actionExecutor = null,
177
+ extractContentTool = null,
178
+ store = null,
179
+ storeOptions = {},
180
+ enableLogging = true
181
+ } = options;
182
+
183
+ // An injected executor belongs to whoever built it (server.js hands us
184
+ // scrape_with_actions'), and destroying it would take that tool's browser
185
+ // down with ours. Only an executor we made ourselves is ours to destroy.
186
+ this._ownsExecutor = !actionExecutor;
187
+ this.actionExecutor = actionExecutor || new ActionExecutor({ enableLogging });
188
+ this.extractContentTool = extractContentTool || new ExtractContentTool();
189
+ this.storeOptions = storeOptions;
190
+ this.store = store || new BrowserSessionStore(storeOptions);
191
+ }
192
+
193
+ async execute(params) {
194
+ const validated = BrowserSessionSchema.parse(params);
195
+ const ownerId = this.ownerId();
196
+
197
+ switch (validated.operation) {
198
+ case 'open': return await this.openSession(validated, ownerId);
199
+ case 'snapshot': return await this.snapshotSession(validated, ownerId);
200
+ case 'act': return await this.actOnSession(validated, ownerId);
201
+ case 'read': return await this.readSession(validated, ownerId);
202
+ case 'screenshot': return await this.screenshotSession(validated, ownerId);
203
+ case 'close': return await this.closeSession(validated, ownerId);
204
+ case 'list': return this.listSessions(ownerId);
205
+ }
206
+ }
207
+
208
+ /**
209
+ * Who the caller is — the identity every session is bound to and every
210
+ * lookup is scoped by.
211
+ *
212
+ * On the hosted REST path the shared secret is not an identity. The website's
213
+ * proxy authenticates to this server with a single X-Internal-Secret
214
+ * (`authenticateRequest` in src/server/transports/streamableHttp.js), so
215
+ * every REST customer arrives as the same internal caller; binding a session
216
+ * to THAT would put all of them inside one tenant, where any customer could
217
+ * name any other customer's session id and be handed their logged-in browser.
218
+ * The per-user owner token is the missing half: the proxy derives it per end
219
+ * user and sends it on X-CrawlForge-Owner, the transport honours it only on a
220
+ * request that already proved the secret, and it lands here as the tenant key.
221
+ *
222
+ * NO FALLBACK WHEN IT IS ABSENT, DELIBERATELY — do not "tidy" the refusal
223
+ * below into a default owner. It is what makes the two repos safe to deploy
224
+ * in either order: an old server ignores the new header, a new server meets
225
+ * an old website that sends none, and in both cases sessions are refused
226
+ * rather than silently collapsing into one shared tenant. A missing owner is
227
+ * a missing tenant boundary, and the honest answer to that is "no session".
228
+ *
229
+ * Everywhere else the install is the tenant: over stdio, and over self-hosted
230
+ * HTTP authenticated with the install's API key or an OAuth token, the same
231
+ * configured key stands behind every request. Its digest is the owner id; the
232
+ * key itself never leaves this method.
233
+ */
234
+ ownerId() {
235
+ if (isInternalRequest()) {
236
+ const ownerToken = internalOwnerToken();
237
+ if (ownerToken) return `${REST_OWNER_PREFIX}${ownerToken}`;
238
+
239
+ throw refuse(
240
+ 'SESSIONS_NOT_AVAILABLE_OVER_REST',
241
+ 'browser_session could not be opened over the CrawlForge REST API: the API proxy ' +
242
+ 'authenticated as a shared internal caller without saying which customer this request ' +
243
+ 'belongs to, so a session id could not be bound to the account that opened it. That is ' +
244
+ 'usually a version skew mid-deploy — try again shortly. Meanwhile, use ' +
245
+ 'scrape_with_actions for a one-shot interaction chain, or run the CrawlForge MCP ' +
246
+ 'server locally (stdio) where sessions work normally.'
247
+ );
248
+ }
249
+
250
+ const apiKey = authManager.getConfig()?.apiKey;
251
+ return apiKey
252
+ ? `key:${createHash('sha256').update(apiKey).digest('hex').slice(0, 16)}`
253
+ : 'local';
254
+ }
255
+
256
+ /**
257
+ * The session this operation names, or the same "session not found" an
258
+ * unknown id gets. The store raises one error for unknown, wrong-owner and
259
+ * expired alike, which is what keeps ids non-enumerable — never answer a
260
+ * wrong owner with "forbidden".
261
+ */
262
+ requireSession(params, ownerId) {
263
+ if (!params.session_id) {
264
+ throw new Error(
265
+ `operation "${params.operation}" requires session_id — the id returned by operation:"open".`
266
+ );
267
+ }
268
+ return this.store.get(params.session_id, ownerId);
269
+ }
270
+
271
+ async openSession(params, ownerId) {
272
+ if (!params.url) {
273
+ throw new Error('operation "open" requires a url to load the session on.');
274
+ }
275
+
276
+ const browserOptions = {
277
+ headless: true,
278
+ viewportWidth: params.viewport?.width,
279
+ viewportHeight: params.viewport?.height,
280
+ timeout: params.timeout,
281
+ respectRobots: params.respect_robots
282
+ };
283
+ if (params.stealth) {
284
+ browserOptions.stealthMode = { enabled: true };
285
+ }
286
+
287
+ // initializePage runs the SSRF guard, then the blocklist/robots gate, and
288
+ // only then creates a page and navigates — closing the page itself if any
289
+ // of that fails. That is the whole of 2.5's gating on `open`, which is why
290
+ // this is not page.goto() behind a gate written here.
291
+ const page = await this.actionExecutor.initializePage(params.url, browserOptions);
292
+ const releasePage = this.releaser(page, params.stealth);
293
+
294
+ let session;
295
+ try {
296
+ session = this.store.create({
297
+ ownerId,
298
+ page,
299
+ releasePage,
300
+ url: page.url(),
301
+ stealth: params.stealth,
302
+ ttlMs: params.ttl === undefined ? undefined : params.ttl * SECOND,
303
+ activityTtlMs: params.activity_ttl === undefined ? undefined : params.activity_ttl * SECOND,
304
+ // undefined for every other owner, which leaves the store's own cap in
305
+ // force — the override exists for the hosted box alone.
306
+ maxPerOwner: ownerId.startsWith(REST_OWNER_PREFIX) ? REST_MAX_SESSIONS_PER_OWNER : undefined
307
+ });
308
+ } catch (error) {
309
+ // A cap refusal arrives with a live page in hand. Give it back before
310
+ // rethrowing, or the refused call leaks the context it just pinned.
311
+ await releasePage();
312
+ throw error;
313
+ }
314
+
315
+ // The session opened; whether the document it landed on is the page is a
316
+ // separate question. g2.com answered `open` with a DataDome 403 whose body
317
+ // was empty, and this returned success:true with no status at all, while
318
+ // `scrape` on the same URL named the vendor and the 403 — the same fault
319
+ // R18 found in scrape_with_actions (2026-09-04), in the one browser tool
320
+ // that never learned the lesson.
321
+ const verdict = await this.pageVerdict(page, { stealth: params.stealth });
322
+
323
+ return {
324
+ success: verdict.success,
325
+ operation: 'open',
326
+ ...sessionInfo(session),
327
+ ...verdictFields(verdict),
328
+ // The page is a wall, but the session behind it is real and holds a
329
+ // browser context — say so, or a caller reading only `success` abandons
330
+ // it to its TTL instead of closing it or acting through the challenge.
331
+ ...(verdict.error
332
+ ? { error: `${verdict.error} The session is open as ${session.id}: act on it, or close it.` }
333
+ : {})
334
+ };
335
+ }
336
+
337
+ /**
338
+ * What the session's page currently is: the page, a bot wall, an HTTP error
339
+ * page, or an error placeholder. One helper, shared with `scrape` and
340
+ * `scrape_with_actions`, so all three name a block identically instead of
341
+ * this tool staying silent about it.
342
+ *
343
+ * `allowEmpty` because a session is routinely opened on an app shell that
344
+ * only paints after the actions the caller is about to send — an empty
345
+ * document is a normal starting state here, not a failure. A real wall still
346
+ * fails on its challenge signature or its HTTP status, which is what the
347
+ * empty-document rule would have caught anyway.
348
+ *
349
+ * Never throws. A verdict is a diagnosis; a page that cannot be read for one
350
+ * (closed, mid-navigation) must not fail the operation being diagnosed.
351
+ */
352
+ async pageVerdict(page, { stealth = false, ...known } = {}) {
353
+ try {
354
+ const title = known.title !== undefined ? known.title : await page.title();
355
+ const html = known.html !== undefined ? known.html : await page.content();
356
+ const text = known.text !== undefined
357
+ ? known.text
358
+ : await page.evaluate(() => document.body?.innerText || '');
359
+
360
+ return stealthDocumentVerdict(
361
+ { url: page.url(), title, text, html, status: page.__crawlforgeNavigation?.status ?? null },
362
+ // The verdict's messages name whatever fetched the document, and its
363
+ // default is the stealth browser. A plain session is not that, and
364
+ // telling someone "the stealth browser did not pass it" when they never
365
+ // asked for stealth hides the one retry that might work.
366
+ { allowEmpty: true, fetcher: stealth ? 'the stealth browser session' : 'the browser session' }
367
+ );
368
+ } catch {
369
+ return { success: true, status: null };
370
+ }
371
+ }
372
+
373
+ async snapshotSession(params, ownerId) {
374
+ const session = this.requireSession(params, ownerId);
375
+
376
+ const snapshot = await captureSnapshot(session.page, {
377
+ interactiveOnly: params.interactive_only,
378
+ maxNodes: params.max_nodes
379
+ });
380
+
381
+ this.store.touch(session, session.page.url());
382
+ return { success: true, operation: 'snapshot', ...sessionInfo(session), snapshot };
383
+ }
384
+
385
+ async actOnSession(params, ownerId) {
386
+ if (!params.actions?.length) {
387
+ throw new Error('operation "act" requires an actions array.');
388
+ }
389
+ const session = this.requireSession(params, ownerId);
390
+
391
+ // D6: arbitrary JavaScript in a browser on OUR infrastructure is a
392
+ // materially different act from the same JavaScript in a browser on the
393
+ // caller's own laptop. Over stdio or loopback the caller is the local user
394
+ // and executeJavaScriptAction's own ALLOW_JAVASCRIPT_EXECUTION flag is the
395
+ // control; served to a network, it is refused outright. Creator mode is the
396
+ // maintainer's own box, so it keeps the local answer.
397
+ if (params.actions.some((action) => action.type === 'executeJavaScript') &&
398
+ isRemoteTransport() && !isCreatorModeVerified()) {
399
+ throw refuse(
400
+ 'JS_EXECUTION_REFUSED_REMOTE',
401
+ 'executeJavaScript is refused in a browser session on a remotely-served CrawlForge ' +
402
+ 'instance: the script would run in a browser on the server, not on your machine. ' +
403
+ 'Use the click / type / select / press actions, or run the MCP server locally over stdio.'
404
+ );
405
+ }
406
+
407
+ // Every `navigate` in here re-runs the SSRF guard and the blocklist/robots
408
+ // gate inside executeNavigateAction — verified, and the reason the gate is
409
+ // not repeated here. A long-lived session is a repeatable navigation
410
+ // primitive, so that per-hop check is what stops it becoming an SSRF hop.
411
+ const result = await this.actionExecutor.executeActionsOnPage(session.page, params.actions, {
412
+ continueOnError: params.continue_on_error,
413
+ timeout: params.timeout,
414
+ browserOptions: { respectRobots: params.respect_robots }
415
+ });
416
+
417
+ this.store.touch(session, result.finalUrl);
418
+ return {
419
+ success: result.success,
420
+ operation: 'act',
421
+ ...sessionInfo(session),
422
+ // The status of the last navigation this chain made, if it made one —
423
+ // a `navigate` action onto a 404 or a wall is otherwise invisible.
424
+ ...(Number.isInteger(session.page.__crawlforgeNavigation?.status)
425
+ ? { httpStatus: session.page.__crawlforgeNavigation.status }
426
+ : {}),
427
+ error: result.error,
428
+ actionResults: result.results.map(withJsResult),
429
+ screenshots: result.screenshots,
430
+ ...(result.capturedStates.length > 0 ? { capturedStates: result.capturedStates } : {}),
431
+ stats: result.stats
432
+ };
433
+ }
434
+
435
+ async readSession(params, ownerId) {
436
+ const session = this.requireSession(params, ownerId);
437
+ const url = session.page.url();
438
+ const html = await session.page.content();
439
+
440
+ const options = {};
441
+ if (params.formats.includes('markdown')) options.outputFormat = 'markdown';
442
+ if (params.formats.includes('html')) options.includeRawHTML = true;
443
+
444
+ // The live post-action DOM is already in hand, so extract_content is handed
445
+ // that rather than the url: a re-fetch would arrive without the session's
446
+ // cookies and before everything the session has clicked, which is the whole
447
+ // point of having one.
448
+ const extracted = await this.extractContentTool.execute({ url, html, options });
449
+
450
+ const content = {};
451
+ if (params.formats.includes('text')) {
452
+ content.text = extracted.content?.text || '';
453
+ }
454
+ if (params.formats.includes('html')) {
455
+ content.html = extracted.content?.html || html;
456
+ }
457
+ if (params.formats.includes('markdown')) {
458
+ // Readability finds no article on most app pages, and then no markdown is
459
+ // produced at all (R20, 2026-09-07). Convert the DOM we hold instead of
460
+ // handing back a placeholder.
461
+ content.markdown = extracted.content?.markdown || htmlToMarkdown(html);
462
+ }
463
+ if (params.formats.includes('json')) {
464
+ content.json = {
465
+ title: extracted.title ?? null,
466
+ metadata: extracted.metadata || {},
467
+ structuredData: extracted.structuredData
468
+ };
469
+ }
470
+
471
+ // Read is where the content is actually handed over, so it is the last
472
+ // place a wall can be named before a caller treats it as the page: g2.com
473
+ // came back here as the single word "g2.com" with success:true. The
474
+ // document is already in hand, so this costs no extra page work.
475
+ const verdict = await this.pageVerdict(session.page, {
476
+ stealth: session.stealth,
477
+ title: extracted.title ?? '',
478
+ text: extracted.content?.text || '',
479
+ html
480
+ });
481
+
482
+ this.store.touch(session, url);
483
+ return {
484
+ success: verdict.success,
485
+ operation: 'read',
486
+ ...sessionInfo(session),
487
+ ...verdictFields(verdict),
488
+ ...(verdict.error ? { error: verdict.error } : {}),
489
+ title: extracted.title ?? null,
490
+ extractionMethod: extracted.extractionMethod,
491
+ content
492
+ };
493
+ }
494
+
495
+ async screenshotSession(params, ownerId) {
496
+ const session = this.requireSession(params, ownerId);
497
+
498
+ const shot = await this.actionExecutor.captureScreenshot(session.page, {
499
+ fullPage: params.full_page,
500
+ format: params.format,
501
+ quality: params.quality,
502
+ selector: params.selector
503
+ });
504
+
505
+ this.store.touch(session, session.page.url());
506
+ // The actionId is what lets the server publish the image as a
507
+ // crawlforge://screenshot/{actionId} resource and drop the base64 from the
508
+ // result — a full-page PNG inline is megabytes (R21, 2026-09-09).
509
+ return {
510
+ success: true,
511
+ operation: 'screenshot',
512
+ ...sessionInfo(session),
513
+ screenshot: { actionId: this.actionExecutor.generateActionId(), ...shot }
514
+ };
515
+ }
516
+
517
+ async closeSession(params, ownerId) {
518
+ if (!params.session_id) {
519
+ throw new Error('operation "close" requires session_id.');
520
+ }
521
+ await this.store.close(params.session_id, ownerId);
522
+ return { success: true, operation: 'close', sessionId: params.session_id, closed: true };
523
+ }
524
+
525
+ listSessions(ownerId) {
526
+ const sessions = this.store.list(ownerId).map(({ id, ...rest }) => ({ sessionId: id, ...rest }));
527
+ return { success: true, operation: 'list', count: sessions.length, sessions };
528
+ }
529
+
530
+ /**
531
+ * How a page goes back: the closure the store calls on close, on expiry and
532
+ * at shutdown.
533
+ *
534
+ * Copied from executeActionChain's `finally`, and the halves are not
535
+ * interchangeable. A stealth page goes through the manager so its pooled
536
+ * context slot is freed as well as the renderer; a standard page must also
537
+ * close the BrowserContext createPage() gave it, because nothing else tracks
538
+ * that one. Getting this wrong pins a context until the process dies, which
539
+ * on the 2 GB Render box is an outage rather than a leak.
540
+ */
541
+ releaser(page, stealth) {
542
+ return async () => {
543
+ if (stealth) {
544
+ await this.actionExecutor.browserProcessor.releaseStealthPage(page);
545
+ return;
546
+ }
547
+ const context = page.context();
548
+ try { await page.close(); } catch (_) { /* ignore close errors */ }
549
+ try { await context.close(); } catch (_) { /* ignore close errors */ }
550
+ };
551
+ }
552
+
553
+ /**
554
+ * Close every live session, leaving the tool able to open more.
555
+ *
556
+ * This is the half the stealth-cleanup lever needs: `stealth_mode`
557
+ * operation:"cleanup" tears down the stealth browser, so every page a stealth
558
+ * session is holding dies with it and those sessions have to go too — but the
559
+ * tool itself must still work afterwards. The store is replaced rather than
560
+ * reused because destroy() also stops its sweep timer, and an injected store
561
+ * is replaced along with the rest.
562
+ */
563
+ async cleanup() {
564
+ await this.store.destroy();
565
+ this.store = new BrowserSessionStore(this.storeOptions);
566
+ }
567
+
568
+ /**
569
+ * Process exit: close the sessions, then the browser — but only if the
570
+ * executor is ours. Sessions always close here; that is why this tool is in
571
+ * server.js's shutdown list even when it shares another tool's executor.
572
+ */
573
+ async destroy() {
574
+ await this.store.destroy();
575
+ if (this._ownsExecutor) await this.actionExecutor.destroy();
576
+ }
577
+ }
578
+
579
+ export default BrowserSessionTool;
@@ -129,6 +129,14 @@ const ExecuteJavaScriptActionSchema = BaseActionSchema.extend({
129
129
  returnResult: z.boolean().default(true)
130
130
  });
131
131
 
132
+ // camelCase fields, matching ActionExecutor's SnapshotActionSchema - this
133
+ // union runs first, so a mismatch here rejects the action before it gets there.
134
+ const SnapshotActionSchema = BaseActionSchema.extend({
135
+ type: z.literal('snapshot'),
136
+ interactiveOnly: z.boolean().default(true),
137
+ maxNodes: z.number().min(1).max(1000).optional()
138
+ });
139
+
132
140
  const ActionSchema = z.union([
133
141
  WaitActionSchema,
134
142
  ClickActionSchema,
@@ -139,7 +147,8 @@ const ActionSchema = z.union([
139
147
  HoverActionSchema,
140
148
  NavigateActionSchema,
141
149
  ScreenshotActionSchema,
142
- ExecuteJavaScriptActionSchema
150
+ ExecuteJavaScriptActionSchema,
151
+ SnapshotActionSchema
143
152
  ]);
144
153
 
145
154
  // Form field schema for auto-fill