@agentium/browser 3.0.2 → 3.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/browser-agent.d.ts +117 -0
- package/dist/browser-agent.d.ts.map +1 -0
- package/dist/browser-provider.d.ts +217 -0
- package/dist/browser-provider.d.ts.map +1 -0
- package/dist/credential-vault.d.ts +37 -0
- package/dist/credential-vault.d.ts.map +1 -0
- package/dist/index.d.ts +5 -801
- package/dist/index.d.ts.map +1 -0
- package/dist/loop-detector.d.ts +83 -0
- package/dist/loop-detector.d.ts.map +1 -0
- package/dist/prompts.d.ts +20 -0
- package/dist/prompts.d.ts.map +1 -0
- package/dist/stealth.d.ts +27 -0
- package/dist/stealth.d.ts.map +1 -0
- package/dist/{index.d.cts → types.d.ts} +16 -385
- package/dist/types.d.ts.map +1 -0
- package/package.json +4 -4
package/dist/index.d.ts
CHANGED
|
@@ -1,801 +1,5 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
*
|
|
7
|
-
* Secrets are stored in memory and NEVER sent to the LLM.
|
|
8
|
-
* The model works with placeholders (e.g. `{{email}}`, `{{password}}`),
|
|
9
|
-
* and the agent resolves them to real values only at execution time.
|
|
10
|
-
*/
|
|
11
|
-
declare class CredentialVault {
|
|
12
|
-
private secrets;
|
|
13
|
-
constructor(initial?: Record<string, string>);
|
|
14
|
-
/** Store a credential. Key names become the placeholder: `{{key}}`. */
|
|
15
|
-
set(key: string, value: string): this;
|
|
16
|
-
/** Retrieve a credential value. Returns undefined if not found. */
|
|
17
|
-
get(key: string): string | undefined;
|
|
18
|
-
has(key: string): boolean;
|
|
19
|
-
/** List available placeholder names (never exposes values). */
|
|
20
|
-
keys(): string[];
|
|
21
|
-
/**
|
|
22
|
-
* Load credentials from environment variables.
|
|
23
|
-
* Maps env var names to placeholder keys.
|
|
24
|
-
*
|
|
25
|
-
* @example
|
|
26
|
-
* vault.fromEnv({ email: "LOGIN_EMAIL", password: "LOGIN_PASS" });
|
|
27
|
-
*/
|
|
28
|
-
fromEnv(mapping: Record<string, string>): this;
|
|
29
|
-
/**
|
|
30
|
-
* Replace `{{key}}` placeholders in text with actual credential values.
|
|
31
|
-
* Used internally by BrowserAgent right before executing a type action.
|
|
32
|
-
*/
|
|
33
|
-
resolve(text: string): string;
|
|
34
|
-
/**
|
|
35
|
-
* Replace any occurrence of real credential values in text with
|
|
36
|
-
* their `{{key}}` placeholder. Used to sanitize logs and action history.
|
|
37
|
-
*/
|
|
38
|
-
mask(text: string): string;
|
|
39
|
-
}
|
|
40
|
-
|
|
41
|
-
/**
|
|
42
|
-
* The complete set of actions a `BrowserAgent` can take during a run.
|
|
43
|
-
*
|
|
44
|
-
* Click / type / scroll accept EITHER an `index` (preferred — resolved
|
|
45
|
-
* against the DOM tree we showed the model) OR `x`/`y` coordinates as a
|
|
46
|
-
* fallback. Indexed actions are far more reliable on dynamic pages because
|
|
47
|
-
* they survive layout shifts, devicePixelRatio mismatches, and "the
|
|
48
|
-
* sibling 4 pixels away" failure modes.
|
|
49
|
-
*/
|
|
50
|
-
type BrowserAction = {
|
|
51
|
-
action: "click";
|
|
52
|
-
/** Element index from the DOM snapshot. Preferred. */
|
|
53
|
-
index?: number;
|
|
54
|
-
/** Fallback x coordinate (CSS pixels). */
|
|
55
|
-
x?: number;
|
|
56
|
-
/** Fallback y coordinate (CSS pixels). */
|
|
57
|
-
y?: number;
|
|
58
|
-
/** Free-form description; if it contains a quoted label, used as a text-locator fallback. */
|
|
59
|
-
description?: string;
|
|
60
|
-
} | {
|
|
61
|
-
action: "type";
|
|
62
|
-
/** Element index from the DOM snapshot. Preferred. */
|
|
63
|
-
index?: number;
|
|
64
|
-
text: string;
|
|
65
|
-
/** Fallback x coordinate (CSS pixels). */
|
|
66
|
-
x?: number;
|
|
67
|
-
/** Fallback y coordinate (CSS pixels). */
|
|
68
|
-
y?: number;
|
|
69
|
-
/** If true (default), clear the field first. */
|
|
70
|
-
clear?: boolean;
|
|
71
|
-
/** If true (default), press Enter after typing when text ends with `\n`. */
|
|
72
|
-
submit?: boolean;
|
|
73
|
-
} | {
|
|
74
|
-
action: "scroll";
|
|
75
|
-
direction: "up" | "down";
|
|
76
|
-
/** Pixels to scroll. Default: 400. */
|
|
77
|
-
amount?: number;
|
|
78
|
-
/** If provided, scroll the element with this index into view instead. */
|
|
79
|
-
index?: number;
|
|
80
|
-
} | {
|
|
81
|
-
action: "navigate";
|
|
82
|
-
url: string;
|
|
83
|
-
} | {
|
|
84
|
-
action: "back";
|
|
85
|
-
} | {
|
|
86
|
-
action: "wait";
|
|
87
|
-
ms: number;
|
|
88
|
-
} | {
|
|
89
|
-
action: "screenshot";
|
|
90
|
-
} | {
|
|
91
|
-
action: "send_keys";
|
|
92
|
-
keys: string;
|
|
93
|
-
} | {
|
|
94
|
-
action: "find_text";
|
|
95
|
-
text: string;
|
|
96
|
-
} | {
|
|
97
|
-
action: "evaluate";
|
|
98
|
-
code: string;
|
|
99
|
-
} | {
|
|
100
|
-
action: "dropdown_options";
|
|
101
|
-
index: number;
|
|
102
|
-
} | {
|
|
103
|
-
action: "select_dropdown";
|
|
104
|
-
index: number;
|
|
105
|
-
text: string;
|
|
106
|
-
} | {
|
|
107
|
-
action: "upload_file";
|
|
108
|
-
index: number;
|
|
109
|
-
path: string;
|
|
110
|
-
} | {
|
|
111
|
-
action: "extract";
|
|
112
|
-
/** Natural-language description of what to extract. */
|
|
113
|
-
query: string;
|
|
114
|
-
/** Include link hrefs in the extracted content. Default: false. */
|
|
115
|
-
extractLinks?: boolean;
|
|
116
|
-
} | {
|
|
117
|
-
action: "tool";
|
|
118
|
-
/** Name of a custom tool registered on the BrowserAgent. */
|
|
119
|
-
name: string;
|
|
120
|
-
args?: Record<string, unknown>;
|
|
121
|
-
} | {
|
|
122
|
-
action: "done";
|
|
123
|
-
result: string;
|
|
124
|
-
} | {
|
|
125
|
-
action: "fail";
|
|
126
|
-
reason: string;
|
|
127
|
-
};
|
|
128
|
-
/**
|
|
129
|
-
* One entry in the DOM snapshot the model sees. Returned by
|
|
130
|
-
* `BrowserProvider.extractDOM()` alongside the human-readable string form.
|
|
131
|
-
*/
|
|
132
|
-
interface DomElement {
|
|
133
|
-
/** Stable 1-based index for this step. Used by indexed actions. */
|
|
134
|
-
index: number;
|
|
135
|
-
/** Center coordinate in CSS pixels (fallback for the model if needed). */
|
|
136
|
-
cx: number;
|
|
137
|
-
cy: number;
|
|
138
|
-
/** ARIA role or tag name. */
|
|
139
|
-
role: string;
|
|
140
|
-
/** Input type, if applicable. */
|
|
141
|
-
type?: string;
|
|
142
|
-
/** Visible label (text / aria-label / placeholder / href / …). */
|
|
143
|
-
label: string;
|
|
144
|
-
/** Tag name. */
|
|
145
|
-
tag: string;
|
|
146
|
-
/** Whether it's an `<input>` / `<textarea>` / `[contenteditable]`. */
|
|
147
|
-
isInput: boolean;
|
|
148
|
-
/** Whether it's a `<select>`. */
|
|
149
|
-
isSelect: boolean;
|
|
150
|
-
/** Whether it's an `<input type="file">`. */
|
|
151
|
-
isFile: boolean;
|
|
152
|
-
/** Origin frame (`"main"` or the iframe `src`/path). */
|
|
153
|
-
frame?: string;
|
|
154
|
-
}
|
|
155
|
-
/**
|
|
156
|
-
* Scroll-context metadata returned alongside `DomElement[]`. Gives the
|
|
157
|
-
* model spatial awareness so it can decide when to scroll vs when more
|
|
158
|
-
* content is below / above the fold.
|
|
159
|
-
*/
|
|
160
|
-
interface DomScrollContext {
|
|
161
|
-
/** Approximate viewports of scrollable content above the current view. */
|
|
162
|
-
pagesAbove: number;
|
|
163
|
-
/** Approximate viewports of scrollable content below the current view. */
|
|
164
|
-
pagesBelow: number;
|
|
165
|
-
/** Total interactive elements found (visible + hidden combined). */
|
|
166
|
-
totalInteractive: number;
|
|
167
|
-
/** Count of interactive elements that exist on the page but aren't in the viewport. */
|
|
168
|
-
hiddenInteractive: number;
|
|
169
|
-
}
|
|
170
|
-
/** Combined return value of `BrowserProvider.extractDOM()`. */
|
|
171
|
-
interface DomSnapshot {
|
|
172
|
-
/** Human-readable string fed to the model. */
|
|
173
|
-
text: string;
|
|
174
|
-
/** Structured list (stable indices) for runtime resolution. */
|
|
175
|
-
elements: DomElement[];
|
|
176
|
-
/** Spatial/scroll context. */
|
|
177
|
-
scroll: DomScrollContext;
|
|
178
|
-
}
|
|
179
|
-
interface BrowserAgentConfig {
|
|
180
|
-
name: string;
|
|
181
|
-
/** Vision-capable model (GPT-4o, Gemini, etc.) */
|
|
182
|
-
model: ModelProvider;
|
|
183
|
-
/**
|
|
184
|
-
* Optional secondary (usually cheaper) model used for the `extract`
|
|
185
|
-
* action and other text-only sub-tasks. Falls back to `model`.
|
|
186
|
-
*/
|
|
187
|
-
pageExtractionLLM?: ModelProvider;
|
|
188
|
-
/**
|
|
189
|
-
* Fallback model used automatically when the primary `model` returns a
|
|
190
|
-
* rate-limit / auth / 5xx error or fails to produce valid JSON several
|
|
191
|
-
* times in a row. Failure-budget aware. Leave unset to disable.
|
|
192
|
-
*/
|
|
193
|
-
fallbackModel?: ModelProvider;
|
|
194
|
-
/**
|
|
195
|
-
* Ask the model to emit a structured thinking/evaluation/memory/next_goal
|
|
196
|
-
* envelope around its action(s). Significantly improves accuracy and
|
|
197
|
-
* self-correction on multi-step tasks. Default: `true`. Set `false` for
|
|
198
|
-
* a `flash_mode` that just returns the raw action(s) — useful for very
|
|
199
|
-
* fast / cheap models that are bad at long outputs.
|
|
200
|
-
*/
|
|
201
|
-
useThinking?: boolean;
|
|
202
|
-
/**
|
|
203
|
-
* Maximum number of recent step turns kept verbatim in the conversation
|
|
204
|
-
* history sent to the model. Older turns are compacted into a single
|
|
205
|
-
* summary line. Default: 6. Set 0 to disable conversation history
|
|
206
|
-
* entirely (each step rebuilt from scratch — v2.0 behaviour).
|
|
207
|
-
*/
|
|
208
|
-
historyWindow?: number;
|
|
209
|
-
/** Extra instructions appended to the default system prompt. */
|
|
210
|
-
instructions?: string;
|
|
211
|
-
/**
|
|
212
|
-
* Append additional instructions to the default system prompt.
|
|
213
|
-
* Alias for `instructions` (browser-use parity). Both are concatenated.
|
|
214
|
-
*/
|
|
215
|
-
extendSystemMessage?: string;
|
|
216
|
-
/**
|
|
217
|
-
* Completely replace the default system prompt with this string.
|
|
218
|
-
* The credentials / DOM / coordinate sections are still appended at the
|
|
219
|
-
* end automatically. Most users should use `instructions` instead.
|
|
220
|
-
*/
|
|
221
|
-
overrideSystemMessage?: string;
|
|
222
|
-
/** Max vision loop iterations. Default: 30 */
|
|
223
|
-
maxSteps?: number;
|
|
224
|
-
/**
|
|
225
|
-
* Maximum number of consecutive step failures (action threw, invalid
|
|
226
|
-
* JSON from model, locator timeout) before the agent gives up. Default: 3.
|
|
227
|
-
*/
|
|
228
|
-
maxFailures?: number;
|
|
229
|
-
/**
|
|
230
|
-
* Maximum number of actions the model can return in a single step.
|
|
231
|
-
* If the model returns an array, we execute them in order until the page
|
|
232
|
-
* navigates or the DOM changes substantially. Default: 3.
|
|
233
|
-
* Set to 1 to force one-action-per-step (the v2.0.x behaviour).
|
|
234
|
-
*/
|
|
235
|
-
maxActionsPerStep?: number;
|
|
236
|
-
/**
|
|
237
|
-
* Actions to run before the LLM loop starts. Useful for boilerplate
|
|
238
|
-
* (cookie-banner click, login flow, scrolling) you already know the
|
|
239
|
-
* answer to — saves vision tokens.
|
|
240
|
-
*/
|
|
241
|
-
initialActions?: BrowserAction[];
|
|
242
|
-
/**
|
|
243
|
-
* Vision mode. Default: `"auto"`.
|
|
244
|
-
* - `true`: always send a screenshot with every step (v2.0.x behaviour).
|
|
245
|
-
* - `false`: never send screenshots; DOM-only operation.
|
|
246
|
-
* - `"auto"`: send a screenshot on the first step and whenever the model
|
|
247
|
-
* used the `screenshot` action in the previous step. Saves vision
|
|
248
|
-
* tokens dramatically on DOM-only-suitable workloads.
|
|
249
|
-
*/
|
|
250
|
-
useVision?: boolean | "auto";
|
|
251
|
-
/**
|
|
252
|
-
* If true (default), detect a URL in the task string and navigate to it
|
|
253
|
-
* before the first LLM call — saves one round trip on simple tasks.
|
|
254
|
-
*/
|
|
255
|
-
directlyOpenUrl?: boolean;
|
|
256
|
-
/** Run browser without visible window. Default: true */
|
|
257
|
-
headless?: boolean;
|
|
258
|
-
/** Browser viewport size. Default: 1280x720 */
|
|
259
|
-
viewport?: {
|
|
260
|
-
width: number;
|
|
261
|
-
height: number;
|
|
262
|
-
};
|
|
263
|
-
/** Initial URL to navigate to before starting the task */
|
|
264
|
-
startUrl?: string;
|
|
265
|
-
/** Milliseconds to wait after each action for the page to settle. Default: 1500 */
|
|
266
|
-
waitAfterAction?: number;
|
|
267
|
-
/** Max consecutive identical actions before the agent auto-fails. Default: 3 */
|
|
268
|
-
maxRepeats?: number;
|
|
269
|
-
/**
|
|
270
|
-
* Include a simplified DOM/accessibility tree (each interactive element
|
|
271
|
-
* tagged with its exact center coordinates) alongside the screenshot.
|
|
272
|
-
* Dramatically improves click accuracy — strongly recommended.
|
|
273
|
-
* Default: true
|
|
274
|
-
*/
|
|
275
|
-
useDOM?: boolean;
|
|
276
|
-
/**
|
|
277
|
-
* Allow the `evaluate` action to run arbitrary JavaScript inside the
|
|
278
|
-
* page. Default: false (security). Only enable if you trust the source
|
|
279
|
-
* of task strings — a malicious task could exfiltrate page contents.
|
|
280
|
-
*/
|
|
281
|
-
allowEvaluate?: boolean;
|
|
282
|
-
/**
|
|
283
|
-
* Restrict navigation to specific domains. Wildcard patterns supported:
|
|
284
|
-
* `"example.com"`, `"*.example.com"`, `"http*://example.com"`.
|
|
285
|
-
* When set, any `navigate` action to a non-matching URL throws.
|
|
286
|
-
*/
|
|
287
|
-
allowedDomains?: string[];
|
|
288
|
-
/**
|
|
289
|
-
* Block navigation to specific domains. Same pattern format as
|
|
290
|
-
* `allowedDomains`. Evaluated AFTER `allowedDomains`, so if both are
|
|
291
|
-
* set a URL must be in `allowedDomains` AND not in `prohibitedDomains`.
|
|
292
|
-
*/
|
|
293
|
-
prohibitedDomains?: string[];
|
|
294
|
-
/**
|
|
295
|
-
* Path to a Playwright storageState JSON file.
|
|
296
|
-
* Restores cookies, localStorage, and sessionStorage from a previous session.
|
|
297
|
-
*/
|
|
298
|
-
storageState?: string;
|
|
299
|
-
/**
|
|
300
|
-
* Connect to an existing browser via Chrome DevTools Protocol instead
|
|
301
|
-
* of launching one. Format: `"http://localhost:9222"`. When set,
|
|
302
|
-
* `headless`, `stealth.args`, `recordVideo`, etc. are ignored — the
|
|
303
|
-
* existing browser's configuration is used.
|
|
304
|
-
*/
|
|
305
|
-
cdpUrl?: string;
|
|
306
|
-
/**
|
|
307
|
-
* Enable video recording of the browser session.
|
|
308
|
-
* Pass `true` for default dir (`./browser-videos`) or `{ dir: "/path" }`.
|
|
309
|
-
*/
|
|
310
|
-
recordVideo?: boolean | {
|
|
311
|
-
dir: string;
|
|
312
|
-
};
|
|
313
|
-
/**
|
|
314
|
-
* Secure credential vault. The LLM never sees real values — only
|
|
315
|
-
* placeholders like `{{email}}`, `{{password}}`. Real values are
|
|
316
|
-
* injected at execution time and scrubbed from all logs.
|
|
317
|
-
*/
|
|
318
|
-
credentials?: CredentialVault;
|
|
319
|
-
/**
|
|
320
|
-
* Enable stealth mode to avoid bot detection.
|
|
321
|
-
* Pass `true` for sensible defaults or a `StealthConfig` object for fine control.
|
|
322
|
-
* Patches navigator.webdriver, plugins, permissions, WebGL, and more.
|
|
323
|
-
*/
|
|
324
|
-
stealth?: boolean | StealthConfig;
|
|
325
|
-
/**
|
|
326
|
-
* Simulate human-like behavior — jittered clicks, variable typing speed,
|
|
327
|
-
* mouse movement curves, random micro-pauses.
|
|
328
|
-
* Pass `true` for defaults or a `HumanizeConfig` for fine control.
|
|
329
|
-
*/
|
|
330
|
-
humanize?: boolean | HumanizeConfig;
|
|
331
|
-
/**
|
|
332
|
-
* Unified memory config — persist browser sessions, decisions, and
|
|
333
|
-
* summaries of past runs. Same config as Agent and VoiceAgent.
|
|
334
|
-
*/
|
|
335
|
-
memory?: UnifiedMemoryConfig;
|
|
336
|
-
/** Skills — pre-packaged or learned tool bundles. */
|
|
337
|
-
skills?: Array<_agentium_core.Skill | string>;
|
|
338
|
-
/**
|
|
339
|
-
* Custom tools that the BrowserAgent itself can invoke during a run.
|
|
340
|
-
* The agent emits `{ "action": "tool", "name": "<tool>", "args": {...} }`
|
|
341
|
-
* and we dispatch to the tool's `execute(args)`. Use this for 2FA codes,
|
|
342
|
-
* API calls, file I/O, calling out to other agents — anything the
|
|
343
|
-
* browser can't do alone.
|
|
344
|
-
*/
|
|
345
|
-
tools?: ToolDef[];
|
|
346
|
-
/** Cost tracker — track vision model token usage and enforce budgets across browser runs. */
|
|
347
|
-
costTracker?: CostTracker;
|
|
348
|
-
logLevel?: LogLevel;
|
|
349
|
-
eventBus?: EventBus;
|
|
350
|
-
}
|
|
351
|
-
interface BrowserRunOpts {
|
|
352
|
-
/** Override startUrl from config */
|
|
353
|
-
startUrl?: string;
|
|
354
|
-
/** Per-run model API key override */
|
|
355
|
-
apiKey?: string;
|
|
356
|
-
/** Session identifier for memory persistence and event tracking */
|
|
357
|
-
sessionId?: string;
|
|
358
|
-
/** User identifier for memory personalization */
|
|
359
|
-
userId?: string;
|
|
360
|
-
/** Path to save storageState (cookies/auth) after the run completes */
|
|
361
|
-
saveStorageState?: string;
|
|
362
|
-
/** Per-run override of `maxSteps`. */
|
|
363
|
-
maxSteps?: number;
|
|
364
|
-
}
|
|
365
|
-
interface BrowserRunOutput {
|
|
366
|
-
/** Final text result produced by the agent */
|
|
367
|
-
result: string;
|
|
368
|
-
/** Whether the task completed successfully (vs maxSteps exhausted or fail) */
|
|
369
|
-
success: boolean;
|
|
370
|
-
/** Full action history with screenshots */
|
|
371
|
-
steps: BrowserStep[];
|
|
372
|
-
/** URL at completion */
|
|
373
|
-
finalUrl: string;
|
|
374
|
-
/** Last screenshot captured */
|
|
375
|
-
finalScreenshot: Buffer;
|
|
376
|
-
/** Total time taken in milliseconds */
|
|
377
|
-
durationMs: number;
|
|
378
|
-
/** Video file path (if recordVideo was enabled) */
|
|
379
|
-
videoPath?: string;
|
|
380
|
-
/** Extracted content from every `extract` action, in chronological order. */
|
|
381
|
-
extractedContent?: string[];
|
|
382
|
-
}
|
|
383
|
-
interface BrowserStep {
|
|
384
|
-
index: number;
|
|
385
|
-
action: BrowserAction;
|
|
386
|
-
/** Screenshot taken before this action was executed. May be empty if useVision=false. */
|
|
387
|
-
screenshot: Buffer;
|
|
388
|
-
pageUrl: string;
|
|
389
|
-
pageTitle: string;
|
|
390
|
-
timestamp: Date;
|
|
391
|
-
/** Simplified DOM snapshot (if useDOM is enabled) */
|
|
392
|
-
dom?: string;
|
|
393
|
-
/** Free-form result string produced by the action (e.g. extract output). */
|
|
394
|
-
output?: string;
|
|
395
|
-
/** Whether this step succeeded (vs threw / failed locator). Default: true. */
|
|
396
|
-
ok?: boolean;
|
|
397
|
-
/** Model's chain-of-thought reasoning (if `useThinking` was on). */
|
|
398
|
-
thinking?: string;
|
|
399
|
-
/** Model's evaluation of whether the previous action met its goal. */
|
|
400
|
-
evaluationPreviousGoal?: string;
|
|
401
|
-
/** Model's running memory of important state. */
|
|
402
|
-
memory?: string;
|
|
403
|
-
/** Model's stated next goal for this step. */
|
|
404
|
-
nextGoal?: string;
|
|
405
|
-
}
|
|
406
|
-
/**
|
|
407
|
-
* Structured envelope the model returns when `useThinking: true`. Inspired
|
|
408
|
-
* by browser-use's `AgentOutput`. Every field is optional from a runtime
|
|
409
|
-
* standpoint — only `action` is required for execution.
|
|
410
|
-
*/
|
|
411
|
-
interface AgentOutput {
|
|
412
|
-
thinking?: string;
|
|
413
|
-
evaluationPreviousGoal?: string;
|
|
414
|
-
memory?: string;
|
|
415
|
-
nextGoal?: string;
|
|
416
|
-
action: BrowserAction | BrowserAction[];
|
|
417
|
-
}
|
|
418
|
-
interface StealthConfig {
|
|
419
|
-
/**
|
|
420
|
-
* Remove `navigator.webdriver` flag and patch common detection vectors
|
|
421
|
-
* (plugins, languages, permissions, WebGL, etc.). Default when stealth=true.
|
|
422
|
-
*/
|
|
423
|
-
patchFingerprint?: boolean;
|
|
424
|
-
/** Custom User-Agent string. A realistic one is used by default. */
|
|
425
|
-
userAgent?: string;
|
|
426
|
-
/** Browser locale. Default: "en-US" */
|
|
427
|
-
locale?: string;
|
|
428
|
-
/** Timezone ID (IANA). Default: "America/New_York" */
|
|
429
|
-
timezone?: string;
|
|
430
|
-
/** Fake geolocation */
|
|
431
|
-
geolocation?: {
|
|
432
|
-
latitude: number;
|
|
433
|
-
longitude: number;
|
|
434
|
-
accuracy?: number;
|
|
435
|
-
};
|
|
436
|
-
/** Ignore HTTPS certificate errors. Default: false (secure). Only enable for local testing. */
|
|
437
|
-
ignoreHTTPSErrors?: boolean;
|
|
438
|
-
/**
|
|
439
|
-
* `window.devicePixelRatio` to emulate. Default: `1`. Set to `2` to mimic
|
|
440
|
-
* a Retina display (sharper screenshots at 2× cost). Setting to 2 on a
|
|
441
|
-
* non-Retina host display can cause the headed window to look zoomed-out
|
|
442
|
-
* or stretched because the OS compositor downsamples a 2× surface.
|
|
443
|
-
*/
|
|
444
|
-
deviceScaleFactor?: number;
|
|
445
|
-
/** HTTP/SOCKS proxy. Format: "http://user:pass@host:port" */
|
|
446
|
-
proxy?: {
|
|
447
|
-
server: string;
|
|
448
|
-
username?: string;
|
|
449
|
-
password?: string;
|
|
450
|
-
};
|
|
451
|
-
}
|
|
452
|
-
interface HumanizeConfig {
|
|
453
|
-
/** Per-character typing delay range in ms. Default: [40, 120] */
|
|
454
|
-
typingDelay?: [number, number];
|
|
455
|
-
/** Random pixel offset added to click coordinates. Default: 3 */
|
|
456
|
-
clickJitter?: number;
|
|
457
|
-
/** Extra random pause between actions in ms range. Default: [200, 800] */
|
|
458
|
-
actionDelay?: [number, number];
|
|
459
|
-
/** Simulate human-like mouse movement to target before clicking. Default: true */
|
|
460
|
-
mouseMovement?: boolean;
|
|
461
|
-
}
|
|
462
|
-
interface PageInfo {
|
|
463
|
-
url: string;
|
|
464
|
-
title: string;
|
|
465
|
-
viewportSize: {
|
|
466
|
-
width: number;
|
|
467
|
-
height: number;
|
|
468
|
-
};
|
|
469
|
-
}
|
|
470
|
-
|
|
471
|
-
declare class BrowserAgent {
|
|
472
|
-
readonly name: string;
|
|
473
|
-
readonly eventBus: EventBus;
|
|
474
|
-
private model;
|
|
475
|
-
private pageExtractionLLM;
|
|
476
|
-
private fallbackModel;
|
|
477
|
-
private useThinking;
|
|
478
|
-
private historyWindow;
|
|
479
|
-
private instructions?;
|
|
480
|
-
private extendSystemMessage?;
|
|
481
|
-
private overrideSystemMessage?;
|
|
482
|
-
private maxSteps;
|
|
483
|
-
private maxFailures;
|
|
484
|
-
private maxActionsPerStep;
|
|
485
|
-
private initialActions;
|
|
486
|
-
private useVision;
|
|
487
|
-
private directlyOpenUrl;
|
|
488
|
-
private headless;
|
|
489
|
-
private viewport;
|
|
490
|
-
private defaultStartUrl?;
|
|
491
|
-
private waitAfterAction;
|
|
492
|
-
private maxRepeats;
|
|
493
|
-
private useDOM;
|
|
494
|
-
private allowEvaluate;
|
|
495
|
-
private allowedDomains?;
|
|
496
|
-
private prohibitedDomains?;
|
|
497
|
-
private storageState?;
|
|
498
|
-
private cdpUrl?;
|
|
499
|
-
private recordVideo?;
|
|
500
|
-
private credentials?;
|
|
501
|
-
private stealth?;
|
|
502
|
-
private humanize?;
|
|
503
|
-
private tools;
|
|
504
|
-
private costTracker;
|
|
505
|
-
private memoryManager;
|
|
506
|
-
private logger;
|
|
507
|
-
/** Access the MemoryManager (if memory is configured). */
|
|
508
|
-
get memory(): MemoryManager | null;
|
|
509
|
-
constructor(config: BrowserAgentConfig);
|
|
510
|
-
run(task: string, opts?: BrowserRunOpts): Promise<BrowserRunOutput>;
|
|
511
|
-
/**
|
|
512
|
-
* Returns a ToolDef that lets a regular Agent delegate browser tasks
|
|
513
|
-
* to this BrowserAgent.
|
|
514
|
-
*/
|
|
515
|
-
asTool(config?: {
|
|
516
|
-
name?: string;
|
|
517
|
-
description?: string;
|
|
518
|
-
}): ToolDef;
|
|
519
|
-
private shouldCaptureVision;
|
|
520
|
-
private detectUrlInTask;
|
|
521
|
-
private assertDomainAllowed;
|
|
522
|
-
/**
|
|
523
|
-
* Build the message array sent to the model: system prompt + a compact
|
|
524
|
-
* summary of older turns (if any) + the most recent `historyWindow`
|
|
525
|
-
* turns verbatim + the current step's user message.
|
|
526
|
-
*
|
|
527
|
-
* Inspired by browser-use's history compaction. Keeps tokens bounded
|
|
528
|
-
* while giving the model meaningful context about what it already
|
|
529
|
-
* tried.
|
|
530
|
-
*/
|
|
531
|
-
private buildMessages;
|
|
532
|
-
/**
|
|
533
|
-
* Call the primary model, retrying once with `fallbackModel` on
|
|
534
|
-
* transient errors (5xx, 429, network). Returns the response or
|
|
535
|
-
* `null` if both models failed.
|
|
536
|
-
*/
|
|
537
|
-
private callModelWithFallback;
|
|
538
|
-
private isTransientError;
|
|
539
|
-
/**
|
|
540
|
-
* Parse the model's raw response into an `AgentOutput`. Tolerant to
|
|
541
|
-
* three shapes:
|
|
542
|
-
* - Full envelope: { thinking, evaluation_previous_goal, action, ... }
|
|
543
|
-
* - Raw action object (legacy / `useThinking: false`)
|
|
544
|
-
* - Raw action array
|
|
545
|
-
*
|
|
546
|
-
* Also strips ```json fences the model occasionally adds.
|
|
547
|
-
*/
|
|
548
|
-
private parseEnvelope;
|
|
549
|
-
/**
|
|
550
|
-
* Quick post-navigation health check. If the page came back blank
|
|
551
|
-
* (no body text and no interactive elements), reload once and wait
|
|
552
|
-
* for stable. This catches the "FreightOS half-loaded font test" class
|
|
553
|
-
* of failure before the LLM ever sees it.
|
|
554
|
-
*/
|
|
555
|
-
private navigationHealthCheck;
|
|
556
|
-
/**
|
|
557
|
-
* Force-finalize the run with whatever partial data the agent has.
|
|
558
|
-
* Called when:
|
|
559
|
-
* - `maxSteps` is exhausted without an explicit `done`,
|
|
560
|
-
* - `maxFailures` is exceeded,
|
|
561
|
-
* - the model can't produce parseable JSON enough times to make
|
|
562
|
-
* forward progress.
|
|
563
|
-
* The result is composed from `extractedContent` + the last few
|
|
564
|
-
* action summaries so the caller gets something useful instead of
|
|
565
|
-
* just a one-line error.
|
|
566
|
-
*/
|
|
567
|
-
private forceDone;
|
|
568
|
-
private finalize;
|
|
569
|
-
/**
|
|
570
|
-
* Execute a single action. Returns `{ output?, didNavigate? }` for the
|
|
571
|
-
* caller's state tracking. May throw — the loop handles failure-budget
|
|
572
|
-
* accounting in that case.
|
|
573
|
-
*/
|
|
574
|
-
private executeAction;
|
|
575
|
-
private sleep;
|
|
576
|
-
/**
|
|
577
|
-
* Parse a quoted target keyword from a click action's `description`.
|
|
578
|
-
* Returns `undefined` for generic / ambiguous labels (login buttons,
|
|
579
|
-
* close, OK, etc.) where a substring text match could fire on the
|
|
580
|
-
* wrong element.
|
|
581
|
-
*/
|
|
582
|
-
private extractClickKeyword;
|
|
583
|
-
}
|
|
584
|
-
|
|
585
|
-
/**
|
|
586
|
-
* Playwright wrapper with stealth anti-detection, human-like behavior,
|
|
587
|
-
* indexed DOM element resolution, and a rich action vocabulary.
|
|
588
|
-
*
|
|
589
|
-
* The `BrowserProvider` is intentionally LLM-agnostic — it exposes the
|
|
590
|
-
* primitives that `BrowserAgent` orchestrates via vision+DOM reasoning.
|
|
591
|
-
*/
|
|
592
|
-
declare class BrowserProvider {
|
|
593
|
-
private browser;
|
|
594
|
-
private context;
|
|
595
|
-
private page;
|
|
596
|
-
private pages;
|
|
597
|
-
private activeTabId;
|
|
598
|
-
private tabCounter;
|
|
599
|
-
private _viewport;
|
|
600
|
-
private _videoDir?;
|
|
601
|
-
private _humanize?;
|
|
602
|
-
/**
|
|
603
|
-
* Most recent DOM snapshot (one per `extractDOM` call). Indexed actions
|
|
604
|
-
* (`clickByIndex`, `inputByIndex`, …) resolve their `index` against this.
|
|
605
|
-
*/
|
|
606
|
-
private _lastDom;
|
|
607
|
-
/** True if we connected over CDP (don't tear down the browser on close). */
|
|
608
|
-
private _attached;
|
|
609
|
-
constructor();
|
|
610
|
-
launch(opts?: {
|
|
611
|
-
headless?: boolean;
|
|
612
|
-
viewport?: {
|
|
613
|
-
width: number;
|
|
614
|
-
height: number;
|
|
615
|
-
};
|
|
616
|
-
storageState?: string;
|
|
617
|
-
recordVideo?: boolean | {
|
|
618
|
-
dir: string;
|
|
619
|
-
};
|
|
620
|
-
stealth?: boolean | StealthConfig;
|
|
621
|
-
humanize?: boolean | HumanizeConfig;
|
|
622
|
-
cdpUrl?: string;
|
|
623
|
-
}): Promise<void>;
|
|
624
|
-
saveStorageState(path: string): Promise<void>;
|
|
625
|
-
navigate(url: string): Promise<void>;
|
|
626
|
-
back(): Promise<void>;
|
|
627
|
-
screenshot(): Promise<Buffer>;
|
|
628
|
-
/** Viewport size in CSS pixels (matches screenshot dimensions). */
|
|
629
|
-
get viewport(): {
|
|
630
|
-
width: number;
|
|
631
|
-
height: number;
|
|
632
|
-
};
|
|
633
|
-
/** Most recent DOM snapshot. Each entry has a stable `index`. */
|
|
634
|
-
get lastDom(): DomElement[];
|
|
635
|
-
click(x: number, y: number): Promise<void>;
|
|
636
|
-
type(text: string): Promise<void>;
|
|
637
|
-
clickAndType(x: number, y: number, text: string): Promise<void>;
|
|
638
|
-
pressKey(key: string): Promise<void>;
|
|
639
|
-
/**
|
|
640
|
-
* Send arbitrary keyboard keys / shortcuts. Accepts a single
|
|
641
|
-
* Playwright key spec (`"Enter"`, `"Control+l"`, `"Shift+ArrowDown"`)
|
|
642
|
-
* or a space-separated sequence (`"Tab Tab Enter"`).
|
|
643
|
-
*/
|
|
644
|
-
sendKeys(keys: string): Promise<void>;
|
|
645
|
-
scroll(direction: "up" | "down", amount?: number): Promise<void>;
|
|
646
|
-
/**
|
|
647
|
-
* Build a Playwright locator for a DOM-snapshot index. Each `extractDOM`
|
|
648
|
-
* call tags surviving elements with `data-bua-idx="<n>"`; we resolve by
|
|
649
|
-
* that attribute. Returns null if the index is unknown.
|
|
650
|
-
*/
|
|
651
|
-
private locatorForIndex;
|
|
652
|
-
/**
|
|
653
|
-
* Click an element by its DOM-snapshot index. The most reliable click
|
|
654
|
-
* path on dynamic pages — survives layout shifts and DPR oddities.
|
|
655
|
-
*/
|
|
656
|
-
clickByIndex(index: number, opts?: {
|
|
657
|
-
timeout?: number;
|
|
658
|
-
}): Promise<boolean>;
|
|
659
|
-
/**
|
|
660
|
-
* Focus an indexed input, optionally clear it, and type. Returns false
|
|
661
|
-
* if the index couldn't be resolved or the input couldn't be focused.
|
|
662
|
-
*/
|
|
663
|
-
inputByIndex(index: number, text: string, opts?: {
|
|
664
|
-
clear?: boolean;
|
|
665
|
-
submit?: boolean;
|
|
666
|
-
timeout?: number;
|
|
667
|
-
}): Promise<boolean>;
|
|
668
|
-
uploadFileByIndex(index: number, path: string): Promise<boolean>;
|
|
669
|
-
/** Scroll the indexed element into view (no click). */
|
|
670
|
-
scrollIntoViewByIndex(index: number): Promise<boolean>;
|
|
671
|
-
/**
|
|
672
|
-
* Deterministic, DOM-based click using Playwright's text locator.
|
|
673
|
-
*
|
|
674
|
-
* Returns `true` if a matching, visible, clickable element was found and
|
|
675
|
-
* clicked within `timeout` ms; `false` otherwise (so the caller can fall
|
|
676
|
-
* back to coordinate clicking). Substring-matches by default — e.g.
|
|
677
|
-
* `clickByText("Cheapest")` matches "Cheapest · 23-28 days · $2,550".
|
|
678
|
-
*/
|
|
679
|
-
clickByText(keyword: string, opts?: {
|
|
680
|
-
timeout?: number;
|
|
681
|
-
}): Promise<boolean>;
|
|
682
|
-
/**
|
|
683
|
-
* Scroll the first occurrence of `text` into view. Returns false if no
|
|
684
|
-
* match was found within the timeout.
|
|
685
|
-
*/
|
|
686
|
-
findText(text: string, opts?: {
|
|
687
|
-
timeout?: number;
|
|
688
|
-
}): Promise<boolean>;
|
|
689
|
-
/**
|
|
690
|
-
* Read the options of a native `<select>` at the given DOM-snapshot
|
|
691
|
-
* index. Returns `[]` if the element is not a `<select>`.
|
|
692
|
-
*/
|
|
693
|
-
dropdownOptions(index: number): Promise<{
|
|
694
|
-
value: string;
|
|
695
|
-
label: string;
|
|
696
|
-
selected: boolean;
|
|
697
|
-
}[]>;
|
|
698
|
-
/**
|
|
699
|
-
* Select an option in a native `<select>` by its visible text or value.
|
|
700
|
-
* Returns false if the element isn't a `<select>` or no option matched.
|
|
701
|
-
*/
|
|
702
|
-
selectDropdown(index: number, text: string): Promise<boolean>;
|
|
703
|
-
/**
|
|
704
|
-
* Run arbitrary JS in the page context. The caller is responsible for
|
|
705
|
-
* gating this behind a config flag — the BrowserAgent only routes the
|
|
706
|
-
* `evaluate` action here when `allowEvaluate: true`.
|
|
707
|
-
*
|
|
708
|
-
* The code is wrapped in `(async () => { ... })()` and the return value
|
|
709
|
-
* is coerced to a string for the model.
|
|
710
|
-
*/
|
|
711
|
-
evaluate(code: string): Promise<string>;
|
|
712
|
-
/**
|
|
713
|
-
* Returns a clean text representation of the visible page body, with
|
|
714
|
-
* optional link extraction. Used by the BrowserAgent's `extract` action
|
|
715
|
-
* — the text is passed to a (usually cheap) LLM with the user's query.
|
|
716
|
-
*/
|
|
717
|
-
pageText(opts?: {
|
|
718
|
-
extractLinks?: boolean;
|
|
719
|
-
maxChars?: number;
|
|
720
|
-
}): Promise<string>;
|
|
721
|
-
/**
|
|
722
|
-
* Snapshot the interactive elements visible in the viewport, tag each
|
|
723
|
-
* with a `data-bua-idx="<n>"` attribute (used by indexed actions), and
|
|
724
|
-
* return:
|
|
725
|
-
* - `text`: a human-readable string fed to the model
|
|
726
|
-
* - `elements`: the structured list with stable indices
|
|
727
|
-
* - `scroll`: spatial context (pages above/below, hidden interactive count)
|
|
728
|
-
*
|
|
729
|
-
* Five properties matter for accuracy:
|
|
730
|
-
* - **Hit-tested**: each listed coordinate / index actually reaches the
|
|
731
|
-
* labeled element (overlays / occlusion skip the entry).
|
|
732
|
-
* - **Visibility filtered (with parent chain)**: an element is dropped
|
|
733
|
-
* if itself OR any ancestor is `display:none`, `visibility:hidden`,
|
|
734
|
-
* `pointer-events:none`, or near-zero opacity.
|
|
735
|
-
* - **Shadow DOM piercing**: traverses open shadow roots so custom
|
|
736
|
-
* elements / web components are visible to the agent.
|
|
737
|
-
* - **Same-origin iframes**: walks into each accessible iframe and
|
|
738
|
-
* includes its interactive elements (offset by the iframe's screen
|
|
739
|
-
* position so the coordinates the model sees are still viewport-
|
|
740
|
-
* relative).
|
|
741
|
-
* - **`cursor: pointer` fallback pass**: catches custom React widgets
|
|
742
|
-
* that have no semantic role/href/onclick but are clickable.
|
|
743
|
-
*/
|
|
744
|
-
extractDOM(opts?: {
|
|
745
|
-
maxElements?: number;
|
|
746
|
-
}): Promise<DomSnapshot>;
|
|
747
|
-
/**
|
|
748
|
-
* Install the `__buaExtract` global on the main page. Idempotent —
|
|
749
|
-
* subsequent calls are no-ops.
|
|
750
|
-
*/
|
|
751
|
-
private installExtractorScript;
|
|
752
|
-
/**
|
|
753
|
-
* The extractor source. Lives in its own method so we can also inject
|
|
754
|
-
* it into iframes that haven't yet had it loaded.
|
|
755
|
-
*
|
|
756
|
-
* This function intentionally runs entirely in the page context. It:
|
|
757
|
-
* - traverses the regular DOM + open shadow roots (deep)
|
|
758
|
-
* - applies a parent-chain visibility filter
|
|
759
|
-
* - applies a `cursor:pointer` second pass for custom widgets
|
|
760
|
-
* - hit-tests each candidate at its center to avoid overlay collisions
|
|
761
|
-
* - returns scroll context (pages above/below, hidden counts)
|
|
762
|
-
* - tags survivors with `data-bua-idx` for indexed actions
|
|
763
|
-
*/
|
|
764
|
-
private extractorScriptSource;
|
|
765
|
-
getPageInfo(): Promise<PageInfo>;
|
|
766
|
-
waitForStable(minWait?: number): Promise<void>;
|
|
767
|
-
newTab(url?: string): Promise<string>;
|
|
768
|
-
switchTab(tabId: string): Promise<void>;
|
|
769
|
-
closeTab(tabId: string): Promise<void>;
|
|
770
|
-
listTabs(): {
|
|
771
|
-
id: string;
|
|
772
|
-
url: string;
|
|
773
|
-
active: boolean;
|
|
774
|
-
}[];
|
|
775
|
-
get currentTabId(): string;
|
|
776
|
-
getVideoPath(tabId?: string): Promise<string | null>;
|
|
777
|
-
get videoDir(): string | undefined;
|
|
778
|
-
close(): Promise<void>;
|
|
779
|
-
/** Add small random offset to coordinates to avoid pixel-perfect bot patterns. */
|
|
780
|
-
private jitter;
|
|
781
|
-
/**
|
|
782
|
-
* Safety net: clamp coordinates returned by the vision model to the actual
|
|
783
|
-
* viewport. If a model occasionally returns image-space coordinates from a
|
|
784
|
-
* 2x screenshot (despite our `scale: "css"` fix), this prevents Playwright
|
|
785
|
-
* from clicking at e.g. (2200, 1300) and either erroring or landing on a
|
|
786
|
-
* random off-screen element.
|
|
787
|
-
*/
|
|
788
|
-
private clampToViewport;
|
|
789
|
-
/**
|
|
790
|
-
* Simulate human mouse movement using smoothstep interpolation.
|
|
791
|
-
*/
|
|
792
|
-
private humanMouseMove;
|
|
793
|
-
/** Small random pause after an interaction. */
|
|
794
|
-
private humanPause;
|
|
795
|
-
private randInt;
|
|
796
|
-
private ensurePage;
|
|
797
|
-
private ensureContext;
|
|
798
|
-
private sleep;
|
|
799
|
-
}
|
|
800
|
-
|
|
801
|
-
export { type AgentOutput, type BrowserAction, BrowserAgent, type BrowserAgentConfig, BrowserProvider, type BrowserRunOpts, type BrowserRunOutput, type BrowserStep, CredentialVault, type DomElement, type DomScrollContext, type DomSnapshot, type HumanizeConfig, type PageInfo, type StealthConfig };
|
|
1
|
+
export { BrowserAgent } from "./browser-agent.js";
|
|
2
|
+
export { BrowserProvider } from "./browser-provider.js";
|
|
3
|
+
export { CredentialVault } from "./credential-vault.js";
|
|
4
|
+
export type { AgentOutput, BrowserAction, BrowserAgentConfig, BrowserRunOpts, BrowserRunOutput, BrowserStep, DomElement, DomScrollContext, DomSnapshot, HumanizeConfig, PageInfo, StealthConfig, } from "./types.js";
|
|
5
|
+
//# sourceMappingURL=index.d.ts.map
|