@agentium/browser 2.0.7 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -1,5 +1,5 @@
1
1
  import * as _agentium_core from '@agentium/core';
2
- import { ModelProvider, UnifiedMemoryConfig, CostTracker, LogLevel, EventBus, MemoryManager, ToolDef } from '@agentium/core';
2
+ import { ModelProvider, UnifiedMemoryConfig, ToolDef, CostTracker, LogLevel, EventBus, MemoryManager } from '@agentium/core';
3
3
 
4
4
  /**
5
5
  * Secure credential store for BrowserAgent.
@@ -38,20 +38,45 @@ declare class CredentialVault {
38
38
  mask(text: string): string;
39
39
  }
40
40
 
41
+ /**
42
+ * The complete set of actions a `BrowserAgent` can take during a run.
43
+ *
44
+ * Click / type / scroll accept EITHER an `index` (preferred — resolved
45
+ * against the DOM tree we showed the model) OR `x`/`y` coordinates as a
46
+ * fallback. Indexed actions are far more reliable on dynamic pages because
47
+ * they survive layout shifts, devicePixelRatio mismatches, and "the
48
+ * sibling 4 pixels away" failure modes.
49
+ */
41
50
  type BrowserAction = {
42
51
  action: "click";
43
- x: number;
44
- y: number;
45
- description: string;
52
+ /** Element index from the DOM snapshot. Preferred. */
53
+ index?: number;
54
+ /** Fallback x coordinate (CSS pixels). */
55
+ x?: number;
56
+ /** Fallback y coordinate (CSS pixels). */
57
+ y?: number;
58
+ /** Free-form description; if it contains a quoted label, used as a text-locator fallback. */
59
+ description?: string;
46
60
  } | {
47
61
  action: "type";
62
+ /** Element index from the DOM snapshot. Preferred. */
63
+ index?: number;
48
64
  text: string;
65
+ /** Fallback x coordinate (CSS pixels). */
49
66
  x?: number;
67
+ /** Fallback y coordinate (CSS pixels). */
50
68
  y?: number;
69
+ /** If true (default), clear the field first. */
70
+ clear?: boolean;
71
+ /** If true (default), press Enter after typing when text ends with `\n`. */
72
+ submit?: boolean;
51
73
  } | {
52
74
  action: "scroll";
53
75
  direction: "up" | "down";
76
+ /** Pixels to scroll. Default: 400. */
54
77
  amount?: number;
78
+ /** If provided, scroll the element with this index into view instead. */
79
+ index?: number;
55
80
  } | {
56
81
  action: "navigate";
57
82
  url: string;
@@ -62,6 +87,37 @@ type BrowserAction = {
62
87
  ms: number;
63
88
  } | {
64
89
  action: "screenshot";
90
+ } | {
91
+ action: "send_keys";
92
+ keys: string;
93
+ } | {
94
+ action: "find_text";
95
+ text: string;
96
+ } | {
97
+ action: "evaluate";
98
+ code: string;
99
+ } | {
100
+ action: "dropdown_options";
101
+ index: number;
102
+ } | {
103
+ action: "select_dropdown";
104
+ index: number;
105
+ text: string;
106
+ } | {
107
+ action: "upload_file";
108
+ index: number;
109
+ path: string;
110
+ } | {
111
+ action: "extract";
112
+ /** Natural-language description of what to extract. */
113
+ query: string;
114
+ /** Include link hrefs in the extracted content. Default: false. */
115
+ extractLinks?: boolean;
116
+ } | {
117
+ action: "tool";
118
+ /** Name of a custom tool registered on the BrowserAgent. */
119
+ name: string;
120
+ args?: Record<string, unknown>;
65
121
  } | {
66
122
  action: "done";
67
123
  result: string;
@@ -69,14 +125,87 @@ type BrowserAction = {
69
125
  action: "fail";
70
126
  reason: string;
71
127
  };
128
+ /**
129
+ * One entry in the DOM snapshot the model sees. Returned by
130
+ * `BrowserProvider.extractDOM()` alongside the human-readable string form.
131
+ */
132
+ interface DomElement {
133
+ /** Stable 1-based index for this step. Used by indexed actions. */
134
+ index: number;
135
+ /** Center coordinate in CSS pixels (fallback for the model if needed). */
136
+ cx: number;
137
+ cy: number;
138
+ /** ARIA role or tag name. */
139
+ role: string;
140
+ /** Input type, if applicable. */
141
+ type?: string;
142
+ /** Visible label (text / aria-label / placeholder / href / …). */
143
+ label: string;
144
+ /** Tag name. */
145
+ tag: string;
146
+ /** Whether it's an `<input>` / `<textarea>` / `[contenteditable]`. */
147
+ isInput: boolean;
148
+ /** Whether it's a `<select>`. */
149
+ isSelect: boolean;
150
+ /** Whether it's an `<input type="file">`. */
151
+ isFile: boolean;
152
+ }
72
153
  interface BrowserAgentConfig {
73
154
  name: string;
74
155
  /** Vision-capable model (GPT-4o, Gemini, etc.) */
75
156
  model: ModelProvider;
76
- /** Extra instructions appended to the system prompt */
157
+ /**
158
+ * Optional secondary (usually cheaper) model used for the `extract`
159
+ * action and other text-only sub-tasks. Falls back to `model`.
160
+ */
161
+ pageExtractionLLM?: ModelProvider;
162
+ /** Extra instructions appended to the default system prompt. */
77
163
  instructions?: string;
164
+ /**
165
+ * Append additional instructions to the default system prompt.
166
+ * Alias for `instructions` (browser-use parity). Both are concatenated.
167
+ */
168
+ extendSystemMessage?: string;
169
+ /**
170
+ * Completely replace the default system prompt with this string.
171
+ * The credentials / DOM / coordinate sections are still appended at the
172
+ * end automatically. Most users should use `instructions` instead.
173
+ */
174
+ overrideSystemMessage?: string;
78
175
  /** Max vision loop iterations. Default: 30 */
79
176
  maxSteps?: number;
177
+ /**
178
+ * Maximum number of consecutive step failures (action threw, invalid
179
+ * JSON from model, locator timeout) before the agent gives up. Default: 3.
180
+ */
181
+ maxFailures?: number;
182
+ /**
183
+ * Maximum number of actions the model can return in a single step.
184
+ * If the model returns an array, we execute them in order until the page
185
+ * navigates or the DOM changes substantially. Default: 3.
186
+ * Set to 1 to force one-action-per-step (the v2.0.x behaviour).
187
+ */
188
+ maxActionsPerStep?: number;
189
+ /**
190
+ * Actions to run before the LLM loop starts. Useful for boilerplate
191
+ * (cookie-banner click, login flow, scrolling) you already know the
192
+ * answer to — saves vision tokens.
193
+ */
194
+ initialActions?: BrowserAction[];
195
+ /**
196
+ * Vision mode. Default: `"auto"`.
197
+ * - `true`: always send a screenshot with every step (v2.0.x behaviour).
198
+ * - `false`: never send screenshots; DOM-only operation.
199
+ * - `"auto"`: send a screenshot on the first step and whenever the model
200
+ * used the `screenshot` action in the previous step. Saves vision
201
+ * tokens dramatically on DOM-only-suitable workloads.
202
+ */
203
+ useVision?: boolean | "auto";
204
+ /**
205
+ * If true (default), detect a URL in the task string and navigate to it
206
+ * before the first LLM call — saves one round trip on simple tasks.
207
+ */
208
+ directlyOpenUrl?: boolean;
80
209
  /** Run browser without visible window. Default: true */
81
210
  headless?: boolean;
82
211
  /** Browser viewport size. Default: 1280x720 */
@@ -97,11 +226,36 @@ interface BrowserAgentConfig {
97
226
  * Default: true
98
227
  */
99
228
  useDOM?: boolean;
229
+ /**
230
+ * Allow the `evaluate` action to run arbitrary JavaScript inside the
231
+ * page. Default: false (security). Only enable if you trust the source
232
+ * of task strings — a malicious task could exfiltrate page contents.
233
+ */
234
+ allowEvaluate?: boolean;
235
+ /**
236
+ * Restrict navigation to specific domains. Wildcard patterns supported:
237
+ * `"example.com"`, `"*.example.com"`, `"http*://example.com"`.
238
+ * When set, any `navigate` action to a non-matching URL throws.
239
+ */
240
+ allowedDomains?: string[];
241
+ /**
242
+ * Block navigation to specific domains. Same pattern format as
243
+ * `allowedDomains`. Evaluated AFTER `allowedDomains`, so if both are
244
+ * set a URL must be in `allowedDomains` AND not in `prohibitedDomains`.
245
+ */
246
+ prohibitedDomains?: string[];
100
247
  /**
101
248
  * Path to a Playwright storageState JSON file.
102
249
  * Restores cookies, localStorage, and sessionStorage from a previous session.
103
250
  */
104
251
  storageState?: string;
252
+ /**
253
+ * Connect to an existing browser via Chrome DevTools Protocol instead
254
+ * of launching one. Format: `"http://localhost:9222"`. When set,
255
+ * `headless`, `stealth.args`, `recordVideo`, etc. are ignored — the
256
+ * existing browser's configuration is used.
257
+ */
258
+ cdpUrl?: string;
105
259
  /**
106
260
  * Enable video recording of the browser session.
107
261
  * Pass `true` for default dir (`./browser-videos`) or `{ dir: "/path" }`.
@@ -134,6 +288,14 @@ interface BrowserAgentConfig {
134
288
  memory?: UnifiedMemoryConfig;
135
289
  /** Skills — pre-packaged or learned tool bundles. */
136
290
  skills?: Array<_agentium_core.Skill | string>;
291
+ /**
292
+ * Custom tools that the BrowserAgent itself can invoke during a run.
293
+ * The agent emits `{ "action": "tool", "name": "<tool>", "args": {...} }`
294
+ * and we dispatch to the tool's `execute(args)`. Use this for 2FA codes,
295
+ * API calls, file I/O, calling out to other agents — anything the
296
+ * browser can't do alone.
297
+ */
298
+ tools?: ToolDef[];
137
299
  /** Cost tracker — track vision model token usage and enforce budgets across browser runs. */
138
300
  costTracker?: CostTracker;
139
301
  logLevel?: LogLevel;
@@ -150,6 +312,8 @@ interface BrowserRunOpts {
150
312
  userId?: string;
151
313
  /** Path to save storageState (cookies/auth) after the run completes */
152
314
  saveStorageState?: string;
315
+ /** Per-run override of `maxSteps`. */
316
+ maxSteps?: number;
153
317
  }
154
318
  interface BrowserRunOutput {
155
319
  /** Final text result produced by the agent */
@@ -166,17 +330,23 @@ interface BrowserRunOutput {
166
330
  durationMs: number;
167
331
  /** Video file path (if recordVideo was enabled) */
168
332
  videoPath?: string;
333
+ /** Extracted content from every `extract` action, in chronological order. */
334
+ extractedContent?: string[];
169
335
  }
170
336
  interface BrowserStep {
171
337
  index: number;
172
338
  action: BrowserAction;
173
- /** Screenshot taken before this action was executed */
339
+ /** Screenshot taken before this action was executed. May be empty if useVision=false. */
174
340
  screenshot: Buffer;
175
341
  pageUrl: string;
176
342
  pageTitle: string;
177
343
  timestamp: Date;
178
344
  /** Simplified DOM snapshot (if useDOM is enabled) */
179
345
  dom?: string;
346
+ /** Free-form result string produced by the action (e.g. extract output). */
347
+ output?: string;
348
+ /** Whether this step succeeded (vs threw / failed locator). Default: true. */
349
+ ok?: boolean;
180
350
  }
181
351
  interface StealthConfig {
182
352
  /**
@@ -198,6 +368,13 @@ interface StealthConfig {
198
368
  };
199
369
  /** Ignore HTTPS certificate errors. Default: false (secure). Only enable for local testing. */
200
370
  ignoreHTTPSErrors?: boolean;
371
+ /**
372
+ * `window.devicePixelRatio` to emulate. Default: `1`. Set to `2` to mimic
373
+ * a Retina display (sharper screenshots at 2× cost). Setting to 2 on a
374
+ * non-Retina host display can cause the headed window to look zoomed-out
375
+ * or stretched because the OS compositor downsamples a 2× surface.
376
+ */
377
+ deviceScaleFactor?: number;
201
378
  /** HTTP/SOCKS proxy. Format: "http://user:pass@host:port" */
202
379
  proxy?: {
203
380
  server: string;
@@ -228,19 +405,32 @@ declare class BrowserAgent {
228
405
  readonly name: string;
229
406
  readonly eventBus: EventBus;
230
407
  private model;
408
+ private pageExtractionLLM;
231
409
  private instructions?;
410
+ private extendSystemMessage?;
411
+ private overrideSystemMessage?;
232
412
  private maxSteps;
413
+ private maxFailures;
414
+ private maxActionsPerStep;
415
+ private initialActions;
416
+ private useVision;
417
+ private directlyOpenUrl;
233
418
  private headless;
234
419
  private viewport;
235
420
  private defaultStartUrl?;
236
421
  private waitAfterAction;
237
422
  private maxRepeats;
238
423
  private useDOM;
424
+ private allowEvaluate;
425
+ private allowedDomains?;
426
+ private prohibitedDomains?;
239
427
  private storageState?;
428
+ private cdpUrl?;
240
429
  private recordVideo?;
241
430
  private credentials?;
242
431
  private stealth?;
243
432
  private humanize?;
433
+ private tools;
244
434
  private costTracker;
245
435
  private memoryManager;
246
436
  private logger;
@@ -256,13 +446,32 @@ declare class BrowserAgent {
256
446
  name?: string;
257
447
  description?: string;
258
448
  }): ToolDef;
449
+ private shouldCaptureVision;
450
+ private detectUrlInTask;
451
+ private assertDomainAllowed;
259
452
  private finalize;
453
+ /**
454
+ * Execute a single action. Returns `{ output?, didNavigate? }` for the
455
+ * caller's state tracking. May throw — the loop handles failure-budget
456
+ * accounting in that case.
457
+ */
260
458
  private executeAction;
261
459
  private sleep;
460
+ /**
461
+ * Parse a quoted target keyword from a click action's `description`.
462
+ * Returns `undefined` for generic / ambiguous labels (login buttons,
463
+ * close, OK, etc.) where a substring text match could fire on the
464
+ * wrong element.
465
+ */
466
+ private extractClickKeyword;
262
467
  }
263
468
 
264
469
  /**
265
- * Playwright wrapper with stealth anti-detection and human-like behavior.
470
+ * Playwright wrapper with stealth anti-detection, human-like behavior,
471
+ * indexed DOM element resolution, and a rich action vocabulary.
472
+ *
473
+ * The `BrowserProvider` is intentionally LLM-agnostic — it exposes the
474
+ * primitives that `BrowserAgent` orchestrates via vision+DOM reasoning.
266
475
  */
267
476
  declare class BrowserProvider {
268
477
  private browser;
@@ -274,6 +483,13 @@ declare class BrowserProvider {
274
483
  private _viewport;
275
484
  private _videoDir?;
276
485
  private _humanize?;
486
+ /**
487
+ * Most recent DOM snapshot (one per `extractDOM` call). Indexed actions
488
+ * (`clickByIndex`, `inputByIndex`, …) resolve their `index` against this.
489
+ */
490
+ private _lastDom;
491
+ /** True if we connected over CDP (don't tear down the browser on close). */
492
+ private _attached;
277
493
  constructor();
278
494
  launch(opts?: {
279
495
  headless?: boolean;
@@ -287,6 +503,7 @@ declare class BrowserProvider {
287
503
  };
288
504
  stealth?: boolean | StealthConfig;
289
505
  humanize?: boolean | HumanizeConfig;
506
+ cdpUrl?: string;
290
507
  }): Promise<void>;
291
508
  saveStorageState(path: string): Promise<void>;
292
509
  navigate(url: string): Promise<void>;
@@ -297,14 +514,114 @@ declare class BrowserProvider {
297
514
  width: number;
298
515
  height: number;
299
516
  };
517
+ /** Most recent DOM snapshot. Each entry has a stable `index`. */
518
+ get lastDom(): DomElement[];
300
519
  click(x: number, y: number): Promise<void>;
301
520
  type(text: string): Promise<void>;
302
521
  clickAndType(x: number, y: number, text: string): Promise<void>;
303
522
  pressKey(key: string): Promise<void>;
523
+ /**
524
+ * Send arbitrary keyboard keys / shortcuts. Accepts a single
525
+ * Playwright key spec (`"Enter"`, `"Control+l"`, `"Shift+ArrowDown"`)
526
+ * or a space-separated sequence (`"Tab Tab Enter"`).
527
+ */
528
+ sendKeys(keys: string): Promise<void>;
304
529
  scroll(direction: "up" | "down", amount?: number): Promise<void>;
530
+ /**
531
+ * Build a Playwright locator for a DOM-snapshot index. Each `extractDOM`
532
+ * call tags surviving elements with `data-bua-idx="<n>"`; we resolve by
533
+ * that attribute. Returns null if the index is unknown.
534
+ */
535
+ private locatorForIndex;
536
+ /**
537
+ * Click an element by its DOM-snapshot index. The most reliable click
538
+ * path on dynamic pages — survives layout shifts and DPR oddities.
539
+ */
540
+ clickByIndex(index: number, opts?: {
541
+ timeout?: number;
542
+ }): Promise<boolean>;
543
+ /**
544
+ * Focus an indexed input, optionally clear it, and type. Returns false
545
+ * if the index couldn't be resolved or the input couldn't be focused.
546
+ */
547
+ inputByIndex(index: number, text: string, opts?: {
548
+ clear?: boolean;
549
+ submit?: boolean;
550
+ timeout?: number;
551
+ }): Promise<boolean>;
552
+ uploadFileByIndex(index: number, path: string): Promise<boolean>;
553
+ /** Scroll the indexed element into view (no click). */
554
+ scrollIntoViewByIndex(index: number): Promise<boolean>;
555
+ /**
556
+ * Deterministic, DOM-based click using Playwright's text locator.
557
+ *
558
+ * Returns `true` if a matching, visible, clickable element was found and
559
+ * clicked within `timeout` ms; `false` otherwise (so the caller can fall
560
+ * back to coordinate clicking). Substring-matches by default — e.g.
561
+ * `clickByText("Cheapest")` matches "Cheapest · 23-28 days · $2,550".
562
+ */
563
+ clickByText(keyword: string, opts?: {
564
+ timeout?: number;
565
+ }): Promise<boolean>;
566
+ /**
567
+ * Scroll the first occurrence of `text` into view. Returns false if no
568
+ * match was found within the timeout.
569
+ */
570
+ findText(text: string, opts?: {
571
+ timeout?: number;
572
+ }): Promise<boolean>;
573
+ /**
574
+ * Read the options of a native `<select>` at the given DOM-snapshot
575
+ * index. Returns `[]` if the element is not a `<select>`.
576
+ */
577
+ dropdownOptions(index: number): Promise<{
578
+ value: string;
579
+ label: string;
580
+ selected: boolean;
581
+ }[]>;
582
+ /**
583
+ * Select an option in a native `<select>` by its visible text or value.
584
+ * Returns false if the element isn't a `<select>` or no option matched.
585
+ */
586
+ selectDropdown(index: number, text: string): Promise<boolean>;
587
+ /**
588
+ * Run arbitrary JS in the page context. The caller is responsible for
589
+ * gating this behind a config flag — the BrowserAgent only routes the
590
+ * `evaluate` action here when `allowEvaluate: true`.
591
+ *
592
+ * The code is wrapped in `(async () => { ... })()` and the return value
593
+ * is coerced to a string for the model.
594
+ */
595
+ evaluate(code: string): Promise<string>;
596
+ /**
597
+ * Returns a clean text representation of the visible page body, with
598
+ * optional link extraction. Used by the BrowserAgent's `extract` action
599
+ * — the text is passed to a (usually cheap) LLM with the user's query.
600
+ */
601
+ pageText(opts?: {
602
+ extractLinks?: boolean;
603
+ maxChars?: number;
604
+ }): Promise<string>;
605
+ /**
606
+ * Snapshot the interactive elements visible in the viewport, tag each
607
+ * with a `data-bua-idx="<n>"` attribute (used by indexed actions), and
608
+ * return BOTH a human-readable string (for the model) and the
609
+ * structured list (for the runtime).
610
+ *
611
+ * Three properties matter for accuracy:
612
+ * - Hit-tested: each listed coordinate / index actually reaches the
613
+ * labeled element (overlays / occlusion skip the entry).
614
+ * - Visibility-filtered: invisible, zero-size, `pointer-events: none`,
615
+ * and near-zero-opacity elements are excluded.
616
+ * - `cursor: pointer` fallback pass: a second scan catches custom React
617
+ * widgets that have no semantic role/href/onclick but are clickable.
618
+ */
305
619
  extractDOM(opts?: {
306
620
  maxElements?: number;
307
- }): Promise<string>;
621
+ }): Promise<{
622
+ text: string;
623
+ elements: DomElement[];
624
+ }>;
308
625
  getPageInfo(): Promise<PageInfo>;
309
626
  waitForStable(minWait?: number): Promise<void>;
310
627
  newTab(url?: string): Promise<string>;
@@ -330,8 +647,7 @@ declare class BrowserProvider {
330
647
  */
331
648
  private clampToViewport;
332
649
  /**
333
- * Simulate human mouse movement using Bézier-like interpolation.
334
- * Moves from the current mouse position to the target in small steps.
650
+ * Simulate human mouse movement using smoothstep interpolation.
335
651
  */
336
652
  private humanMouseMove;
337
653
  /** Small random pause after an interaction. */
@@ -342,4 +658,4 @@ declare class BrowserProvider {
342
658
  private sleep;
343
659
  }
344
660
 
345
- export { type BrowserAction, BrowserAgent, type BrowserAgentConfig, BrowserProvider, type BrowserRunOpts, type BrowserRunOutput, type BrowserStep, CredentialVault, type HumanizeConfig, type PageInfo, type StealthConfig };
661
+ export { type BrowserAction, BrowserAgent, type BrowserAgentConfig, BrowserProvider, type BrowserRunOpts, type BrowserRunOutput, type BrowserStep, CredentialVault, type DomElement, type HumanizeConfig, type PageInfo, type StealthConfig };