@agentium/browser 2.1.1 → 2.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -149,6 +149,32 @@ interface DomElement {
149
149
  isSelect: boolean;
150
150
  /** Whether it's an `<input type="file">`. */
151
151
  isFile: boolean;
152
+ /** Origin frame (`"main"` or the iframe `src`/path). */
153
+ frame?: string;
154
+ }
155
+ /**
156
+ * Scroll-context metadata returned alongside `DomElement[]`. Gives the
157
+ * model spatial awareness so it can decide when to scroll vs when more
158
+ * content is below / above the fold.
159
+ */
160
+ interface DomScrollContext {
161
+ /** Approximate viewports of scrollable content above the current view. */
162
+ pagesAbove: number;
163
+ /** Approximate viewports of scrollable content below the current view. */
164
+ pagesBelow: number;
165
+ /** Total interactive elements found (visible + hidden combined). */
166
+ totalInteractive: number;
167
+ /** Count of interactive elements that exist on the page but aren't in the viewport. */
168
+ hiddenInteractive: number;
169
+ }
170
+ /** Combined return value of `BrowserProvider.extractDOM()`. */
171
+ interface DomSnapshot {
172
+ /** Human-readable string fed to the model. */
173
+ text: string;
174
+ /** Structured list (stable indices) for runtime resolution. */
175
+ elements: DomElement[];
176
+ /** Spatial/scroll context. */
177
+ scroll: DomScrollContext;
152
178
  }
153
179
  interface BrowserAgentConfig {
154
180
  name: string;
@@ -159,6 +185,27 @@ interface BrowserAgentConfig {
159
185
  * action and other text-only sub-tasks. Falls back to `model`.
160
186
  */
161
187
  pageExtractionLLM?: ModelProvider;
188
+ /**
189
+ * Fallback model used automatically when the primary `model` returns a
190
+ * rate-limit / auth / 5xx error or fails to produce valid JSON several
191
+ * times in a row. Failure-budget aware. Leave unset to disable.
192
+ */
193
+ fallbackModel?: ModelProvider;
194
+ /**
195
+ * Ask the model to emit a structured thinking/evaluation/memory/next_goal
196
+ * envelope around its action(s). Significantly improves accuracy and
197
+ * self-correction on multi-step tasks. Default: `true`. Set `false` for
198
+ * a `flash_mode` that just returns the raw action(s) — useful for very
199
+ * fast / cheap models that are bad at long outputs.
200
+ */
201
+ useThinking?: boolean;
202
+ /**
203
+ * Maximum number of recent step turns kept verbatim in the conversation
204
+ * history sent to the model. Older turns are compacted into a single
205
+ * summary line. Default: 6. Set 0 to disable conversation history
206
+ * entirely (each step rebuilt from scratch — v2.0 behaviour).
207
+ */
208
+ historyWindow?: number;
162
209
  /** Extra instructions appended to the default system prompt. */
163
210
  instructions?: string;
164
211
  /**
@@ -347,6 +394,26 @@ interface BrowserStep {
347
394
  output?: string;
348
395
  /** Whether this step succeeded (vs threw / failed locator). Default: true. */
349
396
  ok?: boolean;
397
+ /** Model's chain-of-thought reasoning (if `useThinking` was on). */
398
+ thinking?: string;
399
+ /** Model's evaluation of whether the previous action met its goal. */
400
+ evaluationPreviousGoal?: string;
401
+ /** Model's running memory of important state. */
402
+ memory?: string;
403
+ /** Model's stated next goal for this step. */
404
+ nextGoal?: string;
405
+ }
406
+ /**
407
+ * Structured envelope the model returns when `useThinking: true`. Inspired
408
+ * by browser-use's `AgentOutput`. Every field is optional from a runtime
409
+ * standpoint — only `action` is required for execution.
410
+ */
411
+ interface AgentOutput {
412
+ thinking?: string;
413
+ evaluationPreviousGoal?: string;
414
+ memory?: string;
415
+ nextGoal?: string;
416
+ action: BrowserAction | BrowserAction[];
350
417
  }
351
418
  interface StealthConfig {
352
419
  /**
@@ -406,6 +473,9 @@ declare class BrowserAgent {
406
473
  readonly eventBus: EventBus;
407
474
  private model;
408
475
  private pageExtractionLLM;
476
+ private fallbackModel;
477
+ private useThinking;
478
+ private historyWindow;
409
479
  private instructions?;
410
480
  private extendSystemMessage?;
411
481
  private overrideSystemMessage?;
@@ -449,6 +519,52 @@ declare class BrowserAgent {
449
519
  private shouldCaptureVision;
450
520
  private detectUrlInTask;
451
521
  private assertDomainAllowed;
522
+ /**
523
+ * Build the message array sent to the model: system prompt + a compact
524
+ * summary of older turns (if any) + the most recent `historyWindow`
525
+ * turns verbatim + the current step's user message.
526
+ *
527
+ * Inspired by browser-use's history compaction. Keeps tokens bounded
528
+ * while giving the model meaningful context about what it already
529
+ * tried.
530
+ */
531
+ private buildMessages;
532
+ /**
533
+ * Call the primary model, retrying once with `fallbackModel` on
534
+ * transient errors (5xx, 429, network). Returns the response or
535
+ * `null` if both models failed.
536
+ */
537
+ private callModelWithFallback;
538
+ private isTransientError;
539
+ /**
540
+ * Parse the model's raw response into an `AgentOutput`. Tolerant to
541
+ * three shapes:
542
+ * - Full envelope: { thinking, evaluation_previous_goal, action, ... }
543
+ * - Raw action object (legacy / `useThinking: false`)
544
+ * - Raw action array
545
+ *
546
+ * Also strips ```json fences the model occasionally adds.
547
+ */
548
+ private parseEnvelope;
549
+ /**
550
+ * Quick post-navigation health check. If the page came back blank
551
+ * (no body text and no interactive elements), reload once and wait
552
+ * for stable. This catches the "FreightOS half-loaded font test" class
553
+ * of failure before the LLM ever sees it.
554
+ */
555
+ private navigationHealthCheck;
556
+ /**
557
+ * Force-finalize the run with whatever partial data the agent has.
558
+ * Called when:
559
+ * - `maxSteps` is exhausted without an explicit `done`,
560
+ * - `maxFailures` is exceeded,
561
+ * - the model can't produce parseable JSON enough times to make
562
+ * forward progress.
563
+ * The result is composed from `extractedContent` + the last few
564
+ * action summaries so the caller gets something useful instead of
565
+ * just a one-line error.
566
+ */
567
+ private forceDone;
452
568
  private finalize;
453
569
  /**
454
570
  * Execute a single action. Returns `{ output?, didNavigate? }` for the
@@ -605,23 +721,47 @@ declare class BrowserProvider {
605
721
  /**
606
722
  * Snapshot the interactive elements visible in the viewport, tag each
607
723
  * with a `data-bua-idx="<n>"` attribute (used by indexed actions), and
608
- * return BOTH a human-readable string (for the model) and the
609
- * structured list (for the runtime).
724
+ * return:
725
+ * - `text`: a human-readable string fed to the model
726
+ * - `elements`: the structured list with stable indices
727
+ * - `scroll`: spatial context (pages above/below, hidden interactive count)
610
728
  *
611
- * Three properties matter for accuracy:
612
- * - Hit-tested: each listed coordinate / index actually reaches the
729
+ * Five properties matter for accuracy:
730
+ * - **Hit-tested**: each listed coordinate / index actually reaches the
613
731
  * labeled element (overlays / occlusion skip the entry).
614
- * - Visibility-filtered: invisible, zero-size, `pointer-events: none`,
615
- * and near-zero-opacity elements are excluded.
616
- * - `cursor: pointer` fallback pass: a second scan catches custom React
617
- * widgets that have no semantic role/href/onclick but are clickable.
732
+ * - **Visibility filtered (with parent chain)**: an element is dropped
733
+ * if itself OR any ancestor is `display:none`, `visibility:hidden`,
734
+ * `pointer-events:none`, or near-zero opacity.
735
+ * - **Shadow DOM piercing**: traverses open shadow roots so custom
736
+ * elements / web components are visible to the agent.
737
+ * - **Same-origin iframes**: walks into each accessible iframe and
738
+ * includes its interactive elements (offset by the iframe's screen
739
+ * position so the coordinates the model sees are still viewport-
740
+ * relative).
741
+ * - **`cursor: pointer` fallback pass**: catches custom React widgets
742
+ * that have no semantic role/href/onclick but are clickable.
618
743
  */
619
744
  extractDOM(opts?: {
620
745
  maxElements?: number;
621
- }): Promise<{
622
- text: string;
623
- elements: DomElement[];
624
- }>;
746
+ }): Promise<DomSnapshot>;
747
+ /**
748
+ * Install the `__buaExtract` global on the main page. Idempotent —
749
+ * subsequent calls are no-ops.
750
+ */
751
+ private installExtractorScript;
752
+ /**
753
+ * The extractor source. Lives in its own method so we can also inject
754
+ * it into iframes that haven't yet had it loaded.
755
+ *
756
+ * This function intentionally runs entirely in the page context. It:
757
+ * - traverses the regular DOM + open shadow roots (deep)
758
+ * - applies a parent-chain visibility filter
759
+ * - applies a `cursor:pointer` second pass for custom widgets
760
+ * - hit-tests each candidate at its center to avoid overlay collisions
761
+ * - returns scroll context (pages above/below, hidden counts)
762
+ * - tags survivors with `data-bua-idx` for indexed actions
763
+ */
764
+ private extractorScriptSource;
625
765
  getPageInfo(): Promise<PageInfo>;
626
766
  waitForStable(minWait?: number): Promise<void>;
627
767
  newTab(url?: string): Promise<string>;
@@ -658,4 +798,4 @@ declare class BrowserProvider {
658
798
  private sleep;
659
799
  }
660
800
 
661
- export { type BrowserAction, BrowserAgent, type BrowserAgentConfig, BrowserProvider, type BrowserRunOpts, type BrowserRunOutput, type BrowserStep, CredentialVault, type DomElement, type HumanizeConfig, type PageInfo, type StealthConfig };
801
+ export { type AgentOutput, type BrowserAction, BrowserAgent, type BrowserAgentConfig, BrowserProvider, type BrowserRunOpts, type BrowserRunOutput, type BrowserStep, CredentialVault, type DomElement, type DomScrollContext, type DomSnapshot, type HumanizeConfig, type PageInfo, type StealthConfig };
package/dist/index.d.ts CHANGED
@@ -149,6 +149,32 @@ interface DomElement {
149
149
  isSelect: boolean;
150
150
  /** Whether it's an `<input type="file">`. */
151
151
  isFile: boolean;
152
+ /** Origin frame (`"main"` or the iframe `src`/path). */
153
+ frame?: string;
154
+ }
155
+ /**
156
+ * Scroll-context metadata returned alongside `DomElement[]`. Gives the
157
+ * model spatial awareness so it can decide when to scroll vs when more
158
+ * content is below / above the fold.
159
+ */
160
+ interface DomScrollContext {
161
+ /** Approximate viewports of scrollable content above the current view. */
162
+ pagesAbove: number;
163
+ /** Approximate viewports of scrollable content below the current view. */
164
+ pagesBelow: number;
165
+ /** Total interactive elements found (visible + hidden combined). */
166
+ totalInteractive: number;
167
+ /** Count of interactive elements that exist on the page but aren't in the viewport. */
168
+ hiddenInteractive: number;
169
+ }
170
+ /** Combined return value of `BrowserProvider.extractDOM()`. */
171
+ interface DomSnapshot {
172
+ /** Human-readable string fed to the model. */
173
+ text: string;
174
+ /** Structured list (stable indices) for runtime resolution. */
175
+ elements: DomElement[];
176
+ /** Spatial/scroll context. */
177
+ scroll: DomScrollContext;
152
178
  }
153
179
  interface BrowserAgentConfig {
154
180
  name: string;
@@ -159,6 +185,27 @@ interface BrowserAgentConfig {
159
185
  * action and other text-only sub-tasks. Falls back to `model`.
160
186
  */
161
187
  pageExtractionLLM?: ModelProvider;
188
+ /**
189
+ * Fallback model used automatically when the primary `model` returns a
190
+ * rate-limit / auth / 5xx error or fails to produce valid JSON several
191
+ * times in a row. Failure-budget aware. Leave unset to disable.
192
+ */
193
+ fallbackModel?: ModelProvider;
194
+ /**
195
+ * Ask the model to emit a structured thinking/evaluation/memory/next_goal
196
+ * envelope around its action(s). Significantly improves accuracy and
197
+ * self-correction on multi-step tasks. Default: `true`. Set `false` for
198
+ * a `flash_mode` that just returns the raw action(s) — useful for very
199
+ * fast / cheap models that are bad at long outputs.
200
+ */
201
+ useThinking?: boolean;
202
+ /**
203
+ * Maximum number of recent step turns kept verbatim in the conversation
204
+ * history sent to the model. Older turns are compacted into a single
205
+ * summary line. Default: 6. Set 0 to disable conversation history
206
+ * entirely (each step rebuilt from scratch — v2.0 behaviour).
207
+ */
208
+ historyWindow?: number;
162
209
  /** Extra instructions appended to the default system prompt. */
163
210
  instructions?: string;
164
211
  /**
@@ -347,6 +394,26 @@ interface BrowserStep {
347
394
  output?: string;
348
395
  /** Whether this step succeeded (vs threw / failed locator). Default: true. */
349
396
  ok?: boolean;
397
+ /** Model's chain-of-thought reasoning (if `useThinking` was on). */
398
+ thinking?: string;
399
+ /** Model's evaluation of whether the previous action met its goal. */
400
+ evaluationPreviousGoal?: string;
401
+ /** Model's running memory of important state. */
402
+ memory?: string;
403
+ /** Model's stated next goal for this step. */
404
+ nextGoal?: string;
405
+ }
406
+ /**
407
+ * Structured envelope the model returns when `useThinking: true`. Inspired
408
+ * by browser-use's `AgentOutput`. Every field is optional from a runtime
409
+ * standpoint — only `action` is required for execution.
410
+ */
411
+ interface AgentOutput {
412
+ thinking?: string;
413
+ evaluationPreviousGoal?: string;
414
+ memory?: string;
415
+ nextGoal?: string;
416
+ action: BrowserAction | BrowserAction[];
350
417
  }
351
418
  interface StealthConfig {
352
419
  /**
@@ -406,6 +473,9 @@ declare class BrowserAgent {
406
473
  readonly eventBus: EventBus;
407
474
  private model;
408
475
  private pageExtractionLLM;
476
+ private fallbackModel;
477
+ private useThinking;
478
+ private historyWindow;
409
479
  private instructions?;
410
480
  private extendSystemMessage?;
411
481
  private overrideSystemMessage?;
@@ -449,6 +519,52 @@ declare class BrowserAgent {
449
519
  private shouldCaptureVision;
450
520
  private detectUrlInTask;
451
521
  private assertDomainAllowed;
522
+ /**
523
+ * Build the message array sent to the model: system prompt + a compact
524
+ * summary of older turns (if any) + the most recent `historyWindow`
525
+ * turns verbatim + the current step's user message.
526
+ *
527
+ * Inspired by browser-use's history compaction. Keeps tokens bounded
528
+ * while giving the model meaningful context about what it already
529
+ * tried.
530
+ */
531
+ private buildMessages;
532
+ /**
533
+ * Call the primary model, retrying once with `fallbackModel` on
534
+ * transient errors (5xx, 429, network). Returns the response or
535
+ * `null` if both models failed.
536
+ */
537
+ private callModelWithFallback;
538
+ private isTransientError;
539
+ /**
540
+ * Parse the model's raw response into an `AgentOutput`. Tolerant to
541
+ * three shapes:
542
+ * - Full envelope: { thinking, evaluation_previous_goal, action, ... }
543
+ * - Raw action object (legacy / `useThinking: false`)
544
+ * - Raw action array
545
+ *
546
+ * Also strips ```json fences the model occasionally adds.
547
+ */
548
+ private parseEnvelope;
549
+ /**
550
+ * Quick post-navigation health check. If the page came back blank
551
+ * (no body text and no interactive elements), reload once and wait
552
+ * for stable. This catches the "FreightOS half-loaded font test" class
553
+ * of failure before the LLM ever sees it.
554
+ */
555
+ private navigationHealthCheck;
556
+ /**
557
+ * Force-finalize the run with whatever partial data the agent has.
558
+ * Called when:
559
+ * - `maxSteps` is exhausted without an explicit `done`,
560
+ * - `maxFailures` is exceeded,
561
+ * - the model can't produce parseable JSON enough times to make
562
+ * forward progress.
563
+ * The result is composed from `extractedContent` + the last few
564
+ * action summaries so the caller gets something useful instead of
565
+ * just a one-line error.
566
+ */
567
+ private forceDone;
452
568
  private finalize;
453
569
  /**
454
570
  * Execute a single action. Returns `{ output?, didNavigate? }` for the
@@ -605,23 +721,47 @@ declare class BrowserProvider {
605
721
  /**
606
722
  * Snapshot the interactive elements visible in the viewport, tag each
607
723
  * with a `data-bua-idx="<n>"` attribute (used by indexed actions), and
608
- * return BOTH a human-readable string (for the model) and the
609
- * structured list (for the runtime).
724
+ * return:
725
+ * - `text`: a human-readable string fed to the model
726
+ * - `elements`: the structured list with stable indices
727
+ * - `scroll`: spatial context (pages above/below, hidden interactive count)
610
728
  *
611
- * Three properties matter for accuracy:
612
- * - Hit-tested: each listed coordinate / index actually reaches the
729
+ * Five properties matter for accuracy:
730
+ * - **Hit-tested**: each listed coordinate / index actually reaches the
613
731
  * labeled element (overlays / occlusion skip the entry).
614
- * - Visibility-filtered: invisible, zero-size, `pointer-events: none`,
615
- * and near-zero-opacity elements are excluded.
616
- * - `cursor: pointer` fallback pass: a second scan catches custom React
617
- * widgets that have no semantic role/href/onclick but are clickable.
732
+ * - **Visibility filtered (with parent chain)**: an element is dropped
733
+ * if itself OR any ancestor is `display:none`, `visibility:hidden`,
734
+ * `pointer-events:none`, or near-zero opacity.
735
+ * - **Shadow DOM piercing**: traverses open shadow roots so custom
736
+ * elements / web components are visible to the agent.
737
+ * - **Same-origin iframes**: walks into each accessible iframe and
738
+ * includes its interactive elements (offset by the iframe's screen
739
+ * position so the coordinates the model sees are still viewport-
740
+ * relative).
741
+ * - **`cursor: pointer` fallback pass**: catches custom React widgets
742
+ * that have no semantic role/href/onclick but are clickable.
618
743
  */
619
744
  extractDOM(opts?: {
620
745
  maxElements?: number;
621
- }): Promise<{
622
- text: string;
623
- elements: DomElement[];
624
- }>;
746
+ }): Promise<DomSnapshot>;
747
+ /**
748
+ * Install the `__buaExtract` global on the main page. Idempotent —
749
+ * subsequent calls are no-ops.
750
+ */
751
+ private installExtractorScript;
752
+ /**
753
+ * The extractor source. Lives in its own method so we can also inject
754
+ * it into iframes that haven't yet had it loaded.
755
+ *
756
+ * This function intentionally runs entirely in the page context. It:
757
+ * - traverses the regular DOM + open shadow roots (deep)
758
+ * - applies a parent-chain visibility filter
759
+ * - applies a `cursor:pointer` second pass for custom widgets
760
+ * - hit-tests each candidate at its center to avoid overlay collisions
761
+ * - returns scroll context (pages above/below, hidden counts)
762
+ * - tags survivors with `data-bua-idx` for indexed actions
763
+ */
764
+ private extractorScriptSource;
625
765
  getPageInfo(): Promise<PageInfo>;
626
766
  waitForStable(minWait?: number): Promise<void>;
627
767
  newTab(url?: string): Promise<string>;
@@ -658,4 +798,4 @@ declare class BrowserProvider {
658
798
  private sleep;
659
799
  }
660
800
 
661
- export { type BrowserAction, BrowserAgent, type BrowserAgentConfig, BrowserProvider, type BrowserRunOpts, type BrowserRunOutput, type BrowserStep, CredentialVault, type DomElement, type HumanizeConfig, type PageInfo, type StealthConfig };
801
+ export { type AgentOutput, type BrowserAction, BrowserAgent, type BrowserAgentConfig, BrowserProvider, type BrowserRunOpts, type BrowserRunOutput, type BrowserStep, CredentialVault, type DomElement, type DomScrollContext, type DomSnapshot, type HumanizeConfig, type PageInfo, type StealthConfig };