@tangle-network/browser-agent-driver 0.22.0 → 0.23.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/brain/index.d.ts +21 -0
- package/dist/brain/index.d.ts.map +1 -1
- package/dist/brain/index.js +462 -34
- package/dist/brain/index.js.map +1 -1
- package/dist/browser-launch.d.ts +2 -0
- package/dist/browser-launch.d.ts.map +1 -1
- package/dist/browser-launch.js +12 -5
- package/dist/browser-launch.js.map +1 -1
- package/dist/captcha.d.ts +0 -7
- package/dist/captcha.d.ts.map +1 -1
- package/dist/captcha.js +113 -4
- package/dist/captcha.js.map +1 -1
- package/dist/cli.js +69 -13
- package/dist/cli.js.map +1 -1
- package/dist/config.d.ts +7 -0
- package/dist/config.d.ts.map +1 -1
- package/dist/config.js +1 -0
- package/dist/config.js.map +1 -1
- package/dist/drivers/cursor-overlay.d.ts +1 -1
- package/dist/drivers/cursor-overlay.d.ts.map +1 -1
- package/dist/drivers/cursor-overlay.js +21 -9
- package/dist/drivers/cursor-overlay.js.map +1 -1
- package/dist/drivers/extract-with-index.d.ts +45 -0
- package/dist/drivers/extract-with-index.d.ts.map +1 -0
- package/dist/drivers/extract-with-index.js +141 -0
- package/dist/drivers/extract-with-index.js.map +1 -0
- package/dist/drivers/playwright.d.ts +7 -0
- package/dist/drivers/playwright.d.ts.map +1 -1
- package/dist/drivers/playwright.js +232 -1
- package/dist/drivers/playwright.js.map +1 -1
- package/dist/drivers/som-overlay.d.ts +36 -0
- package/dist/drivers/som-overlay.d.ts.map +1 -0
- package/dist/drivers/som-overlay.js +109 -0
- package/dist/drivers/som-overlay.js.map +1 -0
- package/dist/drivers/types.d.ts +2 -0
- package/dist/drivers/types.d.ts.map +1 -1
- package/dist/recovery.d.ts.map +1 -1
- package/dist/recovery.js +2 -0
- package/dist/recovery.js.map +1 -1
- package/dist/run-state.d.ts +27 -1
- package/dist/run-state.d.ts.map +1 -1
- package/dist/run-state.js +40 -1
- package/dist/run-state.js.map +1 -1
- package/dist/runner/runner.d.ts +21 -0
- package/dist/runner/runner.d.ts.map +1 -1
- package/dist/runner/runner.js +260 -14
- package/dist/runner/runner.js.map +1 -1
- package/dist/supervisor/critic.d.ts.map +1 -1
- package/dist/supervisor/critic.js +1 -0
- package/dist/supervisor/critic.js.map +1 -1
- package/dist/supervisor/policy.js +10 -0
- package/dist/supervisor/policy.js.map +1 -1
- package/dist/test-runner.d.ts.map +1 -1
- package/dist/test-runner.js +8 -3
- package/dist/test-runner.js.map +1 -1
- package/dist/types.d.ts +62 -1
- package/dist/types.d.ts.map +1 -1
- package/package.json +3 -1
package/dist/runner/runner.js
CHANGED
|
@@ -20,10 +20,11 @@ import { detectSupervisorSignal, formatSupervisorSignal } from '../supervisor/po
|
|
|
20
20
|
import { requestSupervisorDirective } from '../supervisor/critic.js';
|
|
21
21
|
import { shouldAcceptFirstPartyBoundaryCompletion } from '../domain-policy.js';
|
|
22
22
|
import { deriveWasteMetrics } from '../run-metrics.js';
|
|
23
|
-
import { RunState } from '../run-state.js';
|
|
23
|
+
import { RunState, DEFAULT_TOKEN_BUDGET } from '../run-state.js';
|
|
24
24
|
import { ContextBudget } from '../context-budget.js';
|
|
25
25
|
import { runOverridePipeline } from '../override-pipeline.js';
|
|
26
26
|
import { withRetry, findElementForRef, safeHostname, pushGoalVerificationEvidence } from './utils.js';
|
|
27
|
+
import { runExtractWithIndex, formatExtractWithIndexResult } from '../drivers/extract-with-index.js';
|
|
27
28
|
import { buildSearchResultsGuidance, buildVisibleLinkRecommendation, getVisibleLinkRecommendation, getRankedVisibleLinkCandidates, rankSearchCandidates } from './search-guidance.js';
|
|
28
29
|
import { buildGoalVerificationClaim, collectSearchWorkflowEvidence, shouldAcceptSearchWorkflowCompletion, shouldAcceptScriptBackedCompletion, detectCompletionContentTypeMismatch } from './goal-verification.js';
|
|
29
30
|
import { verifyExpectedEffect } from './effect-verification.js';
|
|
@@ -130,7 +131,9 @@ const DEFAULT_MAX_TURNS = 20;
|
|
|
130
131
|
const DEFAULT_RETRIES = 3;
|
|
131
132
|
const DEFAULT_RETRY_DELAY_MS = 1000;
|
|
132
133
|
const DEFAULT_MICRO_PLAN_ACTIONS = 2;
|
|
133
|
-
|
|
134
|
+
// Gen 18: clickAt/typeAt added so vision-mode can emit multi-action turns
|
|
135
|
+
// Gen 23: clickLabel/typeLabel for SoM-based actions
|
|
136
|
+
const SAFE_MICRO_ACTIONS = new Set(['click', 'type', 'press', 'hover', 'select', 'scroll', 'wait', 'clickAt', 'typeAt', 'clickLabel', 'typeLabel']);
|
|
134
137
|
const DEFAULT_SUPERVISOR = {
|
|
135
138
|
enabled: true,
|
|
136
139
|
useVision: true,
|
|
@@ -183,6 +186,60 @@ export function hasPlaceholderPattern(text) {
|
|
|
183
186
|
}
|
|
184
187
|
return false;
|
|
185
188
|
}
|
|
189
|
+
/**
|
|
190
|
+
* Gen 9 — runtime two-pass extraction. When the planner emits a single
|
|
191
|
+
* runScript step (per Gen 7.2 rule #7) and that script returns null /
|
|
192
|
+
* empty / whitespace / `{x: null}` / a placeholder pattern, the auto-
|
|
193
|
+
* complete-from-runScript path should NOT fire. Instead the runner should
|
|
194
|
+
* mark the plan as deviated and fall through to the per-action loop where
|
|
195
|
+
* Brain.decide can re-observe the loaded page and emit a smarter action
|
|
196
|
+
* (different selector, click+wait, scroll, etc.).
|
|
197
|
+
*
|
|
198
|
+
* This addresses the failure mode the Gen 8 head-to-head gauntlet
|
|
199
|
+
* surfaced: bad's planner-only path lost to browser-use's per-action loop
|
|
200
|
+
* on tasks where the first runScript pick was wrong (npm, mdn signature,
|
|
201
|
+
* w3c, github, wikipedia variance). Two-pass gives bad's per-action loop
|
|
202
|
+
* the same recovery surface browser-use uses, with the planner's speed
|
|
203
|
+
* advantage on the cases where runScript succeeds first try.
|
|
204
|
+
*
|
|
205
|
+
* "Meaningful" means: not empty/whitespace, not the literal string `null`
|
|
206
|
+
* or `undefined`, and not matching `hasPlaceholderPattern` (which already
|
|
207
|
+
* detects JSON null fields, "<from prior step>" markers, etc.).
|
|
208
|
+
*/
|
|
209
|
+
export function isMeaningfulRunScriptOutput(output) {
|
|
210
|
+
if (typeof output !== 'string')
|
|
211
|
+
return false;
|
|
212
|
+
const trimmed = output.trim();
|
|
213
|
+
if (trimmed.length === 0)
|
|
214
|
+
return false;
|
|
215
|
+
if (trimmed === 'null' || trimmed === 'undefined' || trimmed === '""' || trimmed === "''")
|
|
216
|
+
return false;
|
|
217
|
+
// Empty JSON shells: `{}`, `[]`, `{"x": null}`, `[null, null]`
|
|
218
|
+
if (trimmed === '{}' || trimmed === '[]')
|
|
219
|
+
return false;
|
|
220
|
+
if (hasPlaceholderPattern(trimmed))
|
|
221
|
+
return false;
|
|
222
|
+
// If the output parses as JSON and EVERY top-level value is null/empty,
|
|
223
|
+
// treat it as not meaningful. This catches `{"x": null, "y": ""}` even
|
|
224
|
+
// though the placeholder regex would already catch the null one.
|
|
225
|
+
try {
|
|
226
|
+
const parsed = JSON.parse(trimmed);
|
|
227
|
+
if (parsed && typeof parsed === 'object' && !Array.isArray(parsed)) {
|
|
228
|
+
const values = Object.values(parsed);
|
|
229
|
+
if (values.length > 0) {
|
|
230
|
+
const allEmpty = values.every((v) => v === null || v === undefined || v === '' || v === 0);
|
|
231
|
+
if (allEmpty)
|
|
232
|
+
return false;
|
|
233
|
+
}
|
|
234
|
+
}
|
|
235
|
+
if (Array.isArray(parsed) && parsed.length === 0)
|
|
236
|
+
return false;
|
|
237
|
+
}
|
|
238
|
+
catch {
|
|
239
|
+
// Not JSON, that's fine — fall through to "meaningful" if we got here.
|
|
240
|
+
}
|
|
241
|
+
return true;
|
|
242
|
+
}
|
|
186
243
|
export class BrowserAgent {
|
|
187
244
|
driver;
|
|
188
245
|
brain;
|
|
@@ -225,13 +282,24 @@ export class BrowserAgent {
|
|
|
225
282
|
this.runRegistry = options.runRegistry;
|
|
226
283
|
}
|
|
227
284
|
async run(scenario) {
|
|
228
|
-
|
|
285
|
+
// Gen 14: vision mode gets more turns — each turn takes ~15s (screenshot
|
|
286
|
+
// encode + image tokens) vs ~5s for DOM-first. Without the boost, vision
|
|
287
|
+
// runs out of turns before completing multi-step tasks.
|
|
288
|
+
const isVisionMode = this.config.observationMode === 'vision' || this.config.observationMode === 'hybrid';
|
|
289
|
+
const baseMaxTurns = scenario.maxTurns || DEFAULT_MAX_TURNS;
|
|
290
|
+
// Gen 26: 30 turn minimum for vision. 15/51 failures were turn budget
|
|
291
|
+
// exhaustion at 20. The cost cap (200k tokens) is the real bound.
|
|
292
|
+
const maxTurns = isVisionMode ? Math.max(baseMaxTurns, 30) : baseMaxTurns;
|
|
229
293
|
const retries = this.config.retries ?? DEFAULT_RETRIES;
|
|
230
294
|
const retryDelayMs = this.config.retryDelayMs ?? DEFAULT_RETRY_DELAY_MS;
|
|
231
295
|
const turns = [];
|
|
232
296
|
const startTime = Date.now();
|
|
233
297
|
const phaseTimings = {};
|
|
234
|
-
|
|
298
|
+
// Gen 27: vision+planner mode gets 3× token budget (300k). Gen 26 showed
|
|
299
|
+
// 4 cost_cap failures and 18 turn-exhausted tasks (now getting 30 turns
|
|
300
|
+
// but hitting 200k cap). The timeout (600s) is the real safety net.
|
|
301
|
+
const visionBudgetMultiplier = isVisionMode ? 3 : 1;
|
|
302
|
+
const runState = new RunState(maxTurns, Math.round(DEFAULT_TOKEN_BUDGET * visionBudgetMultiplier));
|
|
235
303
|
const runId = scenario.sessionId
|
|
236
304
|
? `${scenario.sessionId}_${Date.now()}`
|
|
237
305
|
: RunRegistry.generateRunId();
|
|
@@ -278,10 +346,12 @@ export class BrowserAgent {
|
|
|
278
346
|
});
|
|
279
347
|
return agentResult;
|
|
280
348
|
};
|
|
281
|
-
// Wrap onTurn to include mid-run manifest updates (every 3 turns)
|
|
349
|
+
// Wrap onTurn to include mid-run manifest updates (every 3 turns) and to
|
|
350
|
+
// accumulate per-turn token usage for the Gen 10 cost cap.
|
|
282
351
|
const originalOnTurn = this.onTurn;
|
|
283
352
|
this.onTurn = (turn) => {
|
|
284
353
|
originalOnTurn?.(turn);
|
|
354
|
+
runState.recordTokens(turn.tokensUsed);
|
|
285
355
|
if (this.runRegistry && turns.length % 3 === 0) {
|
|
286
356
|
try {
|
|
287
357
|
this.runRegistry.updateRun(runId, {
|
|
@@ -541,6 +611,18 @@ export class BrowserAgent {
|
|
|
541
611
|
totalMs: Date.now() - startTime,
|
|
542
612
|
});
|
|
543
613
|
}
|
|
614
|
+
// Gen 10: hard cost cap. Stops the per-action loop from burning unbounded
|
|
615
|
+
// tokens on cases where recovery isn't converging (the Gen 9 death-spiral
|
|
616
|
+
// failure mode where reddit hit $0.32 / 173K tokens). The cap is enforced
|
|
617
|
+
// BEFORE the next LLM call so the case aborts cleanly with a reason.
|
|
618
|
+
if (runState.isTokenBudgetExhausted) {
|
|
619
|
+
return buildResult({
|
|
620
|
+
success: false,
|
|
621
|
+
reason: `cost_cap_exceeded: ${runState.totalTokensUsed} tokens used, budget ${runState.tokenBudget}`,
|
|
622
|
+
turns,
|
|
623
|
+
totalMs: Date.now() - startTime,
|
|
624
|
+
});
|
|
625
|
+
}
|
|
544
626
|
const turnStart = Date.now();
|
|
545
627
|
this.bus.emitNow({ type: 'turn-started', runId, turn: i });
|
|
546
628
|
try {
|
|
@@ -680,7 +762,11 @@ export class BrowserAgent {
|
|
|
680
762
|
terminalBlocker = detectTerminalBlocker(state);
|
|
681
763
|
}
|
|
682
764
|
}
|
|
683
|
-
catch {
|
|
765
|
+
catch (captchaErr) {
|
|
766
|
+
if (this.config.debug) {
|
|
767
|
+
console.log('[Runner] CAPTCHA solve error:', captchaErr instanceof Error ? captchaErr.message : String(captchaErr));
|
|
768
|
+
}
|
|
769
|
+
}
|
|
684
770
|
}
|
|
685
771
|
}
|
|
686
772
|
if (terminalBlocker) {
|
|
@@ -964,6 +1050,30 @@ export class BrowserAgent {
|
|
|
964
1050
|
if (lastTurn?.action.action === 'runScript' && !lastTurn.error) {
|
|
965
1051
|
ctxBudget.add('extraction-guard', '\nData extracted. If it answers the goal, complete now.\n', 80);
|
|
966
1052
|
}
|
|
1053
|
+
// Gen 27: form stall detection — escalating urgency. Compare by
|
|
1054
|
+
// origin+pathname (ignoring query params) because sites like Google
|
|
1055
|
+
// Flights update URL params with each form interaction without actually
|
|
1056
|
+
// navigating to results.
|
|
1057
|
+
{
|
|
1058
|
+
const urlBase = (u) => { try {
|
|
1059
|
+
const p = new URL(u);
|
|
1060
|
+
return (p.origin + p.pathname).replace(/\/+$/, '');
|
|
1061
|
+
}
|
|
1062
|
+
catch {
|
|
1063
|
+
return u;
|
|
1064
|
+
} };
|
|
1065
|
+
const currentBase = urlBase(state.url);
|
|
1066
|
+
const sameBaseCount = turns.filter(t => t.state?.url && urlBase(t.state.url) === currentBase).length;
|
|
1067
|
+
if (sameBaseCount >= 15) {
|
|
1068
|
+
// Hard stall: 15+ turns on same page. Demand navigation away.
|
|
1069
|
+
// Use DuckDuckGo — Google Search triggers anti-bot CAPTCHAs.
|
|
1070
|
+
ctxBudget.add('form-stall', `\nCRITICAL FORM STALL: You have spent ${sameBaseCount} turns on this page without completing your goal. The form is NOT going to cooperate. Your NEXT action MUST be: navigate to https://duckduckgo.com/?q={your goal as a natural language search query}. Do NOT try the form again. Do NOT use google.com/search (it blocks automated access). Use DuckDuckGo.\n`, 95);
|
|
1071
|
+
}
|
|
1072
|
+
else if (sameBaseCount >= 10) {
|
|
1073
|
+
// Soft stall: suggest fallback but don't force it.
|
|
1074
|
+
ctxBudget.add('form-stall', `\nFORM STALL WARNING: You have been on this page for ${sameBaseCount} turns. Consider navigating to https://duckduckgo.com/?q={your goal as a search query} to find the answer via search results instead. Do NOT use google.com/search (it blocks automated browsers).\n`, 85);
|
|
1075
|
+
}
|
|
1076
|
+
}
|
|
967
1077
|
const extraContext = ctxBudget.build();
|
|
968
1078
|
const forceVision = shouldEscalateVision({
|
|
969
1079
|
config: this.config,
|
|
@@ -1247,6 +1357,9 @@ export class BrowserAgent {
|
|
|
1247
1357
|
: JSON.stringify(scriptResult, null, 2);
|
|
1248
1358
|
runState.firstSufficientEvidenceTurn ??= i;
|
|
1249
1359
|
pushGoalVerificationEvidence(runState.goalVerificationEvidence, `SCRIPT RESULT:\n${stringified ?? '(undefined)'}`);
|
|
1360
|
+
if (typeof stringified === 'string' && stringified.length > 10) {
|
|
1361
|
+
runState.recordEvidence(`EXTRACTED (turn ${i}): ${stringified.slice(0, 500)}`);
|
|
1362
|
+
}
|
|
1250
1363
|
this.brain.injectFeedback(`SCRIPT RESULT:\n${stringified ?? '(undefined)'}`);
|
|
1251
1364
|
}
|
|
1252
1365
|
catch (scriptErr) {
|
|
@@ -1262,6 +1375,39 @@ export class BrowserAgent {
|
|
|
1262
1375
|
this.onTurn?.(turn);
|
|
1263
1376
|
continue;
|
|
1264
1377
|
}
|
|
1378
|
+
// -- 5d. Handle extractWithIndex action (Gen 10) --
|
|
1379
|
+
// Returns a numbered list of every visible element matching `query`,
|
|
1380
|
+
// each with its tag, textContent, key attributes, and a stable
|
|
1381
|
+
// selector. The agent picks elements by index in the next turn.
|
|
1382
|
+
// This is the Gen 10 capability change: pick-by-content instead of
|
|
1383
|
+
// pick-by-selector. Works on data the planner couldn't see at plan
|
|
1384
|
+
// time (XHR-loaded content, dl/dt/dd, deeply-nested wrappers).
|
|
1385
|
+
if (action.action === 'extractWithIndex') {
|
|
1386
|
+
const page = this.driver.getPage?.();
|
|
1387
|
+
if (page) {
|
|
1388
|
+
try {
|
|
1389
|
+
const matches = await runExtractWithIndex(page, action.query, action.contains);
|
|
1390
|
+
const formatted = formatExtractWithIndexResult(matches, action.query, action.contains);
|
|
1391
|
+
runState.firstSufficientEvidenceTurn ??= i;
|
|
1392
|
+
pushGoalVerificationEvidence(runState.goalVerificationEvidence, `EXTRACT RESULT (${matches.length} matches):\n${formatted}`);
|
|
1393
|
+
if (formatted.length > 10) {
|
|
1394
|
+
runState.recordEvidence(`EXTRACTED (turn ${i}): ${formatted.slice(0, 500)}`);
|
|
1395
|
+
}
|
|
1396
|
+
this.brain.injectFeedback(`EXTRACT RESULT (${matches.length} matches for query "${action.query}"${action.contains ? ` containing "${action.contains}"` : ''}):\n${formatted}`);
|
|
1397
|
+
}
|
|
1398
|
+
catch (extractErr) {
|
|
1399
|
+
const msg = extractErr instanceof Error ? extractErr.message : String(extractErr);
|
|
1400
|
+
this.brain.injectFeedback(`EXTRACT ERROR: ${msg}`);
|
|
1401
|
+
}
|
|
1402
|
+
}
|
|
1403
|
+
else {
|
|
1404
|
+
this.brain.injectFeedback('EXTRACT ERROR: Cannot access page — driver does not expose a Playwright page.');
|
|
1405
|
+
}
|
|
1406
|
+
turn.durationMs = Date.now() - turnStart;
|
|
1407
|
+
turns.push(turn);
|
|
1408
|
+
this.onTurn?.(turn);
|
|
1409
|
+
continue;
|
|
1410
|
+
}
|
|
1265
1411
|
// -- 6. Check for terminal actions --
|
|
1266
1412
|
if (action.action === 'complete') {
|
|
1267
1413
|
// Step 1: Goal verification — did the agent actually achieve the goal?
|
|
@@ -1270,6 +1416,7 @@ export class BrowserAgent {
|
|
|
1270
1416
|
const persistentSearchEvidence = collectSearchWorkflowEvidence(scenario.goal, action.result || '', turns);
|
|
1271
1417
|
const verificationEvidence = [
|
|
1272
1418
|
...runState.goalVerificationEvidence,
|
|
1419
|
+
...runState.extractedEvidence,
|
|
1273
1420
|
...persistentSearchEvidence,
|
|
1274
1421
|
];
|
|
1275
1422
|
if (shouldVerifyGoal) {
|
|
@@ -1277,22 +1424,44 @@ export class BrowserAgent {
|
|
|
1277
1424
|
// evidence and had no recent errors. The detailed result text
|
|
1278
1425
|
// (>50 chars) combined with script-extracted evidence means the
|
|
1279
1426
|
// verifier almost always agrees — save the round-trip.
|
|
1427
|
+
//
|
|
1428
|
+
// Gen 12: content-aware gate. gpt-5.4 writes verbose narratives
|
|
1429
|
+
// that admit failure ("could not complete", "not visible", "did
|
|
1430
|
+
// not take effect") yet marks success. The old heuristic (length
|
|
1431
|
+
// + evidence + no errors) rubber-stamped these. Now we scan the
|
|
1432
|
+
// result text for self-contradicting phrases and force LLM
|
|
1433
|
+
// verification when found. This fixes the 6/8 judge disagreement
|
|
1434
|
+
// cases from Gen 11 evolve R2.
|
|
1280
1435
|
const agentResult = action.result || '';
|
|
1281
1436
|
const recentErrors = turns.slice(-2).filter(t => t.error).length;
|
|
1282
1437
|
const hasScriptEvidence = verificationEvidence.some(e => e.startsWith('SCRIPT RESULT:'));
|
|
1438
|
+
// Content-aware gate: detect when the agent's own text admits
|
|
1439
|
+
// failure despite claiming success. These phrases were found in
|
|
1440
|
+
// 6 of 8 false-pass cases on WebVoyager with gpt-5.4.
|
|
1441
|
+
const selfContradicting = /\b(?:could not (?:complete|find|fulfill|verify|confirm|locate|access|extract|retrieve)|not (?:visible|available|found|present|accessible|displayed|shown|confirmed|verified)|did not (?:take effect|work|succeed|load|return)|unable to (?:find|complete|verify|access|extract|retrieve)|no (?:visible (?:answer|result|data|content)|results? (?:found|returned|available))|(?:failed|failure) to (?:find|complete|set|select|navigate)|unfortunately|I (?:was|am) unable|(?:task|request|goal) (?:is|was) (?:not |in)complete)\b/i.test(agentResult);
|
|
1283
1442
|
const fastPathEligible = agentResult.length > 50 &&
|
|
1284
1443
|
recentErrors === 0 &&
|
|
1285
|
-
hasScriptEvidence
|
|
1444
|
+
hasScriptEvidence &&
|
|
1445
|
+
!selfContradicting;
|
|
1286
1446
|
if (fastPathEligible) {
|
|
1287
1447
|
goalResult = {
|
|
1288
1448
|
achieved: true,
|
|
1289
1449
|
confidence: 0.9,
|
|
1290
|
-
evidence: ['Fast-path: agent provided detailed result with script-backed evidence and no
|
|
1450
|
+
evidence: ['Fast-path: agent provided detailed result with script-backed evidence, no recent errors, and no self-contradicting language.'],
|
|
1291
1451
|
missing: [],
|
|
1292
1452
|
};
|
|
1293
1453
|
if (this.config.debug) {
|
|
1294
|
-
console.log('[Runner] Goal verification fast-path: skipped LLM call (strong evidence + no errors)');
|
|
1454
|
+
console.log('[Runner] Goal verification fast-path: skipped LLM call (strong evidence + no errors + no self-contradiction)');
|
|
1455
|
+
}
|
|
1456
|
+
}
|
|
1457
|
+
else if (selfContradicting) {
|
|
1458
|
+
// Force LLM verification — the agent claims success but its
|
|
1459
|
+
// own text suggests failure. The LLM verifier reads the actual
|
|
1460
|
+
// content and makes the right call.
|
|
1461
|
+
if (this.config.debug) {
|
|
1462
|
+
console.log('[Runner] Gen 12: fast-path BLOCKED — agent result contains self-contradicting language, forcing LLM verification');
|
|
1295
1463
|
}
|
|
1464
|
+
goalResult = await this.brain.verifyGoalCompletion(state, scenario.goal, buildGoalVerificationClaim(agentResult, verificationEvidence));
|
|
1296
1465
|
}
|
|
1297
1466
|
else {
|
|
1298
1467
|
goalResult = await this.brain.verifyGoalCompletion(state, scenario.goal, buildGoalVerificationClaim(agentResult, verificationEvidence));
|
|
@@ -1378,10 +1547,18 @@ export class BrowserAgent {
|
|
|
1378
1547
|
runState.verificationRejectionCount++;
|
|
1379
1548
|
turn.verificationFailure = goalResult.missing.join('; ') || 'Goal verification failed';
|
|
1380
1549
|
runState.firstSufficientEvidenceTurn ??= i;
|
|
1381
|
-
//
|
|
1382
|
-
|
|
1383
|
-
|
|
1384
|
-
|
|
1550
|
+
// Gen 19: progressive strategy-shift escalation on rejection.
|
|
1551
|
+
// Each rejection level suggests a MORE different approach.
|
|
1552
|
+
let escalation;
|
|
1553
|
+
if (runState.verificationRejectionCount >= 3) {
|
|
1554
|
+
escalation = ' STRATEGY SHIFT REQUIRED: Your previous approaches have failed 3 times. Try a COMPLETELY different method: use navigate to go to a different search engine or URL, try extractWithIndex instead of runScript, or scroll to look for the data in a different part of the page. Do NOT repeat what you just tried.';
|
|
1555
|
+
}
|
|
1556
|
+
else if (runState.verificationRejectionCount >= 2) {
|
|
1557
|
+
escalation = ' Use runScript or extractWithIndex to extract the exact data from the page and include ALL required values in your completion result.';
|
|
1558
|
+
}
|
|
1559
|
+
else {
|
|
1560
|
+
escalation = ' Re-read the GOAL carefully — your result is missing specific data the goal asked for. Find and include it before completing.';
|
|
1561
|
+
}
|
|
1385
1562
|
this.brain.injectFeedback(`REJECTED (${goalResult.confidence.toFixed(2)}). Missing: ${goalResult.missing.join('; ')}.${escalation}`);
|
|
1386
1563
|
turn.durationMs = Date.now() - turnStart;
|
|
1387
1564
|
turns.push(turn);
|
|
@@ -1555,6 +1732,14 @@ export class BrowserAgent {
|
|
|
1555
1732
|
else {
|
|
1556
1733
|
runState.clearConsecutiveErrors();
|
|
1557
1734
|
executeTimeoutRecoveries = 0; // Reset on successful action
|
|
1735
|
+
// Gen 27: surface form reset warnings from batch fill verification
|
|
1736
|
+
if ('warning' in execResult && typeof execResult.warning === 'string') {
|
|
1737
|
+
const warning = execResult.warning;
|
|
1738
|
+
this.brain.injectFeedback(warning);
|
|
1739
|
+
if (this.config.debug) {
|
|
1740
|
+
console.log(`[Runner] Fill warning: ${warning}`);
|
|
1741
|
+
}
|
|
1742
|
+
}
|
|
1558
1743
|
// Capture element bounding box for replay overlays
|
|
1559
1744
|
if (execResult.bounds) {
|
|
1560
1745
|
turn.actionBounds = execResult.bounds;
|
|
@@ -1814,6 +1999,12 @@ export class BrowserAgent {
|
|
|
1814
1999
|
// runScript runs, so on extraction tasks it fabricates placeholders.
|
|
1815
2000
|
// This deterministic substitution fixes that without an extra LLM call.
|
|
1816
2001
|
let lastRunScriptOutput = null;
|
|
2002
|
+
// Gen 10: track the last extractWithIndex match list. Unlike runScript,
|
|
2003
|
+
// we do NOT auto-substitute this into a placeholder complete — the LLM
|
|
2004
|
+
// must read the formatted match list and pick by index. When the plan
|
|
2005
|
+
// ends with extractWithIndex (or runs out of valid steps), we fall
|
|
2006
|
+
// through to the per-action loop with the match list as feedback.
|
|
2007
|
+
let lastExtractOutput = null;
|
|
1817
2008
|
for (let stepIdx = 0; stepIdx < plan.steps.length; stepIdx++) {
|
|
1818
2009
|
if (scenario.signal?.aborted) {
|
|
1819
2010
|
return {
|
|
@@ -1998,6 +2189,22 @@ export class BrowserAgent {
|
|
|
1998
2189
|
// This is the supply side of the placeholder-substitution fix above.
|
|
1999
2190
|
if (step.action.action === 'runScript' && typeof execResult.data === 'string' && execResult.data.length > 0) {
|
|
2000
2191
|
lastRunScriptOutput = execResult.data;
|
|
2192
|
+
if (execResult.data.length > 10) {
|
|
2193
|
+
runState.recordEvidence(`EXTRACTED (turn ${currentTurnIndex}): ${execResult.data.slice(0, 500)}`);
|
|
2194
|
+
}
|
|
2195
|
+
}
|
|
2196
|
+
// Gen 10: capture extractWithIndex match list for fall-through to the
|
|
2197
|
+
// per-action loop. The LLM must read the list and pick by index — we
|
|
2198
|
+
// do not auto-complete with the raw match list.
|
|
2199
|
+
if (step.action.action === 'extractWithIndex' && typeof execResult.data === 'string' && execResult.data.length > 0) {
|
|
2200
|
+
lastExtractOutput = execResult.data;
|
|
2201
|
+
// Also push as goal verification evidence so the verifier sees what
|
|
2202
|
+
// the agent extracted.
|
|
2203
|
+
runState.firstSufficientEvidenceTurn ??= currentTurnIndex;
|
|
2204
|
+
pushGoalVerificationEvidence(runState.goalVerificationEvidence, `EXTRACT RESULT:\n${execResult.data}`);
|
|
2205
|
+
if (execResult.data.length > 10) {
|
|
2206
|
+
runState.recordEvidence(`EXTRACTED (turn ${currentTurnIndex}): ${execResult.data.slice(0, 500)}`);
|
|
2207
|
+
}
|
|
2001
2208
|
}
|
|
2002
2209
|
// Verify the post-condition. We re-observe to get the post-action
|
|
2003
2210
|
// state, then run the same verifyExpectedEffect helper the per-action
|
|
@@ -2027,6 +2234,7 @@ export class BrowserAgent {
|
|
|
2027
2234
|
|| step.action.action === 'scroll'
|
|
2028
2235
|
|| step.action.action === 'hover'
|
|
2029
2236
|
|| step.action.action === 'runScript'
|
|
2237
|
+
|| step.action.action === 'extractWithIndex'
|
|
2030
2238
|
|| step.action.action === 'evaluate'
|
|
2031
2239
|
|| step.action.action === 'verifyPreview'
|
|
2032
2240
|
|| step.action.action === 'fill'
|
|
@@ -2141,7 +2349,7 @@ export class BrowserAgent {
|
|
|
2141
2349
|
const lastStep = plan.steps[plan.steps.length - 1];
|
|
2142
2350
|
if (lastStep
|
|
2143
2351
|
&& lastStep.action.action === 'runScript'
|
|
2144
|
-
&& lastRunScriptOutput) {
|
|
2352
|
+
&& isMeaningfulRunScriptOutput(lastRunScriptOutput)) {
|
|
2145
2353
|
const synthTurnNumber = currentTurnIndex + 1;
|
|
2146
2354
|
const synthTurn = {
|
|
2147
2355
|
turn: synthTurnNumber,
|
|
@@ -2173,6 +2381,44 @@ export class BrowserAgent {
|
|
|
2173
2381
|
turnsConsumed: plan.steps.length + 1,
|
|
2174
2382
|
};
|
|
2175
2383
|
}
|
|
2384
|
+
// Gen 10: if the plan ended with extractWithIndex, fall through to the
|
|
2385
|
+
// per-action loop with the match list as feedback. The LLM must read
|
|
2386
|
+
// the matches and pick by index — we do NOT auto-complete with the raw
|
|
2387
|
+
// match list. This is the planner-emits-extract path for extraction
|
|
2388
|
+
// tasks like npm/mdn/python-docs where the planner used the new
|
|
2389
|
+
// extractWithIndex action.
|
|
2390
|
+
if (lastExtractOutput) {
|
|
2391
|
+
return {
|
|
2392
|
+
kind: 'deviated',
|
|
2393
|
+
lastState,
|
|
2394
|
+
failedStepIndex: plan.steps.length,
|
|
2395
|
+
reason: `plan completed extractWithIndex but the LLM must read the matches and pick by index. Match list:\n${lastExtractOutput.slice(0, 4000)}\n\nPick the index whose text matches the goal, then emit complete with result: <picked text>`,
|
|
2396
|
+
turnsConsumed: plan.steps.length,
|
|
2397
|
+
};
|
|
2398
|
+
}
|
|
2399
|
+
// Gen 9 (cherry-picked into Gen 10): if the last step WAS a runScript
|
|
2400
|
+
// but the output was NOT meaningful (null, empty, placeholder), DO NOT
|
|
2401
|
+
// auto-complete with garbage. Fall through to the per-action loop with
|
|
2402
|
+
// a deviation reason that names the empty output. In Gen 10 the per-
|
|
2403
|
+
// action loop has TWO new tools that make this recovery actually work:
|
|
2404
|
+
// 1. extractWithIndex (the wide-query content-match action) — see
|
|
2405
|
+
// data-extraction rule #25
|
|
2406
|
+
// 2. cost cap (100k tokens) — bounds any death-spiral if the LLM
|
|
2407
|
+
// can't recover, preventing the Gen 9.1 reddit failure mode
|
|
2408
|
+
if (lastStep
|
|
2409
|
+
&& lastStep.action.action === 'runScript'
|
|
2410
|
+
&& !isMeaningfulRunScriptOutput(lastRunScriptOutput)) {
|
|
2411
|
+
if (this.config.debug) {
|
|
2412
|
+
console.log(`[Runner] Gen 9: runScript returned no meaningful output (${JSON.stringify(lastRunScriptOutput).slice(0, 100)}); falling through to per-action loop for two-pass extraction`);
|
|
2413
|
+
}
|
|
2414
|
+
return {
|
|
2415
|
+
kind: 'deviated',
|
|
2416
|
+
lastState,
|
|
2417
|
+
failedStepIndex: plan.steps.length - 1,
|
|
2418
|
+
reason: `runScript returned no meaningful output (got: ${JSON.stringify(lastRunScriptOutput).slice(0, 200)}). The first-pass extraction failed — re-observe the page and try extractWithIndex with a wide query (e.g. 'p, span, dd, code') and a contains filter naming the expected text fragment. Pick-by-content beats pick-by-selector when the planner couldn't see the data at plan time.`,
|
|
2419
|
+
turnsConsumed: plan.steps.length,
|
|
2420
|
+
};
|
|
2421
|
+
}
|
|
2176
2422
|
// All steps verified BUT the plan ended without an explicit complete/abort.
|
|
2177
2423
|
// This means the planner emitted a finite sequence of "work" steps and
|
|
2178
2424
|
// didn't terminate. The right behavior is NOT to fabricate a complete —
|