zames_pro 2.29.15 → 2.31.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -4,7 +4,7 @@ import { getGitContext, formatGitContext } from './gitTools.js';
4
4
  import { parseXmlToolCalls } from './xml-toolcall.js';
5
5
  import { translate } from './i18n.js';
6
6
  import { normText } from './browser.js';
7
- export async function runAgentLoop({ browser, tools, task, workdir, maxIterations = 0, freshChat = false, sendSystemPrompt = false, transcript = null, attachments = [], onThinking = () => { }, onSendPause = () => { }, onNotice = () => { }, onAssistantThought = () => { }, onToolCall = () => { }, onToolResult = () => { }, onAssistantMessage = () => { }, onChatReady = () => { }, onWarning = () => { }, debugLog = false, locale = 'ru', }) {
7
+ export async function runAgentLoop({ browser, tools, task, workdir, maxIterations = 0, freshChat = false, sendSystemPrompt = false, transcript = null, attachments = [], onThinking = () => { }, onSendPause = () => { }, onNotice = () => { }, onAssistantThought = () => { }, onToolCall = () => { }, onToolResult = () => { }, onAssistantMessage = () => { }, onChatReady = () => { }, onWarning = () => { }, debugLog = false, locale = 'ru', askDeadlineMs = 240_000, maxAfterToolRetries = 6, onAutoCompact = null, autoCompactPct = 95, contextLimit = 1_000_000, getTokenUsage = null, }) {
8
8
  // UI callbacks must NEVER break the agent loop. A rendering error (a huge
9
9
  // tool result, a broken markdown frame, a closed terminal) used to throw
10
10
  // out of the loop right after a tool call — the session looked "stopped
@@ -180,7 +180,16 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
180
180
  // paragraph that is neither a tool call nor a real respond) from a genuine
181
181
  // short answer. We key on the STRUCTURE (work already started), not words.
182
182
  let toolsRanInTask = 0;
183
- const MAX_AFTER_TOOL_RETRIES = 6;
183
+ const MAX_AFTER_TOOL_RETRIES = Math.max(0, Math.floor(maxAfterToolRetries));
184
+ // Token count at which the last auto-compact fired. Re-arm only after
185
+ // the (fresh) chat grows past this plus a margin, so a chat that starts
186
+ // above the threshold does not compact on every tool call.
187
+ let lastAutoCompactTokens = -1;
188
+ // The message of a respond that arrived TOGETHER with real tool calls.
189
+ // It is not delivered immediately (the tools must run first), but if the
190
+ // model then stops without calling respond again, this is the best final
191
+ // message we have and must reach the operator.
192
+ let mixedRespondMsg = '';
184
193
  const iterCap = maxIterations > 0 ? maxIterations : 100000;
185
194
  for (let i = 0; i < iterCap; i++) {
186
195
  // NOTE: the spinner is NOT started here. browser.onSendStart fires it
@@ -199,15 +208,20 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
199
208
  // block the whole loop and look like a silent stop. We race it against a
200
209
  // hard deadline and treat a timeout as a nudge (re-ask), never as a
201
210
  // final answer. The deadline is generous enough for real long answers.
202
- const askDeadlineMs = 240_000;
203
211
  let rawResponse;
204
212
  let askTimer = null;
213
+ // Kept OUTSIDE the race so we can cancel it and wait for it to settle if
214
+ // the watchdog timer wins. Before, the losing ask() kept running (up to a
215
+ // 300s rate-limit wait or the finish loop) while the next iteration
216
+ // started a SECOND ask() against the same page — two sends / two Continue
217
+ // clicks.
218
+ const askPromise = browser.ask(message, {
219
+ agent: !isFirst,
220
+ attachments: isFirst ? attachments : [],
221
+ });
205
222
  try {
206
223
  rawResponse = await Promise.race([
207
- browser.ask(message, {
208
- agent: !isFirst,
209
- attachments: isFirst ? attachments : [],
210
- }),
224
+ askPromise,
211
225
  new Promise((_, reject) => {
212
226
  askTimer = setTimeout(() => reject(new Error('ask() watchdog timeout')), askDeadlineMs);
213
227
  if (askTimer && typeof askTimer.unref === 'function') {
@@ -221,12 +235,27 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
221
235
  catch (e) {
222
236
  if (askTimer)
223
237
  clearTimeout(askTimer);
238
+ // The timer won: cancel the still-running ask() and AWAIT its settle
239
+ // before the next iteration starts another one. Without this the two
240
+ // asks race on the same page. The settle is bounded: a page stuck in a
241
+ // 30s Playwright evaluate must not hang the loop forever.
242
+ browser.cancelPendingAsk?.();
243
+ const settleTimer = new Promise((r) => {
244
+ const t = setTimeout(r, 60_000);
245
+ if (typeof t.unref === 'function')
246
+ t.unref();
247
+ });
248
+ await Promise.race([askPromise.catch(() => { }), settleTimer]);
224
249
  transcript?.log('ask_timeout', {
225
250
  attempt: afterToolRetries,
226
251
  error: e.message,
227
252
  });
253
+ // Show how many watchdog retries remain, so the operator can tell a
254
+ // single hiccup from a genuine stall (the budget used to be invisible).
228
255
  safeWarning(translate(locale)('ds.answer_timeout', {
229
256
  sec: Math.round(askDeadlineMs / 1000),
257
+ attempt: Math.min(afterToolRetries + 1, MAX_AFTER_TOOL_RETRIES),
258
+ max: MAX_AFTER_TOOL_RETRIES,
230
259
  }));
231
260
  if (afterToolRetries < MAX_AFTER_TOOL_RETRIES) {
232
261
  afterToolRetries++;
@@ -469,13 +498,36 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
469
498
  // paragraph "Now let me analyze..." instead of calling a tool). We do not
470
499
  // match words here - the STRUCTURE (toolsRanInTask > 0) is the signal.
471
500
  if (toolsRanInTask > 0) {
501
+ // Before surfacing a reasoning paragraph as a report, make ONE
502
+ // explicit attempt to get a real respond (or a remembered mixed-respond
503
+ // message). This turns "ambiguous text" into an unambiguous final.
504
+ if (!finalRespondAsked) {
505
+ finalRespondAsked = true;
506
+ transcript?.log('final_respond_request', {
507
+ response: rawResponse.slice(0, 500),
508
+ });
509
+ message =
510
+ 'You stopped mid-task without finishing. If the task is DONE — ' +
511
+ 'call the respond tool with the final message to the operator ' +
512
+ '(and nothing else): ' +
513
+ '{"tool": "respond", "args": {"message": "..."}}. ' +
514
+ 'If it is NOT done — reply with exactly one JSON tool-call object, ' +
515
+ 'no text before or after.';
516
+ continue;
517
+ }
472
518
  transcript?.log('protocol_violation_final', {
473
519
  response: rawResponse.slice(0, 500),
474
520
  });
475
521
  safeWarning(translate(locale)('msg.suspicious_stop'));
476
- // The model stopped calling tools mid-task. Surface the last text as
477
- // a READABLE report, but say explicitly that the task may be incomplete:
478
- // never let a reasoning paragraph masquerade as a finished result.
522
+ // The model stopped calling tools mid-task. If it left a real message
523
+ // in a mixed respond earlier, deliver THAT (it is a genuine final);
524
+ // otherwise surface the last text but say explicitly that the task may
525
+ // be incomplete: never let a reasoning paragraph masquerade as a result.
526
+ if (mixedRespondMsg) {
527
+ safeAssistantMessage(mixedRespondMsg);
528
+ transcript?.log('assistant_final', { message: mixedRespondMsg });
529
+ return mixedRespondMsg;
530
+ }
479
531
  const lastText = (rawResponse || '').trim();
480
532
  return lastText
481
533
  ? lastText +
@@ -513,6 +565,14 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
513
565
  transcript?.log('respond_mixed_with_tools', {
514
566
  tools: realCalls.map((c) => c.tool),
515
567
  });
568
+ // A respond mixed with real tools must NOT be dropped silently. Keep
569
+ // its message; if the model then stops WITHOUT calling respond again, we
570
+ // deliver this remembered message instead of a bare reasoning paragraph.
571
+ const m = typeof respondCall.args.message === 'string'
572
+ ? respondCall.args.message
573
+ : String(respondCall.args.message ?? '');
574
+ if (isMeaningfulRespond(m))
575
+ mixedRespondMsg = m;
516
576
  }
517
577
  if (respondCall && realCalls.length === 0) {
518
578
  const msg = typeof respondCall.args.message === 'string'
@@ -619,23 +679,89 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
619
679
  // Real progress was made, so the "unparsed answer" budget is replenished:
620
680
  // a long chain of tools must not run out of it because of earlier hiccups.
621
681
  unparsedRetries = 0;
682
+ // Reset ALL per-task retry counters after a successful tool, not just
683
+ // the three above. stallRetries/looksDoneRetries/malformedRetries used to
684
+ // live for the WHOLE task, so over a long chain of tools their budgets
685
+ // could be exhausted by earlier hiccups and the guards silently stopped
686
+ // protecting against "stopped after a tool call".
687
+ stallRetries = 0;
688
+ looksDoneRetries = 0;
689
+ malformedRetries = 0;
622
690
  finalRespondAsked = false;
623
691
  if (results.length === 1) {
624
692
  const r = results[0];
625
693
  const resultStr = typeof r.result === 'string' ? r.result : JSON.stringify(r.result);
626
- message = `Tool result for ${r.tool}:\n${resultStr.slice(0, 12_000)}`;
694
+ message = `Tool result for ${r.tool}:\n${truncateToolResult(resultStr, 12_000)}`;
627
695
  }
628
696
  else {
629
697
  message = results
630
698
  .map((r) => {
631
699
  const resultStr = typeof r.result === 'string' ? r.result : JSON.stringify(r.result);
632
- return `Tool result for ${r.tool}:\n${resultStr.slice(0, 8000)}`;
700
+ return `Tool result for ${r.tool}:\n${truncateToolResult(resultStr, 8000)}`;
633
701
  })
634
702
  .join('\n\n');
635
703
  }
704
+ // Auto-compact at the ONLY safe seam — after a tool result and before
705
+ // the next send. Never mid-generation, never during a rate-limit wait, and
706
+ // never between the task and the first send (this block runs only after a
707
+ // tool really executed). The callback is awaited, so it serializes with the
708
+ // send throttle. When it returns a NEW chat id we refresh the message's
709
+ // chat and continue in the fresh chat.
710
+ if (onAutoCompact &&
711
+ getTokenUsage &&
712
+ !browser._abort &&
713
+ !browser._stopped) {
714
+ const tokens = getTokenUsage();
715
+ if (tokens !== null && Number.isFinite(tokens) && tokens > 0) {
716
+ const pct = (tokens / contextLimit) * 100;
717
+ const firedRecently = lastAutoCompactTokens >= 0;
718
+ // Re-arm only once the counter grows past the point we compacted at
719
+ // (plus a small margin), so a fresh chat does not compact again on
720
+ // every tool call.
721
+ const armed = !firedRecently || tokens > lastAutoCompactTokens + 1;
722
+ if (pct >= autoCompactPct && armed) {
723
+ lastAutoCompactTokens = tokens;
724
+ safeWarning(translate(locale)('compact.auto_trigger', {
725
+ pct: String(Math.round(pct)),
726
+ tokens: String(tokens),
727
+ }));
728
+ transcript?.log('auto_compact_trigger', { tokens, pct });
729
+ try {
730
+ const newChat = await onAutoCompact();
731
+ if (newChat) {
732
+ transcript?.log('auto_compact_done', { chatId: newChat });
733
+ // Continue in the fresh chat. The next send is a tool-result
734
+ // (agent: true), which is fine: the new chat already holds the
735
+ // system prompt + carryover. `lastAutoCompactTokens` stays at the
736
+ // count we compacted AT, so a chat that starts above the
737
+ // threshold does not compact again until it GROWS past it.
738
+ }
739
+ else {
740
+ transcript?.log('auto_compact_failed', {});
741
+ }
742
+ }
743
+ catch (e) {
744
+ transcript?.log('auto_compact_error', {
745
+ error: e.message,
746
+ });
747
+ }
748
+ }
749
+ }
750
+ }
636
751
  }
637
752
  return 'Iteration limit reached.';
638
753
  }
754
+ // Cap a tool result for the model, but make the truncation EXPLICIT: a bare
755
+ // slice() silently hid the tail and the model had no idea it was looking at a
756
+ // partial output (it would act on a half-read file/log).
757
+ export function truncateToolResult(text, limit) {
758
+ if (text.length <= limit)
759
+ return text;
760
+ const omitted = text.length - limit;
761
+ return (text.slice(0, limit) +
762
+ String.fromCharCode(10) +
763
+ `[...truncated ${omitted} chars]`);
764
+ }
639
765
  // The answer looks like a tool call, but parseToolCall() did not recognize it.
640
766
  // Used as a safeguard against "the agent called a tool and stopped": in that
641
767
  // case runAgentLoop asks the model to resend the call instead of finishing the