zames_pro 2.10.1 → 2.10.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,7 +2,29 @@ import { buildSystemPrompt } from './system-prompt.js';
2
2
  import { getGitContext, formatGitContext } from './gitTools.js';
3
3
  import { parseXmlToolCalls } from './xml-toolcall.js';
4
4
  import { translate } from './i18n.js';
5
+ import { normText } from './browser.js';
5
6
  export async function runAgentLoop({ browser, tools, task, workdir, maxIterations = 40, freshChat = false, sendSystemPrompt = false, transcript = null, attachments = [], onThinking = () => { }, onAssistantThought = () => { }, onToolCall = () => { }, onToolResult = () => { }, onAssistantMessage = () => { }, onChatReady = () => { }, onWarning = () => { }, debugLog = false, locale = 'ru', }) {
7
+ // UI callbacks must NEVER break the agent loop. A rendering error (a huge
8
+ // tool result, a broken markdown frame, a closed terminal) used to throw
9
+ // out of the loop right after a tool call — the session looked "stopped
10
+ // after a tool call", with a tool_call but no tool_result in the log. We
11
+ // wrap every callback so a UI failure is swallowed and the loop continues.
12
+ const safe = (fn) => {
13
+ return (...a) => {
14
+ try {
15
+ fn(...a);
16
+ }
17
+ catch {
18
+ // Intentionally ignored: the loop must survive UI failures.
19
+ }
20
+ };
21
+ };
22
+ const safeThinking = safe(onThinking);
23
+ const safeAssistantThought = safe(onAssistantThought);
24
+ const safeToolCall = safe(onToolCall);
25
+ const safeToolResult = safe(onToolResult);
26
+ const safeAssistantMessage = safe(onAssistantMessage);
27
+ const safeWarning = safe(onWarning);
6
28
  if (freshChat) {
7
29
  await browser.newChat();
8
30
  transcript?.log('new_chat');
@@ -42,7 +64,7 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
42
64
  length: systemPrompt.length,
43
65
  gitContext: gitText,
44
66
  });
45
- onThinking();
67
+ safeThinking();
46
68
  // system-prompt is an agent send: throttled (agent: true).
47
69
  await browser.ask(systemPrompt, { timeout: 60_000, agent: true });
48
70
  await reportChat();
@@ -104,7 +126,7 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
104
126
  let afterToolRetries = 0;
105
127
  const MAX_AFTER_TOOL_RETRIES = 6;
106
128
  for (let i = 0; i < maxIterations; i++) {
107
- onThinking();
129
+ safeThinking();
108
130
  // The first message (task) is user input: no throttle.
109
131
  // Subsequent ones (tool-result and resend requests) are agent sends:
110
132
  // throttled so we don't hit the rate limit.
@@ -139,7 +161,7 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
139
161
  attempt: afterToolRetries,
140
162
  error: e.message,
141
163
  });
142
- onWarning('browser.ask() не вернул ответ за ' +
164
+ safeWarning('browser.ask() не вернул ответ за ' +
143
165
  Math.round(askDeadlineMs / 1000) +
144
166
  'с — повторяю запрос.');
145
167
  if (afterToolRetries < MAX_AFTER_TOOL_RETRIES) {
@@ -150,7 +172,7 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
150
172
  transcript?.log('ask_timeout_exhausted', {
151
173
  message: 'ask() не вернул ответ и лимит повторов исчерпан',
152
174
  });
153
- onWarning('browser.ask() перестал отвечать; лимит повторов исчерпан, ' +
175
+ safeWarning('browser.ask() перестал отвечать; лимит повторов исчерпан, ' +
154
176
  'останавливаюсь. Ответа модели нет — проверьте чат DeepSeek вручную.');
155
177
  return 'ask() watchdog: ответ модели не получен';
156
178
  }
@@ -168,14 +190,22 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
168
190
  // cut-off "Stale. Let me ..." or a truncated JSON). All three mean the
169
191
  // turn is unfinished: nudge instead of stopping.
170
192
  const wdEmpty = !String(rawResponse || '').trim();
193
+ // STALE is compared on NORMALIZED text: DeepSeek often echoes the
194
+ // previous answer with different markdown emphasis (`**done**` vs `done`),
195
+ // which defeated an exact match and made the loop run the SAME tool again —
196
+ // the "stopped after a tool call" signature with a duplicate call.
171
197
  const wdStale = justRanTool &&
172
198
  lastRaw.trim() !== '' &&
173
- rawResponse.trim() === lastRaw.trim();
199
+ (rawResponse.trim() === lastRaw.trim() ||
200
+ normForStale(rawResponse) === normForStale(lastRaw));
174
201
  const wdNoCall = justRanTool &&
175
202
  !wdEmpty &&
176
203
  !wdStale &&
177
204
  parseToolCall(rawResponse) === null &&
178
205
  !responseLooksLikeToolCall(rawResponse);
206
+ // A stale answer is discarded even when it parses to a valid call: it is a
207
+ // duplicate of the previous turn, and re-running the tool would repeat
208
+ // side effects and stall the loop.
179
209
  if (!isFirst && (wdEmpty || wdStale || wdNoCall) && watchdogRetries < MAX_WATCHDOG_RETRIES) {
180
210
  watchdogRetries++;
181
211
  transcript?.log('watchdog_nudge', {
@@ -205,7 +235,7 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
205
235
  if (parsed) {
206
236
  const thought = extractPreToolText(rawResponse);
207
237
  if (thought)
208
- onAssistantThought(thought);
238
+ safeAssistantThought(thought);
209
239
  }
210
240
  const parsedCalls = Array.isArray(parsed) ? parsed : parsed ? [parsed] : [];
211
241
  if (parsedCalls.some((p) => p && p._permissive)) {
@@ -341,7 +371,7 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
341
371
  // not be turned into a tool call even after all retries: warn the
342
372
  // operator instead of silently printing e.g. "Stale. Let me verify".
343
373
  transcript?.log('suspicious_final', { response: rawResponse });
344
- onWarning(translate(locale)('msg.suspicious_stop'));
374
+ safeWarning(translate(locale)('msg.suspicious_stop'));
345
375
  }
346
376
  // A meaningful plain-text answer (e.g. a final report the model forgot to
347
377
  // wrap in respond) is surfaced as-is WITHOUT a warning: after the bounded
@@ -393,11 +423,11 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
393
423
  // the task normally.
394
424
  if (!isMeaningfulRespond(msg)) {
395
425
  transcript?.log('empty_respond_exhausted', { response: rawResponse });
396
- onWarning('Модель вызвала respond без текста, и лимит повторов исчерпан. ' +
426
+ safeWarning('Модель вызвала respond без текста, и лимит повторов исчерпан. ' +
397
427
  'Проверьте чат DeepSeek вручную.');
398
428
  return 'Модель не сформировала итоговое сообщение (пустой respond).';
399
429
  }
400
- onAssistantMessage(msg);
430
+ safeAssistantMessage(msg);
401
431
  transcript?.log('assistant_final', { message: msg });
402
432
  return msg;
403
433
  }
@@ -410,12 +440,12 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
410
440
  const tool = tools.find((t) => t.name === call.tool);
411
441
  if (!tool) {
412
442
  const err = `Неизвестный инструмент: ${call.tool}`;
413
- onToolResult(err);
443
+ safeToolResult(err);
414
444
  transcript?.log('tool_error', { tool: call.tool, error: err });
415
445
  results.push({ tool: call.tool, result: err });
416
446
  continue;
417
447
  }
418
- onToolCall(call.tool, call.args);
448
+ safeToolCall(call.tool, call.args);
419
449
  transcript?.log('tool_call', { tool: call.tool, args: call.args });
420
450
  let result;
421
451
  try {
@@ -424,7 +454,7 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
424
454
  catch (e) {
425
455
  result = `Ошибка: ${e.message}`;
426
456
  }
427
- onToolResult(result);
457
+ safeToolResult(result);
428
458
  transcript?.log('tool_result', {
429
459
  tool: call.tool,
430
460
  result: String(result),
@@ -501,6 +531,16 @@ function isMeaningfulRespond(msg) {
501
531
  return false;
502
532
  return true;
503
533
  }
534
+ // Normalize an answer for STALE comparison: collapse whitespace AND strip
535
+ // markdown emphasis/code markers. DeepSeek echoes the previous turn with a
536
+ // different emphasis (`**done**` vs `done`), which defeated an exact match and
537
+ // made the loop re-run the same tool ("stopped after a tool call").
538
+ function normForStale(s) {
539
+ return normText(s)
540
+ .replace(/[*_`#>]+/g, '')
541
+ .replace(/[ \t]+/g, ' ')
542
+ .trim();
543
+ }
504
544
  // Text that promises a tool call in the future tense but contains no call
505
545
  // itself. DeepSeek regularly "hangs" like this: it writes
506
546
  // "Now update README to mention …", "Let me run the tests", "I'll check now"
package/dist/browser.js CHANGED
@@ -646,15 +646,23 @@ export class DeepSeekBrowser {
646
646
  .evaluate(() => document.body.innerText.length)
647
647
  .catch(() => 0);
648
648
  let started = false;
649
+ // A stale/echo answer (the model repeats the previous text, or the answer
650
+ // legitimately equals it) does NOT change `cur`. In that case the old loop
651
+ // either threw "did not start" after 15s or hung until the full timeout —
652
+ // the operator saw the agent "stop after a tool call". We now also accept
653
+ // the answer when the generation has clearly SETTLED: no Stop button and
654
+ // the text has been stable for a couple of ticks.
655
+ let settledTicks = 0;
656
+ let lastStartCur = '';
649
657
  while (Date.now() < startDeadline) {
650
658
  if (this._abort)
651
659
  return '(прервано пользователем)';
652
660
  const pageText = await this._readPageText();
653
661
  if (isRateLimitText(pageText)) {
654
662
  throw new RateLimitError(pageText.slice(0, 300));
655
- if (isServerBusyText(pageText)) {
656
- throw new ServerBusyError(pageText.slice(0, 300));
657
- }
663
+ }
664
+ if (isServerBusyText(pageText)) {
665
+ throw new ServerBusyError(pageText.slice(0, 300));
658
666
  }
659
667
  const cur = await this._readLastAnswerTextClean().catch(() => '');
660
668
  const bodyLen = await this.page
@@ -678,9 +686,30 @@ export class DeepSeekBrowser {
678
686
  started = true;
679
687
  break;
680
688
  }
689
+ // Fallback for an echo/stale answer: the send happened (lastSentAt was
690
+ // just updated), the Stop button is gone and the text stopped changing.
691
+ // Two stable ticks in a row mean the turn is over even if it equals the
692
+ // previous text — return it instead of hanging/throwing.
693
+ const notGenerating = !(await this._isGenerating());
694
+ if (cur && cur === lastStartCur && notGenerating && cur.trim()) {
695
+ settledTicks++;
696
+ if (settledTicks >= 2) {
697
+ return cur;
698
+ }
699
+ }
700
+ else {
701
+ settledTicks = 0;
702
+ }
703
+ lastStartCur = cur;
681
704
  await this.page.waitForTimeout(300);
682
705
  }
683
706
  if (!started) {
707
+ // Last chance: the answer may have arrived and settled exactly at the
708
+ // deadline. Return the current text instead of a hard error.
709
+ const cur = await this._readLastAnswerTextClean().catch(() => '');
710
+ if (cur && cur.trim() && !(await this._isGenerating())) {
711
+ return cur;
712
+ }
684
713
  // We NO LONGER check the limit over the whole page text — that caused
685
714
  // false positives and 5-minute waits. We just report that
686
715
  // generation did not start.
@@ -718,21 +747,32 @@ export class DeepSeekBrowser {
718
747
  // belongs to the CURRENT send even if the DOM still shows the old text.
719
748
  const isNew = !!cur &&
720
749
  (netFresh || normText(cur) !== normText(beforeText));
721
- if (isNew && cur === last) {
750
+ // An echo/stale answer equals beforeText, so isNew stays false and the
751
+ // old loop waited until the full timeout — the "agent stopped after a
752
+ // tool call" hang. If generation has clearly ENDED (no Stop button) and
753
+ // the text is stable, accept it (even when it repeats the previous one).
754
+ const sameAsBefore = !!cur && !isNew && normText(cur) === normText(beforeText);
755
+ if ((isNew || sameAsBefore) && cur === last) {
722
756
  stable++;
723
- if (stable >= 2)
724
- return cur;
757
+ if (stable >= 2) {
758
+ if (isNew || !(await this._isGenerating()))
759
+ return cur;
760
+ }
725
761
  }
726
762
  else {
727
763
  stable = 0;
728
764
  }
729
- if (isNew)
765
+ if (isNew || sameAsBefore)
730
766
  last = cur;
731
767
  await this.page.waitForTimeout(800);
732
768
  }
733
769
  if (last && (normText(last) !== normText(beforeText) || this._netCapture)) {
734
770
  return last;
735
771
  }
772
+ // Fallback: the turn settled on a text identical to the previous answer.
773
+ if (last && !(await this._isGenerating())) {
774
+ return last;
775
+ }
736
776
  if (this._netCapture && this._netCaptureAt >= this._lastSentAt) {
737
777
  return this._netCapture;
738
778
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "zames_pro",
3
- "version": "2.10.1",
3
+ "version": "2.10.4",
4
4
  "description": "Terminal coding agent over chat.deepseek.com via Playwright",
5
5
  "type": "module",
6
6
  "bin": {