zames_pro 2.9.2 → 2.9.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -80,6 +80,8 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
80
80
  // chatty model.
81
81
  let looksDoneRetries = 0;
82
82
  const MAX_LOOKSDONE_RETRIES = 3;
83
+ let plainTextRetries = 0;
84
+ const MAX_PLAINTEXT_RETRIES = 5;
83
85
  // Watchdog against the agent emitting a tool call and then going silent.
84
86
  // After a tool result the expected next answer is a fresh tool call; if we
85
87
  // instead get an EMPTY answer or the EXACT same answer as the previous turn
@@ -88,16 +90,64 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
88
90
  let justRanTool = false;
89
91
  let watchdogRetries = 0;
90
92
  const MAX_WATCHDOG_RETRIES = 3;
93
+ // After a tool_result the model MUST produce a fresh tool call (or respond).
94
+ // DeepSeek regularly "hangs" right here: the answer comes back empty, or a
95
+ // stale copy of the previous turn, or a fragment that does not parse. This
96
+ // counter collects all such turns so that a single bad turn never becomes a
97
+ // silent finish: at the limit we warn the operator and log the event.
98
+ let afterToolRetries = 0;
99
+ const MAX_AFTER_TOOL_RETRIES = 6;
91
100
  for (let i = 0; i < maxIterations; i++) {
92
101
  onThinking();
93
102
  // The first message (task) is user input: no throttle.
94
103
  // Subsequent ones (tool-result and resend requests) are agent sends:
95
104
  // throttled so we don't hit the rate limit.
96
105
  const isFirst = i === 0;
97
- const rawResponse = await browser.ask(message, {
98
- agent: !isFirst,
99
- attachments: isFirst ? attachments : [],
100
- });
106
+ // Safety net: browser.ask() has its own timeout, but a stuck send used to
107
+ // block the whole loop and look like a silent stop. We race it against a
108
+ // hard deadline and treat a timeout as a nudge (re-ask), never as a
109
+ // final answer. The deadline is generous enough for real long answers.
110
+ const askDeadlineMs = 240_000;
111
+ let rawResponse;
112
+ let askTimer = null;
113
+ try {
114
+ rawResponse = await Promise.race([
115
+ browser.ask(message, {
116
+ agent: !isFirst,
117
+ attachments: isFirst ? attachments : [],
118
+ }),
119
+ new Promise((_, reject) => {
120
+ askTimer = setTimeout(() => reject(new Error('ask() watchdog timeout')), askDeadlineMs);
121
+ if (askTimer && typeof askTimer.unref === 'function') {
122
+ askTimer.unref();
123
+ }
124
+ }),
125
+ ]);
126
+ if (askTimer)
127
+ clearTimeout(askTimer);
128
+ }
129
+ catch (e) {
130
+ if (askTimer)
131
+ clearTimeout(askTimer);
132
+ transcript?.log('ask_timeout', {
133
+ attempt: afterToolRetries,
134
+ error: e.message,
135
+ });
136
+ onWarning('browser.ask() не вернул ответ за ' +
137
+ Math.round(askDeadlineMs / 1000) +
138
+ 'с — повторяю запрос.');
139
+ if (afterToolRetries < MAX_AFTER_TOOL_RETRIES) {
140
+ afterToolRetries++;
141
+ await new Promise((r) => setTimeout(r, 1500));
142
+ continue;
143
+ }
144
+ transcript?.log('ask_timeout_exhausted', {
145
+ message: 'ask() не вернул ответ и лимит повторов исчерпан',
146
+ });
147
+ onWarning('browser.ask() перестал отвечать; лимит повторов исчерпан, ' +
148
+ 'останавливаюсь. Ответа модели нет — проверьте чат DeepSeek вручную.');
149
+ return 'ask() watchdog: ответ модели не получен';
150
+ }
101
151
  await reportChat();
102
152
  transcript?.log('assistant_raw', { response: rawResponse });
103
153
  // The user aborted generation (Esc/Ctrl+C).
@@ -105,19 +155,35 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
105
155
  transcript?.log('user_aborted');
106
156
  return rawResponse;
107
157
  }
108
- // Watchdog: after a tool result we expect a FRESH tool call. If the answer
109
- // is empty or identical to the previous turn (the new message was not
110
- // sent), nudge instead of stopping.
158
+ // Watchdog: after a tool result we expect a FRESH tool call. DeepSeek
159
+ // regularly stops right here; the answer may be (a) empty, (b) an exact
160
+ // copy of the previous turn (the new message was not sent), or (c) a
161
+ // non-empty fragment that parses to nothing and is not a tool call (a
162
+ // cut-off "Stale. Let me ..." or a truncated JSON). All three mean the
163
+ // turn is unfinished: nudge instead of stopping.
111
164
  const wdEmpty = !String(rawResponse || '').trim();
112
- const wdStale = justRanTool && lastRaw.trim() !== '' && rawResponse.trim() === lastRaw.trim();
113
- if (!isFirst && (wdEmpty || wdStale) && watchdogRetries < MAX_WATCHDOG_RETRIES) {
165
+ const wdStale = justRanTool &&
166
+ lastRaw.trim() !== '' &&
167
+ rawResponse.trim() === lastRaw.trim();
168
+ const wdNoCall = justRanTool &&
169
+ !wdEmpty &&
170
+ !wdStale &&
171
+ parseToolCall(rawResponse) === null &&
172
+ !responseLooksLikeToolCall(rawResponse);
173
+ if (!isFirst && (wdEmpty || wdStale || wdNoCall) && watchdogRetries < MAX_WATCHDOG_RETRIES) {
114
174
  watchdogRetries++;
115
175
  transcript?.log('watchdog_nudge', {
116
176
  attempt: watchdogRetries,
117
177
  empty: wdEmpty,
118
178
  stale: wdStale,
179
+ no_call: wdNoCall,
119
180
  response: String(rawResponse || '').slice(0, 200),
120
181
  });
182
+ message =
183
+ 'Ты остановился после результата инструмента. Продолжи работу: ' +
184
+ 'ответь РОВНО одним JSON-объектом вызова инструмента, без текста до и после, ' +
185
+ 'например: {"tool": "Bash", "args": {"command": "..."}}. ' +
186
+ 'Если задача действительно выполнена — вызови respond с итоговым сообщением.';
121
187
  await new Promise((r) => setTimeout(r, 1500));
122
188
  continue;
123
189
  }
@@ -217,16 +283,43 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
217
283
  'сообщением оператору.';
218
284
  continue;
219
285
  }
220
- // All re-ask attempts are exhausted, yet the answer still looks like a
221
- // tool call. Most likely this is a silent stall: we show the operator a
222
- // warning in the terminal (not only in the transcript) so they see the
223
- // problem immediately instead of wondering why the agent stalled.
286
+ // STRICT: only tool calls and respond reach the operator. Plain text is
287
+ // a protocol violation: re-ask for a tool call instead of printing it.
288
+ if (plainTextRetries < MAX_PLAINTEXT_RETRIES) {
289
+ plainTextRetries++;
290
+ transcript?.log('plaintext_retry', {
291
+ attempt: plainTextRetries,
292
+ response: rawResponse.slice(0, 500),
293
+ });
294
+ message =
295
+ 'Only call tools. Do not write plain text. ' +
296
+ 'If the task is done - call respond with the final message. ' +
297
+ 'Otherwise reply with EXACTLY one JSON tool-call object, no text around it.';
298
+ continue;
299
+ }
300
+ // NO SILENT FINISH: we just ran a tool, so the work is NOT done —
301
+ // the model must call another tool or respond. Plain text here is a
302
+ // protocol violation, not a final answer. Re-ask ROWNO one tool-call
303
+ // request (within afterToolRetries) instead of returning to the operator.
304
+ if (justRanTool && afterToolRetries < MAX_AFTER_TOOL_RETRIES) {
305
+ afterToolRetries++;
306
+ transcript?.log('after_tool_retry', {
307
+ attempt: afterToolRetries,
308
+ response: rawResponse.slice(0, 500),
309
+ });
310
+ message =
311
+ 'Ты остановился после вызова инструмента и написал обычный текст. ' +
312
+ 'Задача ещё не завершена. Ответь РОВНО одним JSON-объектом вызова ' +
313
+ 'инструмента, без текста до и после, например: ' +
314
+ '{\"tool\": \"Bash\", \"args\": {\"command\": \"...\"}}. ' +
315
+ 'Если задача действительно выполнена — вызови respond с итоговым сообщением.';
316
+ continue;
317
+ }
224
318
  if (responseLooksLikeToolCall(rawResponse)) {
225
319
  transcript?.log('suspicious_final', { response: rawResponse });
226
- onWarning(translate(locale)('msg.suspicious_stop'));
227
320
  }
228
- onAssistantMessage(rawResponse);
229
- transcript?.log('assistant_final', { message: rawResponse });
321
+ transcript?.log('plaintext_final', { message: rawResponse });
322
+ onWarning(translate(locale)('msg.suspicious_stop'));
230
323
  return rawResponse;
231
324
  }
232
325
  const calls = Array.isArray(parsed) ? parsed : [parsed];
@@ -288,6 +381,9 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
288
381
  // Reset the watchdog so the next empty/repeated answer is nudged.
289
382
  justRanTool = true;
290
383
  watchdogRetries = 0;
384
+ // A fresh tool call just ran: reset the per-tool-result nudge budget so
385
+ // a long chain of tools is not cut off by an earlier bad turn.
386
+ afterToolRetries = 0;
291
387
  if (results.length === 1) {
292
388
  const r = results[0];
293
389
  const resultStr = typeof r.result === 'string' ? r.result : JSON.stringify(r.result);
package/dist/browser.js CHANGED
@@ -668,7 +668,13 @@ export class DeepSeekBrowser {
668
668
  // make us think the new answer had started when in fact nothing was sent
669
669
  // — and then the agent silently "stopped".
670
670
  const changed = cur && normText(cur) !== normText(beforeText);
671
- if (changed || bodyLen > startBodyLen) {
671
+ // Network capture with a fresh timestamp is the STRONGEST proof that a
672
+ // new answer started: it is the raw SSE body for the CURRENT send. Right
673
+ // after a tool result the DOM may still show the previous answer, so
674
+ // 'changed' can stay false for a while — without this check ask() used
675
+ // to hang until the full timeout and the agent appeared to "stop".
676
+ const netStarted = !!this._netCapture && this._netCaptureAt >= this._lastSentAt;
677
+ if (changed || netStarted || bodyLen > startBodyLen) {
672
678
  started = true;
673
679
  break;
674
680
  }
@@ -703,12 +709,15 @@ export class DeepSeekBrowser {
703
709
  }
704
710
  }
705
711
  }
712
+ const netFresh = !!this._netCapture && this._netCaptureAt >= this._lastSentAt;
706
713
  const cur = await this._readLastAnswerTextClean().catch(() => '');
707
714
  // Ignore an "answer" that is identical to what was on the page BEFORE we
708
715
  // sent the message: that is the previous answer, not a new one. Returning
709
716
  // it would make the agent re-process the old tool call (or silently
710
- // stop). We keep waiting instead.
711
- const isNew = cur && normText(cur) !== normText(beforeText);
717
+ // stop). We keep waiting instead. A fresh network capture is exempt: it
718
+ // belongs to the CURRENT send even if the DOM still shows the old text.
719
+ const isNew = !!cur &&
720
+ (netFresh || normText(cur) !== normText(beforeText));
712
721
  if (isNew && cur === last) {
713
722
  stable++;
714
723
  if (stable >= 2)
@@ -721,8 +730,12 @@ export class DeepSeekBrowser {
721
730
  last = cur;
722
731
  await this.page.waitForTimeout(800);
723
732
  }
724
- if (last && normText(last) !== normText(beforeText))
733
+ if (last && (normText(last) !== normText(beforeText) || this._netCapture)) {
725
734
  return last;
735
+ }
736
+ if (this._netCapture && this._netCaptureAt >= this._lastSentAt) {
737
+ return this._netCapture;
738
+ }
726
739
  throw new Error('Новый ответ не получен (на странице остался прежний текст). ' +
727
740
  'Возможно, сообщение не отправилось.');
728
741
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "zames_pro",
3
- "version": "2.9.2",
3
+ "version": "2.9.4",
4
4
  "description": "Terminal coding agent over chat.deepseek.com via Playwright",
5
5
  "type": "module",
6
6
  "bin": {