zames_pro 2.9.3 → 2.9.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -90,16 +90,64 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
90
90
  let justRanTool = false;
91
91
  let watchdogRetries = 0;
92
92
  const MAX_WATCHDOG_RETRIES = 3;
93
+ // After a tool_result the model MUST produce a fresh tool call (or respond).
94
+ // DeepSeek regularly "hangs" right here: the answer comes back empty, or a
95
+ // stale copy of the previous turn, or a fragment that does not parse. This
96
+ // counter collects all such turns so that a single bad turn never becomes a
97
+ // silent finish: at the limit we warn the operator and log the event.
98
+ let afterToolRetries = 0;
99
+ const MAX_AFTER_TOOL_RETRIES = 6;
93
100
  for (let i = 0; i < maxIterations; i++) {
94
101
  onThinking();
95
102
  // The first message (task) is user input: no throttle.
96
103
  // Subsequent ones (tool-result and resend requests) are agent sends:
97
104
  // throttled so we don't hit the rate limit.
98
105
  const isFirst = i === 0;
99
- const rawResponse = await browser.ask(message, {
100
- agent: !isFirst,
101
- attachments: isFirst ? attachments : [],
102
- });
106
+ // Safety net: browser.ask() has its own timeout, but a stuck send used to
107
+ // block the whole loop and look like a silent stop. We race it against a
108
+ // hard deadline and treat a timeout as a nudge (re-ask), never as a
109
+ // final answer. The deadline is generous enough for real long answers.
110
+ const askDeadlineMs = 240_000;
111
+ let rawResponse;
112
+ let askTimer = null;
113
+ try {
114
+ rawResponse = await Promise.race([
115
+ browser.ask(message, {
116
+ agent: !isFirst,
117
+ attachments: isFirst ? attachments : [],
118
+ }),
119
+ new Promise((_, reject) => {
120
+ askTimer = setTimeout(() => reject(new Error('ask() watchdog timeout')), askDeadlineMs);
121
+ if (askTimer && typeof askTimer.unref === 'function') {
122
+ askTimer.unref();
123
+ }
124
+ }),
125
+ ]);
126
+ if (askTimer)
127
+ clearTimeout(askTimer);
128
+ }
129
+ catch (e) {
130
+ if (askTimer)
131
+ clearTimeout(askTimer);
132
+ transcript?.log('ask_timeout', {
133
+ attempt: afterToolRetries,
134
+ error: e.message,
135
+ });
136
+ onWarning('browser.ask() не вернул ответ за ' +
137
+ Math.round(askDeadlineMs / 1000) +
138
+ 'с — повторяю запрос.');
139
+ if (afterToolRetries < MAX_AFTER_TOOL_RETRIES) {
140
+ afterToolRetries++;
141
+ await new Promise((r) => setTimeout(r, 1500));
142
+ continue;
143
+ }
144
+ transcript?.log('ask_timeout_exhausted', {
145
+ message: 'ask() не вернул ответ и лимит повторов исчерпан',
146
+ });
147
+ onWarning('browser.ask() перестал отвечать; лимит повторов исчерпан, ' +
148
+ 'останавливаюсь. Ответа модели нет — проверьте чат DeepSeek вручную.');
149
+ return 'ask() watchdog: ответ модели не получен';
150
+ }
103
151
  await reportChat();
104
152
  transcript?.log('assistant_raw', { response: rawResponse });
105
153
  // The user aborted generation (Esc/Ctrl+C).
@@ -107,19 +155,35 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
107
155
  transcript?.log('user_aborted');
108
156
  return rawResponse;
109
157
  }
110
- // Watchdog: after a tool result we expect a FRESH tool call. If the answer
111
- // is empty or identical to the previous turn (the new message was not
112
- // sent), nudge instead of stopping.
158
+ // Watchdog: after a tool result we expect a FRESH tool call. DeepSeek
159
+ // regularly stops right here; the answer may be (a) empty, (b) an exact
160
+ // copy of the previous turn (the new message was not sent), or (c) a
161
+ // non-empty fragment that parses to nothing and is not a tool call (a
162
+ // cut-off "Stale. Let me ..." or a truncated JSON). All three mean the
163
+ // turn is unfinished: nudge instead of stopping.
113
164
  const wdEmpty = !String(rawResponse || '').trim();
114
- const wdStale = justRanTool && lastRaw.trim() !== '' && rawResponse.trim() === lastRaw.trim();
115
- if (!isFirst && (wdEmpty || wdStale) && watchdogRetries < MAX_WATCHDOG_RETRIES) {
165
+ const wdStale = justRanTool &&
166
+ lastRaw.trim() !== '' &&
167
+ rawResponse.trim() === lastRaw.trim();
168
+ const wdNoCall = justRanTool &&
169
+ !wdEmpty &&
170
+ !wdStale &&
171
+ parseToolCall(rawResponse) === null &&
172
+ !responseLooksLikeToolCall(rawResponse);
173
+ if (!isFirst && (wdEmpty || wdStale || wdNoCall) && watchdogRetries < MAX_WATCHDOG_RETRIES) {
116
174
  watchdogRetries++;
117
175
  transcript?.log('watchdog_nudge', {
118
176
  attempt: watchdogRetries,
119
177
  empty: wdEmpty,
120
178
  stale: wdStale,
179
+ no_call: wdNoCall,
121
180
  response: String(rawResponse || '').slice(0, 200),
122
181
  });
182
+ message =
183
+ 'Ты остановился после результата инструмента. Продолжи работу: ' +
184
+ 'ответь РОВНО одним JSON-объектом вызова инструмента, без текста до и после, ' +
185
+ 'например: {"tool": "Bash", "args": {"command": "..."}}. ' +
186
+ 'Если задача действительно выполнена — вызови respond с итоговым сообщением.';
123
187
  await new Promise((r) => setTimeout(r, 1500));
124
188
  continue;
125
189
  }
@@ -233,6 +297,24 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
233
297
  'Otherwise reply with EXACTLY one JSON tool-call object, no text around it.';
234
298
  continue;
235
299
  }
300
+ // NO SILENT FINISH: we just ran a tool, so the work is NOT done —
301
+ // the model must call another tool or respond. Plain text here is a
302
+ // protocol violation, not a final answer. Re-ask ROWNO one tool-call
303
+ // request (within afterToolRetries) instead of returning to the operator.
304
+ if (justRanTool && afterToolRetries < MAX_AFTER_TOOL_RETRIES) {
305
+ afterToolRetries++;
306
+ transcript?.log('after_tool_retry', {
307
+ attempt: afterToolRetries,
308
+ response: rawResponse.slice(0, 500),
309
+ });
310
+ message =
311
+ 'Ты остановился после вызова инструмента и написал обычный текст. ' +
312
+ 'Задача ещё не завершена. Ответь РОВНО одним JSON-объектом вызова ' +
313
+ 'инструмента, без текста до и после, например: ' +
314
+ '{\"tool\": \"Bash\", \"args\": {\"command\": \"...\"}}. ' +
315
+ 'Если задача действительно выполнена — вызови respond с итоговым сообщением.';
316
+ continue;
317
+ }
236
318
  if (responseLooksLikeToolCall(rawResponse)) {
237
319
  transcript?.log('suspicious_final', { response: rawResponse });
238
320
  }
@@ -299,6 +381,9 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
299
381
  // Reset the watchdog so the next empty/repeated answer is nudged.
300
382
  justRanTool = true;
301
383
  watchdogRetries = 0;
384
+ // A fresh tool call just ran: reset the per-tool-result nudge budget so
385
+ // a long chain of tools is not cut off by an earlier bad turn.
386
+ afterToolRetries = 0;
302
387
  if (results.length === 1) {
303
388
  const r = results[0];
304
389
  const resultStr = typeof r.result === 'string' ? r.result : JSON.stringify(r.result);
package/dist/browser.js CHANGED
@@ -668,7 +668,13 @@ export class DeepSeekBrowser {
668
668
  // make us think the new answer had started when in fact nothing was sent
669
669
  // — and then the agent silently "stopped".
670
670
  const changed = cur && normText(cur) !== normText(beforeText);
671
- if (changed || bodyLen > startBodyLen) {
671
+ // Network capture with a fresh timestamp is the STRONGEST proof that a
672
+ // new answer started: it is the raw SSE body for the CURRENT send. Right
673
+ // after a tool result the DOM may still show the previous answer, so
674
+ // 'changed' can stay false for a while — without this check ask() used
675
+ // to hang until the full timeout and the agent appeared to "stop".
676
+ const netStarted = !!this._netCapture && this._netCaptureAt >= this._lastSentAt;
677
+ if (changed || netStarted || bodyLen > startBodyLen) {
672
678
  started = true;
673
679
  break;
674
680
  }
@@ -703,12 +709,15 @@ export class DeepSeekBrowser {
703
709
  }
704
710
  }
705
711
  }
712
+ const netFresh = !!this._netCapture && this._netCaptureAt >= this._lastSentAt;
706
713
  const cur = await this._readLastAnswerTextClean().catch(() => '');
707
714
  // Ignore an "answer" that is identical to what was on the page BEFORE we
708
715
  // sent the message: that is the previous answer, not a new one. Returning
709
716
  // it would make the agent re-process the old tool call (or silently
710
- // stop). We keep waiting instead.
711
- const isNew = cur && normText(cur) !== normText(beforeText);
717
+ // stop). We keep waiting instead. A fresh network capture is exempt: it
718
+ // belongs to the CURRENT send even if the DOM still shows the old text.
719
+ const isNew = !!cur &&
720
+ (netFresh || normText(cur) !== normText(beforeText));
712
721
  if (isNew && cur === last) {
713
722
  stable++;
714
723
  if (stable >= 2)
@@ -721,8 +730,12 @@ export class DeepSeekBrowser {
721
730
  last = cur;
722
731
  await this.page.waitForTimeout(800);
723
732
  }
724
- if (last && normText(last) !== normText(beforeText))
733
+ if (last && (normText(last) !== normText(beforeText) || this._netCapture)) {
725
734
  return last;
735
+ }
736
+ if (this._netCapture && this._netCaptureAt >= this._lastSentAt) {
737
+ return this._netCapture;
738
+ }
726
739
  throw new Error('Новый ответ не получен (на странице остался прежний текст). ' +
727
740
  'Возможно, сообщение не отправилось.');
728
741
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "zames_pro",
3
- "version": "2.9.3",
3
+ "version": "2.9.4",
4
4
  "description": "Terminal coding agent over chat.deepseek.com via Playwright",
5
5
  "type": "module",
6
6
  "bin": {