zames_pro 2.9.4 → 2.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -72,6 +72,16 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
72
72
  const MAX_MALFORMED_RETRIES = 3;
73
73
  let stallRetries = 0;
74
74
  const MAX_STALL_RETRIES = 5;
75
+ // SINGLE shared budget for "the answer is not a recognized tool call".
76
+ // Before, every guard had its own counter (3+5+3+5+6 = 22 re-asks), so a
77
+ // stuck answer hung the loop for ~20 iterations and then returned an
78
+ // "iteration limit" stub — the exact "agent stopped" symptom. All the
79
+ // guards below now also bump this counter, and once it is exhausted the
80
+ // loop asks for respond exactly once and then finishes with the model's
81
+ // own text (never a stub).
82
+ let unparsedRetries = 0;
83
+ const MAX_UNPARSED_RETRIES = 4;
84
+ let finalRespondAsked = false;
75
85
  // Guard against "the agent stalled": DeepSeek sometimes sends a final text
76
86
  // that merely DESCRIBES the next tool call (or cuts the answer off
77
87
  // mid-word), and the agent silently finishes the task even though the work
@@ -80,8 +90,6 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
80
90
  // chatty model.
81
91
  let looksDoneRetries = 0;
82
92
  const MAX_LOOKSDONE_RETRIES = 3;
83
- let plainTextRetries = 0;
84
- const MAX_PLAINTEXT_RETRIES = 5;
85
93
  // Watchdog against the agent emitting a tool call and then going silent.
86
94
  // After a tool result the expected next answer is a fresh tool call; if we
87
95
  // instead get an EMPTY answer or the EXACT same answer as the previous turn
@@ -90,11 +98,9 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
90
98
  let justRanTool = false;
91
99
  let watchdogRetries = 0;
92
100
  const MAX_WATCHDOG_RETRIES = 3;
93
- // After a tool_result the model MUST produce a fresh tool call (or respond).
94
- // DeepSeek regularly "hangs" right here: the answer comes back empty, or a
95
- // stale copy of the previous turn, or a fragment that does not parse. This
96
- // counter collects all such turns so that a single bad turn never becomes a
97
- // silent finish: at the limit we warn the operator and log the event.
101
+ // Retry budget for a browser.ask() TIMEOUT (not for content): the send did
102
+ // not come back in time. This is separate from unparsedRetries because a
103
+ // timeout is an infrastructure failure, not a model protocol violation.
98
104
  let afterToolRetries = 0;
99
105
  const MAX_AFTER_TOOL_RETRIES = 6;
100
106
  for (let i = 0; i < maxIterations; i++) {
@@ -189,6 +195,13 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
189
195
  }
190
196
  lastRaw = rawResponse;
191
197
  const parsed = parseToolCall(rawResponse);
198
+ // SILENT CONTRACT: pre-tool text around a valid call is NEVER shown to the
199
+ // operator. It is only handed to onAssistantThought, which is a no-op by
200
+ // default and is not wired to the UI in src/index.ts (so ui.assistant /
201
+ // the terminal never receives "Let me check..." / "Сейчас посмотрю").
202
+ // We cannot strip this text from DeepSeek's own chat output with code —
203
+ // that text is generated by the model. We can only (a) forbid it via the
204
+ // system-prompt ("ONLY TOOL CALLS") and (b) not print it here.
192
205
  if (parsed) {
193
206
  const thought = extractPreToolText(rawResponse);
194
207
  if (thought)
@@ -209,6 +222,7 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
209
222
  const looksLikeToolCall = responseLooksLikeToolCall(rawResponse);
210
223
  if (looksLikeToolCall && malformedRetries < MAX_MALFORMED_RETRIES) {
211
224
  malformedRetries++;
225
+ unparsedRetries++;
212
226
  transcript?.log('malformed_toolcall', {
213
227
  attempt: malformedRetries,
214
228
  response: rawResponse,
@@ -240,6 +254,7 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
240
254
  /(messages? too frequent|too many requests|rate limit|server (is )?busy|service (is )?unavailable|слишком часто|try again later)/i.test(trimmed));
241
255
  if (looksService && stallRetries < MAX_STALL_RETRIES) {
242
256
  stallRetries++;
257
+ unparsedRetries++;
243
258
  transcript?.log('stall_retry', {
244
259
  attempt: stallRetries,
245
260
  response: rawResponse,
@@ -264,6 +279,7 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
264
279
  // continue and to actually call a tool this time (or respond if truly done).
265
280
  if (looksLikeUnfinishedWork(trimmed) && looksDoneRetries < MAX_LOOKSDONE_RETRIES) {
266
281
  looksDoneRetries++;
282
+ unparsedRetries++;
267
283
  transcript?.log('unfinished_retry', {
268
284
  attempt: looksDoneRetries,
269
285
  response: rawResponse,
@@ -284,54 +300,75 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
284
300
  continue;
285
301
  }
286
302
  // STRICT: only tool calls and respond reach the operator. Plain text is
287
- // a protocol violation: re-ask for a tool call instead of printing it.
288
- if (plainTextRetries < MAX_PLAINTEXT_RETRIES) {
289
- plainTextRetries++;
303
+ // a protocol violation. The retries are bounded by the SINGLE
304
+ // unparsedRetries budget, so a chatty model cannot hang the loop for
305
+ // 20+ iterations (the old plainTextRetries + afterToolRetries combo).
306
+ if (unparsedRetries < MAX_UNPARSED_RETRIES) {
307
+ unparsedRetries++;
290
308
  transcript?.log('plaintext_retry', {
291
- attempt: plainTextRetries,
309
+ attempt: unparsedRetries,
292
310
  response: rawResponse.slice(0, 500),
293
311
  });
294
312
  message =
295
- 'Only call tools. Do not write plain text. ' +
296
- 'If the task is done - call respond with the final message. ' +
297
- 'Otherwise reply with EXACTLY one JSON tool-call object, no text around it.';
313
+ (justRanTool
314
+ ? 'Ты остановился после вызова инструмента и написал обычный текст. '
315
+ : 'Ты написал обычный текст без вызова инструмента. ') +
316
+ 'Задача ещё не завершена. Ответь РОВНО одним JSON-объектом вызова ' +
317
+ 'инструмента, без текста до и после, например: ' +
318
+ '{\"tool\": \"Bash\", \"args\": {\"command\": \"...\"}}. ' +
319
+ 'Если задача действительно выполнена — вызови respond с итоговым сообщением.';
298
320
  continue;
299
321
  }
300
- // NO SILENT FINISH: we just ran a tool, so the work is NOT done —
301
- // the model must call another tool or respond. Plain text here is a
302
- // protocol violation, not a final answer. Re-ask ROWNO one tool-call
303
- // request (within afterToolRetries) instead of returning to the operator.
304
- if (justRanTool && afterToolRetries < MAX_AFTER_TOOL_RETRIES) {
305
- afterToolRetries++;
306
- transcript?.log('after_tool_retry', {
307
- attempt: afterToolRetries,
322
+ // Budget exhausted. Ask for respond EXACTLY once more; if the model
323
+ // still does not call it, surface its own text as the final answer
324
+ // (with a single warning) instead of looping to the iteration limit.
325
+ if (!finalRespondAsked) {
326
+ finalRespondAsked = true;
327
+ transcript?.log('final_respond_request', {
308
328
  response: rawResponse.slice(0, 500),
309
329
  });
310
330
  message =
311
- 'Ты остановился после вызова инструмента и написал обычный текст. ' +
312
- 'Задача ещё не завершена. Ответь РОВНО одним JSON-объектом вызова ' +
313
- 'инструмента, без текста до и после, например: ' +
314
- '{\"tool\": \"Bash\", \"args\": {\"command\": \"...\"}}. ' +
315
- 'Если задача действительно выполнена — вызови respond с итоговым сообщением.';
331
+ 'Последний шаг: вызови инструмент respond с итоговым сообщением ' +
332
+ 'оператору. Не пиши обычный текст — только вызов respond, например: ' +
333
+ '{\"tool\": \"respond\", \"args\": {\"message\": \"...\"}}';
316
334
  continue;
317
335
  }
318
- if (responseLooksLikeToolCall(rawResponse)) {
336
+ const suspiciousFinal = responseLooksLikeToolCall(rawResponse) ||
337
+ !(rawResponse || '').trim();
338
+ if (suspiciousFinal) {
339
+ // The answer LOOKS like a call (or is empty) but could not be parsed
340
+ // even after all retries: warn the operator.
319
341
  transcript?.log('suspicious_final', { response: rawResponse });
342
+ onWarning(translate(locale)('msg.suspicious_stop'));
320
343
  }
344
+ // A meaningful plain-text answer (e.g. a final report the model forgot to
345
+ // wrap in respond) is surfaced as-is WITHOUT a warning: after the bounded
346
+ // re-asks it is the best available result, and warning here only
347
+ // confused the operator in earlier sessions.
321
348
  transcript?.log('plaintext_final', { message: rawResponse });
322
- onWarning(translate(locale)('msg.suspicious_stop'));
323
349
  return rawResponse;
324
350
  }
325
351
  const calls = Array.isArray(parsed) ? parsed : [parsed];
352
+ // A respond mixed with real tool calls must NOT short-circuit the tools.
353
+ // DeepSeek sometimes returns [{"tool":"Edit",...},{"tool":"respond",...}]
354
+ // in ONE answer; handling respond first would silently DROP the other
355
+ // call and the agent would look "stopped after a tool call". We only
356
+ // finish on respond when it is the SOLE call in the answer.
357
+ const realCalls = calls.filter((c) => c.tool !== 'respond');
326
358
  const respondCall = calls.find((c) => c.tool === 'respond');
327
- if (respondCall) {
359
+ if (respondCall && realCalls.length > 0) {
360
+ transcript?.log('respond_mixed_with_tools', {
361
+ tools: realCalls.map((c) => c.tool),
362
+ });
363
+ }
364
+ if (respondCall && realCalls.length === 0) {
328
365
  const msg = typeof respondCall.args.message === 'string'
329
366
  ? respondCall.args.message
330
367
  : String(respondCall.args.message ?? '');
331
368
  // An empty respond is not final: the model called respond but wrote no
332
369
  // summary. Finishing like this would show the operator nothing and the
333
370
  // task would "hang". We ask it to continue (within stallRetries).
334
- if (!msg.trim() && stallRetries < MAX_STALL_RETRIES) {
371
+ if (!isMeaningfulRespond(msg) && stallRetries < MAX_STALL_RETRIES) {
335
372
  stallRetries++;
336
373
  transcript?.log('empty_respond', { attempt: stallRetries });
337
374
  if (debugLog) {
@@ -347,12 +384,27 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
347
384
  'продолжи работу вызовом инструмента.';
348
385
  continue;
349
386
  }
387
+ // An EMPTY respond must NEVER be a silent final: the operator would see
388
+ // nothing and the task would look "stopped after a tool call". If the
389
+ // retry budget is exhausted, warn the operator and keep the run open
390
+ // instead of returning an empty string. Only a NON-empty respond ends
391
+ // the task normally.
392
+ if (!isMeaningfulRespond(msg)) {
393
+ transcript?.log('empty_respond_exhausted', { response: rawResponse });
394
+ onWarning('Модель вызвала respond без текста, и лимит повторов исчерпан. ' +
395
+ 'Проверьте чат DeepSeek вручную.');
396
+ return 'Модель не сформировала итоговое сообщение (пустой respond).';
397
+ }
350
398
  onAssistantMessage(msg);
351
399
  transcript?.log('assistant_final', { message: msg });
352
400
  return msg;
353
401
  }
354
402
  const results = [];
355
- for (const call of calls) {
403
+ // When respond came together with real tools, skip respond here: its
404
+ // message must NOT be delivered before the tools' results are known.
405
+ // The model will get the tool results and can call respond again.
406
+ const callsToRun = realCalls.length > 0 ? realCalls : calls;
407
+ for (const call of callsToRun) {
356
408
  const tool = tools.find((t) => t.name === call.tool);
357
409
  if (!tool) {
358
410
  const err = `Неизвестный инструмент: ${call.tool}`;
@@ -384,6 +436,10 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
384
436
  // A fresh tool call just ran: reset the per-tool-result nudge budget so
385
437
  // a long chain of tools is not cut off by an earlier bad turn.
386
438
  afterToolRetries = 0;
439
+ // Real progress was made, so the "unparsed answer" budget is replenished:
440
+ // a long chain of tools must not run out of it because of earlier hiccups.
441
+ unparsedRetries = 0;
442
+ finalRespondAsked = false;
387
443
  if (results.length === 1) {
388
444
  const r = results[0];
389
445
  const resultStr = typeof r.result === 'string' ? r.result : JSON.stringify(r.result);
@@ -425,6 +481,24 @@ export function responseLooksLikeToolCall(rawResponse) {
425
481
  /<\s*\|?\s*(DSML|invoke|parameter)/i.test(raw) ||
426
482
  /^\s*\[?\s*\{[^}]*$/.test(raw.trim()));
427
483
  }
484
+ // A respond message is only a real FINAL answer when it carries some meaning.
485
+ // DeepSeek sometimes finishes with a placeholder — "...", "-", "ok", "done",
486
+ // "готово" — which looks like a stop with no report. Such a respond must not
487
+ // end the task silently: it is treated like an empty one (re-ask).
488
+ function isMeaningfulRespond(msg) {
489
+ const t = (msg || '').trim();
490
+ if (!t)
491
+ return false;
492
+ // Only punctuation/dots/ellipses: "...", "---", "…", "?" — not a report.
493
+ if (/^[.…–—_*?!/\s-]+$/.test(t))
494
+ return false;
495
+ // A bare acknowledgement with no content at all. NOTE: "ok"/"done"/
496
+ // "готово" are NOT in this list: a short "готово" is a legitimate final
497
+ // answer for a small task, and dropping it re-opened the loop.
498
+ if (/^(na|null|undefined)[.!]*$/i.test(t))
499
+ return false;
500
+ return true;
501
+ }
428
502
  // Text that promises a tool call in the future tense but contains no call
429
503
  // itself. DeepSeek regularly "hangs" like this: it writes
430
504
  // "Now update README to mention …", "Let me run the tests", "I'll check now"
package/dist/i18n.js CHANGED
@@ -283,6 +283,15 @@ const CATALOG = {
283
283
  en: 'IMPORTANT: reply to the operator in English. All text in the respond tool message field, and any explanations, must be in English.',
284
284
  },
285
285
  'prompt.tools_header': { ru: 'You have access to the following tools:', en: 'You have access to the following tools:' },
286
+ // The hard "ONLY TOOL CALLS" block. All prose around a tool call is a
287
+ // protocol violation: the operator never sees it (only tool calls and the
288
+ // final respond reach the terminal), so it is pure pollution. We cannot
289
+ // stop DeepSeek from generating it INSIDE its chat with code (that is the
290
+ // model's output); we can only forbid it by prompt and hide it here.
291
+ 'prompt.only_tool_calls': {
292
+ ru: '## ТОЛЬКО ВЫЗОВЫ ИНСТРУМЕНТОВ (жёсткое правило)\n\nОбщайся с оператором ТОЛЬКО через вызовы инструментов. Любой обычный текст — объяснения, планы, рассуждения, комментарии, извинения, приветствия, заголовки, списки, markdown, эмодзи — ЗАПРЕЩЁН. Он не читается и считается ошибкой.\n\nТВОЙ ЕДИНСТВЕННЫЙ ВЫВОД — вызов инструмента. В каждом ответе ровно один JSON-объект вызова (или массив независимых вызовов), без единого слова до и после.\n\nНЕЛЬЗЯ писать: «сейчас сделаю», «давай посмотрим», «проверю», планы, объяснения, итоги между шагами.\n\nМОЖНО только вызов инструмента и, в самом конце, когда задача выполнена, respond с итогом.\n\nЕдинственное место, где допускается текст, — поле message внутри respond, и только в самом конце.',
293
+ en: '## ONLY TOOL CALLS (hard rule)\n\nTalk to the operator ONLY through tool calls. Any plain text — explanations, plans, reasoning, comments, apologies, greetings, headings, lists, markdown, emoji — is FORBIDDEN. It is not read and counts as an error.\n\nYOUR ONLY OUTPUT is a tool call. Each turn contains exactly one JSON tool-call object (or an array of independent calls), with not a single word before or after.\n\nYou MUST NOT write: "I will now...", "let us look...", "let me check", plans, explanations, progress notes between steps.\n\nALLOWED: only a tool call and, at the very end, when the task is done, respond with the summary.\n\nThe only place where text is allowed is the message field inside respond, and only at the very end.',
294
+ },
286
295
  };
287
296
  export function translate(locale) {
288
297
  const loc = isLocale(locale) ? locale : DEFAULT_LOCALE;
@@ -45,7 +45,20 @@ stop. If you are still working, emit a tool call instead.
45
45
 
46
46
  So the pattern is: tool call, tool call, tool call, ..., then a single final
47
47
  respond. A bare text message without a tool call ends the task and the
48
- operator will not read it, so never use plain text.\n\nNO PROSE AROUND TOOL CALLS. Each turn must contain ONLY the JSON of the tool\ncall(s) — not a single word before or after, not even a short lead-in like\n"Let me check..." or "Now I'll fix it.". The JSON must be the entire response.\n\nWRONG: "Let me read the file first." then a Read call.\nWRONG: a Read call then "I'll analyze the result next."\nRIGHT: only the JSON of the tool call, nothing else.\n\nIf you feel the urge to explain, do not: put it in the final respond when the\ntask is done (or when you must ask the operator), not between tool calls.
48
+ operator will not read it, so never use plain text.
49
+
50
+ ${t('prompt.only_tool_calls')}
51
+
52
+ NO PROSE AROUND TOOL CALLS. Each turn must contain ONLY the JSON of the tool
53
+ call(s) — not a single word before or after, not even a short lead-in like
54
+ "Let me check..." or "Now I'll fix it.". The JSON must be the entire response.
55
+
56
+ WRONG: "Let me read the file first." then a Read call.
57
+ WRONG: a Read call then "I'll analyze the result next."
58
+ RIGHT: only the JSON of the tool call, nothing else.
59
+
60
+ If you feel the urge to explain, do not: put it in the final respond when the
61
+ task is done (or when you must ask the operator), not between tool calls.
49
62
 
50
63
  You have access to the following tools:
51
64
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "zames_pro",
3
- "version": "2.9.4",
3
+ "version": "2.10.0",
4
4
  "description": "Terminal coding agent over chat.deepseek.com via Playwright",
5
5
  "type": "module",
6
6
  "bin": {