zames_pro 2.9.4 → 2.10.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-loop.js +108 -32
- package/dist/i18n.js +9 -0
- package/dist/system-prompt.js +14 -1
- package/package.json +1 -1
package/dist/agent-loop.js
CHANGED
|
@@ -72,6 +72,16 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
72
72
|
const MAX_MALFORMED_RETRIES = 3;
|
|
73
73
|
let stallRetries = 0;
|
|
74
74
|
const MAX_STALL_RETRIES = 5;
|
|
75
|
+
// SINGLE shared budget for "the answer is not a recognized tool call".
|
|
76
|
+
// Before, every guard had its own counter (3+5+3+5+6 = 22 re-asks), so a
|
|
77
|
+
// stuck answer hung the loop for ~20 iterations and then returned an
|
|
78
|
+
// "iteration limit" stub — the exact "agent stopped" symptom. All the
|
|
79
|
+
// guards below now also bump this counter, and once it is exhausted the
|
|
80
|
+
// loop asks for respond exactly once and then finishes with the model's
|
|
81
|
+
// own text (never a stub).
|
|
82
|
+
let unparsedRetries = 0;
|
|
83
|
+
const MAX_UNPARSED_RETRIES = 4;
|
|
84
|
+
let finalRespondAsked = false;
|
|
75
85
|
// Guard against "the agent stalled": DeepSeek sometimes sends a final text
|
|
76
86
|
// that merely DESCRIBES the next tool call (or cuts the answer off
|
|
77
87
|
// mid-word), and the agent silently finishes the task even though the work
|
|
@@ -80,8 +90,6 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
80
90
|
// chatty model.
|
|
81
91
|
let looksDoneRetries = 0;
|
|
82
92
|
const MAX_LOOKSDONE_RETRIES = 3;
|
|
83
|
-
let plainTextRetries = 0;
|
|
84
|
-
const MAX_PLAINTEXT_RETRIES = 5;
|
|
85
93
|
// Watchdog against the agent emitting a tool call and then going silent.
|
|
86
94
|
// After a tool result the expected next answer is a fresh tool call; if we
|
|
87
95
|
// instead get an EMPTY answer or the EXACT same answer as the previous turn
|
|
@@ -90,11 +98,9 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
90
98
|
let justRanTool = false;
|
|
91
99
|
let watchdogRetries = 0;
|
|
92
100
|
const MAX_WATCHDOG_RETRIES = 3;
|
|
93
|
-
//
|
|
94
|
-
//
|
|
95
|
-
//
|
|
96
|
-
// counter collects all such turns so that a single bad turn never becomes a
|
|
97
|
-
// silent finish: at the limit we warn the operator and log the event.
|
|
101
|
+
// Retry budget for a browser.ask() TIMEOUT (not for content): the send did
|
|
102
|
+
// not come back in time. This is separate from unparsedRetries because a
|
|
103
|
+
// timeout is an infrastructure failure, not a model protocol violation.
|
|
98
104
|
let afterToolRetries = 0;
|
|
99
105
|
const MAX_AFTER_TOOL_RETRIES = 6;
|
|
100
106
|
for (let i = 0; i < maxIterations; i++) {
|
|
@@ -189,6 +195,13 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
189
195
|
}
|
|
190
196
|
lastRaw = rawResponse;
|
|
191
197
|
const parsed = parseToolCall(rawResponse);
|
|
198
|
+
// SILENT CONTRACT: pre-tool text around a valid call is NEVER shown to the
|
|
199
|
+
// operator. It is only handed to onAssistantThought, which is a no-op by
|
|
200
|
+
// default and is not wired to the UI in src/index.ts (so ui.assistant /
|
|
201
|
+
// the terminal never receives "Let me check..." / "Сейчас посмотрю").
|
|
202
|
+
// We cannot strip this text from DeepSeek's own chat output with code —
|
|
203
|
+
// that text is generated by the model. We can only (a) forbid it via the
|
|
204
|
+
// system-prompt ("ONLY TOOL CALLS") and (b) not print it here.
|
|
192
205
|
if (parsed) {
|
|
193
206
|
const thought = extractPreToolText(rawResponse);
|
|
194
207
|
if (thought)
|
|
@@ -209,6 +222,7 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
209
222
|
const looksLikeToolCall = responseLooksLikeToolCall(rawResponse);
|
|
210
223
|
if (looksLikeToolCall && malformedRetries < MAX_MALFORMED_RETRIES) {
|
|
211
224
|
malformedRetries++;
|
|
225
|
+
unparsedRetries++;
|
|
212
226
|
transcript?.log('malformed_toolcall', {
|
|
213
227
|
attempt: malformedRetries,
|
|
214
228
|
response: rawResponse,
|
|
@@ -240,6 +254,7 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
240
254
|
/(messages? too frequent|too many requests|rate limit|server (is )?busy|service (is )?unavailable|слишком часто|try again later)/i.test(trimmed));
|
|
241
255
|
if (looksService && stallRetries < MAX_STALL_RETRIES) {
|
|
242
256
|
stallRetries++;
|
|
257
|
+
unparsedRetries++;
|
|
243
258
|
transcript?.log('stall_retry', {
|
|
244
259
|
attempt: stallRetries,
|
|
245
260
|
response: rawResponse,
|
|
@@ -264,6 +279,7 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
264
279
|
// continue and to actually call a tool this time (or respond if truly done).
|
|
265
280
|
if (looksLikeUnfinishedWork(trimmed) && looksDoneRetries < MAX_LOOKSDONE_RETRIES) {
|
|
266
281
|
looksDoneRetries++;
|
|
282
|
+
unparsedRetries++;
|
|
267
283
|
transcript?.log('unfinished_retry', {
|
|
268
284
|
attempt: looksDoneRetries,
|
|
269
285
|
response: rawResponse,
|
|
@@ -284,54 +300,77 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
284
300
|
continue;
|
|
285
301
|
}
|
|
286
302
|
// STRICT: only tool calls and respond reach the operator. Plain text is
|
|
287
|
-
// a protocol violation
|
|
288
|
-
|
|
289
|
-
|
|
303
|
+
// a protocol violation. The retries are bounded by the SINGLE
|
|
304
|
+
// unparsedRetries budget, so a chatty model cannot hang the loop for
|
|
305
|
+
// 20+ iterations (the old plainTextRetries + afterToolRetries combo).
|
|
306
|
+
if (unparsedRetries < MAX_UNPARSED_RETRIES) {
|
|
307
|
+
unparsedRetries++;
|
|
290
308
|
transcript?.log('plaintext_retry', {
|
|
291
|
-
attempt:
|
|
309
|
+
attempt: unparsedRetries,
|
|
292
310
|
response: rawResponse.slice(0, 500),
|
|
293
311
|
});
|
|
294
312
|
message =
|
|
295
|
-
|
|
296
|
-
'
|
|
297
|
-
'
|
|
313
|
+
(justRanTool
|
|
314
|
+
? 'Ты остановился после вызова инструмента и написал обычный текст. '
|
|
315
|
+
: 'Ты написал обычный текст без вызова инструмента. ') +
|
|
316
|
+
'Задача ещё не завершена. Ответь РОВНО одним JSON-объектом вызова ' +
|
|
317
|
+
'инструмента, без текста до и после, например: ' +
|
|
318
|
+
'{\"tool\": \"Bash\", \"args\": {\"command\": \"...\"}}. ' +
|
|
319
|
+
'Если задача действительно выполнена — вызови respond с итоговым сообщением.';
|
|
298
320
|
continue;
|
|
299
321
|
}
|
|
300
|
-
//
|
|
301
|
-
//
|
|
302
|
-
//
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
transcript?.log('after_tool_retry', {
|
|
307
|
-
attempt: afterToolRetries,
|
|
322
|
+
// Budget exhausted. Ask for respond EXACTLY once more; if the model
|
|
323
|
+
// still does not call it, surface its own text as the final answer
|
|
324
|
+
// (with a single warning) instead of looping to the iteration limit.
|
|
325
|
+
if (!finalRespondAsked) {
|
|
326
|
+
finalRespondAsked = true;
|
|
327
|
+
transcript?.log('final_respond_request', {
|
|
308
328
|
response: rawResponse.slice(0, 500),
|
|
309
329
|
});
|
|
310
330
|
message =
|
|
311
|
-
'
|
|
312
|
-
'
|
|
313
|
-
'
|
|
314
|
-
'{\"tool\": \"Bash\", \"args\": {\"command\": \"...\"}}. ' +
|
|
315
|
-
'Если задача действительно выполнена — вызови respond с итоговым сообщением.';
|
|
331
|
+
'Последний шаг: вызови инструмент respond с итоговым сообщением ' +
|
|
332
|
+
'оператору. Не пиши обычный текст — только вызов respond, например: ' +
|
|
333
|
+
'{\"tool\": \"respond\", \"args\": {\"message\": \"...\"}}';
|
|
316
334
|
continue;
|
|
317
335
|
}
|
|
318
|
-
|
|
336
|
+
const suspiciousFinal = responseLooksLikeToolCall(rawResponse) ||
|
|
337
|
+
looksLikeUnfinishedWork((rawResponse || '').trim()) ||
|
|
338
|
+
!(rawResponse || '').trim();
|
|
339
|
+
if (suspiciousFinal) {
|
|
340
|
+
// The answer LOOKS like a call / promises work / is empty, but could
|
|
341
|
+
// not be turned into a tool call even after all retries: warn the
|
|
342
|
+
// operator instead of silently printing e.g. "Stale. Let me verify".
|
|
319
343
|
transcript?.log('suspicious_final', { response: rawResponse });
|
|
344
|
+
onWarning(translate(locale)('msg.suspicious_stop'));
|
|
320
345
|
}
|
|
346
|
+
// A meaningful plain-text answer (e.g. a final report the model forgot to
|
|
347
|
+
// wrap in respond) is surfaced as-is WITHOUT a warning: after the bounded
|
|
348
|
+
// re-asks it is the best available result, and warning here only
|
|
349
|
+
// confused the operator in earlier sessions.
|
|
321
350
|
transcript?.log('plaintext_final', { message: rawResponse });
|
|
322
|
-
onWarning(translate(locale)('msg.suspicious_stop'));
|
|
323
351
|
return rawResponse;
|
|
324
352
|
}
|
|
325
353
|
const calls = Array.isArray(parsed) ? parsed : [parsed];
|
|
354
|
+
// A respond mixed with real tool calls must NOT short-circuit the tools.
|
|
355
|
+
// DeepSeek sometimes returns [{"tool":"Edit",...},{"tool":"respond",...}]
|
|
356
|
+
// in ONE answer; handling respond first would silently DROP the other
|
|
357
|
+
// call and the agent would look "stopped after a tool call". We only
|
|
358
|
+
// finish on respond when it is the SOLE call in the answer.
|
|
359
|
+
const realCalls = calls.filter((c) => c.tool !== 'respond');
|
|
326
360
|
const respondCall = calls.find((c) => c.tool === 'respond');
|
|
327
|
-
if (respondCall) {
|
|
361
|
+
if (respondCall && realCalls.length > 0) {
|
|
362
|
+
transcript?.log('respond_mixed_with_tools', {
|
|
363
|
+
tools: realCalls.map((c) => c.tool),
|
|
364
|
+
});
|
|
365
|
+
}
|
|
366
|
+
if (respondCall && realCalls.length === 0) {
|
|
328
367
|
const msg = typeof respondCall.args.message === 'string'
|
|
329
368
|
? respondCall.args.message
|
|
330
369
|
: String(respondCall.args.message ?? '');
|
|
331
370
|
// An empty respond is not final: the model called respond but wrote no
|
|
332
371
|
// summary. Finishing like this would show the operator nothing and the
|
|
333
372
|
// task would "hang". We ask it to continue (within stallRetries).
|
|
334
|
-
if (!msg
|
|
373
|
+
if (!isMeaningfulRespond(msg) && stallRetries < MAX_STALL_RETRIES) {
|
|
335
374
|
stallRetries++;
|
|
336
375
|
transcript?.log('empty_respond', { attempt: stallRetries });
|
|
337
376
|
if (debugLog) {
|
|
@@ -347,12 +386,27 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
347
386
|
'продолжи работу вызовом инструмента.';
|
|
348
387
|
continue;
|
|
349
388
|
}
|
|
389
|
+
// An EMPTY respond must NEVER be a silent final: the operator would see
|
|
390
|
+
// nothing and the task would look "stopped after a tool call". If the
|
|
391
|
+
// retry budget is exhausted, warn the operator and keep the run open
|
|
392
|
+
// instead of returning an empty string. Only a NON-empty respond ends
|
|
393
|
+
// the task normally.
|
|
394
|
+
if (!isMeaningfulRespond(msg)) {
|
|
395
|
+
transcript?.log('empty_respond_exhausted', { response: rawResponse });
|
|
396
|
+
onWarning('Модель вызвала respond без текста, и лимит повторов исчерпан. ' +
|
|
397
|
+
'Проверьте чат DeepSeek вручную.');
|
|
398
|
+
return 'Модель не сформировала итоговое сообщение (пустой respond).';
|
|
399
|
+
}
|
|
350
400
|
onAssistantMessage(msg);
|
|
351
401
|
transcript?.log('assistant_final', { message: msg });
|
|
352
402
|
return msg;
|
|
353
403
|
}
|
|
354
404
|
const results = [];
|
|
355
|
-
|
|
405
|
+
// When respond came together with real tools, skip respond here: its
|
|
406
|
+
// message must NOT be delivered before the tools' results are known.
|
|
407
|
+
// The model will get the tool results and can call respond again.
|
|
408
|
+
const callsToRun = realCalls.length > 0 ? realCalls : calls;
|
|
409
|
+
for (const call of callsToRun) {
|
|
356
410
|
const tool = tools.find((t) => t.name === call.tool);
|
|
357
411
|
if (!tool) {
|
|
358
412
|
const err = `Неизвестный инструмент: ${call.tool}`;
|
|
@@ -384,6 +438,10 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
384
438
|
// A fresh tool call just ran: reset the per-tool-result nudge budget so
|
|
385
439
|
// a long chain of tools is not cut off by an earlier bad turn.
|
|
386
440
|
afterToolRetries = 0;
|
|
441
|
+
// Real progress was made, so the "unparsed answer" budget is replenished:
|
|
442
|
+
// a long chain of tools must not run out of it because of earlier hiccups.
|
|
443
|
+
unparsedRetries = 0;
|
|
444
|
+
finalRespondAsked = false;
|
|
387
445
|
if (results.length === 1) {
|
|
388
446
|
const r = results[0];
|
|
389
447
|
const resultStr = typeof r.result === 'string' ? r.result : JSON.stringify(r.result);
|
|
@@ -425,6 +483,24 @@ export function responseLooksLikeToolCall(rawResponse) {
|
|
|
425
483
|
/<\s*\|?\s*(DSML|invoke|parameter)/i.test(raw) ||
|
|
426
484
|
/^\s*\[?\s*\{[^}]*$/.test(raw.trim()));
|
|
427
485
|
}
|
|
486
|
+
// A respond message is only a real FINAL answer when it carries some meaning.
|
|
487
|
+
// DeepSeek sometimes finishes with a placeholder — "...", "-", "ok", "done",
|
|
488
|
+
// "готово" — which looks like a stop with no report. Such a respond must not
|
|
489
|
+
// end the task silently: it is treated like an empty one (re-ask).
|
|
490
|
+
function isMeaningfulRespond(msg) {
|
|
491
|
+
const t = (msg || '').trim();
|
|
492
|
+
if (!t)
|
|
493
|
+
return false;
|
|
494
|
+
// Only punctuation/dots/ellipses: "...", "---", "…", "?" — not a report.
|
|
495
|
+
if (/^[.…–—_*?!/\s-]+$/.test(t))
|
|
496
|
+
return false;
|
|
497
|
+
// A bare acknowledgement with no content at all. NOTE: "ok"/"done"/
|
|
498
|
+
// "готово" are NOT in this list: a short "готово" is a legitimate final
|
|
499
|
+
// answer for a small task, and dropping it re-opened the loop.
|
|
500
|
+
if (/^(na|null|undefined)[.!]*$/i.test(t))
|
|
501
|
+
return false;
|
|
502
|
+
return true;
|
|
503
|
+
}
|
|
428
504
|
// Text that promises a tool call in the future tense but contains no call
|
|
429
505
|
// itself. DeepSeek regularly "hangs" like this: it writes
|
|
430
506
|
// "Now update README to mention …", "Let me run the tests", "I'll check now"
|
package/dist/i18n.js
CHANGED
|
@@ -283,6 +283,15 @@ const CATALOG = {
|
|
|
283
283
|
en: 'IMPORTANT: reply to the operator in English. All text in the respond tool message field, and any explanations, must be in English.',
|
|
284
284
|
},
|
|
285
285
|
'prompt.tools_header': { ru: 'You have access to the following tools:', en: 'You have access to the following tools:' },
|
|
286
|
+
// The hard "ONLY TOOL CALLS" block. All prose around a tool call is a
|
|
287
|
+
// protocol violation: the operator never sees it (only tool calls and the
|
|
288
|
+
// final respond reach the terminal), so it is pure pollution. We cannot
|
|
289
|
+
// stop DeepSeek from generating it INSIDE its chat with code (that is the
|
|
290
|
+
// model's output); we can only forbid it by prompt and hide it here.
|
|
291
|
+
'prompt.only_tool_calls': {
|
|
292
|
+
ru: '## ТОЛЬКО ВЫЗОВЫ ИНСТРУМЕНТОВ (жёсткое правило)\n\nОбщайся с оператором ТОЛЬКО через вызовы инструментов. Любой обычный текст — объяснения, планы, рассуждения, комментарии, извинения, приветствия, заголовки, списки, markdown, эмодзи — ЗАПРЕЩЁН. Он не читается и считается ошибкой.\n\nТВОЙ ЕДИНСТВЕННЫЙ ВЫВОД — вызов инструмента. В каждом ответе ровно один JSON-объект вызова (или массив независимых вызовов), без единого слова до и после.\n\nНЕЛЬЗЯ писать: «сейчас сделаю», «давай посмотрим», «проверю», планы, объяснения, итоги между шагами.\n\nМОЖНО только вызов инструмента и, в самом конце, когда задача выполнена, respond с итогом.\n\nЕдинственное место, где допускается текст, — поле message внутри respond, и только в самом конце.',
|
|
293
|
+
en: '## ONLY TOOL CALLS (hard rule)\n\nTalk to the operator ONLY through tool calls. Any plain text — explanations, plans, reasoning, comments, apologies, greetings, headings, lists, markdown, emoji — is FORBIDDEN. It is not read and counts as an error.\n\nYOUR ONLY OUTPUT is a tool call. Each turn contains exactly one JSON tool-call object (or an array of independent calls), with not a single word before or after.\n\nYou MUST NOT write: "I will now...", "let us look...", "let me check", plans, explanations, progress notes between steps.\n\nALLOWED: only a tool call and, at the very end, when the task is done, respond with the summary.\n\nThe only place where text is allowed is the message field inside respond, and only at the very end.',
|
|
294
|
+
},
|
|
286
295
|
};
|
|
287
296
|
export function translate(locale) {
|
|
288
297
|
const loc = isLocale(locale) ? locale : DEFAULT_LOCALE;
|
package/dist/system-prompt.js
CHANGED
|
@@ -45,7 +45,20 @@ stop. If you are still working, emit a tool call instead.
|
|
|
45
45
|
|
|
46
46
|
So the pattern is: tool call, tool call, tool call, ..., then a single final
|
|
47
47
|
respond. A bare text message without a tool call ends the task and the
|
|
48
|
-
operator will not read it, so never use plain text
|
|
48
|
+
operator will not read it, so never use plain text.
|
|
49
|
+
|
|
50
|
+
${t('prompt.only_tool_calls')}
|
|
51
|
+
|
|
52
|
+
NO PROSE AROUND TOOL CALLS. Each turn must contain ONLY the JSON of the tool
|
|
53
|
+
call(s) — not a single word before or after, not even a short lead-in like
|
|
54
|
+
"Let me check..." or "Now I'll fix it.". The JSON must be the entire response.
|
|
55
|
+
|
|
56
|
+
WRONG: "Let me read the file first." then a Read call.
|
|
57
|
+
WRONG: a Read call then "I'll analyze the result next."
|
|
58
|
+
RIGHT: only the JSON of the tool call, nothing else.
|
|
59
|
+
|
|
60
|
+
If you feel the urge to explain, do not: put it in the final respond when the
|
|
61
|
+
task is done (or when you must ask the operator), not between tool calls.
|
|
49
62
|
|
|
50
63
|
You have access to the following tools:
|
|
51
64
|
|