zames_pro 2.9.2 → 2.9.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-loop.js +112 -16
- package/dist/browser.js +17 -4
- package/package.json +1 -1
package/dist/agent-loop.js
CHANGED
|
@@ -80,6 +80,8 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
80
80
|
// chatty model.
|
|
81
81
|
let looksDoneRetries = 0;
|
|
82
82
|
const MAX_LOOKSDONE_RETRIES = 3;
|
|
83
|
+
let plainTextRetries = 0;
|
|
84
|
+
const MAX_PLAINTEXT_RETRIES = 5;
|
|
83
85
|
// Watchdog against the agent emitting a tool call and then going silent.
|
|
84
86
|
// After a tool result the expected next answer is a fresh tool call; if we
|
|
85
87
|
// instead get an EMPTY answer or the EXACT same answer as the previous turn
|
|
@@ -88,16 +90,64 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
88
90
|
let justRanTool = false;
|
|
89
91
|
let watchdogRetries = 0;
|
|
90
92
|
const MAX_WATCHDOG_RETRIES = 3;
|
|
93
|
+
// After a tool_result the model MUST produce a fresh tool call (or respond).
|
|
94
|
+
// DeepSeek regularly "hangs" right here: the answer comes back empty, or a
|
|
95
|
+
// stale copy of the previous turn, or a fragment that does not parse. This
|
|
96
|
+
// counter collects all such turns so that a single bad turn never becomes a
|
|
97
|
+
// silent finish: at the limit we warn the operator and log the event.
|
|
98
|
+
let afterToolRetries = 0;
|
|
99
|
+
const MAX_AFTER_TOOL_RETRIES = 6;
|
|
91
100
|
for (let i = 0; i < maxIterations; i++) {
|
|
92
101
|
onThinking();
|
|
93
102
|
// The first message (task) is user input: no throttle.
|
|
94
103
|
// Subsequent ones (tool-result and resend requests) are agent sends:
|
|
95
104
|
// throttled so we don't hit the rate limit.
|
|
96
105
|
const isFirst = i === 0;
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
106
|
+
// Safety net: browser.ask() has its own timeout, but a stuck send used to
|
|
107
|
+
// block the whole loop and look like a silent stop. We race it against a
|
|
108
|
+
// hard deadline and treat a timeout as a nudge (re-ask), never as a
|
|
109
|
+
// final answer. The deadline is generous enough for real long answers.
|
|
110
|
+
const askDeadlineMs = 240_000;
|
|
111
|
+
let rawResponse;
|
|
112
|
+
let askTimer = null;
|
|
113
|
+
try {
|
|
114
|
+
rawResponse = await Promise.race([
|
|
115
|
+
browser.ask(message, {
|
|
116
|
+
agent: !isFirst,
|
|
117
|
+
attachments: isFirst ? attachments : [],
|
|
118
|
+
}),
|
|
119
|
+
new Promise((_, reject) => {
|
|
120
|
+
askTimer = setTimeout(() => reject(new Error('ask() watchdog timeout')), askDeadlineMs);
|
|
121
|
+
if (askTimer && typeof askTimer.unref === 'function') {
|
|
122
|
+
askTimer.unref();
|
|
123
|
+
}
|
|
124
|
+
}),
|
|
125
|
+
]);
|
|
126
|
+
if (askTimer)
|
|
127
|
+
clearTimeout(askTimer);
|
|
128
|
+
}
|
|
129
|
+
catch (e) {
|
|
130
|
+
if (askTimer)
|
|
131
|
+
clearTimeout(askTimer);
|
|
132
|
+
transcript?.log('ask_timeout', {
|
|
133
|
+
attempt: afterToolRetries,
|
|
134
|
+
error: e.message,
|
|
135
|
+
});
|
|
136
|
+
onWarning('browser.ask() не вернул ответ за ' +
|
|
137
|
+
Math.round(askDeadlineMs / 1000) +
|
|
138
|
+
'с — повторяю запрос.');
|
|
139
|
+
if (afterToolRetries < MAX_AFTER_TOOL_RETRIES) {
|
|
140
|
+
afterToolRetries++;
|
|
141
|
+
await new Promise((r) => setTimeout(r, 1500));
|
|
142
|
+
continue;
|
|
143
|
+
}
|
|
144
|
+
transcript?.log('ask_timeout_exhausted', {
|
|
145
|
+
message: 'ask() не вернул ответ и лимит повторов исчерпан',
|
|
146
|
+
});
|
|
147
|
+
onWarning('browser.ask() перестал отвечать; лимит повторов исчерпан, ' +
|
|
148
|
+
'останавливаюсь. Ответа модели нет — проверьте чат DeepSeek вручную.');
|
|
149
|
+
return 'ask() watchdog: ответ модели не получен';
|
|
150
|
+
}
|
|
101
151
|
await reportChat();
|
|
102
152
|
transcript?.log('assistant_raw', { response: rawResponse });
|
|
103
153
|
// The user aborted generation (Esc/Ctrl+C).
|
|
@@ -105,19 +155,35 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
105
155
|
transcript?.log('user_aborted');
|
|
106
156
|
return rawResponse;
|
|
107
157
|
}
|
|
108
|
-
// Watchdog: after a tool result we expect a FRESH tool call.
|
|
109
|
-
//
|
|
110
|
-
// sent),
|
|
158
|
+
// Watchdog: after a tool result we expect a FRESH tool call. DeepSeek
|
|
159
|
+
// regularly stops right here; the answer may be (a) empty, (b) an exact
|
|
160
|
+
// copy of the previous turn (the new message was not sent), or (c) a
|
|
161
|
+
// non-empty fragment that parses to nothing and is not a tool call (a
|
|
162
|
+
// cut-off "Stale. Let me ..." or a truncated JSON). All three mean the
|
|
163
|
+
// turn is unfinished: nudge instead of stopping.
|
|
111
164
|
const wdEmpty = !String(rawResponse || '').trim();
|
|
112
|
-
const wdStale = justRanTool &&
|
|
113
|
-
|
|
165
|
+
const wdStale = justRanTool &&
|
|
166
|
+
lastRaw.trim() !== '' &&
|
|
167
|
+
rawResponse.trim() === lastRaw.trim();
|
|
168
|
+
const wdNoCall = justRanTool &&
|
|
169
|
+
!wdEmpty &&
|
|
170
|
+
!wdStale &&
|
|
171
|
+
parseToolCall(rawResponse) === null &&
|
|
172
|
+
!responseLooksLikeToolCall(rawResponse);
|
|
173
|
+
if (!isFirst && (wdEmpty || wdStale || wdNoCall) && watchdogRetries < MAX_WATCHDOG_RETRIES) {
|
|
114
174
|
watchdogRetries++;
|
|
115
175
|
transcript?.log('watchdog_nudge', {
|
|
116
176
|
attempt: watchdogRetries,
|
|
117
177
|
empty: wdEmpty,
|
|
118
178
|
stale: wdStale,
|
|
179
|
+
no_call: wdNoCall,
|
|
119
180
|
response: String(rawResponse || '').slice(0, 200),
|
|
120
181
|
});
|
|
182
|
+
message =
|
|
183
|
+
'Ты остановился после результата инструмента. Продолжи работу: ' +
|
|
184
|
+
'ответь РОВНО одним JSON-объектом вызова инструмента, без текста до и после, ' +
|
|
185
|
+
'например: {"tool": "Bash", "args": {"command": "..."}}. ' +
|
|
186
|
+
'Если задача действительно выполнена — вызови respond с итоговым сообщением.';
|
|
121
187
|
await new Promise((r) => setTimeout(r, 1500));
|
|
122
188
|
continue;
|
|
123
189
|
}
|
|
@@ -217,16 +283,43 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
217
283
|
'сообщением оператору.';
|
|
218
284
|
continue;
|
|
219
285
|
}
|
|
220
|
-
//
|
|
221
|
-
//
|
|
222
|
-
|
|
223
|
-
|
|
286
|
+
// STRICT: only tool calls and respond reach the operator. Plain text is
|
|
287
|
+
// a protocol violation: re-ask for a tool call instead of printing it.
|
|
288
|
+
if (plainTextRetries < MAX_PLAINTEXT_RETRIES) {
|
|
289
|
+
plainTextRetries++;
|
|
290
|
+
transcript?.log('plaintext_retry', {
|
|
291
|
+
attempt: plainTextRetries,
|
|
292
|
+
response: rawResponse.slice(0, 500),
|
|
293
|
+
});
|
|
294
|
+
message =
|
|
295
|
+
'Only call tools. Do not write plain text. ' +
|
|
296
|
+
'If the task is done - call respond with the final message. ' +
|
|
297
|
+
'Otherwise reply with EXACTLY one JSON tool-call object, no text around it.';
|
|
298
|
+
continue;
|
|
299
|
+
}
|
|
300
|
+
// NO SILENT FINISH: we just ran a tool, so the work is NOT done —
|
|
301
|
+
// the model must call another tool or respond. Plain text here is a
|
|
302
|
+
// protocol violation, not a final answer. Re-ask ROWNO one tool-call
|
|
303
|
+
// request (within afterToolRetries) instead of returning to the operator.
|
|
304
|
+
if (justRanTool && afterToolRetries < MAX_AFTER_TOOL_RETRIES) {
|
|
305
|
+
afterToolRetries++;
|
|
306
|
+
transcript?.log('after_tool_retry', {
|
|
307
|
+
attempt: afterToolRetries,
|
|
308
|
+
response: rawResponse.slice(0, 500),
|
|
309
|
+
});
|
|
310
|
+
message =
|
|
311
|
+
'Ты остановился после вызова инструмента и написал обычный текст. ' +
|
|
312
|
+
'Задача ещё не завершена. Ответь РОВНО одним JSON-объектом вызова ' +
|
|
313
|
+
'инструмента, без текста до и после, например: ' +
|
|
314
|
+
'{\"tool\": \"Bash\", \"args\": {\"command\": \"...\"}}. ' +
|
|
315
|
+
'Если задача действительно выполнена — вызови respond с итоговым сообщением.';
|
|
316
|
+
continue;
|
|
317
|
+
}
|
|
224
318
|
if (responseLooksLikeToolCall(rawResponse)) {
|
|
225
319
|
transcript?.log('suspicious_final', { response: rawResponse });
|
|
226
|
-
onWarning(translate(locale)('msg.suspicious_stop'));
|
|
227
320
|
}
|
|
228
|
-
|
|
229
|
-
|
|
321
|
+
transcript?.log('plaintext_final', { message: rawResponse });
|
|
322
|
+
onWarning(translate(locale)('msg.suspicious_stop'));
|
|
230
323
|
return rawResponse;
|
|
231
324
|
}
|
|
232
325
|
const calls = Array.isArray(parsed) ? parsed : [parsed];
|
|
@@ -288,6 +381,9 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
288
381
|
// Reset the watchdog so the next empty/repeated answer is nudged.
|
|
289
382
|
justRanTool = true;
|
|
290
383
|
watchdogRetries = 0;
|
|
384
|
+
// A fresh tool call just ran: reset the per-tool-result nudge budget so
|
|
385
|
+
// a long chain of tools is not cut off by an earlier bad turn.
|
|
386
|
+
afterToolRetries = 0;
|
|
291
387
|
if (results.length === 1) {
|
|
292
388
|
const r = results[0];
|
|
293
389
|
const resultStr = typeof r.result === 'string' ? r.result : JSON.stringify(r.result);
|
package/dist/browser.js
CHANGED
|
@@ -668,7 +668,13 @@ export class DeepSeekBrowser {
|
|
|
668
668
|
// make us think the new answer had started when in fact nothing was sent
|
|
669
669
|
// — and then the agent silently "stopped".
|
|
670
670
|
const changed = cur && normText(cur) !== normText(beforeText);
|
|
671
|
-
|
|
671
|
+
// Network capture with a fresh timestamp is the STRONGEST proof that a
|
|
672
|
+
// new answer started: it is the raw SSE body for the CURRENT send. Right
|
|
673
|
+
// after a tool result the DOM may still show the previous answer, so
|
|
674
|
+
// 'changed' can stay false for a while — without this check ask() used
|
|
675
|
+
// to hang until the full timeout and the agent appeared to "stop".
|
|
676
|
+
const netStarted = !!this._netCapture && this._netCaptureAt >= this._lastSentAt;
|
|
677
|
+
if (changed || netStarted || bodyLen > startBodyLen) {
|
|
672
678
|
started = true;
|
|
673
679
|
break;
|
|
674
680
|
}
|
|
@@ -703,12 +709,15 @@ export class DeepSeekBrowser {
|
|
|
703
709
|
}
|
|
704
710
|
}
|
|
705
711
|
}
|
|
712
|
+
const netFresh = !!this._netCapture && this._netCaptureAt >= this._lastSentAt;
|
|
706
713
|
const cur = await this._readLastAnswerTextClean().catch(() => '');
|
|
707
714
|
// Ignore an "answer" that is identical to what was on the page BEFORE we
|
|
708
715
|
// sent the message: that is the previous answer, not a new one. Returning
|
|
709
716
|
// it would make the agent re-process the old tool call (or silently
|
|
710
|
-
// stop). We keep waiting instead.
|
|
711
|
-
|
|
717
|
+
// stop). We keep waiting instead. A fresh network capture is exempt: it
|
|
718
|
+
// belongs to the CURRENT send even if the DOM still shows the old text.
|
|
719
|
+
const isNew = !!cur &&
|
|
720
|
+
(netFresh || normText(cur) !== normText(beforeText));
|
|
712
721
|
if (isNew && cur === last) {
|
|
713
722
|
stable++;
|
|
714
723
|
if (stable >= 2)
|
|
@@ -721,8 +730,12 @@ export class DeepSeekBrowser {
|
|
|
721
730
|
last = cur;
|
|
722
731
|
await this.page.waitForTimeout(800);
|
|
723
732
|
}
|
|
724
|
-
if (last && normText(last) !== normText(beforeText))
|
|
733
|
+
if (last && (normText(last) !== normText(beforeText) || this._netCapture)) {
|
|
725
734
|
return last;
|
|
735
|
+
}
|
|
736
|
+
if (this._netCapture && this._netCaptureAt >= this._lastSentAt) {
|
|
737
|
+
return this._netCapture;
|
|
738
|
+
}
|
|
726
739
|
throw new Error('Новый ответ не получен (на странице остался прежний текст). ' +
|
|
727
740
|
'Возможно, сообщение не отправилось.');
|
|
728
741
|
}
|