zames_pro 2.9.3 → 2.9.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-loop.js +94 -9
- package/dist/browser.js +17 -4
- package/package.json +1 -1
package/dist/agent-loop.js
CHANGED
|
@@ -90,16 +90,64 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
90
90
|
let justRanTool = false;
|
|
91
91
|
let watchdogRetries = 0;
|
|
92
92
|
const MAX_WATCHDOG_RETRIES = 3;
|
|
93
|
+
// After a tool_result the model MUST produce a fresh tool call (or respond).
|
|
94
|
+
// DeepSeek regularly "hangs" right here: the answer comes back empty, or a
|
|
95
|
+
// stale copy of the previous turn, or a fragment that does not parse. This
|
|
96
|
+
// counter collects all such turns so that a single bad turn never becomes a
|
|
97
|
+
// silent finish: at the limit we warn the operator and log the event.
|
|
98
|
+
let afterToolRetries = 0;
|
|
99
|
+
const MAX_AFTER_TOOL_RETRIES = 6;
|
|
93
100
|
for (let i = 0; i < maxIterations; i++) {
|
|
94
101
|
onThinking();
|
|
95
102
|
// The first message (task) is user input: no throttle.
|
|
96
103
|
// Subsequent ones (tool-result and resend requests) are agent sends:
|
|
97
104
|
// throttled so we don't hit the rate limit.
|
|
98
105
|
const isFirst = i === 0;
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
106
|
+
// Safety net: browser.ask() has its own timeout, but a stuck send used to
|
|
107
|
+
// block the whole loop and look like a silent stop. We race it against a
|
|
108
|
+
// hard deadline and treat a timeout as a nudge (re-ask), never as a
|
|
109
|
+
// final answer. The deadline is generous enough for real long answers.
|
|
110
|
+
const askDeadlineMs = 240_000;
|
|
111
|
+
let rawResponse;
|
|
112
|
+
let askTimer = null;
|
|
113
|
+
try {
|
|
114
|
+
rawResponse = await Promise.race([
|
|
115
|
+
browser.ask(message, {
|
|
116
|
+
agent: !isFirst,
|
|
117
|
+
attachments: isFirst ? attachments : [],
|
|
118
|
+
}),
|
|
119
|
+
new Promise((_, reject) => {
|
|
120
|
+
askTimer = setTimeout(() => reject(new Error('ask() watchdog timeout')), askDeadlineMs);
|
|
121
|
+
if (askTimer && typeof askTimer.unref === 'function') {
|
|
122
|
+
askTimer.unref();
|
|
123
|
+
}
|
|
124
|
+
}),
|
|
125
|
+
]);
|
|
126
|
+
if (askTimer)
|
|
127
|
+
clearTimeout(askTimer);
|
|
128
|
+
}
|
|
129
|
+
catch (e) {
|
|
130
|
+
if (askTimer)
|
|
131
|
+
clearTimeout(askTimer);
|
|
132
|
+
transcript?.log('ask_timeout', {
|
|
133
|
+
attempt: afterToolRetries,
|
|
134
|
+
error: e.message,
|
|
135
|
+
});
|
|
136
|
+
onWarning('browser.ask() не вернул ответ за ' +
|
|
137
|
+
Math.round(askDeadlineMs / 1000) +
|
|
138
|
+
'с — повторяю запрос.');
|
|
139
|
+
if (afterToolRetries < MAX_AFTER_TOOL_RETRIES) {
|
|
140
|
+
afterToolRetries++;
|
|
141
|
+
await new Promise((r) => setTimeout(r, 1500));
|
|
142
|
+
continue;
|
|
143
|
+
}
|
|
144
|
+
transcript?.log('ask_timeout_exhausted', {
|
|
145
|
+
message: 'ask() не вернул ответ и лимит повторов исчерпан',
|
|
146
|
+
});
|
|
147
|
+
onWarning('browser.ask() перестал отвечать; лимит повторов исчерпан, ' +
|
|
148
|
+
'останавливаюсь. Ответа модели нет — проверьте чат DeepSeek вручную.');
|
|
149
|
+
return 'ask() watchdog: ответ модели не получен';
|
|
150
|
+
}
|
|
103
151
|
await reportChat();
|
|
104
152
|
transcript?.log('assistant_raw', { response: rawResponse });
|
|
105
153
|
// The user aborted generation (Esc/Ctrl+C).
|
|
@@ -107,19 +155,35 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
107
155
|
transcript?.log('user_aborted');
|
|
108
156
|
return rawResponse;
|
|
109
157
|
}
|
|
110
|
-
// Watchdog: after a tool result we expect a FRESH tool call.
|
|
111
|
-
//
|
|
112
|
-
// sent),
|
|
158
|
+
// Watchdog: after a tool result we expect a FRESH tool call. DeepSeek
|
|
159
|
+
// regularly stops right here; the answer may be (a) empty, (b) an exact
|
|
160
|
+
// copy of the previous turn (the new message was not sent), or (c) a
|
|
161
|
+
// non-empty fragment that parses to nothing and is not a tool call (a
|
|
162
|
+
// cut-off "Stale. Let me ..." or a truncated JSON). All three mean the
|
|
163
|
+
// turn is unfinished: nudge instead of stopping.
|
|
113
164
|
const wdEmpty = !String(rawResponse || '').trim();
|
|
114
|
-
const wdStale = justRanTool &&
|
|
115
|
-
|
|
165
|
+
const wdStale = justRanTool &&
|
|
166
|
+
lastRaw.trim() !== '' &&
|
|
167
|
+
rawResponse.trim() === lastRaw.trim();
|
|
168
|
+
const wdNoCall = justRanTool &&
|
|
169
|
+
!wdEmpty &&
|
|
170
|
+
!wdStale &&
|
|
171
|
+
parseToolCall(rawResponse) === null &&
|
|
172
|
+
!responseLooksLikeToolCall(rawResponse);
|
|
173
|
+
if (!isFirst && (wdEmpty || wdStale || wdNoCall) && watchdogRetries < MAX_WATCHDOG_RETRIES) {
|
|
116
174
|
watchdogRetries++;
|
|
117
175
|
transcript?.log('watchdog_nudge', {
|
|
118
176
|
attempt: watchdogRetries,
|
|
119
177
|
empty: wdEmpty,
|
|
120
178
|
stale: wdStale,
|
|
179
|
+
no_call: wdNoCall,
|
|
121
180
|
response: String(rawResponse || '').slice(0, 200),
|
|
122
181
|
});
|
|
182
|
+
message =
|
|
183
|
+
'Ты остановился после результата инструмента. Продолжи работу: ' +
|
|
184
|
+
'ответь РОВНО одним JSON-объектом вызова инструмента, без текста до и после, ' +
|
|
185
|
+
'например: {"tool": "Bash", "args": {"command": "..."}}. ' +
|
|
186
|
+
'Если задача действительно выполнена — вызови respond с итоговым сообщением.';
|
|
123
187
|
await new Promise((r) => setTimeout(r, 1500));
|
|
124
188
|
continue;
|
|
125
189
|
}
|
|
@@ -233,6 +297,24 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
233
297
|
'Otherwise reply with EXACTLY one JSON tool-call object, no text around it.';
|
|
234
298
|
continue;
|
|
235
299
|
}
|
|
300
|
+
// NO SILENT FINISH: we just ran a tool, so the work is NOT done —
|
|
301
|
+
// the model must call another tool or respond. Plain text here is a
|
|
302
|
+
// protocol violation, not a final answer. Re-ask ROWNO one tool-call
|
|
303
|
+
// request (within afterToolRetries) instead of returning to the operator.
|
|
304
|
+
if (justRanTool && afterToolRetries < MAX_AFTER_TOOL_RETRIES) {
|
|
305
|
+
afterToolRetries++;
|
|
306
|
+
transcript?.log('after_tool_retry', {
|
|
307
|
+
attempt: afterToolRetries,
|
|
308
|
+
response: rawResponse.slice(0, 500),
|
|
309
|
+
});
|
|
310
|
+
message =
|
|
311
|
+
'Ты остановился после вызова инструмента и написал обычный текст. ' +
|
|
312
|
+
'Задача ещё не завершена. Ответь РОВНО одним JSON-объектом вызова ' +
|
|
313
|
+
'инструмента, без текста до и после, например: ' +
|
|
314
|
+
'{\"tool\": \"Bash\", \"args\": {\"command\": \"...\"}}. ' +
|
|
315
|
+
'Если задача действительно выполнена — вызови respond с итоговым сообщением.';
|
|
316
|
+
continue;
|
|
317
|
+
}
|
|
236
318
|
if (responseLooksLikeToolCall(rawResponse)) {
|
|
237
319
|
transcript?.log('suspicious_final', { response: rawResponse });
|
|
238
320
|
}
|
|
@@ -299,6 +381,9 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
299
381
|
// Reset the watchdog so the next empty/repeated answer is nudged.
|
|
300
382
|
justRanTool = true;
|
|
301
383
|
watchdogRetries = 0;
|
|
384
|
+
// A fresh tool call just ran: reset the per-tool-result nudge budget so
|
|
385
|
+
// a long chain of tools is not cut off by an earlier bad turn.
|
|
386
|
+
afterToolRetries = 0;
|
|
302
387
|
if (results.length === 1) {
|
|
303
388
|
const r = results[0];
|
|
304
389
|
const resultStr = typeof r.result === 'string' ? r.result : JSON.stringify(r.result);
|
package/dist/browser.js
CHANGED
|
@@ -668,7 +668,13 @@ export class DeepSeekBrowser {
|
|
|
668
668
|
// make us think the new answer had started when in fact nothing was sent
|
|
669
669
|
// — and then the agent silently "stopped".
|
|
670
670
|
const changed = cur && normText(cur) !== normText(beforeText);
|
|
671
|
-
|
|
671
|
+
// Network capture with a fresh timestamp is the STRONGEST proof that a
|
|
672
|
+
// new answer started: it is the raw SSE body for the CURRENT send. Right
|
|
673
|
+
// after a tool result the DOM may still show the previous answer, so
|
|
674
|
+
// 'changed' can stay false for a while — without this check ask() used
|
|
675
|
+
// to hang until the full timeout and the agent appeared to "stop".
|
|
676
|
+
const netStarted = !!this._netCapture && this._netCaptureAt >= this._lastSentAt;
|
|
677
|
+
if (changed || netStarted || bodyLen > startBodyLen) {
|
|
672
678
|
started = true;
|
|
673
679
|
break;
|
|
674
680
|
}
|
|
@@ -703,12 +709,15 @@ export class DeepSeekBrowser {
|
|
|
703
709
|
}
|
|
704
710
|
}
|
|
705
711
|
}
|
|
712
|
+
const netFresh = !!this._netCapture && this._netCaptureAt >= this._lastSentAt;
|
|
706
713
|
const cur = await this._readLastAnswerTextClean().catch(() => '');
|
|
707
714
|
// Ignore an "answer" that is identical to what was on the page BEFORE we
|
|
708
715
|
// sent the message: that is the previous answer, not a new one. Returning
|
|
709
716
|
// it would make the agent re-process the old tool call (or silently
|
|
710
|
-
// stop). We keep waiting instead.
|
|
711
|
-
|
|
717
|
+
// stop). We keep waiting instead. A fresh network capture is exempt: it
|
|
718
|
+
// belongs to the CURRENT send even if the DOM still shows the old text.
|
|
719
|
+
const isNew = !!cur &&
|
|
720
|
+
(netFresh || normText(cur) !== normText(beforeText));
|
|
712
721
|
if (isNew && cur === last) {
|
|
713
722
|
stable++;
|
|
714
723
|
if (stable >= 2)
|
|
@@ -721,8 +730,12 @@ export class DeepSeekBrowser {
|
|
|
721
730
|
last = cur;
|
|
722
731
|
await this.page.waitForTimeout(800);
|
|
723
732
|
}
|
|
724
|
-
if (last && normText(last) !== normText(beforeText))
|
|
733
|
+
if (last && (normText(last) !== normText(beforeText) || this._netCapture)) {
|
|
725
734
|
return last;
|
|
735
|
+
}
|
|
736
|
+
if (this._netCapture && this._netCaptureAt >= this._lastSentAt) {
|
|
737
|
+
return this._netCapture;
|
|
738
|
+
}
|
|
726
739
|
throw new Error('Новый ответ не получен (на странице остался прежний текст). ' +
|
|
727
740
|
'Возможно, сообщение не отправилось.');
|
|
728
741
|
}
|