zames_pro 2.10.1 → 2.10.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-loop.js +52 -12
- package/dist/browser.js +47 -7
- package/package.json +1 -1
package/dist/agent-loop.js
CHANGED
|
@@ -2,7 +2,29 @@ import { buildSystemPrompt } from './system-prompt.js';
|
|
|
2
2
|
import { getGitContext, formatGitContext } from './gitTools.js';
|
|
3
3
|
import { parseXmlToolCalls } from './xml-toolcall.js';
|
|
4
4
|
import { translate } from './i18n.js';
|
|
5
|
+
import { normText } from './browser.js';
|
|
5
6
|
export async function runAgentLoop({ browser, tools, task, workdir, maxIterations = 40, freshChat = false, sendSystemPrompt = false, transcript = null, attachments = [], onThinking = () => { }, onAssistantThought = () => { }, onToolCall = () => { }, onToolResult = () => { }, onAssistantMessage = () => { }, onChatReady = () => { }, onWarning = () => { }, debugLog = false, locale = 'ru', }) {
|
|
7
|
+
// UI callbacks must NEVER break the agent loop. A rendering error (a huge
|
|
8
|
+
// tool result, a broken markdown frame, a closed terminal) used to throw
|
|
9
|
+
// out of the loop right after a tool call — the session looked "stopped
|
|
10
|
+
// after a tool call", with a tool_call but no tool_result in the log. We
|
|
11
|
+
// wrap every callback so a UI failure is swallowed and the loop continues.
|
|
12
|
+
const safe = (fn) => {
|
|
13
|
+
return (...a) => {
|
|
14
|
+
try {
|
|
15
|
+
fn(...a);
|
|
16
|
+
}
|
|
17
|
+
catch {
|
|
18
|
+
// Intentionally ignored: the loop must survive UI failures.
|
|
19
|
+
}
|
|
20
|
+
};
|
|
21
|
+
};
|
|
22
|
+
const safeThinking = safe(onThinking);
|
|
23
|
+
const safeAssistantThought = safe(onAssistantThought);
|
|
24
|
+
const safeToolCall = safe(onToolCall);
|
|
25
|
+
const safeToolResult = safe(onToolResult);
|
|
26
|
+
const safeAssistantMessage = safe(onAssistantMessage);
|
|
27
|
+
const safeWarning = safe(onWarning);
|
|
6
28
|
if (freshChat) {
|
|
7
29
|
await browser.newChat();
|
|
8
30
|
transcript?.log('new_chat');
|
|
@@ -42,7 +64,7 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
42
64
|
length: systemPrompt.length,
|
|
43
65
|
gitContext: gitText,
|
|
44
66
|
});
|
|
45
|
-
|
|
67
|
+
safeThinking();
|
|
46
68
|
// system-prompt is an agent send: throttled (agent: true).
|
|
47
69
|
await browser.ask(systemPrompt, { timeout: 60_000, agent: true });
|
|
48
70
|
await reportChat();
|
|
@@ -104,7 +126,7 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
104
126
|
let afterToolRetries = 0;
|
|
105
127
|
const MAX_AFTER_TOOL_RETRIES = 6;
|
|
106
128
|
for (let i = 0; i < maxIterations; i++) {
|
|
107
|
-
|
|
129
|
+
safeThinking();
|
|
108
130
|
// The first message (task) is user input: no throttle.
|
|
109
131
|
// Subsequent ones (tool-result and resend requests) are agent sends:
|
|
110
132
|
// throttled so we don't hit the rate limit.
|
|
@@ -139,7 +161,7 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
139
161
|
attempt: afterToolRetries,
|
|
140
162
|
error: e.message,
|
|
141
163
|
});
|
|
142
|
-
|
|
164
|
+
safeWarning('browser.ask() не вернул ответ за ' +
|
|
143
165
|
Math.round(askDeadlineMs / 1000) +
|
|
144
166
|
'с — повторяю запрос.');
|
|
145
167
|
if (afterToolRetries < MAX_AFTER_TOOL_RETRIES) {
|
|
@@ -150,7 +172,7 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
150
172
|
transcript?.log('ask_timeout_exhausted', {
|
|
151
173
|
message: 'ask() не вернул ответ и лимит повторов исчерпан',
|
|
152
174
|
});
|
|
153
|
-
|
|
175
|
+
safeWarning('browser.ask() перестал отвечать; лимит повторов исчерпан, ' +
|
|
154
176
|
'останавливаюсь. Ответа модели нет — проверьте чат DeepSeek вручную.');
|
|
155
177
|
return 'ask() watchdog: ответ модели не получен';
|
|
156
178
|
}
|
|
@@ -168,14 +190,22 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
168
190
|
// cut-off "Stale. Let me ..." or a truncated JSON). All three mean the
|
|
169
191
|
// turn is unfinished: nudge instead of stopping.
|
|
170
192
|
const wdEmpty = !String(rawResponse || '').trim();
|
|
193
|
+
// STALE is compared on NORMALIZED text: DeepSeek often echoes the
|
|
194
|
+
// previous answer with different markdown emphasis (`**done**` vs `done`),
|
|
195
|
+
// which defeated an exact match and made the loop run the SAME tool again —
|
|
196
|
+
// the "stopped after a tool call" signature with a duplicate call.
|
|
171
197
|
const wdStale = justRanTool &&
|
|
172
198
|
lastRaw.trim() !== '' &&
|
|
173
|
-
rawResponse.trim() === lastRaw.trim()
|
|
199
|
+
(rawResponse.trim() === lastRaw.trim() ||
|
|
200
|
+
normForStale(rawResponse) === normForStale(lastRaw));
|
|
174
201
|
const wdNoCall = justRanTool &&
|
|
175
202
|
!wdEmpty &&
|
|
176
203
|
!wdStale &&
|
|
177
204
|
parseToolCall(rawResponse) === null &&
|
|
178
205
|
!responseLooksLikeToolCall(rawResponse);
|
|
206
|
+
// A stale answer is discarded even when it parses to a valid call: it is a
|
|
207
|
+
// duplicate of the previous turn, and re-running the tool would repeat
|
|
208
|
+
// side effects and stall the loop.
|
|
179
209
|
if (!isFirst && (wdEmpty || wdStale || wdNoCall) && watchdogRetries < MAX_WATCHDOG_RETRIES) {
|
|
180
210
|
watchdogRetries++;
|
|
181
211
|
transcript?.log('watchdog_nudge', {
|
|
@@ -205,7 +235,7 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
205
235
|
if (parsed) {
|
|
206
236
|
const thought = extractPreToolText(rawResponse);
|
|
207
237
|
if (thought)
|
|
208
|
-
|
|
238
|
+
safeAssistantThought(thought);
|
|
209
239
|
}
|
|
210
240
|
const parsedCalls = Array.isArray(parsed) ? parsed : parsed ? [parsed] : [];
|
|
211
241
|
if (parsedCalls.some((p) => p && p._permissive)) {
|
|
@@ -341,7 +371,7 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
341
371
|
// not be turned into a tool call even after all retries: warn the
|
|
342
372
|
// operator instead of silently printing e.g. "Stale. Let me verify".
|
|
343
373
|
transcript?.log('suspicious_final', { response: rawResponse });
|
|
344
|
-
|
|
374
|
+
safeWarning(translate(locale)('msg.suspicious_stop'));
|
|
345
375
|
}
|
|
346
376
|
// A meaningful plain-text answer (e.g. a final report the model forgot to
|
|
347
377
|
// wrap in respond) is surfaced as-is WITHOUT a warning: after the bounded
|
|
@@ -393,11 +423,11 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
393
423
|
// the task normally.
|
|
394
424
|
if (!isMeaningfulRespond(msg)) {
|
|
395
425
|
transcript?.log('empty_respond_exhausted', { response: rawResponse });
|
|
396
|
-
|
|
426
|
+
safeWarning('Модель вызвала respond без текста, и лимит повторов исчерпан. ' +
|
|
397
427
|
'Проверьте чат DeepSeek вручную.');
|
|
398
428
|
return 'Модель не сформировала итоговое сообщение (пустой respond).';
|
|
399
429
|
}
|
|
400
|
-
|
|
430
|
+
safeAssistantMessage(msg);
|
|
401
431
|
transcript?.log('assistant_final', { message: msg });
|
|
402
432
|
return msg;
|
|
403
433
|
}
|
|
@@ -410,12 +440,12 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
410
440
|
const tool = tools.find((t) => t.name === call.tool);
|
|
411
441
|
if (!tool) {
|
|
412
442
|
const err = `Неизвестный инструмент: ${call.tool}`;
|
|
413
|
-
|
|
443
|
+
safeToolResult(err);
|
|
414
444
|
transcript?.log('tool_error', { tool: call.tool, error: err });
|
|
415
445
|
results.push({ tool: call.tool, result: err });
|
|
416
446
|
continue;
|
|
417
447
|
}
|
|
418
|
-
|
|
448
|
+
safeToolCall(call.tool, call.args);
|
|
419
449
|
transcript?.log('tool_call', { tool: call.tool, args: call.args });
|
|
420
450
|
let result;
|
|
421
451
|
try {
|
|
@@ -424,7 +454,7 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
424
454
|
catch (e) {
|
|
425
455
|
result = `Ошибка: ${e.message}`;
|
|
426
456
|
}
|
|
427
|
-
|
|
457
|
+
safeToolResult(result);
|
|
428
458
|
transcript?.log('tool_result', {
|
|
429
459
|
tool: call.tool,
|
|
430
460
|
result: String(result),
|
|
@@ -501,6 +531,16 @@ function isMeaningfulRespond(msg) {
|
|
|
501
531
|
return false;
|
|
502
532
|
return true;
|
|
503
533
|
}
|
|
534
|
+
// Normalize an answer for STALE comparison: collapse whitespace AND strip
|
|
535
|
+
// markdown emphasis/code markers. DeepSeek echoes the previous turn with a
|
|
536
|
+
// different emphasis (`**done**` vs `done`), which defeated an exact match and
|
|
537
|
+
// made the loop re-run the same tool ("stopped after a tool call").
|
|
538
|
+
function normForStale(s) {
|
|
539
|
+
return normText(s)
|
|
540
|
+
.replace(/[*_`#>]+/g, '')
|
|
541
|
+
.replace(/[ \t]+/g, ' ')
|
|
542
|
+
.trim();
|
|
543
|
+
}
|
|
504
544
|
// Text that promises a tool call in the future tense but contains no call
|
|
505
545
|
// itself. DeepSeek regularly "hangs" like this: it writes
|
|
506
546
|
// "Now update README to mention …", "Let me run the tests", "I'll check now"
|
package/dist/browser.js
CHANGED
|
@@ -646,15 +646,23 @@ export class DeepSeekBrowser {
|
|
|
646
646
|
.evaluate(() => document.body.innerText.length)
|
|
647
647
|
.catch(() => 0);
|
|
648
648
|
let started = false;
|
|
649
|
+
// A stale/echo answer (the model repeats the previous text, or the answer
|
|
650
|
+
// legitimately equals it) does NOT change `cur`. In that case the old loop
|
|
651
|
+
// either threw "did not start" after 15s or hung until the full timeout —
|
|
652
|
+
// the operator saw the agent "stop after a tool call". We now also accept
|
|
653
|
+
// the answer when the generation has clearly SETTLED: no Stop button and
|
|
654
|
+
// the text has been stable for a couple of ticks.
|
|
655
|
+
let settledTicks = 0;
|
|
656
|
+
let lastStartCur = '';
|
|
649
657
|
while (Date.now() < startDeadline) {
|
|
650
658
|
if (this._abort)
|
|
651
659
|
return '(прервано пользователем)';
|
|
652
660
|
const pageText = await this._readPageText();
|
|
653
661
|
if (isRateLimitText(pageText)) {
|
|
654
662
|
throw new RateLimitError(pageText.slice(0, 300));
|
|
655
|
-
|
|
656
|
-
|
|
657
|
-
|
|
663
|
+
}
|
|
664
|
+
if (isServerBusyText(pageText)) {
|
|
665
|
+
throw new ServerBusyError(pageText.slice(0, 300));
|
|
658
666
|
}
|
|
659
667
|
const cur = await this._readLastAnswerTextClean().catch(() => '');
|
|
660
668
|
const bodyLen = await this.page
|
|
@@ -678,9 +686,30 @@ export class DeepSeekBrowser {
|
|
|
678
686
|
started = true;
|
|
679
687
|
break;
|
|
680
688
|
}
|
|
689
|
+
// Fallback for an echo/stale answer: the send happened (lastSentAt was
|
|
690
|
+
// just updated), the Stop button is gone and the text stopped changing.
|
|
691
|
+
// Two stable ticks in a row mean the turn is over even if it equals the
|
|
692
|
+
// previous text — return it instead of hanging/throwing.
|
|
693
|
+
const notGenerating = !(await this._isGenerating());
|
|
694
|
+
if (cur && cur === lastStartCur && notGenerating && cur.trim()) {
|
|
695
|
+
settledTicks++;
|
|
696
|
+
if (settledTicks >= 2) {
|
|
697
|
+
return cur;
|
|
698
|
+
}
|
|
699
|
+
}
|
|
700
|
+
else {
|
|
701
|
+
settledTicks = 0;
|
|
702
|
+
}
|
|
703
|
+
lastStartCur = cur;
|
|
681
704
|
await this.page.waitForTimeout(300);
|
|
682
705
|
}
|
|
683
706
|
if (!started) {
|
|
707
|
+
// Last chance: the answer may have arrived and settled exactly at the
|
|
708
|
+
// deadline. Return the current text instead of a hard error.
|
|
709
|
+
const cur = await this._readLastAnswerTextClean().catch(() => '');
|
|
710
|
+
if (cur && cur.trim() && !(await this._isGenerating())) {
|
|
711
|
+
return cur;
|
|
712
|
+
}
|
|
684
713
|
// We NO LONGER check the limit over the whole page text — that caused
|
|
685
714
|
// false positives and 5-minute waits. We just report that
|
|
686
715
|
// generation did not start.
|
|
@@ -718,21 +747,32 @@ export class DeepSeekBrowser {
|
|
|
718
747
|
// belongs to the CURRENT send even if the DOM still shows the old text.
|
|
719
748
|
const isNew = !!cur &&
|
|
720
749
|
(netFresh || normText(cur) !== normText(beforeText));
|
|
721
|
-
|
|
750
|
+
// An echo/stale answer equals beforeText, so isNew stays false and the
|
|
751
|
+
// old loop waited until the full timeout — the "agent stopped after a
|
|
752
|
+
// tool call" hang. If generation has clearly ENDED (no Stop button) and
|
|
753
|
+
// the text is stable, accept it (even when it repeats the previous one).
|
|
754
|
+
const sameAsBefore = !!cur && !isNew && normText(cur) === normText(beforeText);
|
|
755
|
+
if ((isNew || sameAsBefore) && cur === last) {
|
|
722
756
|
stable++;
|
|
723
|
-
if (stable >= 2)
|
|
724
|
-
|
|
757
|
+
if (stable >= 2) {
|
|
758
|
+
if (isNew || !(await this._isGenerating()))
|
|
759
|
+
return cur;
|
|
760
|
+
}
|
|
725
761
|
}
|
|
726
762
|
else {
|
|
727
763
|
stable = 0;
|
|
728
764
|
}
|
|
729
|
-
if (isNew)
|
|
765
|
+
if (isNew || sameAsBefore)
|
|
730
766
|
last = cur;
|
|
731
767
|
await this.page.waitForTimeout(800);
|
|
732
768
|
}
|
|
733
769
|
if (last && (normText(last) !== normText(beforeText) || this._netCapture)) {
|
|
734
770
|
return last;
|
|
735
771
|
}
|
|
772
|
+
// Fallback: the turn settled on a text identical to the previous answer.
|
|
773
|
+
if (last && !(await this._isGenerating())) {
|
|
774
|
+
return last;
|
|
775
|
+
}
|
|
736
776
|
if (this._netCapture && this._netCaptureAt >= this._lastSentAt) {
|
|
737
777
|
return this._netCapture;
|
|
738
778
|
}
|