zames_pro 2.10.2 → 2.10.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-loop.js +20 -1
- package/dist/browser.js +92 -8
- package/package.json +1 -1
package/dist/agent-loop.js
CHANGED
|
@@ -2,6 +2,7 @@ import { buildSystemPrompt } from './system-prompt.js';
|
|
|
2
2
|
import { getGitContext, formatGitContext } from './gitTools.js';
|
|
3
3
|
import { parseXmlToolCalls } from './xml-toolcall.js';
|
|
4
4
|
import { translate } from './i18n.js';
|
|
5
|
+
import { normText } from './browser.js';
|
|
5
6
|
export async function runAgentLoop({ browser, tools, task, workdir, maxIterations = 40, freshChat = false, sendSystemPrompt = false, transcript = null, attachments = [], onThinking = () => { }, onAssistantThought = () => { }, onToolCall = () => { }, onToolResult = () => { }, onAssistantMessage = () => { }, onChatReady = () => { }, onWarning = () => { }, debugLog = false, locale = 'ru', }) {
|
|
6
7
|
// UI callbacks must NEVER break the agent loop. A rendering error (a huge
|
|
7
8
|
// tool result, a broken markdown frame, a closed terminal) used to throw
|
|
@@ -189,14 +190,22 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
189
190
|
// cut-off "Stale. Let me ..." or a truncated JSON). All three mean the
|
|
190
191
|
// turn is unfinished: nudge instead of stopping.
|
|
191
192
|
const wdEmpty = !String(rawResponse || '').trim();
|
|
193
|
+
// STALE is compared on NORMALIZED text: DeepSeek often echoes the
|
|
194
|
+
// previous answer with different markdown emphasis (`**done**` vs `done`),
|
|
195
|
+
// which defeated an exact match and made the loop run the SAME tool again —
|
|
196
|
+
// the "stopped after a tool call" signature with a duplicate call.
|
|
192
197
|
const wdStale = justRanTool &&
|
|
193
198
|
lastRaw.trim() !== '' &&
|
|
194
|
-
rawResponse.trim() === lastRaw.trim()
|
|
199
|
+
(rawResponse.trim() === lastRaw.trim() ||
|
|
200
|
+
normForStale(rawResponse) === normForStale(lastRaw));
|
|
195
201
|
const wdNoCall = justRanTool &&
|
|
196
202
|
!wdEmpty &&
|
|
197
203
|
!wdStale &&
|
|
198
204
|
parseToolCall(rawResponse) === null &&
|
|
199
205
|
!responseLooksLikeToolCall(rawResponse);
|
|
206
|
+
// A stale answer is discarded even when it parses to a valid call: it is a
|
|
207
|
+
// duplicate of the previous turn, and re-running the tool would repeat
|
|
208
|
+
// side effects and stall the loop.
|
|
200
209
|
if (!isFirst && (wdEmpty || wdStale || wdNoCall) && watchdogRetries < MAX_WATCHDOG_RETRIES) {
|
|
201
210
|
watchdogRetries++;
|
|
202
211
|
transcript?.log('watchdog_nudge', {
|
|
@@ -522,6 +531,16 @@ function isMeaningfulRespond(msg) {
|
|
|
522
531
|
return false;
|
|
523
532
|
return true;
|
|
524
533
|
}
|
|
534
|
+
// Normalize an answer for STALE comparison: collapse whitespace AND strip
|
|
535
|
+
// markdown emphasis/code markers. DeepSeek echoes the previous turn with a
|
|
536
|
+
// different emphasis (`**done**` vs `done`), which defeated an exact match and
|
|
537
|
+
// made the loop re-run the same tool ("stopped after a tool call").
|
|
538
|
+
function normForStale(s) {
|
|
539
|
+
return normText(s)
|
|
540
|
+
.replace(/[*_`#>]+/g, '')
|
|
541
|
+
.replace(/[ \t]+/g, ' ')
|
|
542
|
+
.trim();
|
|
543
|
+
}
|
|
525
544
|
// Text that promises a tool call in the future tense but contains no call
|
|
526
545
|
// itself. DeepSeek regularly "hangs" like this: it writes
|
|
527
546
|
// "Now update README to mention …", "Let me run the tests", "I'll check now"
|
package/dist/browser.js
CHANGED
|
@@ -520,9 +520,23 @@ export class DeepSeekBrowser {
|
|
|
520
520
|
.replace(/\r\n/g, '\n')
|
|
521
521
|
.replace(/\u00a0/g, ' ')
|
|
522
522
|
.trim();
|
|
523
|
-
|
|
523
|
+
// Verify the WHOLE text landed in the input, not just that it is non-empty.
|
|
524
|
+
// A partial paste (DeepSeek input limits, lost chars) used to pass the
|
|
525
|
+
// old "not empty" check, so a TRUNCATED message was sent, the model
|
|
526
|
+
// replied to the wrong thing (or nothing), and the loop looked stalled.
|
|
527
|
+
if (norm(got) !== norm(text)) {
|
|
524
528
|
await input.click();
|
|
529
|
+
await this.page.keyboard.press("Control+A");
|
|
530
|
+
await this.page.keyboard.press("Delete");
|
|
525
531
|
await this.page.keyboard.insertText(text);
|
|
532
|
+
const got2 = await input.evaluate((el) => {
|
|
533
|
+
if (el.tagName.toLowerCase() === "textarea" || el.tagName.toLowerCase() === "input")
|
|
534
|
+
return el.value;
|
|
535
|
+
return el.innerText || el.textContent || "";
|
|
536
|
+
});
|
|
537
|
+
if (norm(got2) !== norm(text)) {
|
|
538
|
+
throw new Error("Не удалось вставить текст в поле ввода DeepSeek целиком (вставлено " + norm(got2).length + " из " + norm(text).length + " символов). Сообщение не отправлено, чтобы не отправить обрезанный текст.");
|
|
539
|
+
}
|
|
526
540
|
}
|
|
527
541
|
}
|
|
528
542
|
// Attach files/images to the chat via the hidden <input type=file> of the
|
|
@@ -602,6 +616,18 @@ export class DeepSeekBrowser {
|
|
|
602
616
|
console.error(theme.warn(`⏳ пауза ${Math.ceil(gap / 1000)}с перед отправкой`));
|
|
603
617
|
await this.page.waitForTimeout(gap);
|
|
604
618
|
}
|
|
619
|
+
// DEBUG: append ask() phases to ~/.zames/ask-debug.log so a stall can be
|
|
620
|
+
// diagnosed from the field (what the DOM/network looked like at each step).
|
|
621
|
+
_askDebug(msg) {
|
|
622
|
+
if (!process.env.ZAMES_ASK_DEBUG)
|
|
623
|
+
return;
|
|
624
|
+
try {
|
|
625
|
+
const line = new Date().toISOString() + ' ' + msg + String.fromCharCode(10);
|
|
626
|
+
const dir = path.join(os.homedir(), '.zames');
|
|
627
|
+
void fs.appendFile(path.join(dir, 'ask-debug.log'), line).catch(() => { });
|
|
628
|
+
}
|
|
629
|
+
catch { }
|
|
630
|
+
}
|
|
605
631
|
async _askOnce(prompt, { timeout, agent, attachments = [], }) {
|
|
606
632
|
// We reset the abort flag ONLY at the very start of the send.
|
|
607
633
|
this._abort = false;
|
|
@@ -610,6 +636,7 @@ export class DeepSeekBrowser {
|
|
|
610
636
|
throw new Error('Не найдено поле ввода. Запустите /debug-dom и поправьте INPUT_SELECTORS.');
|
|
611
637
|
}
|
|
612
638
|
const beforeText = await this._readLastAnswerTextClean().catch(() => '');
|
|
639
|
+
this._askDebug('SEND agent=' + agent + ' len=' + prompt.length + ' beforeLen=' + beforeText.length + ' beforeHead=' + JSON.stringify(beforeText.slice(0, 60)));
|
|
613
640
|
await this._waitForSendSlot(agent);
|
|
614
641
|
this._netCapture = '';
|
|
615
642
|
this._netCaptureAt = 0;
|
|
@@ -638,6 +665,7 @@ export class DeepSeekBrowser {
|
|
|
638
665
|
await this.page.keyboard.press('Enter');
|
|
639
666
|
}
|
|
640
667
|
this._lastSentAt = Date.now();
|
|
668
|
+
this._askDebug('SENT at=' + this._lastSentAt);
|
|
641
669
|
// Wait for the start: either Stop appeared, or the answer text changed,
|
|
642
670
|
// or the total amount of text on the page grew. In parallel we catch
|
|
643
671
|
// the rate-limit toast (only toasts, not the whole body).
|
|
@@ -646,15 +674,23 @@ export class DeepSeekBrowser {
|
|
|
646
674
|
.evaluate(() => document.body.innerText.length)
|
|
647
675
|
.catch(() => 0);
|
|
648
676
|
let started = false;
|
|
677
|
+
// A stale/echo answer (the model repeats the previous text, or the answer
|
|
678
|
+
// legitimately equals it) does NOT change `cur`. In that case the old loop
|
|
679
|
+
// either threw "did not start" after 15s or hung until the full timeout —
|
|
680
|
+
// the operator saw the agent "stop after a tool call". We now also accept
|
|
681
|
+
// the answer when the generation has clearly SETTLED: no Stop button and
|
|
682
|
+
// the text has been stable for a couple of ticks.
|
|
683
|
+
let settledTicks = 0;
|
|
684
|
+
let lastStartCur = '';
|
|
649
685
|
while (Date.now() < startDeadline) {
|
|
650
686
|
if (this._abort)
|
|
651
687
|
return '(прервано пользователем)';
|
|
652
688
|
const pageText = await this._readPageText();
|
|
653
689
|
if (isRateLimitText(pageText)) {
|
|
654
690
|
throw new RateLimitError(pageText.slice(0, 300));
|
|
655
|
-
|
|
656
|
-
|
|
657
|
-
|
|
691
|
+
}
|
|
692
|
+
if (isServerBusyText(pageText)) {
|
|
693
|
+
throw new ServerBusyError(pageText.slice(0, 300));
|
|
658
694
|
}
|
|
659
695
|
const cur = await this._readLastAnswerTextClean().catch(() => '');
|
|
660
696
|
const bodyLen = await this.page
|
|
@@ -676,11 +712,43 @@ export class DeepSeekBrowser {
|
|
|
676
712
|
const netStarted = !!this._netCapture && this._netCaptureAt >= this._lastSentAt;
|
|
677
713
|
if (changed || netStarted || bodyLen > startBodyLen) {
|
|
678
714
|
started = true;
|
|
715
|
+
this._askDebug('STARTED changed=' + changed + ' netStarted=' + netStarted + ' bodyGrew=' + (bodyLen > startBodyLen));
|
|
679
716
|
break;
|
|
680
717
|
}
|
|
718
|
+
// Fallback for an echo: the text equals beforeText, so it is the OLD
|
|
719
|
+
// answer still on screen, NOT a new one. We must NOT return it (that
|
|
720
|
+
// made the loop re-run the previous tool call). We only return when the
|
|
721
|
+
// text DIFFERS from beforeText and has settled, or when a fresh network
|
|
722
|
+
// capture proves a new answer exists. If it stays equal, keep waiting.
|
|
723
|
+
const notGenerating = !(await this._isGenerating());
|
|
724
|
+
const differs = !!cur && normText(cur) !== normText(beforeText);
|
|
725
|
+
if (differs && cur === lastStartCur && notGenerating) {
|
|
726
|
+
settledTicks++;
|
|
727
|
+
if (settledTicks >= 2) {
|
|
728
|
+
this._askDebug('SETTLED-differs return len=' + cur.length);
|
|
729
|
+
return cur;
|
|
730
|
+
}
|
|
731
|
+
}
|
|
732
|
+
else {
|
|
733
|
+
settledTicks = 0;
|
|
734
|
+
}
|
|
735
|
+
lastStartCur = cur;
|
|
736
|
+
this._askDebug('START-loop curLen=' + cur.length + ' changed=' + changed + ' netStarted=' + netStarted + ' bodyLen=' + bodyLen + ' settled=' + settledTicks + ' generating=' + notGenerating);
|
|
681
737
|
await this.page.waitForTimeout(300);
|
|
682
738
|
}
|
|
683
739
|
if (!started) {
|
|
740
|
+
// Last chance: accept the text only when it DIFFERS from beforeText
|
|
741
|
+
// (otherwise it is the old answer on screen) or a fresh network capture
|
|
742
|
+
// proves a new answer. Returning an equal text made the loop re-run the
|
|
743
|
+
// previous tool call.
|
|
744
|
+
const cur = await this._readLastAnswerTextClean().catch(() => '');
|
|
745
|
+
const fresh = !!this._netCapture && this._netCaptureAt >= this._lastSentAt;
|
|
746
|
+
if (cur && cur.trim() && normText(cur) !== normText(beforeText) && !(await this._isGenerating())) {
|
|
747
|
+
return cur;
|
|
748
|
+
}
|
|
749
|
+
if (fresh) {
|
|
750
|
+
return this._netCapture;
|
|
751
|
+
}
|
|
684
752
|
// We NO LONGER check the limit over the whole page text — that caused
|
|
685
753
|
// false positives and 5-minute waits. We just report that
|
|
686
754
|
// generation did not start.
|
|
@@ -718,24 +786,40 @@ export class DeepSeekBrowser {
|
|
|
718
786
|
// belongs to the CURRENT send even if the DOM still shows the old text.
|
|
719
787
|
const isNew = !!cur &&
|
|
720
788
|
(netFresh || normText(cur) !== normText(beforeText));
|
|
721
|
-
|
|
789
|
+
// An echo/stale answer equals beforeText, so isNew stays false and the
|
|
790
|
+
// old loop waited until the full timeout — the "agent stopped after a
|
|
791
|
+
// tool call" hang. If generation has clearly ENDED (no Stop button) and
|
|
792
|
+
// the text is stable, accept it (even when it repeats the previous one).
|
|
793
|
+
const sameAsBefore = !!cur && !isNew && normText(cur) === normText(beforeText);
|
|
794
|
+
if ((isNew || sameAsBefore) && cur === last) {
|
|
722
795
|
stable++;
|
|
723
|
-
if (stable >= 2)
|
|
724
|
-
|
|
796
|
+
if (stable >= 2) {
|
|
797
|
+
if (isNew || !(await this._isGenerating())) {
|
|
798
|
+
this._askDebug('RETURN stable curLen=' + cur.length);
|
|
799
|
+
return cur;
|
|
800
|
+
}
|
|
801
|
+
}
|
|
725
802
|
}
|
|
726
803
|
else {
|
|
727
804
|
stable = 0;
|
|
728
805
|
}
|
|
729
|
-
if (isNew)
|
|
806
|
+
if (isNew || sameAsBefore)
|
|
730
807
|
last = cur;
|
|
808
|
+
this._askDebug('FIN-loop isNew=' + isNew + ' sameAsBefore=' + sameAsBefore + ' stable=' + stable + ' curLen=' + cur.length + ' lastLen=' + last.length + ' netFresh=' + netFresh);
|
|
731
809
|
await this.page.waitForTimeout(800);
|
|
732
810
|
}
|
|
733
811
|
if (last && (normText(last) !== normText(beforeText) || this._netCapture)) {
|
|
812
|
+
this._askDebug('RETURN last len=' + last.length);
|
|
813
|
+
return last;
|
|
814
|
+
}
|
|
815
|
+
// Fallback: the turn settled on a text identical to the previous answer.
|
|
816
|
+
if (last && !(await this._isGenerating())) {
|
|
734
817
|
return last;
|
|
735
818
|
}
|
|
736
819
|
if (this._netCapture && this._netCaptureAt >= this._lastSentAt) {
|
|
737
820
|
return this._netCapture;
|
|
738
821
|
}
|
|
822
|
+
this._askDebug('THROW no-new-answer lastLen=' + last.length + ' beforeLen=' + beforeText.length);
|
|
739
823
|
throw new Error('Новый ответ не получен (на странице остался прежний текст). ' +
|
|
740
824
|
'Возможно, сообщение не отправилось.');
|
|
741
825
|
}
|