zames_pro 2.15.3 → 2.15.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-loop.js +34 -4
- package/dist/browser.js +12 -5
- package/dist/config.js +4 -4
- package/dist/i18n.js +2 -2
- package/dist/index.js +2 -2
- package/dist/net-capture.js +5 -0
- package/package.json +1 -1
package/dist/agent-loop.js
CHANGED
|
@@ -4,7 +4,7 @@ import { getGitContext, formatGitContext } from './gitTools.js';
|
|
|
4
4
|
import { parseXmlToolCalls } from './xml-toolcall.js';
|
|
5
5
|
import { translate } from './i18n.js';
|
|
6
6
|
import { normText } from './browser.js';
|
|
7
|
-
export async function runAgentLoop({ browser, tools, task, workdir, maxIterations =
|
|
7
|
+
export async function runAgentLoop({ browser, tools, task, workdir, maxIterations = 0, freshChat = false, sendSystemPrompt = false, transcript = null, attachments = [], onThinking = () => { }, onAssistantThought = () => { }, onToolCall = () => { }, onToolResult = () => { }, onAssistantMessage = () => { }, onChatReady = () => { }, onWarning = () => { }, debugLog = false, locale = 'ru', }) {
|
|
8
8
|
// UI callbacks must NEVER break the agent loop. A rendering error (a huge
|
|
9
9
|
// tool result, a broken markdown frame, a closed terminal) used to throw
|
|
10
10
|
// out of the loop right after a tool call — the session looked "stopped
|
|
@@ -83,8 +83,11 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
83
83
|
commands: context?.commands.map((c) => c.name) ?? [],
|
|
84
84
|
});
|
|
85
85
|
safeThinking();
|
|
86
|
-
// system-prompt is
|
|
87
|
-
|
|
86
|
+
// system-prompt is the FIRST message of a fresh chat: the rate limit only
|
|
87
|
+
// applies to a rapid back-and-forth, so this send is not throttled
|
|
88
|
+
// (agent: false). Throttling it used to add a useless 15s pause at the
|
|
89
|
+
// start of every new session.
|
|
90
|
+
await browser.ask(systemPrompt, { timeout: 60_000, agent: false });
|
|
88
91
|
await reportChat();
|
|
89
92
|
}
|
|
90
93
|
let message = task;
|
|
@@ -150,8 +153,14 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
150
153
|
// not come back in time. This is separate from unparsedRetries because a
|
|
151
154
|
// timeout is an infrastructure failure, not a model protocol violation.
|
|
152
155
|
let afterToolRetries = 0;
|
|
156
|
+
// toolsRanInTask: how many tools actually ran in THIS task. The protocol
|
|
157
|
+
// guard uses it to tell "started, then slipped into chat mode" (a reasoning
|
|
158
|
+
// paragraph that is neither a tool call nor a real respond) from a genuine
|
|
159
|
+
// short answer. We key on the STRUCTURE (work already started), not words.
|
|
160
|
+
let toolsRanInTask = 0;
|
|
153
161
|
const MAX_AFTER_TOOL_RETRIES = 6;
|
|
154
|
-
|
|
162
|
+
const iterCap = maxIterations > 0 ? maxIterations : 100000;
|
|
163
|
+
for (let i = 0; i < iterCap; i++) {
|
|
155
164
|
// NOTE: the spinner is NOT started here. browser.onSendStart fires it
|
|
156
165
|
// right when the message is actually typed/sent (after the send-pause),
|
|
157
166
|
// so no spinner runs during the pre-send phase.
|
|
@@ -413,6 +422,26 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
413
422
|
'{\"tool\": \"respond\", \"args\": {\"message\": \"...\"}}';
|
|
414
423
|
continue;
|
|
415
424
|
}
|
|
425
|
+
// STRUCTURAL guard against "the model slipped into chat mode":
|
|
426
|
+
// after work has already started (a tool really ran in this task), a
|
|
427
|
+
// plain-text answer that is neither a tool call nor a real respond must
|
|
428
|
+
// NOT be returned to the operator as the final result. Returning it is
|
|
429
|
+
// exactly the "agent stopped mid-task" symptom (e.g. it writes a reasoning
|
|
430
|
+
// paragraph "Now let me analyze..." instead of calling a tool). We do not
|
|
431
|
+
// match words here - the STRUCTURE (toolsRanInTask > 0) is the signal.
|
|
432
|
+
if (toolsRanInTask > 0) {
|
|
433
|
+
transcript?.log('protocol_violation_final', {
|
|
434
|
+
response: rawResponse.slice(0, 500),
|
|
435
|
+
});
|
|
436
|
+
safeWarning(translate(locale)('msg.suspicious_stop'));
|
|
437
|
+
// The model stopped calling tools mid-task. Surface the last text as
|
|
438
|
+
// a READABLE report, but say explicitly that the task may be incomplete:
|
|
439
|
+
// never let a reasoning paragraph masquerade as a finished result.
|
|
440
|
+
const lastText = (rawResponse || '').trim();
|
|
441
|
+
return (lastText
|
|
442
|
+
? lastText + String.fromCharCode(10) + String.fromCharCode(10) + '(The model stopped calling tools before finishing. The task may be incomplete - check the DeepSeek chat.)'
|
|
443
|
+
: 'The model stopped calling tools before finishing the task.');
|
|
444
|
+
}
|
|
416
445
|
const suspiciousFinal = responseLooksLikeToolCall(rawResponse) ||
|
|
417
446
|
looksLikeUnfinishedWork((rawResponse || '').trim()) ||
|
|
418
447
|
!(rawResponse || '').trim();
|
|
@@ -514,6 +543,7 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
|
|
|
514
543
|
// A tool just ran — the next answer is expected to be a fresh tool call.
|
|
515
544
|
// Reset the watchdog so the next empty/repeated answer is nudged.
|
|
516
545
|
justRanTool = true;
|
|
546
|
+
toolsRanInTask++;
|
|
517
547
|
// The tool really executed on this turn. The NEXT iteration will treat
|
|
518
548
|
// a repeat of this answer as stale only because of this flag (see NEXT,
|
|
519
549
|
// not the current, justRanTool check).
|
package/dist/browser.js
CHANGED
|
@@ -144,7 +144,7 @@ export class DeepSeekBrowser {
|
|
|
144
144
|
// a generation really begins, not during the pre-send phase (chat open,
|
|
145
145
|
// throttle wait), which used to show a spinner with no work in flight.
|
|
146
146
|
onSendStart;
|
|
147
|
-
constructor({ headless = false, debug = false, channel = 'chrome', answerTimeoutMs = 180000, askRetries = 3, stabilityChecks =
|
|
147
|
+
constructor({ headless = false, debug = false, channel = 'chrome', answerTimeoutMs = 180000, askRetries = 3, stabilityChecks = 2, stabilityDelayMs = 400, minSendIntervalMs = 15000, rateLimitWaitMs = 300000, maxRateLimitRetries = 6, maxServerBusyRetries = 5, serverBusyWaitMs = 3000, } = {}) {
|
|
148
148
|
this.headless = headless;
|
|
149
149
|
this.debug = debug;
|
|
150
150
|
this.channel = channel;
|
|
@@ -180,7 +180,10 @@ export class DeepSeekBrowser {
|
|
|
180
180
|
async _launchOnce() {
|
|
181
181
|
const options = {
|
|
182
182
|
headless: this.headless,
|
|
183
|
-
slowMo
|
|
183
|
+
// 30ms slowMo per Playwright action added up over thousands of DOM
|
|
184
|
+
// actions per run. The waits we need are explicit; keep a small value
|
|
185
|
+
// for stability of clicks/typing.
|
|
186
|
+
slowMo: 10,
|
|
184
187
|
args: ['--disable-blink-features=AutomationControlled'],
|
|
185
188
|
};
|
|
186
189
|
if (this.channel)
|
|
@@ -694,7 +697,7 @@ export class DeepSeekBrowser {
|
|
|
694
697
|
await this._attachFiles(attachments);
|
|
695
698
|
}
|
|
696
699
|
await this._setInputText(input, prompt);
|
|
697
|
-
await this.page.waitForTimeout(
|
|
700
|
+
await this.page.waitForTimeout(50);
|
|
698
701
|
let sent = false;
|
|
699
702
|
for (const sel of SEND_SELECTORS) {
|
|
700
703
|
const btn = this.page.locator(sel).last();
|
|
@@ -867,7 +870,11 @@ export class DeepSeekBrowser {
|
|
|
867
870
|
const sameAsBefore = !!cur && !isNew && normText(cur) === normText(beforeText);
|
|
868
871
|
if (isNew && cur === last) {
|
|
869
872
|
stable++;
|
|
870
|
-
|
|
873
|
+
// `stabilityChecks` / `stabilityDelayMs` are the real knobs here.
|
|
874
|
+
// They used to be dead config (hardcoded 2 checks with an 800ms tick),
|
|
875
|
+
// so the FIN-loop always cost ~1.6s per answer. Defaults are now
|
|
876
|
+
// 2 checks x 400ms (~0.8s saved per turn) and the values are honored.
|
|
877
|
+
if (stable >= Math.max(1, this.stabilityChecks - 1)) {
|
|
871
878
|
if (isNew || !(await this._isGenerating())) {
|
|
872
879
|
this._askDebug('RETURN stable curLen=' + cur.length);
|
|
873
880
|
return cur;
|
|
@@ -880,7 +887,7 @@ export class DeepSeekBrowser {
|
|
|
880
887
|
if (isNew)
|
|
881
888
|
last = cur;
|
|
882
889
|
this._askDebug('FIN-loop isNew=' + isNew + ' sameAsBefore=' + sameAsBefore + ' stable=' + stable + ' curLen=' + cur.length + ' lastLen=' + last.length + ' netFresh=' + netFresh);
|
|
883
|
-
await this.page.waitForTimeout(
|
|
890
|
+
await this.page.waitForTimeout(Math.max(0, this.stabilityDelayMs));
|
|
884
891
|
}
|
|
885
892
|
if (last && (normText(last) !== normText(beforeText) || this._netCapture)) {
|
|
886
893
|
this._askDebug('RETURN last len=' + last.length);
|
package/dist/config.js
CHANGED
|
@@ -6,7 +6,7 @@ const ZAMES_HOME = path.join(os.homedir(), '.zames');
|
|
|
6
6
|
const HOME_CONFIG = path.join(ZAMES_HOME, 'config.json');
|
|
7
7
|
const PROJECT_CONFIG = path.join(process.cwd(), '.zamesrc.json');
|
|
8
8
|
export const DEFAULTS = {
|
|
9
|
-
maxIterations:
|
|
9
|
+
maxIterations: 0,
|
|
10
10
|
headless: false,
|
|
11
11
|
debug: false,
|
|
12
12
|
browserChannel: null,
|
|
@@ -36,8 +36,8 @@ export const DEFAULTS = {
|
|
|
36
36
|
browser: {
|
|
37
37
|
answerTimeoutMs: 180000,
|
|
38
38
|
askRetries: 3,
|
|
39
|
-
stabilityChecks:
|
|
40
|
-
stabilityDelayMs:
|
|
39
|
+
stabilityChecks: 2,
|
|
40
|
+
stabilityDelayMs: 400,
|
|
41
41
|
minSendIntervalMs: 15000,
|
|
42
42
|
rateLimitWaitMs: 300000,
|
|
43
43
|
maxRateLimitRetries: 6,
|
|
@@ -77,7 +77,7 @@ function deepMerge(target, source) {
|
|
|
77
77
|
}
|
|
78
78
|
export const CONFIG_SCHEMA = [
|
|
79
79
|
{ path: 'ui.locale', type: 'enum', values: ['ru', 'en'], labelKey: 'cfg.f.ui_locale', groupKey: 'cfg.group.ui' },
|
|
80
|
-
{ path: 'maxIterations', type: 'number', min:
|
|
80
|
+
{ path: 'maxIterations', type: 'number', min: 0, max: 100000, labelKey: 'cfg.f.maxIterations', groupKey: 'cfg.group.agent' },
|
|
81
81
|
{ path: 'headless', type: 'boolean', labelKey: 'cfg.f.headless', groupKey: 'cfg.group.agent' },
|
|
82
82
|
{ path: 'debug', type: 'boolean', labelKey: 'cfg.f.debug', groupKey: 'cfg.group.agent' },
|
|
83
83
|
{ path: 'hotReload', type: 'boolean', labelKey: 'cfg.f.hotReload', groupKey: 'cfg.group.agent' },
|
package/dist/i18n.js
CHANGED
|
@@ -45,7 +45,7 @@ const CATALOG = {
|
|
|
45
45
|
'help.opt.resume_last': { ru: 'вернуться в последний сохранённый чат', en: 'resume the last saved chat' },
|
|
46
46
|
'help.opt.new_chat': { ru: 'начать новый чат (по умолчанию)', en: 'start a new chat (default)' },
|
|
47
47
|
'help.opt.resend_prompt': { ru: 'дослать system-prompt в существующий чат', en: 'resend system-prompt into an existing chat' },
|
|
48
|
-
'help.opt.max_iter': { ru: 'лимит
|
|
48
|
+
'help.opt.max_iter': { ru: 'лимит итераций, 0 = без лимита ({n})', en: 'iteration limit, 0 = unlimited ({n})' },
|
|
49
49
|
'help.opt.headless': { ru: 'браузер без UI', en: 'headless browser' },
|
|
50
50
|
'help.opt.debug': { ru: 'подробный лог', en: 'verbose log' },
|
|
51
51
|
'help.opt.calibrate': { ru: 'режим калибровки селекторов', en: 'selector calibration mode' },
|
|
@@ -144,7 +144,7 @@ const CATALOG = {
|
|
|
144
144
|
'status.last_chat': { ru: 'Last chat: {v}', en: 'Last chat: {v}' },
|
|
145
145
|
'status.sessions': { ru: 'Сессии: {v}', en: 'Sessions: {v}' },
|
|
146
146
|
'status.dev': { ru: 'Dev mode (auto-reload): {v}', en: 'Dev mode (auto-reload): {v}' },
|
|
147
|
-
'status.max_iter': { ru: 'Лимит итераций: {v}', en: 'Iteration limit: {v}' },
|
|
147
|
+
'status.max_iter': { ru: 'Лимит итераций: {v} (0 = без лимита)', en: 'Iteration limit: {v} (0 = unlimited)' },
|
|
148
148
|
'status.headless': { ru: 'Headless: {v}', en: 'Headless: {v}' },
|
|
149
149
|
'status.debug': { ru: 'Debug: {v}', en: 'Debug: {v}' },
|
|
150
150
|
'status.undo': { ru: 'Undo: {v}', en: 'Undo: {v}' },
|
package/dist/index.js
CHANGED
|
@@ -53,8 +53,8 @@ const t = (key, params) => translate(currentLocale)(key, params);
|
|
|
53
53
|
const headless = hasFlag('--headless') || config.headless;
|
|
54
54
|
const debug = hasFlag('--debug') || config.debug;
|
|
55
55
|
const calibrate = hasFlag('--calibrate');
|
|
56
|
-
const
|
|
57
|
-
|
|
56
|
+
const maxIterArg = getArg('--max-iter', null);
|
|
57
|
+
const maxIter = maxIterArg !== null ? Number(maxIterArg) : config.maxIterations;
|
|
58
58
|
const positional = getPositional();
|
|
59
59
|
const task = getArg('--task', positional.join(' ').trim() || null);
|
|
60
60
|
const chatIdArg = getArg('--chat', null);
|
package/dist/net-capture.js
CHANGED
|
@@ -133,7 +133,12 @@ export function extractAnswer(body) {
|
|
|
133
133
|
}
|
|
134
134
|
// Saves the DeepSeek network response body to disk for post-mortem analysis.
|
|
135
135
|
// The files live in ~/.zames/net-log — from them the real answer format is visible.
|
|
136
|
+
// DEBUG ONLY: disabled unless ZAMES_NET_DEBUG=1. It writes a file per network
|
|
137
|
+
// response (thousands of files / tens of MB) and is not needed for the agent
|
|
138
|
+
// to work — the answer is taken from extractAnswer() in memory.
|
|
136
139
|
export async function dumpNetBody(url, body) {
|
|
140
|
+
if (!process.env.ZAMES_NET_DEBUG)
|
|
141
|
+
return;
|
|
137
142
|
try {
|
|
138
143
|
const dir = path.join(os.homedir(), '.zames', 'net-log');
|
|
139
144
|
await fs.mkdir(dir, { recursive: true });
|