zames_pro 2.15.3 → 2.15.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -4,7 +4,7 @@ import { getGitContext, formatGitContext } from './gitTools.js';
4
4
  import { parseXmlToolCalls } from './xml-toolcall.js';
5
5
  import { translate } from './i18n.js';
6
6
  import { normText } from './browser.js';
7
- export async function runAgentLoop({ browser, tools, task, workdir, maxIterations = 200, freshChat = false, sendSystemPrompt = false, transcript = null, attachments = [], onThinking = () => { }, onAssistantThought = () => { }, onToolCall = () => { }, onToolResult = () => { }, onAssistantMessage = () => { }, onChatReady = () => { }, onWarning = () => { }, debugLog = false, locale = 'ru', }) {
7
+ export async function runAgentLoop({ browser, tools, task, workdir, maxIterations = 0, freshChat = false, sendSystemPrompt = false, transcript = null, attachments = [], onThinking = () => { }, onAssistantThought = () => { }, onToolCall = () => { }, onToolResult = () => { }, onAssistantMessage = () => { }, onChatReady = () => { }, onWarning = () => { }, debugLog = false, locale = 'ru', }) {
8
8
  // UI callbacks must NEVER break the agent loop. A rendering error (a huge
9
9
  // tool result, a broken markdown frame, a closed terminal) used to throw
10
10
  // out of the loop right after a tool call — the session looked "stopped
@@ -83,8 +83,11 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
83
83
  commands: context?.commands.map((c) => c.name) ?? [],
84
84
  });
85
85
  safeThinking();
86
- // system-prompt is an agent send: throttled (agent: true).
87
- await browser.ask(systemPrompt, { timeout: 60_000, agent: true });
86
+ // system-prompt is the FIRST message of a fresh chat: the rate limit only
87
+ // applies to a rapid back-and-forth, so this send is not throttled
88
+ // (agent: false). Throttling it used to add a useless 15s pause at the
89
+ // start of every new session.
90
+ await browser.ask(systemPrompt, { timeout: 60_000, agent: false });
88
91
  await reportChat();
89
92
  }
90
93
  let message = task;
@@ -150,8 +153,14 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
150
153
  // not come back in time. This is separate from unparsedRetries because a
151
154
  // timeout is an infrastructure failure, not a model protocol violation.
152
155
  let afterToolRetries = 0;
156
+ // toolsRanInTask: how many tools actually ran in THIS task. The protocol
157
+ // guard uses it to tell "started, then slipped into chat mode" (a reasoning
158
+ // paragraph that is neither a tool call nor a real respond) from a genuine
159
+ // short answer. We key on the STRUCTURE (work already started), not words.
160
+ let toolsRanInTask = 0;
153
161
  const MAX_AFTER_TOOL_RETRIES = 6;
154
- for (let i = 0; i < maxIterations; i++) {
162
+ const iterCap = maxIterations > 0 ? maxIterations : 100000;
163
+ for (let i = 0; i < iterCap; i++) {
155
164
  // NOTE: the spinner is NOT started here. browser.onSendStart fires it
156
165
  // right when the message is actually typed/sent (after the send-pause),
157
166
  // so no spinner runs during the pre-send phase.
@@ -413,6 +422,26 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
413
422
  '{\"tool\": \"respond\", \"args\": {\"message\": \"...\"}}';
414
423
  continue;
415
424
  }
425
+ // STRUCTURAL guard against "the model slipped into chat mode":
426
+ // after work has already started (a tool really ran in this task), a
427
+ // plain-text answer that is neither a tool call nor a real respond must
428
+ // NOT be returned to the operator as the final result. Returning it is
429
+ // exactly the "agent stopped mid-task" symptom (e.g. it writes a reasoning
430
+ // paragraph "Now let me analyze..." instead of calling a tool). We do not
431
+ // match words here - the STRUCTURE (toolsRanInTask > 0) is the signal.
432
+ if (toolsRanInTask > 0) {
433
+ transcript?.log('protocol_violation_final', {
434
+ response: rawResponse.slice(0, 500),
435
+ });
436
+ safeWarning(translate(locale)('msg.suspicious_stop'));
437
+ // The model stopped calling tools mid-task. Surface the last text as
438
+ // a READABLE report, but say explicitly that the task may be incomplete:
439
+ // never let a reasoning paragraph masquerade as a finished result.
440
+ const lastText = (rawResponse || '').trim();
441
+ return (lastText
442
+ ? lastText + String.fromCharCode(10) + String.fromCharCode(10) + '(The model stopped calling tools before finishing. The task may be incomplete - check the DeepSeek chat.)'
443
+ : 'The model stopped calling tools before finishing the task.');
444
+ }
416
445
  const suspiciousFinal = responseLooksLikeToolCall(rawResponse) ||
417
446
  looksLikeUnfinishedWork((rawResponse || '').trim()) ||
418
447
  !(rawResponse || '').trim();
@@ -514,6 +543,7 @@ export async function runAgentLoop({ browser, tools, task, workdir, maxIteration
514
543
  // A tool just ran — the next answer is expected to be a fresh tool call.
515
544
  // Reset the watchdog so the next empty/repeated answer is nudged.
516
545
  justRanTool = true;
546
+ toolsRanInTask++;
517
547
  // The tool really executed on this turn. The NEXT iteration will treat
518
548
  // a repeat of this answer as stale only because of this flag (see NEXT,
519
549
  // not the current, justRanTool check).
package/dist/browser.js CHANGED
@@ -144,7 +144,7 @@ export class DeepSeekBrowser {
144
144
  // a generation really begins, not during the pre-send phase (chat open,
145
145
  // throttle wait), which used to show a spinner with no work in flight.
146
146
  onSendStart;
147
- constructor({ headless = false, debug = false, channel = 'chrome', answerTimeoutMs = 180000, askRetries = 3, stabilityChecks = 3, stabilityDelayMs = 1000, minSendIntervalMs = 15000, rateLimitWaitMs = 300000, maxRateLimitRetries = 6, maxServerBusyRetries = 5, serverBusyWaitMs = 3000, } = {}) {
147
+ constructor({ headless = false, debug = false, channel = 'chrome', answerTimeoutMs = 180000, askRetries = 3, stabilityChecks = 2, stabilityDelayMs = 400, minSendIntervalMs = 15000, rateLimitWaitMs = 300000, maxRateLimitRetries = 6, maxServerBusyRetries = 5, serverBusyWaitMs = 3000, } = {}) {
148
148
  this.headless = headless;
149
149
  this.debug = debug;
150
150
  this.channel = channel;
@@ -180,7 +180,10 @@ export class DeepSeekBrowser {
180
180
  async _launchOnce() {
181
181
  const options = {
182
182
  headless: this.headless,
183
- slowMo: 30,
183
+ // 30ms slowMo per Playwright action added up over thousands of DOM
184
+ // actions per run. The waits we need are explicit; keep a small value
185
+ // for stability of clicks/typing.
186
+ slowMo: 10,
184
187
  args: ['--disable-blink-features=AutomationControlled'],
185
188
  };
186
189
  if (this.channel)
@@ -694,7 +697,7 @@ export class DeepSeekBrowser {
694
697
  await this._attachFiles(attachments);
695
698
  }
696
699
  await this._setInputText(input, prompt);
697
- await this.page.waitForTimeout(200);
700
+ await this.page.waitForTimeout(50);
698
701
  let sent = false;
699
702
  for (const sel of SEND_SELECTORS) {
700
703
  const btn = this.page.locator(sel).last();
@@ -867,7 +870,11 @@ export class DeepSeekBrowser {
867
870
  const sameAsBefore = !!cur && !isNew && normText(cur) === normText(beforeText);
868
871
  if (isNew && cur === last) {
869
872
  stable++;
870
- if (stable >= 2) {
873
+ // `stabilityChecks` / `stabilityDelayMs` are the real knobs here.
874
+ // They used to be dead config (hardcoded 2 checks with an 800ms tick),
875
+ // so the FIN-loop always cost ~1.6s per answer. Defaults are now
876
+ // 2 checks x 400ms (~0.8s saved per turn) and the values are honored.
877
+ if (stable >= Math.max(1, this.stabilityChecks - 1)) {
871
878
  if (isNew || !(await this._isGenerating())) {
872
879
  this._askDebug('RETURN stable curLen=' + cur.length);
873
880
  return cur;
@@ -880,7 +887,7 @@ export class DeepSeekBrowser {
880
887
  if (isNew)
881
888
  last = cur;
882
889
  this._askDebug('FIN-loop isNew=' + isNew + ' sameAsBefore=' + sameAsBefore + ' stable=' + stable + ' curLen=' + cur.length + ' lastLen=' + last.length + ' netFresh=' + netFresh);
883
- await this.page.waitForTimeout(800);
890
+ await this.page.waitForTimeout(Math.max(0, this.stabilityDelayMs));
884
891
  }
885
892
  if (last && (normText(last) !== normText(beforeText) || this._netCapture)) {
886
893
  this._askDebug('RETURN last len=' + last.length);
package/dist/config.js CHANGED
@@ -6,7 +6,7 @@ const ZAMES_HOME = path.join(os.homedir(), '.zames');
6
6
  const HOME_CONFIG = path.join(ZAMES_HOME, 'config.json');
7
7
  const PROJECT_CONFIG = path.join(process.cwd(), '.zamesrc.json');
8
8
  export const DEFAULTS = {
9
- maxIterations: 200,
9
+ maxIterations: 0,
10
10
  headless: false,
11
11
  debug: false,
12
12
  browserChannel: null,
@@ -36,8 +36,8 @@ export const DEFAULTS = {
36
36
  browser: {
37
37
  answerTimeoutMs: 180000,
38
38
  askRetries: 3,
39
- stabilityChecks: 3,
40
- stabilityDelayMs: 1000,
39
+ stabilityChecks: 2,
40
+ stabilityDelayMs: 400,
41
41
  minSendIntervalMs: 15000,
42
42
  rateLimitWaitMs: 300000,
43
43
  maxRateLimitRetries: 6,
@@ -77,7 +77,7 @@ function deepMerge(target, source) {
77
77
  }
78
78
  export const CONFIG_SCHEMA = [
79
79
  { path: 'ui.locale', type: 'enum', values: ['ru', 'en'], labelKey: 'cfg.f.ui_locale', groupKey: 'cfg.group.ui' },
80
- { path: 'maxIterations', type: 'number', min: 1, max: 100000, labelKey: 'cfg.f.maxIterations', groupKey: 'cfg.group.agent' },
80
+ { path: 'maxIterations', type: 'number', min: 0, max: 100000, labelKey: 'cfg.f.maxIterations', groupKey: 'cfg.group.agent' },
81
81
  { path: 'headless', type: 'boolean', labelKey: 'cfg.f.headless', groupKey: 'cfg.group.agent' },
82
82
  { path: 'debug', type: 'boolean', labelKey: 'cfg.f.debug', groupKey: 'cfg.group.agent' },
83
83
  { path: 'hotReload', type: 'boolean', labelKey: 'cfg.f.hotReload', groupKey: 'cfg.group.agent' },
package/dist/i18n.js CHANGED
@@ -45,7 +45,7 @@ const CATALOG = {
45
45
  'help.opt.resume_last': { ru: 'вернуться в последний сохранённый чат', en: 'resume the last saved chat' },
46
46
  'help.opt.new_chat': { ru: 'начать новый чат (по умолчанию)', en: 'start a new chat (default)' },
47
47
  'help.opt.resend_prompt': { ru: 'дослать system-prompt в существующий чат', en: 'resend system-prompt into an existing chat' },
48
- 'help.opt.max_iter': { ru: 'лимит итераций (по умолчанию {n})', en: 'iteration limit (default {n})' },
48
+ 'help.opt.max_iter': { ru: 'лимит итераций, 0 = без лимита ({n})', en: 'iteration limit, 0 = unlimited ({n})' },
49
49
  'help.opt.headless': { ru: 'браузер без UI', en: 'headless browser' },
50
50
  'help.opt.debug': { ru: 'подробный лог', en: 'verbose log' },
51
51
  'help.opt.calibrate': { ru: 'режим калибровки селекторов', en: 'selector calibration mode' },
@@ -144,7 +144,7 @@ const CATALOG = {
144
144
  'status.last_chat': { ru: 'Last chat: {v}', en: 'Last chat: {v}' },
145
145
  'status.sessions': { ru: 'Сессии: {v}', en: 'Sessions: {v}' },
146
146
  'status.dev': { ru: 'Dev mode (auto-reload): {v}', en: 'Dev mode (auto-reload): {v}' },
147
- 'status.max_iter': { ru: 'Лимит итераций: {v}', en: 'Iteration limit: {v}' },
147
+ 'status.max_iter': { ru: 'Лимит итераций: {v} (0 = без лимита)', en: 'Iteration limit: {v} (0 = unlimited)' },
148
148
  'status.headless': { ru: 'Headless: {v}', en: 'Headless: {v}' },
149
149
  'status.debug': { ru: 'Debug: {v}', en: 'Debug: {v}' },
150
150
  'status.undo': { ru: 'Undo: {v}', en: 'Undo: {v}' },
package/dist/index.js CHANGED
@@ -53,8 +53,8 @@ const t = (key, params) => translate(currentLocale)(key, params);
53
53
  const headless = hasFlag('--headless') || config.headless;
54
54
  const debug = hasFlag('--debug') || config.debug;
55
55
  const calibrate = hasFlag('--calibrate');
56
- const maxIter = Number(getArg('--max-iter', String(config.maxIterations))) ||
57
- config.maxIterations;
56
+ const maxIterArg = getArg('--max-iter', null);
57
+ const maxIter = maxIterArg !== null ? Number(maxIterArg) : config.maxIterations;
58
58
  const positional = getPositional();
59
59
  const task = getArg('--task', positional.join(' ').trim() || null);
60
60
  const chatIdArg = getArg('--chat', null);
@@ -133,7 +133,12 @@ export function extractAnswer(body) {
133
133
  }
134
134
  // Saves the DeepSeek network response body to disk for post-mortem analysis.
135
135
  // The files live in ~/.zames/net-log — from them the real answer format is visible.
136
+ // DEBUG ONLY: disabled unless ZAMES_NET_DEBUG=1. It writes a file per network
137
+ // response (thousands of files / tens of MB) and is not needed for the agent
138
+ // to work — the answer is taken from extractAnswer() in memory.
136
139
  export async function dumpNetBody(url, body) {
140
+ if (!process.env.ZAMES_NET_DEBUG)
141
+ return;
137
142
  try {
138
143
  const dir = path.join(os.homedir(), '.zames', 'net-log');
139
144
  await fs.mkdir(dir, { recursive: true });
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "zames_pro",
3
- "version": "2.15.3",
3
+ "version": "2.15.5",
4
4
  "description": "Terminal coding agent over chat.deepseek.com via Playwright",
5
5
  "type": "module",
6
6
  "bin": {