zames_pro 2.18.0 → 2.19.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -138,8 +138,8 @@ config commands (/new, /chats, /resume, /cd, /status, /config,
138
138
  Codex CLI:
139
139
 
140
140
  - /diff [--staged] — show the working-tree git diff (--staged for the index).
141
- - /cost (alias /usage) — session stats: tasks, tool calls, duration
142
- (DeepSeek web does not expose token counts).
141
+ - /cost (alias /usage) — session stats: tasks, tool calls, duration, and the
142
+ context size in tokens (DeepSeek's `accumulated_token_usage`).
143
143
  - /export [file] — write the session transcript to a Markdown file
144
144
  (zames-export-<stamp>.md by default).
145
145
  - /doctor — diagnose node, git, config, browser, clipboard and MCP.
@@ -152,6 +152,14 @@ Codex CLI:
152
152
  last 20 messages are shown (`RESTORED_HISTORY_LIMIT`).
153
153
  - /review [focus] [--staged] — ask the agent to review uncommitted changes
154
154
  and report findings (no code changes).
155
+ - /compact — ask DeepSeek to compress the current chat into a handover
156
+ summary, then open a NEW chat, resend the system prompt and post the summary
157
+ as the carried-over context. Use it when the context gets long.
158
+
159
+ The token context is also shown live: the status line above the input has the
160
+ spinner/text on the left and the context on the right (e.g. `125k · 13%`,
161
+ percent of a 1M context). It comes from DeepSeek's `accumulated_token_usage`
162
+ and is hidden until the first answer delivers it.
155
163
 
156
164
 
157
165
  ## Project context, skills and memory
package/dist/browser.js CHANGED
@@ -1,5 +1,5 @@
1
1
  import { chromium, } from 'playwright';
2
- import { extractAnswer, dumpNetBody } from './net-capture.js';
2
+ import { extractAnswer, extractTokenUsage, dumpNetBody } from './net-capture.js';
3
3
  import path from 'path';
4
4
  import os from 'os';
5
5
  import fs from 'fs/promises';
@@ -277,6 +277,11 @@ export class DeepSeekBrowser {
277
277
  _netCapture;
278
278
  _netCaptureAt;
279
279
  _netChatId;
280
+ // The latest CONTEXT size (in tokens) DeepSeek reported in the current
281
+ // chat. It is the `accumulated_token_usage` counter taken from the SSE
282
+ // completion stream and from /api/v0/chat/history_messages. Null until the
283
+ // first answer (or history fetch) delivers it. Shown by /cost and /status.
284
+ _lastTokenUsage;
280
285
  // Authorization / PoW headers sniffed from DeepSeek's own API requests, so
281
286
  // fetchChatMessages can replay them (a bare fetch does not get them).
282
287
  _apiAuth;
@@ -328,6 +333,7 @@ export class DeepSeekBrowser {
328
333
  this._netCapture = '';
329
334
  this._netCaptureAt = 0;
330
335
  this._netChatId = null;
336
+ this._lastTokenUsage = null;
331
337
  this._apiAuth = '';
332
338
  this._apiPow = '';
333
339
  this._lastHistoryError = '';
@@ -472,6 +478,12 @@ export class DeepSeekBrowser {
472
478
  if (this._netSniff.length > this._netSniffLimit)
473
479
  this._netSniff.shift();
474
480
  void dumpNetBody(url, body);
481
+ // Context size (tokens) reported by DeepSeek for this answer. Kept
482
+ // even when extractAnswer() returns nothing (a history_messages
483
+ // response carries the counter but no answer text).
484
+ const usage = extractTokenUsage(body);
485
+ if (usage !== null)
486
+ this._lastTokenUsage = usage;
475
487
  const extracted = extractAnswer(body);
476
488
  if (extracted) {
477
489
  this._netCapture = extracted;
@@ -856,6 +868,20 @@ export class DeepSeekBrowser {
856
868
  if (this._netCapture && this._netCaptureAt >= this._lastSentAt) {
857
869
  return this._netCapture;
858
870
  }
871
+ return await this._readLastAnswerTextDom();
872
+ }
873
+ // The answer as rendered on the page, ALWAYS from the DOM (never from the
874
+ // network capture). This is what the "has a new answer started?" checks
875
+ // must compare against `beforeText`.
876
+ //
877
+ // Why a separate reader: `_readLastAnswerText()` prefers `_netCapture` when
878
+ // it is fresh. `beforeText` was taken through it, so after a send the
879
+ // network capture (the SAME text) made `cur` equal to `beforeText` — the
880
+ // "changed" signal never fired and ask() ended with "no new answer" and
881
+ // retried for minutes, while the operator saw the agent "stop after a tool
882
+ // call". The DOM reader has no such self-comparison problem: before the
883
+ // send the DOM shows the OLD answer, after it the NEW one.
884
+ async _readLastAnswerTextDom() {
859
885
  return await this.page.evaluate((sels) => {
860
886
  // DeepSeek stores the model's reasoning in .ds-think-content blocks.
861
887
  // They are NOT the answer and must never be picked up as the answer
@@ -922,6 +948,19 @@ export class DeepSeekBrowser {
922
948
  }
923
949
  async _readLastAnswerTextClean() {
924
950
  const raw = await this._readLastAnswerText().catch(() => '');
951
+ return this._cleanAnswer(raw);
952
+ }
953
+ // The same "clean" filter as _readLastAnswerTextClean, but read from the
954
+ // DOM only. Used by _askOnce for the before/after comparison: the network
955
+ // capture must not be compared against itself (see _readLastAnswerTextDom).
956
+ async _readLastAnswerTextCleanDom() {
957
+ const raw = await this._readLastAnswerTextDom().catch(() => '');
958
+ return this._cleanAnswer(raw);
959
+ }
960
+ // Drop service placeholders ("Reading…", "Thinking…") so they are not
961
+ // mistaken for an answer; keep everything else as-is (including whitespace
962
+ // the tool-call relies on).
963
+ _cleanAnswer(raw) {
925
964
  const t = (raw || '').trim();
926
965
  if (!t)
927
966
  return '';
@@ -1321,7 +1360,12 @@ export class DeepSeekBrowser {
1321
1360
  if (!input) {
1322
1361
  throw new Error(this._t('ds.input_missing'));
1323
1362
  }
1324
- const beforeText = await this._readLastAnswerTextClean().catch(() => '');
1363
+ // `beforeText` MUST come from the DOM, not from _readLastAnswerText():
1364
+ // that reader prefers the network capture, so comparing `cur` (also the
1365
+ // capture) against `beforeText` compared the capture against itself and
1366
+ // the "new answer" check never fired — the source of the
1367
+ // ds.send_no_new_answer flapping.
1368
+ const beforeText = await this._readLastAnswerTextCleanDom().catch(() => '');
1325
1369
  this._askDebug('SEND agent=' + agent + ' len=' + prompt.length + ' beforeLen=' + beforeText.length + ' beforeHead=' + JSON.stringify(beforeText.slice(0, 60)));
1326
1370
  await this._waitForSendSlot(agent);
1327
1371
  // Esc/Ctrl+C pressed during the pause — do not send anything.
@@ -1395,7 +1439,7 @@ export class DeepSeekBrowser {
1395
1439
  if (isServerBusyText(pageText)) {
1396
1440
  throw new ServerBusyError(pageText.slice(0, 300));
1397
1441
  }
1398
- const cur = await this._readLastAnswerTextClean().catch(() => '');
1442
+ const cur = await this._readLastAnswerTextCleanDom().catch(() => '');
1399
1443
  const bodyLen = await this.page
1400
1444
  .evaluate(() => document.body.innerText.length)
1401
1445
  .catch(() => 0);
@@ -1444,7 +1488,7 @@ export class DeepSeekBrowser {
1444
1488
  // (otherwise it is the old answer on screen) or a fresh network capture
1445
1489
  // proves a new answer. Returning an equal text made the loop re-run the
1446
1490
  // previous tool call.
1447
- const cur = await this._readLastAnswerTextClean().catch(() => '');
1491
+ const cur = await this._readLastAnswerTextCleanDom().catch(() => '');
1448
1492
  const fresh = !!this._netCapture && this._netCaptureAt >= this._lastSentAt;
1449
1493
  if (cur && cur.trim() && normText(cur) !== normText(beforeText) && !(await this._isGenerating())) {
1450
1494
  return cur;
@@ -1464,7 +1508,7 @@ export class DeepSeekBrowser {
1464
1508
  while (Date.now() < retryDeadline) {
1465
1509
  if (this._abort)
1466
1510
  return '(прервано пользователем)';
1467
- const cur2 = await this._readLastAnswerTextClean().catch(() => '');
1511
+ const cur2 = await this._readLastAnswerTextCleanDom().catch(() => '');
1468
1512
  const net2 = !!this._netCapture && this._netCaptureAt >= this._lastSentAt;
1469
1513
  const grew2 = (await this.page
1470
1514
  .evaluate(() => document.body.innerText.length)
@@ -1506,7 +1550,7 @@ export class DeepSeekBrowser {
1506
1550
  }
1507
1551
  }
1508
1552
  const netFresh = !!this._netCapture && this._netCaptureAt >= this._lastSentAt;
1509
- const cur = await this._readLastAnswerTextClean().catch(() => '');
1553
+ const cur = await this._readLastAnswerTextCleanDom().catch(() => '');
1510
1554
  // Ignore an "answer" that is identical to what was on the page BEFORE we
1511
1555
  // sent the message: that is the previous answer, not a new one. Returning
1512
1556
  // it would make the agent re-process the old tool call (or silently
@@ -1700,7 +1744,13 @@ export class DeepSeekBrowser {
1700
1744
  }
1701
1745
  const NL = String.fromCharCode(10);
1702
1746
  const out = [];
1747
+ // The context size: the LATEST accumulated_token_usage in the chat
1748
+ // (each message carries the running counter).
1749
+ let usage = null;
1703
1750
  for (const m of messages) {
1751
+ if (m && typeof m.accumulated_token_usage === 'number') {
1752
+ usage = m.accumulated_token_usage;
1753
+ }
1704
1754
  const role = m && m.role === 'ASSISTANT' ? 'assistant' : 'user';
1705
1755
  const want = role === 'assistant' ? 'RESPONSE' : 'REQUEST';
1706
1756
  let text = '';
@@ -1714,7 +1764,7 @@ export class DeepSeekBrowser {
1714
1764
  if (text)
1715
1765
  out.push({ role, text });
1716
1766
  }
1717
- return { error: '', list: out };
1767
+ return { error: '', list: out, usage };
1718
1768
  }
1719
1769
  catch (e) {
1720
1770
  return { error: 'fetch failed: ' + e.message, list: [] };
@@ -1725,6 +1775,11 @@ export class DeepSeekBrowser {
1725
1775
  list: [],
1726
1776
  }));
1727
1777
  this._lastHistoryError = res?.error || '';
1778
+ // Pick up the context size the history carries, so /resume (and /cost
1779
+ // right after it) shows a real number even before the first answer.
1780
+ const usage = res?.usage;
1781
+ if (typeof usage === 'number')
1782
+ this._lastTokenUsage = usage;
1728
1783
  const list = (res?.list || []);
1729
1784
  if (list.length || !res?.error)
1730
1785
  return list;
@@ -1743,6 +1798,9 @@ export class DeepSeekBrowser {
1743
1798
  const NL = String.fromCharCode(10);
1744
1799
  const out = [];
1745
1800
  for (const m of messages) {
1801
+ if (m && typeof m.accumulated_token_usage === 'number') {
1802
+ this._lastTokenUsage = m.accumulated_token_usage;
1803
+ }
1746
1804
  const role = m && m.role === 'ASSISTANT' ? 'assistant' : 'user';
1747
1805
  const want = role === 'assistant' ? 'RESPONSE' : 'REQUEST';
1748
1806
  let text = '';
@@ -1874,6 +1932,11 @@ export class DeepSeekBrowser {
1874
1932
  return this._netChatId;
1875
1933
  }
1876
1934
  }
1935
+ // The latest context size (in tokens) DeepSeek reported for the current
1936
+ // chat, or null when nothing has been seen yet. Used by /cost and /status.
1937
+ getLastTokenUsage() {
1938
+ return this._lastTokenUsage;
1939
+ }
1877
1940
  async close() {
1878
1941
  try {
1879
1942
  if (this.context)
package/dist/commands.js CHANGED
@@ -77,9 +77,15 @@ export function formatDuration(ms) {
77
77
  return m + 'm ' + pad(s) + 's';
78
78
  return s + 's';
79
79
  }
80
- export function renderCost(stats, transcriptFile) {
80
+ export function renderCost(stats, transcriptFile, tokenUsage = null) {
81
81
  const lines = [];
82
- lines.push('Session stats (DeepSeek web does not expose token counts):');
82
+ lines.push('Session stats:');
83
+ if (typeof tokenUsage === 'number') {
84
+ lines.push(' context: ~' + tokenUsage + ' tokens (DeepSeek accumulated_token_usage)');
85
+ }
86
+ else {
87
+ lines.push(' context: unknown (DeepSeek reports it after the first answer in a chat)');
88
+ }
83
89
  lines.push(' tasks: ' + stats.turns);
84
90
  lines.push(' tool calls: ' + stats.toolCalls);
85
91
  const top = Object.entries(stats.toolCounts).sort((a, b) => b[1] - a[1]);
@@ -258,6 +264,51 @@ export function formatRestoredHistory(messages, opts = {}) {
258
264
  }
259
265
  return out.join(NL + NL);
260
266
  }
267
+ // ---------- /compact ----------
268
+ /**
269
+ * The prompt that asks the model to compress the current chat into a handover
270
+ * summary. It is sent to the OLD chat before a new one is opened; the answer
271
+ * (the summary) is then carried over as the context of the new chat.
272
+ *
273
+ * The summary must be self-sufficient: the new chat sees ONLY this text (plus
274
+ * the system prompt), so the model is told to keep facts, decisions, file
275
+ * paths, commands and the exact current state of the work.
276
+ */
277
+ export function buildCompactPrompt(locale = 'ru') {
278
+ if (locale === 'en') {
279
+ return ('Compact the conversation so far into a handover summary for a NEW chat. ' +
280
+ 'This summary is the ONLY context the new chat will start with, so it must be self-sufficient. ' +
281
+ 'Include: (1) the user goal and constraints; (2) what has been done so far; ' +
282
+ '(3) the exact current state (files changed, commands run, their results); ' +
283
+ '(4) open questions and the next concrete steps. ' +
284
+ 'Keep file paths, function/identifier names, commands and error texts verbatim. ' +
285
+ 'Be concise but complete — no small talk, no code dumps beyond short essential snippets.');
286
+ }
287
+ return ('Сожми историю диалога в краткое резюме для НОВОГО чата. ' +
288
+ 'Это резюме будет ЕДИНСТВЕННЫМ контекстом, с которым новый чат начнёт работу, поэтому оно должно быть самодостаточным. ' +
289
+ 'Включи: (1) цель пользователя и ограничения; (2) что уже сделано; ' +
290
+ '(3) точное текущее состояние (изменённые файлы, выполненные команды и их результаты); ' +
291
+ '(4) открытые вопросы и следующие конкретные шаги. ' +
292
+ 'Пути к файлам, имена функций/идентификаторов, команды и тексты ошибок сохраняй дословно. ' +
293
+ 'Пиши кратко, но полно — без воды и без больших дампов кода (только короткие важные фрагменты).');
294
+ }
295
+ /**
296
+ * Wrap the model's summary into the text posted as the first message of the
297
+ * NEW chat. The system prompt is sent separately (sendSystemPrompt), so here
298
+ * we only mark the block as a carried-over context and add the operator's
299
+ * original goal so the model does not lose it.
300
+ */
301
+ export function buildCompactCarryover(summary, task) {
302
+ const body = String(summary ?? '').trim();
303
+ const goal = String(task ?? '').trim();
304
+ let out = 'Context carried over from a previous chat (compacted). ' +
305
+ 'Treat it as the history of our work so far and continue from the current state.';
306
+ out += NL + NL + body;
307
+ if (goal) {
308
+ out += NL + NL + 'Original task: ' + goal;
309
+ }
310
+ return out;
311
+ }
261
312
  // ---------- /review ----------
262
313
  export function buildReviewPrompt(focus, hasStaged = false) {
263
314
  const scope = hasStaged ? 'staged' : 'uncommitted';
package/dist/i18n.js CHANGED
@@ -86,6 +86,13 @@ const CATALOG = {
86
86
  'help.cmd.permissions': { ru: '/permissions настройки подтверждений', en: '/permissions confirmation settings' },
87
87
  'help.cmd.add_dir': { ru: '/add-dir <path> проверить директорию', en: '/add-dir <path> validate a directory' },
88
88
  'help.cmd.review': { ru: '/review [focus] ревью незакоммиченных изменений', en: '/review [focus] review uncommitted changes' },
89
+ 'help.cmd.compact': { ru: '/compact сжать историю и открыть новый чат с резюме', en: '/compact compact the history and open a new chat with the summary' },
90
+ 'compact.start': { ru: '🗜️ Сжимаю историю чата (DeepSeek)...', en: '🗜️ Compacting the chat history (DeepSeek)...' },
91
+ 'compact.empty': { ru: 'Нечего сжимать: в чате ещё нет ответов.', en: 'Nothing to compact: the chat has no answers yet.' },
92
+ 'compact.summary_failed': { ru: 'Не удалось получить резюме от модели: {v}', en: 'Could not get the summary from the model: {v}' },
93
+ 'compact.done': { ru: '✅ История сжата, открыт новый чат с резюме.', en: '✅ History compacted, a new chat with the summary is open.' },
94
+ 'compact.report': { ru: 'Резюме перенесено в новый чат ({chars} символов, токенов было ~{tokens}).', en: 'The summary was carried into the new chat ({chars} chars, ~{tokens} tokens before).' },
95
+ 'compact.no_chat': { ru: 'Чат ещё не создан — сжимать нечего.', en: 'No chat created yet — nothing to compact.' },
89
96
  'diff.not_repo': { ru: 'Не git-репозиторий.', en: 'Not a git repository.' },
90
97
  'export.done': { ru: 'Сессия выгружена: {v}', en: 'Session exported: {v}' },
91
98
  'export.outside': { ru: 'Путь вне рабочей директории.', en: 'Path is outside the working directory.' },
@@ -316,6 +323,7 @@ const CATALOG = {
316
323
  'chats.current_id': { ru: 'Текущий chat id: {v}', en: 'Current chat id: {v}' },
317
324
  'chats.not_created': { ru: 'Чат ещё не создан.', en: 'No chat created yet.' },
318
325
  'chats.history_title': { ru: 'Диалог чата:', en: 'Chat dialogue:' },
326
+ 'chats.history_tokens': { ru: 'Контекст чата: ~{v} токенов', en: 'Chat context: ~{v} tokens' },
319
327
  'chats.history_empty': { ru: 'Диалог пуст или не удалось прочитать сообщения.', en: 'The dialogue is empty or the messages could not be read.' },
320
328
  'chats.history_service_only': { ru: 'В этом чате нет пользовательских реплик — только служебные сообщения агента (вызовы инструментов).', en: 'This chat has no user turns — only the agent service messages (tool calls).' },
321
329
  'chats.history_truncated': { ru: '… показаны последние {n} сообщений.', en: '… showing the last {n} messages.' },
@@ -355,6 +363,10 @@ const CATALOG = {
355
363
  'mcp.title': { ru: 'MCP-серверы (инструментов: {n}):', en: 'MCP servers ({n} tools):' },
356
364
  'mcp.status_error': { ru: '(ошибка: {v})', en: '(error: {v})' },
357
365
  'status.mcp': { ru: 'MCP-инструменты: {v}', en: 'MCP tools: {v}' },
366
+ 'status.tokens': {
367
+ ru: 'Контекст (токенов): {v}',
368
+ en: 'Context (tokens): {v}',
369
+ },
358
370
  // ---------- config menu ----------
359
371
  'cfg.group.ui': { ru: 'Интерфейс', en: 'Interface' },
360
372
  'cfg.group.agent': { ru: 'Агент', en: 'Agent' },
package/dist/index.js CHANGED
@@ -16,7 +16,7 @@ import { translate, normalizeLocale, localeDisplayName, isLocale, } from './i18n
16
16
  import { Transcript } from './transcript.js';
17
17
  import { UndoStore } from './undo.js';
18
18
  import { selfReview, selfDiff, selfApply, selfList } from './self-review.js';
19
- import { formatDiff, diffGitArgs, parseTranscript, summarizeTranscript, renderCost, formatExport, defaultExportPath, renderDoctor, renderPermissions, resolveExtraDir, buildReviewPrompt, trimRestoredMessages, RESTORED_HISTORY_LIMIT, } from './commands.js';
19
+ import { formatDiff, diffGitArgs, parseTranscript, summarizeTranscript, renderCost, formatExport, defaultExportPath, renderDoctor, renderPermissions, resolveExtraDir, buildReviewPrompt, trimRestoredMessages, RESTORED_HISTORY_LIMIT, buildCompactPrompt, buildCompactCarryover, } from './commands.js';
20
20
  import { renderMarkdown } from './markdown.js';
21
21
  import { closeWeb } from './web.js';
22
22
  import { saveSession, loadLastSession, listSessions, sessionsDir, } from './sessions.js';
@@ -244,6 +244,7 @@ ${theme.bold(t('help.commands'))}
244
244
  ${t('help.cmd.permissions')}
245
245
  ${t('help.cmd.add_dir')}
246
246
  ${t('help.cmd.review')}
247
+ ${t('help.cmd.compact')}
247
248
  ${t('help.cmd.config')}
248
249
  ${t('help.cmd.lang')}
249
250
  ${t('help.cmd.debug_dom')}
@@ -299,6 +300,7 @@ const SLASH_COMMANDS = [
299
300
  { name: '/permissions', key: 'help.cmd.permissions' },
300
301
  { name: '/add-dir', key: 'help.cmd.add_dir' },
301
302
  { name: '/review', key: 'help.cmd.review' },
303
+ { name: '/compact', key: 'help.cmd.compact' },
302
304
  { name: '/config', key: 'help.cmd.config' },
303
305
  { name: '/skills', key: 'help.cmd.skills' },
304
306
  { name: '/memory', key: 'help.cmd.memory' },
@@ -836,6 +838,10 @@ async function printRestoredHistory(browser, ui, chatId = null) {
836
838
  console.log(text);
837
839
  };
838
840
  out(theme.system(t('chats.history_title')));
841
+ const restoredTokens = browser.getLastTokenUsage();
842
+ if (typeof restoredTokens === 'number') {
843
+ out(theme.dim(t('chats.history_tokens', { v: String(restoredTokens) })));
844
+ }
839
845
  for (const m of messages) {
840
846
  if (m.role === 'user') {
841
847
  out(theme.user('❯ ' + t('chats.history_you') + ': ') + m.text.trim());
@@ -1195,6 +1201,10 @@ async function main() {
1195
1201
  });
1196
1202
  editor = ed;
1197
1203
  ed.setTmpDir(TMP_DIR);
1204
+ // The token context right-aligned on the status line (above the input).
1205
+ // The editor pulls the number on every render, so it follows the live
1206
+ // DeepSeek counter (accumulated_token_usage) without a polling timer.
1207
+ ed.onContextQuery = () => browser.getLastTokenUsage();
1198
1208
  ed.onAttach = async (raw) => {
1199
1209
  // Case 1: the paste is the image data itself (data URL / base64 blob).
1200
1210
  const image = parseImagePaste(raw);
@@ -2107,6 +2117,10 @@ async function main() {
2107
2117
  console.log(theme.system(t('status.mcp', {
2108
2118
  v: mcpPool ? String(mcpPool.status().toolCount) : t('common.none'),
2109
2119
  })));
2120
+ const tokens = browser.getLastTokenUsage();
2121
+ console.log(theme.system(t('status.tokens', {
2122
+ v: tokens === null ? t('common.unknown') : String(tokens),
2123
+ })));
2110
2124
  continue;
2111
2125
  }
2112
2126
  if (lower === '/config' || lower.startsWith('/config ')) {
@@ -2140,7 +2154,7 @@ async function main() {
2140
2154
  // best-effort
2141
2155
  }
2142
2156
  }
2143
- console.log(theme.system(renderCost(stats, transcript.file)));
2157
+ console.log(theme.system(renderCost(stats, transcript.file, browser.getLastTokenUsage())));
2144
2158
  continue;
2145
2159
  }
2146
2160
  if (lower === '/export' || lower.startsWith('/export ')) {
@@ -2234,6 +2248,91 @@ async function main() {
2234
2248
  console.log(theme.system(t('adddir.note', { v: res.path })));
2235
2249
  continue;
2236
2250
  }
2251
+ if (lower === '/compact') {
2252
+ // Compaction: ask DeepSeek (in the CURRENT chat) to compress the
2253
+ // history into a handover summary, then start a NEW chat, resend the
2254
+ // system prompt and post the summary as the carried-over context.
2255
+ // This keeps the model working with a small context while nothing is
2256
+ // lost: the summary plus the system prompt are all the new chat needs.
2257
+ if (!currentChatId) {
2258
+ currentChatId = await browser.getCurrentChatId();
2259
+ }
2260
+ if (!currentChatId) {
2261
+ console.error(theme.warn(t('compact.no_chat')));
2262
+ continue;
2263
+ }
2264
+ const beforeTokens = browser.getLastTokenUsage();
2265
+ if (editor)
2266
+ editor.lock(t('msg.input_locked'));
2267
+ try {
2268
+ console.log(theme.system(t('compact.start')));
2269
+ // 1) Ask the OLD chat to summarize itself. agent: true - this is a
2270
+ // real back-and-forth, so the send throttle applies.
2271
+ let summary = '';
2272
+ try {
2273
+ summary = await browser.ask(buildCompactPrompt(currentLocale), {
2274
+ agent: true,
2275
+ timeout: Math.max(60_000, config.browser.answerTimeoutMs),
2276
+ });
2277
+ }
2278
+ catch (e) {
2279
+ console.error(theme.error(t('compact.summary_failed', { v: e.message })));
2280
+ continue;
2281
+ }
2282
+ summary = String(summary || '').trim();
2283
+ // A model "answer" that is actually an error/abort sentinel is not a
2284
+ // summary - do not carry it over.
2285
+ if (!summary || /^\(прервано пользователем\)$/.test(summary)) {
2286
+ console.error(theme.error(t('compact.summary_failed', { v: summary || t('common.unknown') })));
2287
+ continue;
2288
+ }
2289
+ transcript.log('compact_summary', {
2290
+ chars: summary.length,
2291
+ beforeTokens,
2292
+ });
2293
+ // 2) New chat + system prompt + the summary as the first message.
2294
+ await browser.newChat();
2295
+ await browser.ask(mod.buildSystemPrompt({
2296
+ workdir: currentWorkdir,
2297
+ tools: mod.createTools(currentWorkdir, { undo }),
2298
+ locale: currentLocale,
2299
+ }), { timeout: 60_000, agent: false });
2300
+ await browser.ask(buildCompactCarryover(summary, task ?? undefined), {
2301
+ timeout: 60_000,
2302
+ agent: false,
2303
+ });
2304
+ currentChatId = await browser.getCurrentChatId();
2305
+ saveLastChat(currentChatId, currentWorkdir);
2306
+ freshChatNext = false;
2307
+ // The new chat already carries the system prompt and the context.
2308
+ sendSystemPromptNext = false;
2309
+ console.log(theme.assistant(t('compact.done') +
2310
+ String.fromCharCode(10) +
2311
+ t('compact.report', {
2312
+ chars: summary.length,
2313
+ tokens: beforeTokens === null
2314
+ ? t('common.unknown')
2315
+ : String(beforeTokens),
2316
+ })));
2317
+ if (editor) {
2318
+ editor.printAbove(theme.dim(String.fromCharCode(10) +
2319
+ '--- compacted context ---' +
2320
+ String.fromCharCode(10) +
2321
+ summary +
2322
+ String.fromCharCode(10) +
2323
+ '--- end ---' +
2324
+ String.fromCharCode(10)));
2325
+ }
2326
+ }
2327
+ catch (e) {
2328
+ console.error(theme.error(e.message));
2329
+ }
2330
+ finally {
2331
+ if (editor)
2332
+ editor.unlock();
2333
+ }
2334
+ continue;
2335
+ }
2237
2336
  if (lower === '/review' || lower.startsWith('/review ')) {
2238
2337
  const rest = trimmed.slice('/review'.length).trim();
2239
2338
  const staged = rest.indexOf("--staged") !== -1;
package/dist/input.js CHANGED
@@ -170,6 +170,31 @@ export function layoutInput(promptStr, buf, cursor, cols) {
170
170
  }
171
171
  return { rows, cursorRow, cursorCol };
172
172
  }
173
+ // Format a token count for the status line: compact (10k, 125k) plus the
174
+ // percentage of CONTEXT_LIMIT. Exported so it is unit-tested without a live
175
+ // editor. A null/undefined/NaN count renders an empty string (no status).
176
+ export const CONTEXT_LIMIT = 1_000_000;
177
+ export function formatTokenStatus(tokens, limit = CONTEXT_LIMIT) {
178
+ if (typeof tokens !== 'number' || !Number.isFinite(tokens) || tokens < 0) {
179
+ return '';
180
+ }
181
+ const n = Math.round(tokens);
182
+ let compact;
183
+ if (n >= 1_000_000) {
184
+ const m = n / 1_000_000;
185
+ compact = (Number.isInteger(m) ? String(m) : m.toFixed(1)) + 'M';
186
+ }
187
+ else if (n >= 1_000) {
188
+ const k = n / 1_000;
189
+ compact = (k >= 100 ? String(Math.round(k)) : k.toFixed(1).replace(/\.0$/, '')) + 'k';
190
+ }
191
+ else {
192
+ compact = String(n);
193
+ }
194
+ const pct = Math.max(0, (n / limit) * 100);
195
+ const pctStr = pct >= 10 ? String(Math.round(pct)) : pct.toFixed(1);
196
+ return compact + ' · ' + pctStr + '%';
197
+ }
173
198
  export class LineEditor {
174
199
  promptStr;
175
200
  buf;
@@ -218,6 +243,13 @@ export class LineEditor {
218
243
  // Interface language for the editor's own labels (hint, answer marker,
219
244
  // pause status). Everything the OPERATOR sees must be localized.
220
245
  locale;
246
+ // Token context for the status line: the compact count (10k/125k) plus the
247
+ // percentage of the context limit, right-aligned above the input line. Null
248
+ // hides it. Updated by the caller from browser.getLastTokenUsage().
249
+ contextStatus;
250
+ // A callback the editor calls to fetch the CURRENT token count before every
251
+ // status render, so the status line stays fresh without the caller polling.
252
+ onContextQuery;
221
253
  constructor({ prompt = '> ', commands = [], locale = 'ru' } = {}) {
222
254
  this.locale = locale;
223
255
  this.promptStr = prompt;
@@ -250,6 +282,22 @@ export class LineEditor {
250
282
  this.onAttach = null;
251
283
  this.onClipboard = null;
252
284
  this.locked = false;
285
+ this.contextStatus = null;
286
+ this.onContextQuery = null;
287
+ }
288
+ // The token status for the CURRENT render: refreshed from onContextQuery
289
+ // when wired, otherwise the last value passed to setContextStatus().
290
+ _contextForRender() {
291
+ if (this.onContextQuery) {
292
+ try {
293
+ const n = this.onContextQuery();
294
+ this.contextStatus = n === null || n === undefined ? null : formatTokenStatus(n) || null;
295
+ }
296
+ catch {
297
+ // A broken callback must never break the render.
298
+ }
299
+ }
300
+ return this.contextStatus;
253
301
  }
254
302
  // Read the OS clipboard for an image and insert its marker. Used when the
255
303
  // terminal sends no usable paste data (Ctrl+V / right-click / empty paste).
@@ -370,6 +418,11 @@ export class LineEditor {
370
418
  this.promptStr = str;
371
419
  this._render();
372
420
  }
421
+ // Update the token-context text shown at the right of the status line.
422
+ setContextStatus(tokens) {
423
+ this.contextStatus = formatTokenStatus(tokens) || null;
424
+ this._render();
425
+ }
373
426
  // Update the interface language (labels: hint, answer marker, pause).
374
427
  setLocale(locale) {
375
428
  this.locale = locale;
@@ -402,11 +455,29 @@ export class LineEditor {
402
455
  const cols = process.stdout.columns || 80;
403
456
  let out = '';
404
457
  let top = 0;
458
+ // Status line above the input: the spinner/answer text on the left and the
459
+ // token context right-aligned on the SAME row (10k · 12%). The context is
460
+ // refreshed from onContextQuery() on every render, so it follows the live
461
+ // DeepSeek counter without a polling timer of its own.
462
+ const ctx = this._contextForRender();
463
+ const ctxText = ctx ? theme.dim(ctx) : '';
405
464
  if (this.statusText) {
406
- out += this.statusText + NL;
465
+ if (ctxText) {
466
+ const pad = Math.max(1, cols - visLen(this.statusText) - visLen(ctxText));
467
+ out += this.statusText + ' '.repeat(pad) + ctxText + NL;
468
+ }
469
+ else {
470
+ out += this.statusText + NL;
471
+ }
407
472
  // The status may wrap onto several lines — we account for this,
408
473
  // otherwise the block erase misses and statuses pile up.
409
- top = visRows(this.statusText, cols);
474
+ top = visRows(this.statusText + (ctxText ? ' '.repeat(2) + ctxText : ''), cols);
475
+ }
476
+ else if (ctxText) {
477
+ // Idle: no spinner, but the context still belongs on its own line just
478
+ // above the input, right-aligned.
479
+ out += ' '.repeat(Math.max(0, cols - visLen(ctxText))) + ctxText + NL;
480
+ top = visRows(ctxText, cols);
410
481
  }
411
482
  const lay = layoutInput(this.promptStr, this.buf, this.cursor, cols);
412
483
  out += lay.rows.map((r) => r.prefix + r.text).join(NL);
@@ -159,6 +159,53 @@ export function extractAnswer(body) {
159
159
  return sse;
160
160
  return extractFromJson(body);
161
161
  }
162
+ // The CONTEXT size (in tokens) DeepSeek reports for the current answer.
163
+ //
164
+ // chat.deepseek.com does not expose prompt_tokens/completion_tokens the way
165
+ // the API does. What it sends instead is `accumulated_token_usage` — a
166
+ // CUMULATIVE counter of the whole chat so far, present both in the SSE
167
+ // completion stream and (per message) in /api/v0/chat/history_messages. It
168
+ // is the number the operator wants for "how much context is used": the
169
+ // latest value is the current size of the chat context in tokens.
170
+ //
171
+ // SSE placement:
172
+ // * the initial fragment: v.response.accumulated_token_usage
173
+ // * an update chunk: {"p":"response","o":"BATCH",
174
+ // "v":[{"p":"accumulated_token_usage","v":N}, ...]}
175
+ // We take the LAST value seen (the freshest).
176
+ //
177
+ // Returns null when the body carries no counter (e.g. an OpenAI-shaped
178
+ // response or a non-answer endpoint) so the caller can keep the old value.
179
+ export function extractTokenUsage(body) {
180
+ let found = null;
181
+ const consider = (n) => {
182
+ if (typeof n === 'number' && Number.isFinite(n) && n >= 0)
183
+ found = n;
184
+ };
185
+ for (const obj of parseDataLines(body)) {
186
+ if (obj == null || typeof obj !== 'object')
187
+ continue;
188
+ const o = obj;
189
+ // The initial fragment: v.response.accumulated_token_usage.
190
+ const v = o.v;
191
+ const resp = v && typeof v === 'object'
192
+ ? v.response
193
+ : undefined;
194
+ if (resp)
195
+ consider(resp.accumulated_token_usage);
196
+ // A BATCH update: v is an array of {"p":"accumulated_token_usage","v":N}.
197
+ if (Array.isArray(o.v)) {
198
+ for (const item of o.v) {
199
+ if (item && typeof item === 'object') {
200
+ const it = item;
201
+ if (it.p === 'accumulated_token_usage')
202
+ consider(it.v);
203
+ }
204
+ }
205
+ }
206
+ }
207
+ return found;
208
+ }
162
209
  // Saves the DeepSeek network response body to disk for post-mortem analysis.
163
210
  // The files live in ~/.zames/net-log — from them the real answer format is visible.
164
211
  // DEBUG ONLY: disabled unless ZAMES_NET_DEBUG=1. It writes a file per network
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "zames_pro",
3
- "version": "2.18.0",
3
+ "version": "2.19.0",
4
4
  "description": "Terminal coding agent over chat.deepseek.com via Playwright",
5
5
  "type": "module",
6
6
  "bin": {