nexrall-code 0.5.116 → 0.5.118

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,2715 +0,0 @@
1
- import * as readline from 'readline';
2
- import * as path from 'path';
3
- import * as fs from 'fs';
4
- import * as os from 'os';
5
- import * as crypto from 'crypto';
6
- import { createRequire } from 'module';
7
- import { fileURLToPath } from 'url';
8
- import { execSync, execFileSync } from 'child_process';
9
- import chalk from 'chalk';
10
- import { runAgentLoop, getBalance, describeAttachment, CheckpointManager, loadPlugins, isDestructiveBash, compactMessagesForResume, readAllMemory, readMemory, memoryStats, clearMemory, loadSkills, findSkill, expandSkill, userInvokableSkills, salvageHistory, loadAgentTypesWithWarnings, McpManager, contextWindowFor, compactionThresholds, estimateTokensRough, estimateBodyBytes, getModelCatalogue, liveLabelFor, liveCostMultiplierFor, liveSupportsImageInput, liveSupportsPdfInput, liveReasoningEffortStyle, liveVendorFor, liveSelectableModelIds, createWorktree, removeWorktree, enterWorktree, findWorktreeByName, listWorktrees, worktreeHasWork, registerPeer, listPeers, startPeerInbox, } from '@nexrall/code-core';
11
- import { formatToolUse, formatToolResult, formatUsage, formatThinking, MarkdownStreamRenderer, ToolStreamPrinter } from '../ui/theme';
12
- import { requestPermission, setAutoApprove, initPermissions, setMode, setReadlineInterface, isYoloMode } from '../permissions/handler';
13
- import { currentAudit, enableAudit } from '../audit';
14
- import { describeMode, resetSessionSubAgentBudget } from '@nexrall/code-core';
15
- import { listSessions, loadSession, saveSession, lastSession } from './sessions';
16
- import { decideTerminalSetup, detectMacOSMajor, installEditorKeybinding, enableAppleTerminalOptionAsMeta, SHIFT_ENTER_SEQUENCE, MACOS_NATIVE_SHIFT_RETURN_MAJOR, } from '../terminal/terminalSetup';
17
- import { updateCommand } from './update';
18
- import { prepareSessionScreen, padToBottom, startRowCount, stopRowCount, addRowCount, renderBanner as renderBannerCard, } from '../ui/screen';
19
- import { detectTrustSignals, trustGrantedByEnv, TRUST_ENV_VAR } from '../trust';
20
- import { askForTrust } from '../ui/trustPrompt';
21
- import { startInkTerminal, stopInkTerminal, getInkTerminal, flushInkFrame, footerText } from '../ui/inkTerminal';
22
- import { InkReadlineAdapter } from '../ui/inkReadlineAdapter';
23
- // CLI version for the startup banner, read from package.json — the single
24
- // source of truth — rather than hand-typed here.
25
- //
26
- // This WAS `const CLI_VERSION = '0.5.63'`, kept in sync by hand per the release
27
- // checklist. index.ts had the identical literal and was fixed to read
28
- // package.json after it shipped a lagging `nex --version`; this copy was missed
29
- // and silently drifted SEVEN releases (banner said 0.5.63 while the installed
30
- // package was 0.5.70). A release checklist step that must be remembered is not
31
- // a guarantee — deriving it is.
32
- // Resolved by WALKING UP to the nearest package.json rather than a fixed
33
- // relative path, because this file runs from two different layouts and no
34
- // single literal is correct in both:
35
- // • built: bundled by esbuild into packages/cli/dist/index.js → '../package.json'
36
- // • source: executed by tsx from packages/cli/src/commands/ → '../../package.json'
37
- // Hardcoding either one leaves the other throwing MODULE_NOT_FOUND at startup
38
- // (the test suite catches exactly that, which is how this was found).
39
- const require = createRequire(import.meta.url);
40
- const CLI_VERSION = (() => {
41
- let dir = path.dirname(fileURLToPath(import.meta.url));
42
- // Bounded walk: the package root is 1-2 levels up in both layouts; the guard
43
- // stops at the filesystem root instead of looping if something is unexpected.
44
- for (let i = 0; i < 6; i++) {
45
- const candidate = path.join(dir, 'package.json');
46
- if (fs.existsSync(candidate)) {
47
- try {
48
- const v = require(candidate).version;
49
- if (typeof v === 'string' && v)
50
- return v;
51
- }
52
- catch { /* unreadable — keep walking */ }
53
- }
54
- const parent = path.dirname(dir);
55
- if (parent === dir)
56
- break;
57
- dir = parent;
58
- }
59
- // Never block startup over a banner string.
60
- return 'unknown';
61
- })();
62
- // ─── Model Map ────────────────────────────────────────────────────────────────
63
- // NOTE: alias → real model ID mapping lives on the BACKEND (backend/routes/code.js
64
- // CODE_MODEL_MAP: turbo→claude-sonnet-5, pro→claude-opus-4-8, ultra→claude-fable-5).
65
- // The CLI sends the alias as-is; only display labels are resolved locally.
66
- // Display labels for the models the picker offers, plus the legacy tier aliases
67
- // an older config may still hold.
68
- //
69
- // The labels are the real model names now: with more than one provider, "Pro"
70
- // says nothing about what is running or what it costs, while "Claude Opus 5"
71
- // and "GPT-5.4" do. Aliases stay accepted on the wire (the backend resolves
72
- // them) but are shown under their real name so the two never look like
73
- // different things.
74
- const MODEL_LABELS = {
75
- 'claude-sonnet-5': 'Claude Sonnet 5',
76
- 'claude-opus-5': 'Claude Opus 5',
77
- 'claude-fable-5': 'Claude Fable 5',
78
- 'claude-haiku-4-5-20251001': 'Claude Haiku 4.5',
79
- 'gpt-5.6-sol': 'GPT-5.6 Sol',
80
- 'gpt-5.6-terra': 'GPT-5.6 Terra',
81
- 'gpt-5.6-luna': 'GPT-5.6 Luna',
82
- 'gpt-5.4': 'GPT-5.4',
83
- 'gpt-5.4-mini': 'GPT-5.4 Mini',
84
- 'gpt-4.1': 'GPT-4.1',
85
- 'deepseek-flash': 'DeepSeek Flash',
86
- 'qwen3.7-max': 'Qwen3.7 Max',
87
- 'glm-5.3': 'GLM 5.3',
88
- };
89
- /** Legacy tier alias -> real model id. Kept so existing configs keep working. */
90
- const LEGACY_MODEL_ALIASES = {
91
- turbo: 'claude-sonnet-5',
92
- auto: 'claude-sonnet-5',
93
- pro: 'claude-opus-5',
94
- power: 'claude-opus-5',
95
- ultra: 'claude-fable-5',
96
- fast: 'claude-haiku-4-5-20251001',
97
- };
98
- /** Models offered by `/model` and `--model`, in display order (grouped by provider). */
99
- const SELECTABLE_MODELS = [
100
- 'claude-sonnet-5', 'claude-opus-5', 'claude-fable-5',
101
- 'gpt-5.6-sol', 'gpt-5.6-terra', 'gpt-5.6-luna',
102
- 'gpt-5.4', 'gpt-5.4-mini', 'gpt-4.1',
103
- 'deepseek-flash',
104
- 'qwen3.7-max',
105
- 'glm-5.3',
106
- ];
107
- // Relative cost vs gpt-4.1 / qwen3.7-max (=1x), for the "which one's cheaper"
108
- // hint printed by `/model` and shown as a badge in the VS Code model menu.
109
- // Rounded to friendly round numbers off backend/migrations/113_multi_provider_
110
- // pricing.sql and 114_deepseek_qwen_pricing.sql's per-1M-token pricing — a
111
- // relative signal only, not a live price feed. It will drift if pricing
112
- // changes without this table being updated by hand (no shared source between
113
- // this repo and the backend's model_pricing table).
114
- const MODEL_COST_MULTIPLIER = {
115
- 'claude-sonnet-5': 1.5,
116
- 'claude-opus-5': 3,
117
- 'claude-fable-5': 5,
118
- // GPT-5.6 family (2026-07 launch, Terra/Luna repriced 2026-07-30) — same
119
- // relative-to-gpt-4.1/qwen3.7-max (=1x) signal as every other row here,
120
- // off backend migration 132's per-1M-token seed prices (Sol $5/$30, Terra
121
- // $2/$12, Luna $0.20/$1.20 vs gpt-4.1's $2/$8).
122
- 'gpt-5.6-sol': 2,
123
- 'gpt-5.6-terra': 1.2,
124
- 'gpt-5.6-luna': 0.1,
125
- 'gpt-5.4': 2,
126
- 'gpt-5.4-mini': 0.5,
127
- 'gpt-4.1': 1,
128
- 'deepseek-flash': 0.1,
129
- 'qwen3.7-max': 1,
130
- // $1.40/$4.40 per 1M in/out (docs.z.ai, 2026-08-30) vs gpt-4.1's $2.00/$8.00
131
- // (migration 113) — roughly 0.6x on a blended 75/25 in/out turn. Rounded to
132
- // a friendly number like every other row; re-check backend/migrations/
133
- // 130_glm_5_3_pricing.sql's own caveat if Z.ai ships a GLM-5.3-specific rate.
134
- 'glm-5.3': 0.6,
135
- };
136
- function modelWithCostHint(id) {
137
- // Live catalogue first (GET /api/code/models, fetched once at session
138
- // start — see fetchLiveModelCatalogue below), falling back to this
139
- // hand-maintained table when offline or on an older backend.
140
- const mult = liveCostMultiplierFor(id, MODEL_COST_MULTIPLIER[id]);
141
- return mult === undefined ? id : `${id} (${mult}x)`;
142
- }
143
- // Models with NO image/vision input support, mirroring backend
144
- // services/providers/modelRegistry.js's `supportsImageInput: false` rows.
145
- // This is now the FALLBACK, not the source of truth: needsVisionSidecar()
146
- // below checks the live catalogue (GET /api/code/models) first, so a model
147
- // added to the backend registry gets the CORRECT gating even before this
148
- // table is updated by hand for a new nexrall-code release. Kept as a fallback
149
- // for offline use / older backends without the route.
150
- //
151
- // 2026-08-11+: no longer a hard refusal. A model in this set still cannot see
152
- // the raw bytes, but claude-sonnet-5 transcribes the attachment into a text
153
- // spec BEFORE it is ever pushed into `messages`, so the turn proceeds on the
154
- // model the user actually picked — see describeAttachment's header in
155
- // @nexrall/code-core for why this only needs to happen once, not per turn.
156
- //
157
- // deepseek-flash (DeepSeek-V4.1-Flash, renamed from deepseek-v4-pro 2026-09)
158
- // is DELIBERATELY absent here — DeepSeek's own pricing page marks Vision as
159
- // supported for this model (unlike the retired deepseek-v4-pro), so it sees
160
- // the raw image bytes directly and needs no sidecar. See backend
161
- // services/providers/modelRegistry.js's own comment on this id.
162
- const NO_VISION_MODELS = new Set(['qwen3.7-max', 'glm-5.3']);
163
- // Models with NO PDF/`document`-block support at all. Wider than
164
- // NO_VISION_MODELS: none of the OpenAI-compatible models (GPT included) speak
165
- // Anthropic's `document` content-block shape — there is no equivalent field in
166
- // `chat/completions` — so the backend degrades it to a text placeholder for
167
- // EVERY one of them (see backend streamFactory.js's `document` branch),
168
- // regardless of whether that model has vision support. deepseek-flash is
169
- // listed explicitly here (not inherited via NO_VISION_MODELS) since it has
170
- // vision but still has no PDF/document-block support.
171
- const NO_PDF_MODELS = new Set([
172
- 'gpt-5.6-sol', 'gpt-5.6-terra', 'gpt-5.6-luna',
173
- 'gpt-5.4', 'gpt-5.4-mini', 'gpt-4.1', 'gpt-4o-mini',
174
- 'deepseek-flash',
175
- ...NO_VISION_MODELS,
176
- ]);
177
- /**
178
- * Does `model` need the blind-attachment sidecar for an image? Live catalogue
179
- * first (populated by fetchLiveModelCatalogue at session start), static
180
- * NO_VISION_MODELS fallback otherwise — see that Set's own doc comment.
181
- */
182
- function needsVisionSidecar(model) {
183
- return !liveSupportsImageInput(model, !NO_VISION_MODELS.has(model));
184
- }
185
- /** Same live-first, static-fallback pattern as needsVisionSidecar, for PDFs. */
186
- function needsPdfSidecar(model) {
187
- return !liveSupportsPdfInput(model, !NO_PDF_MODELS.has(model));
188
- }
189
- /**
190
- * The models `/model` and `--model` offer, in display order. Live catalogue
191
- * first (already grouped/ordered by the backend the same way
192
- * SELECTABLE_MODELS is by hand), static fallback otherwise.
193
- */
194
- function selectableModelIds() {
195
- return liveSelectableModelIds() ?? SELECTABLE_MODELS;
196
- }
197
- export function normaliseModelId(model) {
198
- const raw = (model ?? '').trim();
199
- if (!raw)
200
- return 'claude-sonnet-5';
201
- return LEGACY_MODEL_ALIASES[raw.toLowerCase()] ?? raw;
202
- }
203
- /** Keyed by exact model id — mirrors backend modelRegistry.js's REASONING_EFFORT_STYLE. */
204
- const MODEL_EFFORT_STYLE = {
205
- // GPT-5.6 Sol/Terra/Luna hit the SAME "reasoning_effort + tools" 400 as
206
- // gpt-5.4/mini below (confirmed by multiple third-party integrations
207
- // 2026-07: github.com/BerriAI/litellm#33221, github.com/danny-avila/
208
- // LibreChat#14231) — OPENAI_GATED already handles that (skip when tools
209
- // are present), so no new style is needed for this family.
210
- 'gpt-5.6-sol': 'openai_gated',
211
- 'gpt-5.6-terra': 'openai_gated',
212
- 'gpt-5.6-luna': 'openai_gated',
213
- 'gpt-5.4': 'openai_gated',
214
- 'gpt-5.4-mini': 'openai_gated',
215
- 'gpt-4.1': 'none',
216
- 'gpt-4o-mini': 'none',
217
- 'deepseek-flash': 'deepseek',
218
- 'qwen3.7-max': 'qwen',
219
- 'glm-5.3': 'glm',
220
- // Everything else (all claude-* ids) falls through to 'anthropic' below.
221
- };
222
- const EFFORT_STYLE_CONFIG = {
223
- // Anthropic's real output_config.effort enum is low|medium|high|xhigh|max
224
- // (docs.claude.com extended output). 'extra'/'ultra' are legacy UI labels
225
- // this codebase invented, translated server-side by routes/code.js's
226
- // EFFORT_MAP (extra->xhigh, ultra->max) — kept as the wire values here
227
- // (rather than switching to 'xhigh'/'max' directly) only for backward
228
- // compatibility with saved sessions/scripts already passing them.
229
- anthropic: { levels: ['low', 'medium', 'high', 'extra', 'ultra'], names: ['Low', 'Medium', 'High', 'Extra High', 'Max'] },
230
- // DeepSeek V4 has THREE real buckets — live-verified against
231
- // api.deepseek.com 2026-08-30 (see streamFactory.test.js's DEEPSEEK case):
232
- // low->low, medium/high/xhigh->high, max->max.
233
- deepseek: { levels: ['low', 'high', 'max'], names: ['Low', 'Standard', 'Max'] },
234
- // Qwen3.7-Max's knob is a boolean (enable_thinking), not a graded scale.
235
- qwen: { levels: ['low', 'high'], names: ['Thinking Off', 'Thinking On'] },
236
- // GLM-5.3 cannot disable reasoning at all (a hard 400 if you try) and has
237
- // 3 real buckets (low/high/max) via a FLAT top-level `reasoning_effort`
238
- // string — live-verified against api.z.ai 2026-08-30 (see
239
- // streamFactory.test.js's GLM_THINKING_LEVEL case).
240
- glm: { levels: ['low', 'high', 'max'], names: ['Low', 'High', 'Max'] },
241
- // Effectively a no-op today whenever the turn has tools (almost always true
242
- // for nexrall-code) — see backend modelRegistry.js's OPENAI_GATED doc
243
- // comment. Still forwards a real value for the rare tool-less turn.
244
- openai_gated: { levels: ['low', 'medium', 'high'], names: ['Low', 'Medium', 'High'] },
245
- // No reasoning-effort concept on this model at all (gpt-4.1, gpt-4o-mini).
246
- none: { levels: ['medium'], names: ['N/A'] },
247
- };
248
- /**
249
- * Backend REASONING_EFFORT_STYLE value (modelRegistry.js) -> this file's
250
- * EffortStyle.
251
- */
252
- const _BACKEND_STYLE_TO_LOCAL = {
253
- openai_gated: 'openai_gated',
254
- deepseek: 'deepseek',
255
- qwen_thinking_toggle: 'qwen',
256
- glm_thinking_level: 'glm',
257
- none: 'none',
258
- };
259
- function effortStyleFor(modelId) {
260
- const id = normaliseModelId(modelId);
261
- // Anthropic's five-level lever isn't `reasoningEffortStyle` at all — the
262
- // backend reports 'none' for EVERY Anthropic model (Claude has no
263
- // `reasoning_effort` field; its lever is the separate output_config.effort
264
- // mechanism this file models as 'anthropic'). Detect by VENDOR rather than
265
- // by style string, mirroring webview/main.js's applyModelCatalogue (same
266
- // exception, same reason) — checking the static table instead would wrongly
267
- // fall through to 'none' for a brand-new Anthropic model this table has
268
- // never seen.
269
- if (liveVendorFor(id) === 'Anthropic')
270
- return 'anthropic';
271
- const liveStyle = liveReasoningEffortStyle(id);
272
- if (liveStyle && _BACKEND_STYLE_TO_LOCAL[liveStyle])
273
- return _BACKEND_STYLE_TO_LOCAL[liveStyle];
274
- return MODEL_EFFORT_STYLE[id] ?? 'anthropic';
275
- }
276
- export function effortConfigFor(modelId) {
277
- return EFFORT_STYLE_CONFIG[effortStyleFor(modelId)];
278
- }
279
- /**
280
- * Clamp a stored/requested effort token to one this model's scale actually
281
- * has. Without this, switching from Claude (5 levels) to Qwen (2 levels)
282
- * would leave an unsupported token in state — the CLI would echo back
283
- * "Effort → extra" for a model that silently treats it as its default.
284
- *
285
- * @returns a wire value guaranteed to be in `effortConfigFor(modelId).levels`
286
- */
287
- function clampEffortForModel(effort, modelId) {
288
- const config = effortConfigFor(modelId);
289
- if (config.levels.includes(effort))
290
- return effort;
291
- return defaultEffortForModel(modelId);
292
- }
293
- /** The "normal" notch for a model's scale — its SECOND level where one exists. */
294
- function defaultEffortForModel(modelId) {
295
- const levels = effortConfigFor(modelId).levels;
296
- return levels[Math.min(1, levels.length - 1)];
297
- }
298
- /**
299
- * Label for a model id or legacy alias.
300
- *
301
- * An UNKNOWN id echoes back as-is rather than silently displaying the default
302
- * model's name: this build ships independently of the backend, so a model added
303
- * server-side is valid before this table knows about it. Showing "Claude Sonnet
304
- * 5" for a session actually running something else would be a lie in the one
305
- * place the user looks to check.
306
- */
307
- function resolveModelLabel(alias) {
308
- const id = normaliseModelId(alias);
309
- return liveLabelFor(id, MODEL_LABELS[id] ?? id);
310
- }
311
- // Session helpers imported from sessions.ts
312
- // ─── Env collection ───────────────────────────────────────────────────────────
313
- function tryExec(cmd, cwd) {
314
- try {
315
- return execSync(cmd, { cwd, encoding: 'utf-8', stdio: ['ignore', 'pipe', 'ignore'] }).trim() || undefined;
316
- }
317
- catch {
318
- return undefined;
319
- }
320
- }
321
- // ─── Low/zero balance topup notice ────────────────────────────────────────────
322
- // Same deep-link the VS Code extension's gear-icon "Payment & Billing" and the
323
- // web app's ?settings=billing param both already use.
324
- const BILLING_URL = 'https://app.nexrall.com/?settings=billing';
325
- // OSC 8 terminal hyperlink — terminals that support it (iTerm2, Kitty, modern
326
- // Windows Terminal, VS Code's integrated terminal, etc.) render `label` as a
327
- // clickable link; terminals that don't just print `label` with the escape
328
- // codes stripped by their parser, so this degrades gracefully everywhere.
329
- function hyperlink(label, url) {
330
- return `\x1b]8;;${url}\x1b\\${label}\x1b]8;;\x1b\\`;
331
- }
332
- // Deduped per-process: only re-print when the state actually changes (low →
333
- // zero, or after a topup brings the balance back up and it drops again later)
334
- // — mirrors CanvasChat.jsx's lastBalanceNoticeRef, so a long session doesn't
335
- // reprint this on every single turn while the user stays under the threshold.
336
- let lastBalanceNoticeState = null;
337
- function printBalanceNotice(balance, zero) {
338
- const state = zero ? 'zero' : 'low';
339
- if (lastBalanceNoticeState === state)
340
- return;
341
- lastBalanceNoticeState = state;
342
- const title = zero ? 'Out of balance' : 'Balance running low';
343
- const detail = zero
344
- ? "You're out of funds — top up to keep the agent working."
345
- : `Your wallet is under $5${typeof balance === 'number' ? ` ($${balance.toFixed(2)} left)` : ''}. Top up before it runs out.`;
346
- console.log();
347
- console.log(chalk.bgYellow.black(` ${title} `) + ' ' + chalk.dim(detail));
348
- console.log(' ' + chalk.yellow(hyperlink('→ Top up', BILLING_URL)) + chalk.dim(` (${BILLING_URL})`));
349
- console.log();
350
- }
351
- /** OS username fallback for the "Welcome back <name>!" banner greeting. */
352
- function currentUserLabel() {
353
- try {
354
- const u = os.userInfo().username;
355
- return u ? u.charAt(0).toUpperCase() + u.slice(1) : 'there';
356
- }
357
- catch {
358
- return 'there';
359
- }
360
- }
361
- export function collectEnv(workDir) {
362
- return {
363
- cwd: workDir,
364
- platform: `${os.platform()} ${os.release()}`,
365
- shell: process.env.SHELL ?? process.env.ComSpec ?? 'unknown',
366
- date: new Date().toISOString().slice(0, 10),
367
- gitBranch: tryExec('git branch --show-current', workDir),
368
- gitStatus: tryExec('git status --short', workDir),
369
- gitDiff: tryExec('git diff --stat HEAD', workDir),
370
- recentCommits: tryExec('git log --oneline -5', workDir),
371
- };
372
- }
373
- // ─── nexrall.md reader — project + global instructions + memory ───────────────
374
- const NEXRALL_MD_CANDIDATES = [
375
- 'nexrall.md',
376
- 'NEXRALL.md',
377
- '.nexrall/instructions.md',
378
- '.nexrall/config.md',
379
- ];
380
- const tryRead = (p) => {
381
- try {
382
- return fs.existsSync(p) ? fs.readFileSync(p, 'utf-8').trim() : null;
383
- }
384
- catch {
385
- return null;
386
- }
387
- };
388
- export function readNexrallMd(workDir) {
389
- const parts = [];
390
- // 0. Persistent memory (project + global, see agent/memory.ts) — project-scoped
391
- // memory is keyed by workDir so it never leaks into an unrelated project's session.
392
- const memContent = readAllMemory(workDir);
393
- if (memContent)
394
- parts.push(memContent);
395
- // 1. Global user-level: ~/.nexrall/nexrall.md (applies to every project)
396
- const globalContent = tryRead(path.join(os.homedir(), '.nexrall', 'nexrall.md'));
397
- if (globalContent)
398
- parts.push(`[Global instructions]\n${globalContent}`);
399
- // 2. Project-level nexrall.md
400
- for (const candidate of NEXRALL_MD_CANDIDATES) {
401
- const content = tryRead(path.join(workDir, candidate));
402
- if (content) {
403
- parts.push(`[Project instructions — ${candidate}]\n${content}`);
404
- break;
405
- }
406
- }
407
- return parts.length > 0 ? parts.join('\n\n---\n\n') : undefined;
408
- }
409
- // ─── Spinner ──────────────────────────────────────────────────────────────────
410
- // How much of the model's live thinking text to keep on the one-line status row.
411
- // Short on purpose: it shares the row with the spinner, the elapsed timer and the
412
- // token counter, and the point is a glanceable sign of activity, not readable
413
- // prose — the full thinking block is printed properly by onThinking afterwards.
414
- const THINK_TAIL_CHARS = 48;
415
- /**
416
- * Compose the live status row. Pure + exported so the formatting rules can be
417
- * tested without an Ink instance, a real terminal or a running timer — the class
418
- * below owns only the timing/lifecycle, which is what makes this testable at all.
419
- */
420
- export function composeStatusLine(opts) {
421
- const timer = chalk.dim(` (${opts.elapsedSec}s)`);
422
- // Past 15s with zero output, a plain "running…" reads as hung even with the
423
- // counter ticking — add an explicit reassurance so the user knows nothing is
424
- // wrong. Suppressed when a detail is present: a live token counter is already
425
- // proof of life, so the sentence would be noise.
426
- const hint = opts.hint !== false && !opts.detail && opts.elapsedSec >= 15
427
- ? chalk.dim(' · still running, no output yet is normal for quiet commands')
428
- : '';
429
- const detail = opts.detail ? chalk.dim(` ${opts.detail}`) : '';
430
- return `${chalk.cyan(opts.frame)} ${chalk.dim(opts.text)}${timer}${detail}${hint}`;
431
- }
432
- /** Format a cumulative output-token count for the status row. */
433
- export function formatProgressTokens(tokens) {
434
- if (!Number.isFinite(tokens) || tokens <= 0)
435
- return '';
436
- return tokens >= 1000 ? `~${(tokens / 1000).toFixed(1)}k tokens` : `~${tokens} tokens`;
437
- }
438
- // A bare "running bash…" spinner with no elapsed-time counter is
439
- // indistinguishable from a genuinely hung process — a silent long-running
440
- // command (a directory scan, a slow network call with no progress output)
441
- // looked identical to a frozen terminal, which is what users reported as
442
- // "đứng yên không thấy tác động gì". Claude Code's own status line always
443
- // shows a running elapsed-seconds counter for exactly this reason: proof of
444
- // life doesn't require the underlying command to print anything itself.
445
- //
446
- // Renders through Ink's `setLive()` (an ephemeral, non-<Static> status line
447
- // the App component already exposes for exactly this) instead of a raw
448
- // `\r\x1b[K` in-place repaint — Ink owns the whole frame, so a spinner
449
- // fighting it for direct terminal control would just get overwritten by
450
- // Ink's own next render. In one-shot/headless runs there is no Ink instance
451
- // (getInkTerminal() returns null) and the spinner is silently a no-op, which
452
- // is correct: runTurn() is never invoked on that path (see runTurnHeadless).
453
- class Spinner {
454
- interval = null;
455
- frames = ['◐', '◓', '◑', '◒'];
456
- i = 0;
457
- startedAt = 0;
458
- baseText = '';
459
- // Extra live detail appended after the elapsed timer (e.g. "~1.2k tokens").
460
- // Kept separate from `baseText` and mutated through setDetail() so a stream of
461
- // progress events can refresh it WITHOUT restarting the spinner — calling
462
- // start() again would reset `startedAt` and the elapsed counter would sit at 0s
463
- // forever, which is precisely the "is it frozen?" signal this class exists to
464
- // remove.
465
- detail = '';
466
- // Whether the reassurance hint applies. Off for phases where slowness is
467
- // already explained by the detail text (e.g. a live token counter is itself
468
- // proof of life, so "no output yet is normal" would be noise).
469
- hintEnabled = true;
470
- start(text, opts) {
471
- this.stop();
472
- this.baseText = text;
473
- this.detail = '';
474
- this.hintEnabled = opts?.hint !== false;
475
- this.startedAt = Date.now();
476
- this.render();
477
- this.interval = setInterval(() => this.render(), 100);
478
- }
479
- /**
480
- * Replace the trailing detail text in place, keeping the elapsed timer running.
481
- *
482
- * No-op when the spinner isn't active: progress events can arrive a tick after
483
- * something else (a tool row, the first text delta) legitimately stopped it, and
484
- * resurrecting the spinner there would fight the printed output for the status line.
485
- *
486
- * Deliberately does NOT call render() immediately. onThinkingDelta/onThinkingProgress
487
- * can fire many times per second while the model streams (one SSE delta each), and
488
- * each of those used to trigger its own Ink re-render (setLive -> React state update ->
489
- * full-frame clear/redraw) layered on TOP of the interval's own 100ms redraw already
490
- * running in start(). Two independent, unsynchronized render sources hitting the same
491
- * frame is exactly what produced the visible jitter users reported ("dòng working...
492
- * giật giật liên tục") — every fast delta briefly repainted the status line out of
493
- * step with the spinner's own tick. Just updating the field here and letting the
494
- * existing setInterval pick it up on its next 100ms tick caps the redraw rate at a
495
- * steady 10fps with zero perceptible added latency (worst case: one stale frame for
496
- * <100ms), and removes the race entirely.
497
- */
498
- setDetail(detail) {
499
- if (this.interval === null)
500
- return;
501
- this.detail = detail;
502
- }
503
- /** Swap the label without resetting the elapsed timer (phase change within one turn). Same reasoning as setDetail() above: no immediate render, the interval tick picks it up. */
504
- setText(text) {
505
- if (this.interval === null)
506
- return;
507
- this.baseText = text;
508
- }
509
- render() {
510
- getInkTerminal()?.setLive(composeStatusLine({
511
- frame: this.frames[this.i % this.frames.length],
512
- text: this.baseText,
513
- elapsedSec: Math.floor((Date.now() - this.startedAt) / 1000),
514
- detail: this.detail,
515
- hint: this.hintEnabled,
516
- }));
517
- this.i++;
518
- }
519
- stop() {
520
- if (this.interval !== null) {
521
- clearInterval(this.interval);
522
- this.interval = null;
523
- this.detail = '';
524
- getInkTerminal()?.setLive('');
525
- }
526
- }
527
- get active() { return this.interval !== null; }
528
- }
529
- // ─── MCP (Model Context Protocol) ──────────────────────────────────────────────
530
- //
531
- // Historically only the VS Code extension connected to configured MCP servers
532
- // (ChatPanel._initMcp) — the CLI never read .nexrall/mcp.json at all, so any
533
- // MCP tool a user configured silently didn't exist as far as the CLI's agent
534
- // loop was concerned (no error, the model just never saw those tools). This
535
- // module-level singleton is initialized once per process in startChatSession
536
- // and threaded into every runTurn/runTurnHeadless call below, so MCP tools
537
- // now work identically across both surfaces.
538
- //
539
- // NOT ported: the VS Code extension's OAuth 2.1 device-flow for remote
540
- // (HTTP/SSE) servers lacking an explicit Authorization header (it stores
541
- // tokens in vscode.SecretStorage, which has no CLI equivalent). A remote MCP
542
- // server that needs OAuth still works here if the user supplies a static
543
- // Authorization header in mcp.json; interactive OAuth login is VS Code-only
544
- // for now (tracked as a follow-up, not silently broken — /mcp reports it).
545
- let _mcpManager = null;
546
- let _mcpConfig = null;
547
- async function initMcp(workDir) {
548
- _mcpConfig = McpManager.loadConfig(workDir);
549
- if (Object.keys(_mcpConfig.mcpServers).length === 0) {
550
- _mcpManager = null;
551
- return;
552
- }
553
- const manager = new McpManager();
554
- await manager.connectAll(_mcpConfig);
555
- _mcpManager = manager;
556
- }
557
- /** Human-readable MCP status for the `/mcp` slash command. */
558
- function formatMcpStatus() {
559
- if (!_mcpConfig || Object.keys(_mcpConfig.mcpServers).length === 0) {
560
- return chalk.dim(' No MCP servers configured. Add one to .nexrall/mcp.json or ~/.nexrall/mcp.json.');
561
- }
562
- if (!_mcpManager)
563
- return chalk.dim(' MCP servers configured but not connected.');
564
- const rows = _mcpManager.getStatus();
565
- const lines = rows.map((s) => {
566
- if (s.connected) {
567
- return ' ' + chalk.green('●') + ' ' + chalk.bold(s.name) + chalk.dim(` (${s.transport}) — ${s.tools.length} tool(s)`);
568
- }
569
- const authNote = s.needsAuth ? chalk.yellow(' [needs OAuth — not supported in CLI yet, add a static Authorization header]') : '';
570
- return ' ' + chalk.red('●') + ' ' + chalk.bold(s.name) + chalk.dim(` (${s.transport}) — `) + chalk.red(s.error ?? 'failed to connect') + authNote;
571
- });
572
- return lines.join('\n');
573
- }
574
- /**
575
- * `/context` — show current conversation size against the model's context
576
- * window and the auto-prune/auto-compact thresholds, so a long session's
577
- * housekeeping (which fires silently mid-run) is visible on demand instead of
578
- * being a surprise. Uses the same rough token estimate the auto-compact guard
579
- * itself relies on for a resumed session (no extra API call needed).
580
- */
581
- function formatContextUsage(messages, modelAlias) {
582
- const window = contextWindowFor(modelAlias);
583
- const tokens = estimateTokensRough(messages);
584
- const bytes = estimateBodyBytes(messages);
585
- const { prune, compact } = compactionThresholds();
586
- const pct = Math.min(100, (tokens / window) * 100);
587
- const barWidth = 30;
588
- const filled = Math.round((pct / 100) * barWidth);
589
- const bar = '█'.repeat(filled) + '░'.repeat(barWidth - filled);
590
- const barColor = pct >= compact * 100 ? chalk.red : pct >= prune * 100 ? chalk.yellow : chalk.green;
591
- const lines = [
592
- '',
593
- ' ' + chalk.bold('Context usage'),
594
- ' ' + barColor(bar) + chalk.dim(` ~${pct.toFixed(1)}%`),
595
- chalk.dim(` ~${tokens.toLocaleString()} / ${window.toLocaleString()} tokens (rough estimate) · ${(bytes / 1024).toFixed(0)}KB serialized · ${messages.length} messages`),
596
- chalk.dim(` auto-prune at ~${(prune * 100).toFixed(0)}% · auto-compact (summarise) at ~${(compact * 100).toFixed(0)}%`),
597
- '',
598
- ];
599
- return lines.join('\n');
600
- }
601
- /** `/agents` — list discovered custom sub-agent types (built-in + project + global + plugin). */
602
- function formatAgentsList(workDir) {
603
- const { types, warnings } = loadAgentTypesWithWarnings(workDir);
604
- if (!types.length)
605
- return chalk.dim(' No agent types found (this should not happen — builtins always register).');
606
- const lines = types.map((t) => {
607
- const tools = t.tools ? chalk.dim(` — tools: ${t.tools.join(', ')}`) : chalk.dim(' — ALL tools (no allowlist)');
608
- const model = t.model ? chalk.dim(` — model: ${t.model}`) : '';
609
- const scope = chalk.dim(` (${t.source})`);
610
- return ' ' + chalk.cyan(t.name.padEnd(16)) + chalk.white(t.description) + scope + tools + model;
611
- });
612
- // Problems are shown right where the user is looking at agents. Each of these
613
- // silently changed what an agent could do, which is precisely why they must not
614
- // stay quiet.
615
- if (warnings.length) {
616
- lines.push('');
617
- lines.push(' ' + chalk.yellow.bold(`⚠ ${warnings.length} problem${warnings.length > 1 ? 's' : ''} in your agent definitions:`));
618
- for (const w of warnings) {
619
- lines.push(' ' + chalk.yellow(`• ${w.agent}`) + chalk.dim(` — ${w.message}`));
620
- lines.push(' ' + chalk.dim(w.file));
621
- }
622
- }
623
- return lines.join('\n');
624
- }
625
- // ─── Run One Turn via Agent Loop ──────────────────────────────────────────────
626
- async function runTurn(messages, modelAlias, workDir, abortSignal, env, nexrallMd, mode, effort, checkpointManager, onProgress,
627
- // Optional Claude-Code-style mid-turn follow-ups (see the InkReadlineAdapter
628
- // comment for why `rl.pause()` no longer blocks typing): when provided,
629
- // the loop drains these at every turn boundary and folds them into the
630
- // conversation instead of waiting for the whole turn to finish. Omitted by
631
- // callers that run a turn OUTSIDE the main REPL loop (e.g. /compact, /init)
632
- // where there is no live rl to queue against.
633
- takePendingInput,
634
- // Active worktree isolation, if this session was started with --worktree
635
- // or resumed into one — threaded straight into AgentLoopOptions.worktree
636
- // so core's hard enforcement (checkWorktreeIsolation) is actually engaged.
637
- // Undefined for an ordinary session, matching every existing call site.
638
- worktreeState,
639
- // Cross-session messaging context, set up once in startChatSession when
640
- // this session registers as a discoverable peer (interactive sessions
641
- // only — see peerHandle's own comment there). Bundled into one object
642
- // (rather than three more positional params on an already-long signature)
643
- // since all three always travel together — see runAgentLoop's identical
644
- // fields in types.ts for what each one actually does.
645
- peerCtx) {
646
- let lastUsage;
647
- const spinner = new Spinner();
648
- const mdRender = new MarkdownStreamRenderer();
649
- const toolStream = new ToolStreamPrinter();
650
- let toolStartTime = 0;
651
- let lastToolName = '';
652
- let thinkingTokens = 0;
653
- // Live one-line detail for the status line while the model is working.
654
- // Composed from the two independent progress signals the backend sends
655
- // (a cumulative output-token count, and the thinking text itself) so the
656
- // status line reflects whichever has arrived.
657
- let liveThinkTail = '';
658
- const pushStatus = () => {
659
- const parts = [];
660
- const tok = formatProgressTokens(thinkingTokens);
661
- if (tok)
662
- parts.push(tok);
663
- if (liveThinkTail)
664
- parts.push(liveThinkTail);
665
- spinner.setDetail(parts.join(' · '));
666
- };
667
- // Resume the "model is working" status line after something printed over it.
668
- //
669
- // Every printed row (a tool call, a tool result, a notice) stops the spinner,
670
- // but the agent loop then goes straight back to the model — which can spend a
671
- // long time on prompt processing before the next token. Without restarting the
672
- // spinner there, the terminal falls silent again after every single tool round,
673
- // which is the bulk of a long agentic run.
674
- const resumeWorking = (label = 'working…') => {
675
- if (abortSignal.aborted)
676
- return;
677
- thinkingTokens = 0;
678
- liveThinkTail = '';
679
- spinner.start(label, { hint: false });
680
- };
681
- // Apply the active mode to the permission gate so 'plan' hard-refuses writes
682
- // (matches the VS Code panel — enforcement, not just a system-prompt request).
683
- setMode(mode);
684
- console.log();
685
- // Proof of life from the very first millisecond of the turn.
686
- //
687
- // Nothing used to be shown between the user pressing Enter and the first text
688
- // token, and on a large context that gap is genuinely long (the client allows
689
- // FIRST_EVENT_TIMEOUT_MS = 300 s for prompt processing before it even calls it a
690
- // stall). A dead terminal for tens of seconds is indistinguishable from a hang,
691
- // which is the single biggest reason this CLI *felt* slower than it is.
692
- //
693
- // `hint: false` because the reassurance copy is written for quiet shell commands
694
- // ("no output yet is normal"); here the live token counter below is itself the
695
- // proof of life, so the extra sentence would just be noise.
696
- spinner.start('thinking…', { hint: false });
697
- const result = await runAgentLoop(messages, {
698
- workDir,
699
- model: modelAlias,
700
- env,
701
- nexrallMd,
702
- mode,
703
- // `/mode plan` previously only hinted to the backend — the footer said "plan
704
- // mode on" while every write tool stayed live, so the one mode whose entire
705
- // promise is "I will not touch anything" was the one not enforced. This
706
- // makes it real: the loop refuses mutating tools before the permission
707
- // prompt, and sub-agents inherit the lock.
708
- planMode: mode === 'plan',
709
- // Same "enforced, not just hinted" reasoning as planMode directly above:
710
- // when set, core's checkWorktreeIsolation refuses any write/bash-cwd/git-
711
- // redirect that targets outside this worktree, BEFORE the permission
712
- // prompt — undefined (the default) changes nothing for an ordinary session.
713
- worktree: worktreeState,
714
- // Cross-session messaging — see peerCtx's own doc comment above for why
715
- // these three travel together. Undefined (spread of an undefined
716
- // object's properties is a no-op) for a session that isn't registered
717
- // as a peer, changing nothing for that case.
718
- selfPeer: peerCtx?.selfPeer,
719
- drainPeerMessages: peerCtx?.drainPeerMessages,
720
- onPeerMessage: peerCtx?.onPeerMessage,
721
- effort,
722
- // Claude-Code-style follow-ups: text the user typed and sent WHILE this
723
- // turn was already running. inkTerminal.tsx echoes each one into the
724
- // transcript itself the moment it's submitted (chronological, matching
725
- // the VS Code panel's queueMessage) — so onInjectedInput is intentionally
726
- // a no-op here rather than printing it again.
727
- takePendingInput,
728
- onInjectedInput: () => { },
729
- // The backend streams a cumulative output-token count (routes/code.js's
730
- // sendProgress, throttled to ≤5/s) covering thinking, visible text AND
731
- // tool-argument JSON. This used to be stored in a variable and rendered only
732
- // at message_complete — i.e. after the turn was already over — so during the
733
- // long phase it exists to describe, it showed nothing at all.
734
- onThinkingProgress: (tokens) => {
735
- thinkingTokens = tokens;
736
- pushStatus();
737
- },
738
- // Live thinking text. Previously a hard no-op with the note "shown on
739
- // onThinking" — but onThinking only fires at message_complete, so a long
740
- // reasoning phase rendered nothing whatsoever until it had finished. Show a
741
- // short rolling tail on the status line so the user can see it actively
742
- // reasoning, then let onThinking print the proper summary block at the end.
743
- onThinkingDelta: (text) => {
744
- if (abortSignal.aborted)
745
- return;
746
- // Collapse to a single line: the status line is one row, and a raw newline
747
- // would tear Ink's frame.
748
- const flat = text.replace(/\s+/g, ' ');
749
- liveThinkTail = (liveThinkTail + flat).slice(-THINK_TAIL_CHARS);
750
- pushStatus();
751
- },
752
- onThinking: (text) => {
753
- if (abortSignal.aborted)
754
- return;
755
- spinner.stop();
756
- console.log(formatThinking(text, thinkingTokens));
757
- thinkingTokens = 0;
758
- liveThinkTail = '';
759
- // The model keeps generating after a thinking block (text, or a tool call
760
- // whose arguments can take a while to stream) — keep the status line alive
761
- // instead of going dark until the next event lands.
762
- resumeWorking();
763
- },
764
- onText: (text) => {
765
- if (abortSignal.aborted)
766
- return;
767
- spinner.stop();
768
- mdRender.feed(text); // stream through markdown renderer
769
- },
770
- // System notices (mid-run auto-prune/auto-compact housekeeping) are NOT part
771
- // of the model's own reply — used to go through onText, which spliced
772
- // "♻️ Trimmed ~0.3MB…" straight into the markdown stream renderer as if the
773
- // model itself had said it. Print it as its own dim line instead (same
774
- // treatment as the resume-time compaction notice below).
775
- onNotice: (text) => {
776
- if (abortSignal.aborted)
777
- return;
778
- mdRender.flush();
779
- spinner.stop();
780
- console.log(chalk.dim(` ${text}`));
781
- resumeWorking();
782
- },
783
- onToolUse: (name, input) => {
784
- if (abortSignal.aborted)
785
- return;
786
- mdRender.flush(); // finalize any in-progress line before tool
787
- spinner.stop();
788
- toolStream.reset();
789
- console.log();
790
- console.log(formatToolUse(name, input));
791
- lastToolName = name;
792
- toolStartTime = Date.now();
793
- spinner.start(`running ${name}…`);
794
- },
795
- // Live progress for foreground bash: the FIRST chunk stops the "running…"
796
- // spinner (which would otherwise overwrite/interleave badly with printed
797
- // lines) and every subsequent chunk streams straight to the terminal —
798
- // see ToolStreamPrinter for the line-buffering/cap logic.
799
- onToolStreamChunk: (_name, chunk) => {
800
- if (abortSignal.aborted)
801
- return;
802
- if (spinner.active)
803
- spinner.stop();
804
- toolStream.feed(chunk);
805
- },
806
- onToolResult: (_name, res) => {
807
- spinner.stop();
808
- toolStream.flush(); // emit any trailing partial line the stream held back
809
- const durationMs = Date.now() - toolStartTime;
810
- console.log(formatToolResult(lastToolName, res, durationMs));
811
- toolStartTime = 0;
812
- // Back to the model with this result — that round trip is often the longest
813
- // silent stretch in an agentic run, so keep the status line up.
814
- resumeWorking();
815
- },
816
- // Ignore a `partial` report: it belongs to a cut-short attempt that was restarted,
817
- // and the replacement attempt reports the turn's real totals. Letting it through
818
- // would print a token count for output the user never saw.
819
- onUsage: (u, partial) => { if (!partial)
820
- lastUsage = u; },
821
- // A transient disconnect (network drop, machine sleep/wake, overloaded upstream)
822
- // is retried transparently by the network layer — without this, that pause was
823
- // completely invisible: the CLI just appeared to freeze and then resume with no
824
- // explanation. Reuse the same spinner to show what's actually happening.
825
- onRetry: (attempt, _maxAttempts, reason) => {
826
- if (abortSignal.aborted)
827
- return;
828
- spinner.start(`${reason}… (attempt ${attempt})`);
829
- },
830
- onRetryResolved: () => {
831
- spinner.stop();
832
- },
833
- // The stream died after part of the answer had already been printed, and the
834
- // turn is being restarted from the top. Providing this handler is what OPTS US
835
- // IN to post-render restarts at all (see AgentLoopOptions.onStreamRestart) —
836
- // without it a mid-answer disconnect kills the whole turn.
837
- //
838
- // A terminal can't unprint scrolled-away output, so instead of pretending the
839
- // fragment never happened we (a) reset the markdown parser so the dead attempt's
840
- // half-open code fence/table can't corrupt everything the retry prints, and
841
- // (b) draw an explicit marker so the user understands why the answer restarts.
842
- onStreamRestart: (reason, discardedChars) => {
843
- if (abortSignal.aborted)
844
- return;
845
- spinner.stop();
846
- mdRender.reset();
847
- if (discardedChars > 0) {
848
- console.log('\n' + chalk.yellow(' ↺ Connection dropped mid-answer — restarting this response.') +
849
- chalk.dim(`\n (${reason}. The ${discardedChars} characters above are incomplete; the full answer follows.)`) + '\n');
850
- }
851
- },
852
- onBalanceStatus: (balance, zero) => {
853
- spinner.stop();
854
- mdRender.flush();
855
- printBalanceNotice(balance, zero);
856
- },
857
- requestPermission,
858
- checkpointManager,
859
- onProgress,
860
- mcpManager: _mcpManager ?? undefined,
861
- // undefined unless --audit was passed, in which case the loop's behaviour
862
- // is unchanged. Read per turn (not captured once) so /clear and /resume,
863
- // which reassign sessionId, are reflected in the trail immediately.
864
- audit: currentAudit(),
865
- });
866
- mdRender.flush(); // flush any remaining buffered text
867
- spinner.stop();
868
- if (lastUsage) {
869
- console.log(chalk.dim(' ' + formatUsage(lastUsage.output_tokens)));
870
- }
871
- return { messages: result, usage: lastUsage };
872
- }
873
- // ─── Headless Turn (JSON output modes) ─────────────────────────────────────
874
- //
875
- // For scripting/CI: `nex -p "..." --output-format json` emits a single JSON
876
- // object on stdout; `stream-json` emits one JSON object per line as events
877
- // happen (NDJSON). All permissions are auto-approved (headless implies yolo —
878
- // there is no TTY to prompt on) and nothing decorative is written to stdout.
879
- async function runTurnHeadless(messages, modelAlias, workDir, _abortSignal, env, nexrallMd, mode, effort, format, checkpointManager, onProgress,
880
- // See runTurn's identical parameter for what this threads through to.
881
- worktreeState) {
882
- let lastUsage;
883
- let resultText = '';
884
- let toolCallCount = 0;
885
- const emit = (obj) => {
886
- if (format === 'stream-json')
887
- process.stdout.write(JSON.stringify(obj) + '\n');
888
- };
889
- emit({ type: 'start', model: modelAlias, workDir });
890
- const result = await runAgentLoop(messages, {
891
- workDir,
892
- model: modelAlias,
893
- env,
894
- nexrallMd,
895
- mode,
896
- planMode: mode === 'plan',
897
- worktree: worktreeState,
898
- effort,
899
- abortSignal: _abortSignal,
900
- onText: (text) => { resultText += text; emit({ type: 'text', text }); },
901
- // ── Reasoning-phase liveness ──────────────────────────────────────
902
- // These used to be UNWIRED here while runTurn() (interactive) wired all
903
- // three, and the asymmetry was not cosmetic: on a reasoning model at high
904
- // effort the thinking phase emits no text and no tool calls, so a headless
905
- // consumer saw ZERO events for the entire phase and could not distinguish
906
- // "actively reasoning" from "process wedged".
907
- //
908
- // Measured on a Terminal-Bench 4.0 trial (2026-08-30, deepseek-v4-pro
909
- // --effort max): nex-output.jsonl sat at 4 lines for 65 MINUTES while
910
- // tcpdump inside the container's netns showed ~25 packets/s still flowing
911
- // and the turn ultimately reported 310,872 output tokens. Nothing was
912
- // actually wrong — but every signal available to the harness (log line
913
- // count, file mtime) said "hung", and the socket/CPU forensics needed to
914
- // prove otherwise are not something a CI wrapper can do.
915
- //
916
- // `thinking_progress` is the load-bearing one: it carries the backend's
917
- // cumulative output-token count (routes/code.js's sendProgress, already
918
- // throttled to <=5/s server-side, so this cannot flood the log) and fires
919
- // DURING the phase. That makes it a real heartbeat.
920
- onThinkingProgress: (tokens) => { emit({ type: 'thinking_progress', tokens }); },
921
- // Fires once per turn at message_complete with the full reasoning text.
922
- // Typed `thinking` deliberately: the bench's ATIF converter (atif.py)
923
- // already routes exactly this event type into the trajectory's reasoning
924
- // buffer, so wiring it here also fills in reasoning that was previously
925
- // dropped on the floor for every headless run.
926
- onThinking: (text) => { emit({ type: 'thinking', text }); },
927
- // onThinkingDelta is intentionally NOT wired. Its chunks concatenate to the
928
- // same string `onThinking` emits in full above, so emitting both would
929
- // duplicate the entire reasoning trace — on the 310k-token turn measured
930
- // above that is megabytes of redundant NDJSON — while adding no liveness
931
- // signal `thinking_progress` does not already provide.
932
- // System notice (mid-run auto-prune/auto-compact) — emit as its own event
933
- // type instead of falling through to onText, so a stream-json consumer
934
- // doesn't see compaction housekeeping text mixed into the model's `text`
935
- // events or accumulated into resultText.
936
- onNotice: (text) => { emit({ type: 'notice', text }); },
937
- onToolUse: (name, input) => {
938
- emit({ type: 'tool_use', tool: name, input });
939
- },
940
- // Live bash progress for scripted/CI consumers — lets a wrapper tail a
941
- // long-running build/test command instead of blocking silently until
942
- // tool_result. Chunk is raw and un-truncated (unlike the final result,
943
- // which is head/tail-capped), so a chatty command can emit many of these.
944
- onToolStreamChunk: (name, chunk) => {
945
- emit({ type: 'tool_stream', tool: name, chunk });
946
- },
947
- onToolResult: (name, res) => {
948
- toolCallCount++;
949
- emit({ type: 'tool_result', tool: name, ok: res.error === undefined, ...(res.error ? { error: res.error } : {}) });
950
- },
951
- // Headless consumers get the partial report too — tagged, so a script can account
952
- // for the real cost of a restarted turn — but it never becomes `lastUsage`, which
953
- // represents the turn's actual output.
954
- onUsage: (u, partial) => {
955
- if (!partial)
956
- lastUsage = u;
957
- emit({ type: 'usage', usage: u, ...(partial ? { partial: true } : {}) });
958
- },
959
- onRetry: (attempt, maxAttempts, reason) => {
960
- emit({ type: 'retry', attempt, max_attempts: maxAttempts, reason });
961
- },
962
- // Headless consumers parse NDJSON, so a restart is trivially clean for them:
963
- // they simply drop every `text`/`thinking` event seen since the turn started.
964
- // Emitting it also opts headless mode into post-render restarts, so a scripted
965
- // /CI run survives a blip instead of exiting non-zero halfway through.
966
- onStreamRestart: (reason, discardedChars) => {
967
- emit({ type: 'stream_restart', reason, discarded_chars: discardedChars });
968
- },
969
- onBalanceStatus: (balance, zero) => {
970
- emit({ type: 'balance_status', balance, zero, billing_url: BILLING_URL });
971
- },
972
- // Headless → auto-approve (no TTY to ask on). Destructive/irreversible
973
- // commands (DB drops, force-push, terraform destroy…) fail CLOSED here: with
974
- // no human to confirm, they are denied unless NEXRALL_ALLOW_DESTRUCTIVE=1 is
975
- // explicitly set for this CI run.
976
- requestPermission: async ({ tool, input }) => {
977
- const d = isDestructiveBash(tool, input);
978
- if (d && process.env.NEXRALL_ALLOW_DESTRUCTIVE !== '1') {
979
- emit({ type: 'error', error: `Destructive command blocked in headless mode (${d.category}): ${d.reason}. Set NEXRALL_ALLOW_DESTRUCTIVE=1 to permit.` });
980
- return false;
981
- }
982
- return true;
983
- },
984
- checkpointManager,
985
- onProgress,
986
- mcpManager: _mcpManager ?? undefined,
987
- // Threaded here as well as in runTurn(). This is the path where the trail
988
- // matters MOST — headless auto-approves every tool call because there is no
989
- // TTY to ask on, so nothing else records what an unattended CI run touched.
990
- audit: currentAudit(),
991
- });
992
- if (format === 'json') {
993
- process.stdout.write(JSON.stringify({
994
- type: 'result',
995
- result: resultText.trim(),
996
- tool_calls: toolCallCount,
997
- usage: lastUsage ?? null,
998
- num_messages: result.length,
999
- }) + '\n');
1000
- }
1001
- else {
1002
- emit({ type: 'result', result: resultText.trim(), usage: lastUsage ?? null });
1003
- }
1004
- return { messages: result, usage: lastUsage };
1005
- }
1006
- // ─── Slash Command Help ───────────────────────────────────────────────────────
1007
- function printHelp() {
1008
- console.log();
1009
- console.log(chalk.bold(' Slash commands:'));
1010
- const cmds = [
1011
- ['/exit, /quit', 'Exit Nexrall Code'],
1012
- ['/clear', 'Clear conversation history'],
1013
- ['/sessions', 'List saved sessions'],
1014
- ['/resume [id]', 'Resume last or specific saved session'],
1015
- ['/save', 'Save current session now'],
1016
- ['/rewind [id]', 'List file checkpoints, or roll back to one'],
1017
- ['/compact', 'Summarize conversation to save tokens'],
1018
- ['/init', 'Generate nexrall.md for this project'],
1019
- ['/memory [global] [archived] [clear]', 'View persistent memory (project or global); archived also shows superseded facts; clear wipes it'],
1020
- ['/worktree [list]', 'Show this session\'s worktree isolation status, or list every worktree (start one with `nex --worktree [name]`)'],
1021
- ['/peers [send <name> <msg>|broadcast <msg>]', 'List other discoverable Nexrall Code sessions, message one directly, or broadcast to all'],
1022
- ['/help', 'Show this help'],
1023
- ['/model [name]', 'Switch model (e.g. claude-opus-5, gpt-5.4)'],
1024
- ['/mode [ask|edit|plan|auto]', 'Set agent mode'],
1025
- ['/effort [level]', 'Set thinking effort level (options depend on the current model — run /effort with no argument to see them)'],
1026
- ['/yolo', 'Auto-approve all permissions'],
1027
- ['/balance', 'Show wallet balance'],
1028
- ['/add <filepath>', 'Add a file to conversation context'],
1029
- ['/image <filepath> [caption]', 'Attach an image or PDF (png/jpg/gif/webp/pdf, max 10MB) and send it to the model'],
1030
- ['/skills', 'List skills (.nexrall/skills/<name>/SKILL.md or .nexrall/commands/*.md) — the model can also auto-invoke these'],
1031
- ['/plugins', 'List installed plugins (.nexrall/plugins/)'],
1032
- ['/mcp', 'Show MCP server connection status'],
1033
- ['/agents', 'List available sub-agent types (for the task tool)'],
1034
- ['/context', 'Show context window usage for this conversation'],
1035
- ['/terminal-setup', 'Configure this terminal so Shift+Enter inserts a newline'],
1036
- ['/update', 'Update nex to the latest version'],
1037
- ];
1038
- for (const [cmd, desc] of cmds) {
1039
- console.log(' ' + chalk.cyan(cmd.padEnd(35)) + chalk.dim(desc));
1040
- }
1041
- // The footer advertises "/help for shortcuts", but this only ever listed
1042
- // slash commands — so the keys that need explaining most (how to type a
1043
- // newline without submitting) were documented nowhere at all.
1044
- console.log();
1045
- console.log(chalk.bold(' Keys:'));
1046
- const keys = [
1047
- ['Enter', 'Send the message'],
1048
- ['Ctrl+J', 'Insert a newline — works in every terminal, no setup'],
1049
- ['\\ then Enter', 'Continue on the next line'],
1050
- ['Shift+Enter', 'Insert a newline (run /terminal-setup once if it sends instead)'],
1051
- ['Option+Enter', 'Insert a newline (macOS; /terminal-setup enables it)'],
1052
- ['Ctrl+C', 'Stop the current turn, or press twice to exit'],
1053
- ['←/→', 'Move the cursor within the line'],
1054
- ];
1055
- for (const [key, desc] of keys) {
1056
- console.log(' ' + chalk.cyan(key.padEnd(35)) + chalk.dim(desc));
1057
- }
1058
- console.log();
1059
- }
1060
- // ─── Session List Display ─────────────────────────────────────────────────────
1061
- function printSessions() {
1062
- const sessions = listSessions().slice(0, 20);
1063
- if (!sessions.length) {
1064
- console.log(chalk.dim(' No saved sessions yet.'));
1065
- return;
1066
- }
1067
- console.log();
1068
- console.log(chalk.bold(' Saved sessions:'));
1069
- for (const s of sessions) {
1070
- const when = new Date(s.updatedAt).toLocaleString();
1071
- const title = s.title.length > 50 ? s.title.slice(0, 50) + '…' : s.title;
1072
- console.log(' ' + chalk.cyan(s.id.slice(0, 8).padEnd(10)) +
1073
- chalk.dim(`${when} `) +
1074
- chalk.white(title));
1075
- }
1076
- console.log(chalk.dim('\n Use /resume <id> to continue a session.'));
1077
- console.log();
1078
- }
1079
- // ─── Start Chat Session ───────────────────────────────────────────────────────
1080
- export async function startChatSession(options) {
1081
- // Headless JSON output → stdout must contain ONLY JSON. Suppress all decor.
1082
- const headless = options.outputFormat === 'json' || options.outputFormat === 'stream-json';
1083
- const requestedWorkDir = options.workDir;
1084
- // Reassigned below, ONLY when --worktree is active, to the worktree's own
1085
- // directory — every downstream use of `workDir` in this function (env
1086
- // collection, nexrall.md, MCP config, checkpoints, session storage, and the
1087
- // AgentLoopOptions.workDir passed to runAgentLoop) then transparently
1088
- // operates on the isolated directory instead of the main checkout, with no
1089
- // other code in this file needing to know isolation is active. `let`, not
1090
- // `const`, purely to allow that one reassignment — nothing else in this
1091
- // function mutates it.
1092
- let workDir = requestedWorkDir;
1093
- // Set once a worktree is created/resumed below; threaded into every
1094
- // runAgentLoop call's `worktree:` option so the hard enforcement in
1095
- // core/agent/loop.ts (checkWorktreeIsolation) is actually active for this
1096
- // session. Undefined for an ordinary (non-isolated) session — the default,
1097
- // unchanged behaviour.
1098
- let worktreeState;
1099
- // Set below (interactive sessions only — see its own registration block)
1100
- // once this session registers itself as a discoverable peer. Threaded into
1101
- // AgentLoopOptions as `selfPeer`/`drainPeerMessages` so the
1102
- // list_peer_sessions/message_peer_session tools and the inbox-drain hook
1103
- // in loop.ts actually work. Undefined for a one-shot/headless run — a
1104
- // process that exits before any OTHER session could plausibly reach it
1105
- // has nothing to gain from the registration overhead, mirroring why
1106
- // Claude Code's own docs describe messaging as meaningful between
1107
- // LONG-LIVED sessions.
1108
- let peerHandle;
1109
- let peerInbox;
1110
- const peerInboxQueue = [];
1111
- let modelAlias = options.model;
1112
- // Kick off the live model catalogue fetch (GET /api/code/models) in the
1113
- // BACKGROUND, not awaited: every call site that reads it
1114
- // (modelWithCostHint/resolveModelLabel/effortStyleFor/needsVisionSidecar/
1115
- // needsPdfSidecar/selectableModelIds, all in this file) falls back to its
1116
- // hand-maintained table when the cache isn't populated yet, so there is
1117
- // nothing to block on. Started this early so the round-trip has the
1118
- // longest possible head start — the trust prompt and startup banner below
1119
- // give it several hundred ms to resolve before the user can reach `/model`.
1120
- // Failures are swallowed inside getModelCatalogue itself; nothing here
1121
- // needs a .catch — but one is defensive against a future change removing
1122
- // that guarantee, since an unhandled rejection would otherwise crash a CLI
1123
- // session over a model-picker enrichment call.
1124
- void getModelCatalogue().catch(() => { });
1125
- // ── Workspace trust gate ──────────────────────────────────────────────────
1126
- //
1127
- // This MUST run before anything reads the working directory. Opening a
1128
- // folder is not passive: `.nexrall/mcp.json` causes `spawn()` of the
1129
- // commands it names, `nexrall.md` is injected into the system prompt as
1130
- // trusted instructions, and `.nexrall/{skills,commands,permissions.json}`
1131
- // add playbooks and pre-approved tool permissions. Without this gate,
1132
- // `git clone` of a hostile repo followed by `nex` runs attacker-chosen
1133
- // commands before the user has typed anything.
1134
- //
1135
- // Note the ordering constraint: readNexrallMd/loadSkills/initMcp all appear
1136
- // BELOW this point on purpose. Moving any of them above it would reintroduce
1137
- // the hole this closes.
1138
- //
1139
- // The answer is deliberately NOT remembered between sessions (see trust.ts):
1140
- // trust would be granted to a path, but the risk lives in the folder's
1141
- // contents, and a later `git pull` can add a .nexrall/mcp.json that would
1142
- // then run without ever being announced.
1143
- //
1144
- // Only interactive TTY sessions can show a prompt. Non-interactive runs
1145
- // (piped stdin, one-shot `-p`, headless JSON) have no way to ask, so rather
1146
- // than hanging on a prompt nobody can see — or silently proceeding — they
1147
- // refuse, unless the invoker pre-authorised the run via the environment.
1148
- const canPrompt = !headless &&
1149
- !options.prompt &&
1150
- !options.stdinText &&
1151
- process.stdin.isTTY === true &&
1152
- process.stdout.isTTY === true;
1153
- if (!trustGrantedByEnv()) {
1154
- if (!canPrompt) {
1155
- process.stderr.write(`Refusing to run in an unconfirmed folder:\n ${workDir}\n\n` +
1156
- 'This folder can configure the agent (.nexrall/mcp.json can run local ' +
1157
- 'commands, nexrall.md steers the agent).\n' +
1158
- `Set ${TRUST_ENV_VAR}=1 to confirm non-interactively, or run \`nex\` ` +
1159
- 'interactively here to review it.\n');
1160
- process.exit(1);
1161
- }
1162
- const signals = detectTrustSignals(workDir);
1163
- const trusted = await askForTrust(workDir, signals);
1164
- if (!trusted) {
1165
- // Ink clears its final frame as it unmounts, and process.exit() can cut
1166
- // off a write that hasn't flushed yet. Defer the message to the next
1167
- // tick so it lands after the unmount and is actually visible.
1168
- await new Promise((r) => setImmediate(r));
1169
- process.stdout.write('Not trusted — exiting without loading this folder.\n');
1170
- process.exit(0);
1171
- }
1172
- // ── Close the raw-mode gap between the trust prompt and the main UI ──────
1173
- //
1174
- // askForTrust() runs in its OWN short-lived Ink instance. When the user
1175
- // answers, that instance's exit() synchronously unmounts, and Ink's
1176
- // unmount() calls `stdin.setRawMode(false)` — handing the terminal back to
1177
- // the OS in cooked mode WITH LOCAL ECHO. The main UI's Ink instance
1178
- // (startInkTerminal(), below) doesn't re-enable raw mode until its own
1179
- // render() runs, which happens after collectEnv/readNexrallMd/
1180
- // prepareSessionScreen() below.
1181
- //
1182
- // That gap is real wall-clock time (measured ~150-300ms with a synchronous
1183
- // pty harness), and terminal echo is done by the OS tty driver, not by
1184
- // Node — so any key typed during it is echoed by the KERNEL at whatever
1185
- // row the cursor happened to land on after the trust prompt's last frame
1186
- // (i.e. one row below "Enter to confirm..."), not drawn into the input box
1187
- // Ink hasn't mounted yet. That is the exact "text appears below the
1188
- // chrome, only reappearing correctly once enough is typed" symptom this
1189
- // fixes: once the main Ink instance's useInput() effect fires and
1190
- // re-enables raw mode, it starts rendering correctly again, so short
1191
- // bursts of fast typing right after confirming trust were the ones
1192
- // visibly torn.
1193
- //
1194
- // We can enable raw mode here unconditionally: reaching this line means
1195
- // `canPrompt` was true, which requires an interactive TTY session, so
1196
- // `interactive` is guaranteed to be true below. Re-enabling here just
1197
- // closes the window; Ink's own setRawMode(true) call once it mounts is a
1198
- // harmless no-op on top of this (see ink's App.js rawModeEnabledCount).
1199
- //
1200
- // MUST be deferred past the current microtask queue via setImmediate, not
1201
- // called synchronously right after `await askForTrust(...)`. Ink's own
1202
- // raw-mode teardown for the JUST-UNMOUNTED trust prompt is itself queued
1203
- // with `queueMicrotask()` (see ink's App.js `handleSetRawMode`'s disable
1204
- // branch) — it hasn't run yet at this point, it only runs once the
1205
- // microtask queue drains. Calling `setRawMode(true)` synchronously here
1206
- // races that pending microtask: ordering isn't guaranteed to put ours
1207
- // last, so the trust prompt's own delayed `setRawMode(false)` can still
1208
- // fire AFTER ours and undo it, or the two calls can otherwise leave
1209
- // stdin's raw-mode/`readable` listener bookkeeping inconsistent — this
1210
- // was reproduced directly as `nex` hanging forever after confirming
1211
- // trust (this line was reached, but the session never rendered).
1212
- // `setImmediate` schedules a macrotask, which only runs after Node fully
1213
- // drains the microtask queue — guaranteeing Ink's pending teardown has
1214
- // already completed by the time we re-enable raw mode.
1215
- await new Promise((resolve) => setImmediate(resolve));
1216
- if (process.stdin.isTTY)
1217
- process.stdin.setRawMode(true);
1218
- }
1219
- // ── Worktree isolation (--worktree) ─────────────────────────────────────
1220
- //
1221
- // Deliberately AFTER the trust gate above (isolation is a property of an
1222
- // already-trusted folder, not a substitute for trusting it) and BEFORE
1223
- // collectEnv/readNexrallMd below, so every subsequent read of `workDir` in
1224
- // this function — env, nexrall.md, MCP config, checkpoints, session
1225
- // storage, and the AgentLoopOptions passed to runAgentLoop — transparently
1226
- // operates on the worktree instead of the main checkout.
1227
- if (options.worktree !== undefined) {
1228
- const requestedName = typeof options.worktree === 'string' ? options.worktree : undefined;
1229
- const existing = requestedName ? findWorktreeByName(requestedWorkDir, requestedName) : null;
1230
- if (existing) {
1231
- const entered = enterWorktree(existing);
1232
- if (!entered.ok) {
1233
- process.stderr.write(`Failed to resume worktree "${requestedName}": ${entered.error}\n`);
1234
- process.exit(1);
1235
- }
1236
- worktreeState = existing;
1237
- if (!headless)
1238
- console.log(chalk.dim(` Resumed worktree: ${chalk.bold(existing.name)} (${existing.worktreePath})`));
1239
- }
1240
- else {
1241
- const created = createWorktree(requestedWorkDir, requestedName);
1242
- if (!created.ok || !created.state) {
1243
- process.stderr.write(`Failed to create worktree: ${created.error ?? 'unknown error'}\n`);
1244
- process.exit(1);
1245
- }
1246
- worktreeState = created.state;
1247
- if (!headless) {
1248
- console.log(chalk.green(` Created worktree: ${chalk.bold(worktreeState.name)}`));
1249
- console.log(chalk.dim(` ${worktreeState.worktreePath}`));
1250
- if (worktreeState.branch)
1251
- console.log(chalk.dim(` Branch: ${worktreeState.branch}`));
1252
- console.log();
1253
- }
1254
- }
1255
- // Every later use of `workDir` in this function now transparently
1256
- // resolves inside the isolated directory — this is the ONE reassignment
1257
- // `let workDir` above exists to allow.
1258
- workDir = worktreeState.worktreePath;
1259
- // Auto-cleanup on process exit — mirrors Claude Code's own worktree
1260
- // lifecycle ("When the agent finishes and the worktree is clean, it's
1261
- // automatically cleaned up"). Registered on `process.on('exit', …)`
1262
- // rather than at any of this function's many individual
1263
- // `process.exit()` call sites, because 'exit' fires for every one of
1264
- // them uniformly (Node still runs 'exit' listeners even after
1265
- // `process.exit()` is called) — one registration point instead of
1266
- // needing to thread cleanup through 15+ separate exit paths.
1267
- //
1268
- // Deliberately SYNCHRONOUS only: Node does not run the event loop during
1269
- // 'exit', so nothing async here would ever complete — worktreeHasWork
1270
- // and removeWorktree are both plain sync fs/spawnSync calls for exactly
1271
- // this reason (see worktree.ts).
1272
- process.on('exit', () => {
1273
- if (!worktreeState)
1274
- return;
1275
- const hadWork = worktreeHasWork(worktreeState);
1276
- if (hadWork) {
1277
- // Matches Claude Code's own choice here: a worktree with real,
1278
- // uncommitted work is left in place rather than silently discarded,
1279
- // with the exact resume incantation printed so nothing is lost.
1280
- process.stderr.write(`\nWorktree "${worktreeState.name}" has uncommitted work and was left in place:\n` +
1281
- ` ${worktreeState.worktreePath}\n` +
1282
- `Resume it with: nex --worktree ${worktreeState.name}\n`);
1283
- return;
1284
- }
1285
- removeWorktree(worktreeState);
1286
- });
1287
- }
1288
- // Collect runtime env
1289
- const env = collectEnv(workDir);
1290
- const nexrallMd = readNexrallMd(workDir);
1291
- // Interactive-only (no one-shot prompt, no headless output) sessions take
1292
- // over the screen: both the visible screen and the scrollback are cleared so
1293
- // the session starts on a clean slate with the banner at the top, and
1294
- // scrolling up later reveals the banner and earlier turns rather than the
1295
- // shell prompts that preceded `nex`. See ui/screen.ts's prepareSessionScreen
1296
- // for why this uses the normal screen buffer rather than the alternate one.
1297
- //
1298
- // One-shot/headless runs never clear anything: they're meant to compose with
1299
- // other commands in a pipeline and must leave the user's scrollback intact.
1300
- const interactive = !headless && !options.prompt && !options.stdinText;
1301
- // ── Cross-session peer discovery + messaging ────────────────────────────
1302
- //
1303
- // Register this session as a discoverable peer BEFORE the REPL starts, so
1304
- // it is visible to any other session's list_peer_sessions from the very
1305
- // first turn. Interactive-only, deliberately (see peerHandle's own field
1306
- // comment for why). No flag to opt in/out — matches Claude Code's own
1307
- // default (cross-session messaging is on by default, refusable per-message
1308
- // via the target's own inbound policy, not a global toggle a user has to
1309
- // discover first).
1310
- let peerHeartbeatTimer;
1311
- if (interactive) {
1312
- // registerPeer mints the id first; startPeerInbox's socket is named
1313
- // after that SAME id, so registration happens before the inbox starts.
1314
- // The record is then updated with the real socketPath and re-persisted
1315
- // (via heartbeat()) so other sessions' list_peer_sessions see a
1316
- // reachable peer immediately, not one whose socketPath is still empty.
1317
- const handle = registerPeer({ clientType: 'cli', workDir: requestedWorkDir });
1318
- const inbox = startPeerInbox(handle.record.id, { onMessage: (m) => peerInboxQueue.push(m) });
1319
- handle.record.socketPath = inbox.socketPath;
1320
- handle.heartbeat();
1321
- peerHandle = handle;
1322
- peerInbox = inbox;
1323
- peerHeartbeatTimer = setInterval(() => handle.heartbeat(), 10_000);
1324
- // Timers keep the process alive even with nothing else pending — must
1325
- // not prevent normal exit once the REPL itself decides to quit.
1326
- peerHeartbeatTimer.unref?.();
1327
- process.on('exit', () => {
1328
- peerInbox?.close();
1329
- peerHandle?.unregister();
1330
- if (peerHeartbeatTimer)
1331
- clearInterval(peerHeartbeatTimer);
1332
- });
1333
- }
1334
- // Built once — passed to every runTurn call site below (see runTurn's
1335
- // peerCtx parameter). undefined selfPeer/drainPeerMessages/onPeerMessage
1336
- // when peerHandle never got set (non-interactive) makes every field a
1337
- // no-op in the loop, identical to today's behaviour.
1338
- const peerCtx = {
1339
- selfPeer: peerHandle ? { id: peerHandle.record.id, name: peerHandle.record.name } : undefined,
1340
- drainPeerMessages: peerHandle ? () => peerInboxQueue.splice(0) : undefined,
1341
- onPeerMessage: peerHandle
1342
- ? (msgs) => {
1343
- for (const m of msgs) {
1344
- if (!headless)
1345
- console.log(chalk.dim(`\n › Message from @${m.from}: `) + chalk.white(m.text));
1346
- }
1347
- }
1348
- : undefined,
1349
- };
1350
- if (interactive) {
1351
- prepareSessionScreen();
1352
- process.on('exit', () => { stopInkTerminal(); });
1353
- // ── Close the SAME raw-mode gap as above, for the "already trusted"
1354
- // fast path ──────────────────────────────────────────────────────────
1355
- //
1356
- // The `setRawMode(true)` fix a few lines up (search "Close the raw-mode
1357
- // gap between the trust prompt and the main UI") only runs when the
1358
- // trust prompt was actually shown. When the workspace is already
1359
- // trusted — `NEXRALL_TRUST_WORKSPACE=1`, or trust remembered from an
1360
- // earlier `nex` invocation in this folder — the entire `if
1361
- // (!trustGrantedByEnv())` block above is skipped, so stdin is left at
1362
- // whatever mode the process started in (cooked, with the kernel doing
1363
- // local echo) all the way from process start until Ink's own
1364
- // `useInput()` effect fires inside `startInkTerminal()` below. That is
1365
- // the exact same "typed text appears one row below the input box, only
1366
- // reappearing correctly once enough is typed" symptom, just triggered on
1367
- // ordinary startup instead of right after a trust decision — reachable
1368
- // on literally every `nex` run in an already-trusted folder, not just
1369
- // the first one. Setting raw mode here, synchronously and unconditionally
1370
- // right before Ink mounts, closes it: there is no pending Ink
1371
- // teardown microtask to race against here (unlike the post-trust-prompt
1372
- // case), since no other Ink instance has run yet in this process.
1373
- if (process.stdin.isTTY)
1374
- process.stdin.setRawMode(true);
1375
- // Boot the Ink app FIRST, before anything (including the banner below)
1376
- // prints — Ink's own default `patchConsole: true` behavior (see
1377
- // ink/build/render.js) only takes effect once `render()` has run, so
1378
- // every console.log call after this point transparently prints above
1379
- // the input box/footer this component owns. The Spinner class and
1380
- // MarkdownStreamRenderer/ToolStreamPrinter (theme.ts) — the only other
1381
- // interactive-path writers — use console.log / Ink's own setLive() API,
1382
- // so there is no separate bridge needed for raw process.stdout.write().
1383
- await startInkTerminal();
1384
- // Begin tallying startup output so padToBottom() below knows how far to
1385
- // push the input line + footer down. This must come AFTER
1386
- // startInkTerminal(): Ink's patchConsole replaces console.log during
1387
- // render(), so counting first would have Ink's patch overwrite the
1388
- // counting wrapper and every line would go uncounted.
1389
- startRowCount();
1390
- }
1391
- if (options.showBanner !== false && !headless) {
1392
- const branch = tryExec('git branch --show-current', workDir);
1393
- const gitStatus = tryExec('git status --short', workDir);
1394
- const changedLines = gitStatus ? gitStatus.split('\n').filter(Boolean).length : 0;
1395
- const tips = ['Run /init to create a nexrall.md file with instructions for Nexrall Code.'];
1396
- if (branch) {
1397
- tips.push(`On branch ${branch}${changedLines > 0 ? ` (${changedLines} changed)` : ''}.`);
1398
- }
1399
- // Box-drawn "Welcome back" card matching Claude Code's own startup
1400
- // screen (chalk/Unicode only — see conversation: Ink's Yoga layout
1401
- // engine can't be bundled into the single-file esbuild output `nex`
1402
- // ships as, so this reimplements just the visual layout by hand).
1403
- //
1404
- // Everything BUT `workDir` in this info bundle is captured once here and
1405
- // reused verbatim by the regenerator below — none of it changes for the
1406
- // life of the session (version/user/tips are fixed at startup, and the
1407
- // model label only changes via /model, which doesn't re-print the
1408
- // banner). Only `columns` varies per call, which is the ENTIRE reason
1409
- // this needs to be regenerable at all — see setBannerRegenerator's doc.
1410
- const bannerInfo = {
1411
- version: CLI_VERSION,
1412
- userLabel: currentUserLabel(),
1413
- modelLabel: resolveModelLabel(modelAlias),
1414
- workDir,
1415
- nexrallMdLoaded: !!nexrallMd,
1416
- tips,
1417
- };
1418
- if (interactive) {
1419
- // Routed through print(isBanner: true) instead of a plain console.log
1420
- // so a later resize can find and replace it — see
1421
- // AppHandle.setBannerRegenerator's own doc for the full mechanism
1422
- // (verified end-to-end against a real `claude` binary in a pty
1423
- // harness: banner/transcript/footer all reprint at the CURRENT column
1424
- // count on every resize, not just at startup).
1425
- const bannerCard = renderBannerCard({ ...bannerInfo, columns: process.stdout.columns ?? 80 });
1426
- getInkTerminal().print(bannerCard, { isBanner: true });
1427
- getInkTerminal().setBannerRegenerator((columns) => renderBannerCard({ ...bannerInfo, columns }));
1428
- // print() above goes through Ink's <Static> list, not console.log, so
1429
- // the startRowCount() tally (which only patches console.log/error) is
1430
- // otherwise blind to the banner's ~15-18 rows entirely — see
1431
- // addRowCount's own doc for the padToBottom() breakage that caused.
1432
- addRowCount(bannerCard);
1433
- }
1434
- else {
1435
- console.log(renderBannerCard({ ...bannerInfo, columns: process.stdout.columns ?? 80 }));
1436
- }
1437
- console.log();
1438
- }
1439
- let messages = [];
1440
- let sessionId = crypto.randomBytes(8).toString('hex');
1441
- let sessionTitle = '';
1442
- let agentMode = 'auto';
1443
- // Effort is now validated PER MODEL (effortConfigFor's levels), not against
1444
- // one fixed 4-level list shared by every provider — see the "Per-model
1445
- // effort scale" section above for why. A caller-supplied value (the
1446
- // -e/--effort CLI flag, or a headless harness's CLI_FLAGS) that isn't on
1447
- // the CURRENT model's scale clamps to that model's "normal" notch instead
1448
- // of sending a value the backend/model has never seen.
1449
- let effortLevel = options.effort
1450
- ? clampEffortForModel(options.effort, modelAlias)
1451
- : defaultEffortForModel(modelAlias);
1452
- const permRules = initPermissions(workDir);
1453
- const ruleCount = permRules.allow.length + permRules.ask.length + permRules.deny.length;
1454
- if (ruleCount && !headless)
1455
- console.log(chalk.dim(` ${ruleCount} permission rule(s) loaded`));
1456
- let skills = loadSkills(workDir);
1457
- if (skills.length && !headless)
1458
- console.log(chalk.dim(` ${skills.length} skill(s) loaded`));
1459
- // MCP servers configured in .nexrall/mcp.json (project or global) — connect
1460
- // once up front, same as the VS Code extension's _initMcp. Errors per-server
1461
- // are non-fatal (recorded in getStatus() for `/mcp`); a server that fails to
1462
- // connect just contributes no tools rather than blocking the whole session.
1463
- //
1464
- // ── Why an INTERACTIVE session does not await this ────────────────────────
1465
- //
1466
- // connectAll() spawns every stdio server and performs a full JSON-RPC
1467
- // handshake (initialize + tools/list) against every remote one, in parallel
1468
- // but bounded by the SLOWEST of them. Measured against a real 5-server
1469
- // mcp.json: 8.0s, dominated by one server whose endpoint now returns 410.
1470
- //
1471
- // Awaiting that here — between Ink mounting and padToBottom() below — is the
1472
- // direct cause of the reported "footer starts in the wrong place and only
1473
- // drops to the bottom after a while": the input box and footer paint
1474
- // immediately (measured 170ms) wherever the banner ended, then sit there
1475
- // for the whole handshake because the padding that pushes them down is
1476
- // queued behind this await. Measured end-to-end: chrome visible at 170ms,
1477
- // final position at 8651ms — an 8.5s window in which the UI looks wrong
1478
- // and typed characters land against a stale frame.
1479
- //
1480
- // Interactive sessions therefore START the connect and carry on painting.
1481
- // Nothing is lost: `_mcpManager` is read at CALL time when each turn is
1482
- // dispatched (see the `mcpManager: _mcpManager ?? undefined` argument in
1483
- // runTurn/runTurnHeadless), never captured at startup, and the REPL awaits
1484
- // `mcpReady()` before the first turn — so the model can never see a
1485
- // half-connected tool list. The only visible difference is that the
1486
- // "N/M MCP server(s) connected" line now arrives when it's true rather
1487
- // than holding the whole UI hostage until it is.
1488
- //
1489
- // Headless/one-shot runs still await: they dispatch their single turn
1490
- // immediately with no REPL to defer to, and printing progress into a
1491
- // machine-readable stream is not a concern there.
1492
- const mcpStarted = (async () => {
1493
- try {
1494
- await initMcp(workDir);
1495
- const connected = _mcpManager?.serverNames.length ?? 0;
1496
- const total = _mcpConfig ? Object.keys(_mcpConfig.mcpServers).length : 0;
1497
- if (total && !headless) {
1498
- console.log(chalk.dim(` ${connected}/${total} MCP server(s) connected`) + chalk.dim(' (/mcp for details)'));
1499
- }
1500
- }
1501
- catch { /* non-fatal — session works fine with zero MCP tools */ }
1502
- })();
1503
- // Awaited before the first turn (and by /mcp) so no turn ever runs against a
1504
- // half-connected tool list. Never rejects — the IIFE above swallows errors.
1505
- const mcpReady = () => mcpStarted;
1506
- if (!interactive)
1507
- await mcpStarted;
1508
- // Headless output only makes sense with a one-shot prompt (or piped stdin).
1509
- if (headless && !options.prompt && !options.stdinText) {
1510
- process.stdout.write(JSON.stringify({ type: 'error', error: '--output-format json/stream-json requires a one-shot prompt (or piped stdin).' }) + '\n');
1511
- process.exit(1);
1512
- }
1513
- // ── Resume session ────────────────────────────────────────────────────────
1514
- if (options.resume) {
1515
- const stored = options.resume === 'last'
1516
- ? lastSession()
1517
- : loadSession(options.resume);
1518
- if (!stored) {
1519
- console.error(chalk.red(` Session not found: ${options.resume}`));
1520
- console.error(chalk.dim(' Use /sessions to list saved sessions.'));
1521
- process.exit(1);
1522
- }
1523
- messages = stored.messages;
1524
- sessionId = stored.id;
1525
- sessionTitle = stored.title;
1526
- // Resuming an old session and sending one message used to resend the WHOLE
1527
- // stored history at full price: the in-loop auto-compact guard only reacts
1528
- // to lastPromptTokens from a PREVIOUS turn's usage event, which doesn't
1529
- // exist yet on turn 0 of a freshly-resumed session. Proactively compact
1530
- // here, before the first new message is even sent.
1531
- try {
1532
- const compacted = await compactMessagesForResume(messages, {
1533
- workDir,
1534
- model: modelAlias,
1535
- clientType: 'cli',
1536
- onNotice: (text) => { if (!headless)
1537
- console.log(chalk.dim(text.trim())); },
1538
- });
1539
- void compacted;
1540
- }
1541
- catch { /* non-fatal — worst case the session resends uncompacted */ }
1542
- if (!headless) {
1543
- console.log(chalk.green(` Resumed session: ${chalk.bold(stored.title.slice(0, 60))}`));
1544
- console.log(chalk.dim(` ${messages.length} messages restored`));
1545
- console.log();
1546
- }
1547
- }
1548
- // Checkpoints are scoped per session so resuming a session (even after a
1549
- // process restart) restores its rewind history from disk.
1550
- const checkpoints = new CheckpointManager(workDir, sessionId);
1551
- // Enabled HERE, next to the checkpoint manager, because both key off
1552
- // sessionId and this is the first point where it is final: the --resume
1553
- // branch above replaces it, so enabling any earlier would print (and, for
1554
- // the default path, derive) a trail location for a session that was then
1555
- // discarded. Thereafter the id is read LIVE per turn — /clear and /resume
1556
- // reassign it mid-session, and a trail stamped with a stale id would
1557
- // attribute work to the wrong conversation.
1558
- if (options.audit) {
1559
- const auditPath = enableAudit({
1560
- filePath: typeof options.audit === 'string' ? options.audit : undefined,
1561
- workDir,
1562
- getSessionId: () => sessionId,
1563
- });
1564
- if (!headless)
1565
- console.log(chalk.dim(` Audit trail: ${auditPath}`));
1566
- }
1567
- // Helper: derive session title from first user message
1568
- const updateTitle = () => {
1569
- if (sessionTitle)
1570
- return;
1571
- const first = messages.find(m => m.role === 'user');
1572
- const raw = first?.content[0]?.text ?? '';
1573
- sessionTitle = raw.length > 60 ? raw.slice(0, 60) + '…' : raw;
1574
- };
1575
- // A turn died (dead stream, upstream error, crashed tool). The agent loop works
1576
- // on its own copy of the history, so WITHOUT this the REPL would fall back to a
1577
- // `messages` array that still contains only the original user message — every
1578
- // completed tool round of a long run silently evaporates, even though the user
1579
- // was already billed for it and "continue" then has nothing to continue FROM.
1580
- // salvageHistory() returns the loop's partial history when real progress was
1581
- // made, so we adopt it, persist it, and tell the user it's safe to resume.
1582
- const recoverTurn = (err, current) => {
1583
- const salvaged = salvageHistory(err);
1584
- console.error('\n' + chalk.red('Error: ') + String(err.message));
1585
- if (!salvaged)
1586
- return current;
1587
- // progressCount is already a count of completed tool ROUNDS, tracked by the loop.
1588
- // Deriving it from array lengths here would be wrong: auto-compaction can splice the
1589
- // history shorter mid-run, so the delta doesn't reflect what the user watched happen.
1590
- const rounds = err.progressCount ?? 1;
1591
- console.error(chalk.dim(` Kept the ${rounds} step${rounds === 1 ? '' : 's'} completed before the interruption — ` +
1592
- `send "continue" to pick up where it stopped.`));
1593
- return salvaged;
1594
- };
1595
- // Incremental session persistence: fired at every turn boundary inside the
1596
- // agent loop so a crash / kill hours into a long run only loses the in-flight
1597
- // step. Debounced (≤ every 3s) to keep disk writes cheap on fast tool rounds.
1598
- let lastProgressSave = 0;
1599
- const saveProgress = (live) => {
1600
- const now = Date.now();
1601
- if (now - lastProgressSave < 3_000)
1602
- return;
1603
- lastProgressSave = now;
1604
- try {
1605
- updateTitle();
1606
- if (live.length)
1607
- saveSession(sessionId, sessionTitle || 'Untitled', workDir, live);
1608
- }
1609
- catch { /* best-effort — never let persistence break the run */ }
1610
- };
1611
- // ── One-shot mode ─────────────────────────────────────────────────────────
1612
- if (options.prompt || options.stdinText) {
1613
- // If stdin is piped, prepend it to the prompt
1614
- let text = options.prompt ?? '';
1615
- if (options.stdinText) {
1616
- text = options.stdinText + (text ? `\n\n${text}` : '');
1617
- }
1618
- checkpoints.beginTurn(text, messages.length);
1619
- messages.push({ role: 'user', content: [{ type: 'text', text }] });
1620
- const abortSignal = { aborted: false };
1621
- const onSigint = () => { abortSignal.aborted = true; };
1622
- process.once('SIGINT', onSigint);
1623
- try {
1624
- const fmt = options.outputFormat;
1625
- const result = (fmt === 'json' || fmt === 'stream-json')
1626
- ? await runTurnHeadless(messages, modelAlias, workDir, abortSignal, env, nexrallMd, agentMode, effortLevel, fmt, checkpoints, saveProgress, worktreeState)
1627
- : await runTurn(messages, modelAlias, workDir, abortSignal, env, nexrallMd, agentMode, effortLevel, checkpoints, saveProgress, undefined, worktreeState, peerCtx);
1628
- messages = result.messages;
1629
- checkpoints.commitTurn();
1630
- updateTitle();
1631
- saveSession(sessionId, sessionTitle, workDir, messages);
1632
- // One-shot must exit explicitly on success, matching the error path below
1633
- // (process.exit(1)). Without this, Node's event loop only exits once every
1634
- // handle is gone — but a background shell started via run_in_background
1635
- // (bash tool, detached:true, stdio:['ignore','pipe','pipe']) keeps its
1636
- // stdout/stderr pipes open on OUR side without ever being unref()'d. If the
1637
- // agent's turn left behind a live background process — e.g. a dev/gRPC/pypi
1638
- // server started with run_in_background:true and deliberately never killed,
1639
- // because it's meant to keep serving — those open pipes count as active
1640
- // handles and the process hangs forever after a fully successful turn,
1641
- // even though the model already produced its final `result` event. This
1642
- // was confirmed live: a --output-format stream-json run against
1643
- // terminal-bench tasks that start a background server (kv-store-grpc,
1644
- // pypi-server) emitted the final `result` event and then never exited,
1645
- // hanging the wrapping `docker compose exec` for the rest of the task
1646
- // timeout. Exiting explicitly here makes success behave like failure
1647
- // always did — the process ends the instant the turn is done, regardless
1648
- // of what background children it leaves running.
1649
- //
1650
- // On Linux, stdout to a pipe (exactly the `nex ... | tee file` setup used
1651
- // by CI/benchmark harnesses) is non-blocking — process.stdout.write() can
1652
- // return before the OS has actually flushed the bytes. process.exit() does
1653
- // not wait for that; see the identical note on the trust-prompt exit above.
1654
- // The headless JSON/stream-json result event was just written synchronously
1655
- // above (inside runTurnHeadless), so defer one tick to let it actually drain
1656
- // before tearing the process down.
1657
- await new Promise((r) => setImmediate(r));
1658
- process.exit(0);
1659
- }
1660
- catch (err) {
1661
- // One-shot exits the process, so the ONLY way a partial run survives is on
1662
- // disk — persist the salvaged history before exiting so `nex --resume`
1663
- // resumes the work instead of silently restarting it from zero.
1664
- const salvaged = salvageHistory(err);
1665
- if (salvaged) {
1666
- messages = salvaged;
1667
- try {
1668
- updateTitle();
1669
- saveSession(sessionId, sessionTitle || 'Untitled', workDir, messages);
1670
- // Commit the checkpoint turn as well. Only the success path does this, so
1671
- // without it a salvaged run would resume its conversation via --resume but
1672
- // the file snapshots taken during the interrupted turn would never become a
1673
- // /rewind point — history and undo would disagree about what happened.
1674
- checkpoints.commitTurn();
1675
- }
1676
- catch { /* best-effort — never mask the real error */ }
1677
- }
1678
- if (options.outputFormat === 'json' || options.outputFormat === 'stream-json') {
1679
- process.stdout.write(JSON.stringify({
1680
- type: 'error',
1681
- error: String(err.message),
1682
- ...(salvaged ? { resumable: true, messagesCompleted: salvaged.length } : {}),
1683
- }) + '\n');
1684
- }
1685
- else {
1686
- console.error(chalk.red('\nError: ') + String(err.message));
1687
- if (salvaged) {
1688
- console.error(chalk.dim(' Progress was saved — run `nex --resume` (or `nex -r`) to resume this session.'));
1689
- }
1690
- }
1691
- process.exit(1);
1692
- }
1693
- finally {
1694
- process.removeListener('SIGINT', onSigint);
1695
- }
1696
- return;
1697
- }
1698
- // ── Interactive REPL ──────────────────────────────────────────────────────
1699
- console.log(chalk.dim(' Type a message, /help for commands, or Ctrl+C to exit.'));
1700
- console.log();
1701
- // All startup output is on screen now, so insert exactly enough blank rows
1702
- // for the input line + footer to come to rest on the terminal's last row.
1703
- // Without this they sat directly beneath the banner with a block of empty
1704
- // rows below them.
1705
- //
1706
- // Nothing may print between here and the REPL: any later line would be
1707
- // counted by neither the tally nor the padding and would push the chrome one
1708
- // row past the bottom, scrolling the top of the banner away.
1709
- //
1710
- // `startupPadded` records whether padToBottom actually printed filler rows
1711
- // — those rows are real, permanent scrollback (a terminal is append-only,
1712
- // so nothing printed later can ever "fill" the gap instead of just
1713
- // appending after it) and have to be collapsed once real conversation
1714
- // content starts, or they stay wedged between the banner and the first
1715
- // turn for the rest of the session. See the first `rl.on('line', …)`
1716
- // handling below, and collapseStartupPadding's own doc in inkTerminal.tsx.
1717
- let startupPadded = false;
1718
- if (interactive) {
1719
- // Wait for Ink to commit the frame containing everything printed above
1720
- // before measuring against it. Ink renders on a timer, so without this the
1721
- // padding is computed against a stale frame and the result is
1722
- // nondeterministic — measured: identical terminal sizes kept the banner on
1723
- // some runs and scrolled its top row away on others.
1724
- await flushInkFrame();
1725
- // Pass the real footer string so its true height is measured rather than
1726
- // assumed — it wraps to two rows on a terminal narrower than ~56 columns,
1727
- // and under-counting pushes the banner's top row off the screen.
1728
- startupPadded = padToBottom(stopRowCount(), footerText({ mode: agentMode, autoApprove: isYoloMode() }));
1729
- }
1730
- let agentRunning = false;
1731
- const abortSignal = { aborted: false };
1732
- // Handles Ctrl+C for the NON-interactive path only (headless/one-shot's
1733
- // real readline.Interface, not running under Ink's raw mode). A pressed
1734
- // Ctrl+C there is translated into a genuine SIGINT by the kernel's tty
1735
- // line discipline, which this listens for directly.
1736
- //
1737
- // Interactive sessions never reach this: Ink puts stdin into raw mode for
1738
- // the whole session, which disables that kernel-level Ctrl+C→SIGINT
1739
- // translation entirely (see inkTerminal.tsx's useInput setting rawMode via
1740
- // Ink internally) — so the '\x03' byte only ever exists as a keystroke for
1741
- // useInput's own handler to see, never as a process signal. That handler's
1742
- // `onInterrupt` callback (wired below, once `rl` exists) is the only path
1743
- // left to interrupt a running turn interactively; this listener would
1744
- // simply never fire for it.
1745
- process.on('SIGINT', () => {
1746
- if (agentRunning) {
1747
- console.log('\n' + chalk.yellow(' Interrupted.'));
1748
- abortSignal.aborted = true;
1749
- agentRunning = false;
1750
- }
1751
- else {
1752
- updateTitle();
1753
- if (messages.length)
1754
- saveSession(sessionId, sessionTitle || 'Untitled', workDir, messages);
1755
- console.log('\n' + chalk.dim(' Goodbye.'));
1756
- process.exit(0);
1757
- }
1758
- });
1759
- // Interactive sessions get a readline.Interface-SHAPED adapter backed by
1760
- // Ink (see ui/inkReadlineAdapter.ts) instead of a real readline.Interface —
1761
- // chat.ts's business logic below (`rl.on('line', …)`, `rl.pause()`,
1762
- // `rl.resume()`, `rl.close()`) is unchanged; only the rendering/input
1763
- // layer underneath it moved to Ink's React-driven renderer. One-shot/
1764
- // headless runs (no Ink instance) fall back to a real readline.Interface,
1765
- // same as before.
1766
- const rl = interactive
1767
- ? new InkReadlineAdapter(getInkTerminal())
1768
- : readline.createInterface({ input: process.stdin, output: process.stdout, terminal: true });
1769
- // Interactive-only: Ctrl+C pressed WHILE a turn is running, routed through
1770
- // Ink (see the SIGINT comment above for why the kernel signal never fires
1771
- // here). Mirrors the SIGINT branch above exactly — same message, same
1772
- // abortSignal/agentRunning reset — just triggered via the adapter's
1773
- // 'interrupt' event instead of a process signal.
1774
- if (rl instanceof InkReadlineAdapter) {
1775
- rl.on('interrupt', () => {
1776
- if (agentRunning) {
1777
- console.log('\n' + chalk.yellow(' Interrupted.'));
1778
- abortSignal.aborted = true;
1779
- agentRunning = false;
1780
- }
1781
- });
1782
- }
1783
- // Route every permission y/n prompt through THIS same interface instead of
1784
- // letting askUser() spin up a second readline.Interface on process.stdin.
1785
- // rl.pause() (used below before each agent turn) doesn't remove this
1786
- // interface's stdin listeners, so a second Interface listening concurrently
1787
- // caused every keystroke — including the permission answer itself — to be
1788
- // processed and re-drawn by both, which is what produced the duplicated/
1789
- // "lấn" characters and broken line-wrapping users were seeing.
1790
- setReadlineInterface(rl);
1791
- // The footer ("auto mode on · /help for shortcuts · /agents for agents")
1792
- // reflects agentMode/yolo state, which slash commands below mutate
1793
- // directly — unlike before (where it redrew implicitly on every
1794
- // rl.prompt()), Ink only re-renders when its own state changes, so each
1795
- // place that changes agentMode/auto-approve must call this explicitly.
1796
- const updateFooter = () => {
1797
- if (interactive)
1798
- getInkTerminal().setFooter({ mode: agentMode, autoApprove: isYoloMode() });
1799
- };
1800
- updateFooter();
1801
- // Passed to every runTurn() call below as its takePendingInput param — see
1802
- // that param's own comment. `rl instanceof InkReadlineAdapter` is false for
1803
- // headless/one-shot's plain readline.Interface (no queueing surface there,
1804
- // and one-shot exits right after its single turn so there is no "typing
1805
- // during" to speak of), in which case runTurn just gets `undefined` and
1806
- // behaves exactly as it always did.
1807
- const takePendingInput = rl instanceof InkReadlineAdapter
1808
- ? () => rl.takePendingInput()
1809
- : undefined;
1810
- let inputBuffer = '';
1811
- rl.on('line', async (rawLine) => {
1812
- // Multi-line continuation (trailing backslash)
1813
- if (rawLine.endsWith('\\')) {
1814
- inputBuffer += rawLine.slice(0, -1) + '\n';
1815
- if (!interactive)
1816
- process.stdout.write(chalk.dim('… '));
1817
- return;
1818
- }
1819
- const userInput = (inputBuffer + rawLine).trim();
1820
- inputBuffer = '';
1821
- if (!userInput) {
1822
- rl.prompt();
1823
- return;
1824
- }
1825
- // Collapse the startup padding gap (see `startupPadded`'s own comment
1826
- // above) exactly once, on the FIRST real line submitted — right before
1827
- // anything from this line gets appended to the transcript. Every
1828
- // branch below (a slash command or an ordinary message) is about to
1829
- // print something, so this is the last point at which "nothing has been
1830
- // appended below the padding yet" is still true; any later point would
1831
- // let real content print into the gap first and defeat the collapse.
1832
- if (startupPadded) {
1833
- startupPadded = false;
1834
- // Awaited — see collapseStartupPadding's own doc for why firing the raw
1835
- // clear before Ink has flushed the just-echoed line to the terminal can
1836
- // lose that line (the "first message sometimes doesn't show up" bug this
1837
- // fixes). Every branch below only prints AFTER this resolves, so nothing
1838
- // races the flush this waits for.
1839
- await getInkTerminal().collapseStartupPadding();
1840
- }
1841
- // ── Slash commands ────────────────────────────────────────────────────
1842
- if (userInput.startsWith('/')) {
1843
- const parts = userInput.split(/\s+/);
1844
- const cmd = parts[0];
1845
- const arg = parts.slice(1).join(' ');
1846
- switch (cmd) {
1847
- case '/exit':
1848
- case '/quit':
1849
- updateTitle();
1850
- if (messages.length)
1851
- saveSession(sessionId, sessionTitle || 'Untitled', workDir, messages);
1852
- console.log(chalk.dim(' Goodbye.'));
1853
- rl.close();
1854
- process.exit(0);
1855
- case '/clear':
1856
- updateTitle();
1857
- if (messages.length)
1858
- saveSession(sessionId, sessionTitle || 'Untitled', workDir, messages);
1859
- messages = [];
1860
- sessionId = crypto.randomBytes(8).toString('hex');
1861
- sessionTitle = '';
1862
- // `/clear` starts a new session in a REUSED process, so the sub-agent budget has to
1863
- // start over with it — otherwise it would only ever be a per-invocation cap, and a
1864
- // long interactive session would silently lose delegation.
1865
- resetSessionSubAgentBudget();
1866
- console.log(chalk.dim(' Conversation cleared. New session started.'));
1867
- rl.prompt();
1868
- return;
1869
- case '/save':
1870
- updateTitle();
1871
- saveSession(sessionId, sessionTitle || 'Untitled', workDir, messages);
1872
- console.log(chalk.green(` Session saved: ${chalk.bold(sessionTitle || sessionId.slice(0, 8))}`));
1873
- rl.prompt();
1874
- return;
1875
- case '/sessions':
1876
- printSessions();
1877
- rl.prompt();
1878
- return;
1879
- case '/resume': {
1880
- if (agentRunning) {
1881
- console.log(chalk.red(' Cannot resume while agent is running.'));
1882
- rl.prompt();
1883
- return;
1884
- }
1885
- // Save current session first
1886
- updateTitle();
1887
- if (messages.length)
1888
- saveSession(sessionId, sessionTitle || 'Untitled', workDir, messages);
1889
- const stored = arg ? loadSession(arg) : lastSession();
1890
- if (!stored) {
1891
- console.log(chalk.red(` Session not found${arg ? `: ${arg}` : ''}.`));
1892
- console.log(chalk.dim(' Use /sessions to list.'));
1893
- rl.prompt();
1894
- return;
1895
- }
1896
- messages = stored.messages;
1897
- sessionId = stored.id;
1898
- sessionTitle = stored.title;
1899
- // Same resume-time cost guard as the --resume startup flag (see above):
1900
- // proactively compact BEFORE the next message is sent, since the in-loop
1901
- // guard's lastPromptTokens is stale/zero for a session resumed mid-process.
1902
- try {
1903
- await compactMessagesForResume(messages, {
1904
- workDir,
1905
- model: modelAlias,
1906
- clientType: 'cli',
1907
- onNotice: (text) => console.log(chalk.dim(text.trim())),
1908
- });
1909
- }
1910
- catch { /* non-fatal */ }
1911
- console.log(chalk.green(` Resumed: ${chalk.bold(stored.title.slice(0, 60))}`));
1912
- console.log(chalk.dim(` ${messages.length} messages restored`));
1913
- rl.prompt();
1914
- return;
1915
- }
1916
- case '/rewind': {
1917
- const cps = checkpoints.list();
1918
- if (!cps.length) {
1919
- console.log(chalk.dim(' No checkpoints yet — file edits create them automatically.'));
1920
- rl.prompt();
1921
- return;
1922
- }
1923
- if (!arg) {
1924
- console.log();
1925
- console.log(chalk.bold(' Checkpoints:'));
1926
- for (const c of cps) {
1927
- const when = c.createdAt.slice(11, 19);
1928
- console.log(' ' + chalk.cyan(`#${c.id}`.padEnd(5)) +
1929
- chalk.dim(`${when} `) +
1930
- chalk.yellow(`${c.fileCount} file${c.fileCount !== 1 ? 's' : ''}`.padEnd(9)) +
1931
- chalk.white(c.label));
1932
- }
1933
- console.log(chalk.dim(' Use /rewind <id> to restore.'));
1934
- console.log();
1935
- rl.prompt();
1936
- return;
1937
- }
1938
- const id = parseInt(arg, 10);
1939
- if (Number.isNaN(id)) {
1940
- console.log(chalk.red(` Invalid id: ${arg}`));
1941
- rl.prompt();
1942
- return;
1943
- }
1944
- const res = checkpoints.restore(id);
1945
- if (!res) {
1946
- console.log(chalk.red(` Checkpoint #${id} not found.`));
1947
- rl.prompt();
1948
- return;
1949
- }
1950
- messages = messages.slice(0, res.messageIndex);
1951
- console.log(chalk.green(` Rewound to #${id}: restored ${res.restored.length} file(s), ${messages.length} messages kept.`));
1952
- for (const p of res.restored)
1953
- console.log(chalk.dim(' ↩ ' + path.relative(workDir, p)));
1954
- if (res.bashCount > 0) {
1955
- console.log(chalk.yellow(` ⚠ ${res.bashCount} shell command(s) ran in the rewound turns — their side effects were NOT undone.`));
1956
- for (const h of res.gitStashHashes) {
1957
- console.log(chalk.dim(` Recovery point (pre-bash git snapshot): git stash apply ${h}`));
1958
- }
1959
- }
1960
- rl.prompt();
1961
- return;
1962
- }
1963
- case '/compact': {
1964
- if (messages.length < 4) {
1965
- console.log(chalk.dim(' Nothing to compact.'));
1966
- rl.prompt();
1967
- return;
1968
- }
1969
- rl.pause();
1970
- console.log(chalk.dim(' Compacting conversation…'));
1971
- const KEEP = 4;
1972
- const toSummarize = messages.slice(0, -KEEP);
1973
- const kept = messages.slice(-KEEP);
1974
- const convText = toSummarize
1975
- .map(m => `${m.role.toUpperCase()}: ${m.content[0]?.text ?? '(tool content)'}`)
1976
- .join('\n\n');
1977
- const summaryPrompt = `Summarize the key facts, decisions, code changes, and context from this conversation in concise bullet points (max 300 words):\n\n${convText}`;
1978
- try {
1979
- abortSignal.aborted = false;
1980
- agentRunning = true;
1981
- const r = await runTurn([{ role: 'user', content: [{ type: 'text', text: summaryPrompt }] }], modelAlias, workDir, abortSignal, env, nexrallMd, agentMode, effortLevel, undefined, undefined, takePendingInput, worktreeState, peerCtx);
1982
- agentRunning = false;
1983
- const summaryText = [...r.messages].reverse().find(m => m.role === 'assistant')?.content[0]?.text ?? '';
1984
- messages = [
1985
- { role: 'user', content: [{ type: 'text', text: `[Compacted ${toSummarize.length} messages]\n\nSummary:\n${summaryText}` }] },
1986
- { role: 'assistant', content: [{ type: 'text', text: 'Got it — I have the summary of our earlier work.' }] },
1987
- ...kept,
1988
- ];
1989
- // MUST persist immediately: without this, the on-disk transcript still
1990
- // holds the FULL pre-compact history until the next natural save point
1991
- // (sending the very next message), which then overwrites the file with
1992
- // this shortened `messages` array — permanently discarding everything
1993
- // /compact just summarized away, with no backup. Same bug, same fix,
1994
- // as ChatPanel.ts's identical /compact handler.
1995
- saveSession(sessionId, sessionTitle || 'Untitled', workDir, messages);
1996
- console.log(chalk.green(` Compacted ${toSummarize.length} messages → ${messages.length} messages.`));
1997
- }
1998
- catch (err) {
1999
- agentRunning = false;
2000
- console.error(chalk.red(' Compact failed: ') + String(err.message));
2001
- }
2002
- rl.resume();
2003
- rl.prompt();
2004
- return;
2005
- }
2006
- case '/init': {
2007
- // Generate nexrall.md for the current project
2008
- const nexrallMdPath = path.join(workDir, 'nexrall.md');
2009
- if (fs.existsSync(nexrallMdPath)) {
2010
- // Must reuse `rl` (the shared interface) rather than spinning up a
2011
- // second readline.Interface on process.stdin. In interactive
2012
- // sessions `rl` is an InkReadlineAdapter and Ink owns stdin in raw
2013
- // mode; a concurrent readline.Interface becomes a competing stdin
2014
- // consumer, so keystrokes get consumed/echoed by both and
2015
- // characters go missing or duplicated — the same class of bug the
2016
- // comment above setReadlineInterface() describes.
2017
- const answer = await new Promise(resolve => rl.question(chalk.yellow(' nexrall.md already exists. Overwrite? [y/n] '), a => resolve(a.trim())));
2018
- if (answer !== 'y' && answer !== 'yes') {
2019
- console.log(chalk.dim(' Cancelled.'));
2020
- rl.prompt();
2021
- return;
2022
- }
2023
- }
2024
- const readme = tryRead(path.join(workDir, 'README.md'))?.slice(0, 2000) ?? '';
2025
- const pkgJson = tryRead(path.join(workDir, 'package.json'))?.slice(0, 1000) ?? '';
2026
- const dirList = tryExec('ls -la', workDir)?.slice(0, 500) ?? '';
2027
- const initPrompt = `Analyze this project and write a concise nexrall.md (project-specific instructions for an AI coding assistant).\nInclude: project overview, tech stack, coding conventions, important files/folders, what to avoid.\nKeep it under 400 words. Use markdown. Be specific and actionable.\n\nREADME:\n${readme || '(none)'}\n\npackage.json:\n${pkgJson || '(none)'}\n\nDirectory:\n${dirList}`;
2028
- console.log(chalk.dim(' Generating nexrall.md…'));
2029
- rl.pause();
2030
- abortSignal.aborted = false;
2031
- agentRunning = true;
2032
- try {
2033
- const r = await runTurn([{ role: 'user', content: [{ type: 'text', text: initPrompt }] }], modelAlias, workDir, abortSignal, env, nexrallMd, 'auto', effortLevel, undefined, undefined, takePendingInput, worktreeState, peerCtx);
2034
- agentRunning = false;
2035
- const content = [...r.messages].reverse().find(m => m.role === 'assistant')?.content[0]?.text ?? '';
2036
- if (content) {
2037
- fs.writeFileSync(nexrallMdPath, content, 'utf-8');
2038
- console.log(chalk.green(` nexrall.md written to ${nexrallMdPath}`));
2039
- }
2040
- }
2041
- catch (err) {
2042
- agentRunning = false;
2043
- console.error(chalk.red(' Init failed: ') + String(err.message));
2044
- }
2045
- rl.resume();
2046
- rl.prompt();
2047
- return;
2048
- }
2049
- case '/worktree': {
2050
- // Usage: /worktree → show this session's isolation status
2051
- // /worktree list → list every worktree under requestedWorkDir
2052
- const sub = arg.trim().toLowerCase();
2053
- console.log();
2054
- if (sub === 'list') {
2055
- const all = listWorktrees(requestedWorkDir);
2056
- if (!all.length) {
2057
- console.log(chalk.dim(' No worktrees under this project.'));
2058
- }
2059
- else {
2060
- console.log(chalk.bold(` Worktrees (${all.length})`));
2061
- for (const w of all) {
2062
- const dirty = worktreeHasWork(w) ? chalk.yellow('dirty') : chalk.dim('clean');
2063
- console.log(` ${chalk.cyan(w.name)} ${chalk.dim(w.worktreePath)} [${dirty}]`);
2064
- }
2065
- console.log();
2066
- console.log(chalk.dim(' Resume one with: nex --worktree <name>'));
2067
- }
2068
- }
2069
- else if (worktreeState) {
2070
- console.log(chalk.bold(' This session is isolated in a worktree:'));
2071
- console.log(` ${chalk.cyan(worktreeState.name)} ${chalk.dim(worktreeState.worktreePath)}`);
2072
- console.log(chalk.dim(` Backend: ${worktreeState.backend}${worktreeState.branch ? ` · Branch: ${worktreeState.branch}` : ''}`));
2073
- console.log(chalk.dim(` Main checkout: ${worktreeState.mainCheckout}`));
2074
- console.log();
2075
- console.log(chalk.dim(' Writes/bash outside this worktree are refused by this session automatically.'));
2076
- }
2077
- else {
2078
- console.log(chalk.dim(' This session is NOT isolated in a worktree.'));
2079
- console.log(chalk.dim(' Start one with: nex --worktree [name]'));
2080
- }
2081
- console.log();
2082
- rl.prompt();
2083
- return;
2084
- }
2085
- case '/peers': {
2086
- // Usage: /peers → list other discoverable sessions
2087
- // /peers send <name> <message> → message one directly
2088
- // /peers broadcast <message> → message every other reachable session at once
2089
- const rest = arg.trim();
2090
- console.log();
2091
- if (!peerHandle) {
2092
- console.log(chalk.dim(' This session is not registered as a discoverable peer (non-interactive sessions skip registration).'));
2093
- console.log();
2094
- rl.prompt();
2095
- return;
2096
- }
2097
- if (rest.startsWith('send ')) {
2098
- const afterSend = rest.slice('send '.length);
2099
- const spaceIdx = afterSend.indexOf(' ');
2100
- const targetName = spaceIdx === -1 ? afterSend : afterSend.slice(0, spaceIdx);
2101
- const text = spaceIdx === -1 ? '' : afterSend.slice(spaceIdx + 1).trim();
2102
- if (!targetName || !text) {
2103
- console.log(chalk.dim(' Usage: /peers send <name> <message>'));
2104
- }
2105
- else {
2106
- const target = listPeers().find((p) => p.name === targetName);
2107
- if (!target) {
2108
- console.log(chalk.red(` No live session named "${targetName}". Use /peers to list who's reachable.`));
2109
- }
2110
- else if (!target.socketPath) {
2111
- console.log(chalk.red(` Session "${targetName}" is not currently reachable for messaging.`));
2112
- }
2113
- else {
2114
- const { sendPeerMessage } = await import('@nexrall/code-core');
2115
- const result = await sendPeerMessage(target.socketPath, { from: peerHandle.record.name, fromId: peerHandle.record.id, text });
2116
- console.log(result.ok ? chalk.green(` Sent to ${targetName}.`) : chalk.red(` Failed: ${result.error}`));
2117
- }
2118
- }
2119
- }
2120
- else if (rest.startsWith('broadcast ')) {
2121
- const text = rest.slice('broadcast '.length).trim();
2122
- if (!text) {
2123
- console.log(chalk.dim(' Usage: /peers broadcast <message>'));
2124
- }
2125
- else {
2126
- const { broadcastPeerMessage } = await import('@nexrall/code-core');
2127
- const targets = listPeers(peerHandle.record.id)
2128
- .filter((p) => p.id !== peerHandle.record.id && p.socketPath && p.inbound !== 'refuse')
2129
- .map((p) => ({ name: p.name, socketPath: p.socketPath }));
2130
- if (!targets.length) {
2131
- console.log(chalk.dim(' No other reachable Nexrall Code sessions to broadcast to.'));
2132
- }
2133
- else {
2134
- const results = await broadcastPeerMessage(targets, { from: peerHandle.record.name, fromId: peerHandle.record.id, text });
2135
- const ok = results.filter((r) => r.ok);
2136
- console.log(chalk.green(` Broadcast to ${results.length} session(s): ${ok.length} delivered.`));
2137
- for (const r of results.filter((r) => !r.ok))
2138
- console.log(chalk.red(` ✗ ${r.name}: ${r.error}`));
2139
- }
2140
- }
2141
- }
2142
- else {
2143
- const peers = listPeers(peerHandle.record.id);
2144
- if (peers.length <= 1) {
2145
- console.log(chalk.dim(' No other Nexrall Code sessions are currently discoverable on this machine.'));
2146
- }
2147
- else {
2148
- console.log(chalk.bold(` This session: ${chalk.cyan(peerHandle.record.name)}`));
2149
- console.log();
2150
- for (const p of peers) {
2151
- if (p.id === peerHandle.record.id)
2152
- continue;
2153
- const status = p.status === 'running' ? chalk.yellow('busy') : chalk.dim('idle');
2154
- console.log(` ${chalk.cyan(p.name)} ${chalk.dim(p.clientType)} [${status}] ${chalk.dim(p.workDir)}`);
2155
- }
2156
- console.log();
2157
- console.log(chalk.dim(' Message one with: /peers send <name> <message>'));
2158
- }
2159
- }
2160
- console.log();
2161
- rl.prompt();
2162
- return;
2163
- }
2164
- case '/memory': {
2165
- // Usage: /memory → show project + global ACTIVE memory merged
2166
- // /memory global → show ONLY global memory
2167
- // /memory archived → also include superseded/archived facts
2168
- // /memory clear → wipe project memory (with confirm)
2169
- // /memory global clear → wipe global memory (with confirm)
2170
- const memArgs = arg.toLowerCase().split(/\s+/).filter(Boolean);
2171
- const wantsGlobalOnly = memArgs.includes('global');
2172
- const wantsClear = memArgs.includes('clear');
2173
- const wantsArchived = memArgs.includes('archived');
2174
- const memScope = wantsGlobalOnly ? 'global' : 'project';
2175
- if (wantsClear) {
2176
- // Reuse the shared `rl` — see the /init note above: a second
2177
- // readline.Interface on process.stdin competes with Ink for
2178
- // keystrokes. (Also note rl.pause() would suppress the very
2179
- // answer we're about to ask for, so don't pause around this.)
2180
- const label = memScope === 'global' ? 'GLOBAL' : "this PROJECT's";
2181
- const answer = await new Promise(resolve => rl.question(chalk.yellow(` Clear ${label} memory? This cannot be undone. [y/n] `), a => resolve(a.trim())));
2182
- if (answer === 'y' || answer === 'yes') {
2183
- clearMemory(memScope, workDir);
2184
- console.log(chalk.green(` ${memScope === 'global' ? 'Global' : 'Project'} memory cleared.`));
2185
- }
2186
- else {
2187
- console.log(chalk.dim(' Cancelled.'));
2188
- }
2189
- rl.resume();
2190
- rl.prompt();
2191
- return;
2192
- }
2193
- console.log();
2194
- if (wantsGlobalOnly) {
2195
- const stats = memoryStats('global');
2196
- const content = readMemory('global', undefined, { includeArchived: wantsArchived });
2197
- const archivedNote = stats.archived > 0 ? `, ${stats.archived} archived` : '';
2198
- console.log(chalk.bold(' Global memory ') + chalk.dim(`(${stats.entries} entries${archivedNote}, ${(stats.bytes / 1024).toFixed(1)}KB) — ${stats.file}`));
2199
- console.log();
2200
- console.log(content || chalk.dim(' No global memories saved yet.'));
2201
- }
2202
- else {
2203
- const projStats = memoryStats('project', workDir);
2204
- const globalStats = memoryStats('global');
2205
- const projArchived = projStats.archived > 0 ? `+${projStats.archived} archived ` : '';
2206
- const globalArchived = globalStats.archived > 0 ? `+${globalStats.archived} archived ` : '';
2207
- console.log(chalk.bold(' Memory') + chalk.dim(` — project: ${projStats.entries} ${projArchived}entries (${(projStats.bytes / 1024).toFixed(1)}KB) · global: ${globalStats.entries} ${globalArchived}entries (${(globalStats.bytes / 1024).toFixed(1)}KB)`));
2208
- console.log();
2209
- const merged = readAllMemory(workDir, { includeArchived: wantsArchived });
2210
- console.log(merged || chalk.dim(' No memories saved yet.'));
2211
- }
2212
- console.log();
2213
- console.log(chalk.dim(' Tip: /memory global · /memory archived · /memory clear · /memory global clear'));
2214
- rl.prompt();
2215
- return;
2216
- }
2217
- case '/help':
2218
- printHelp();
2219
- rl.prompt();
2220
- return;
2221
- case '/model': {
2222
- // NOT lowercased: model ids are case-sensitive tokens, and forcing case
2223
- // here would corrupt any id containing uppercase. normaliseModelId does
2224
- // the case-insensitive match for the legacy ALIASES only.
2225
- const requested = arg.trim();
2226
- const models = selectableModelIds();
2227
- if (!requested) {
2228
- console.log(chalk.dim(` Current: ${chalk.cyan(resolveModelLabel(modelAlias))}`));
2229
- console.log(chalk.dim(` Options: ${models.map(modelWithCostHint).join(' · ')}`));
2230
- }
2231
- else if (models.includes(normaliseModelId(requested))) {
2232
- // Store the RESOLVED id, so `/model turbo` and `/model claude-sonnet-5`
2233
- // leave the session in identical state rather than two spellings that
2234
- // drift apart in later comparisons.
2235
- modelAlias = normaliseModelId(requested);
2236
- // Re-clamp effort onto the NEW model's scale — switching from
2237
- // Claude (5 levels, e.g. 'extra') to Qwen (2 levels) would
2238
- // otherwise leave effortLevel holding a wire value Qwen has never
2239
- // seen, silently treated as its default by the backend with
2240
- // nothing on screen saying so.
2241
- const clamped = clampEffortForModel(effortLevel, modelAlias);
2242
- const effortChanged = clamped !== effortLevel;
2243
- effortLevel = clamped;
2244
- console.log(chalk.green(` Model → ${chalk.bold(resolveModelLabel(modelAlias))}`));
2245
- if (effortChanged) {
2246
- console.log(chalk.dim(` Effort reset → ${effortLevel} (previous level not available on this model)`));
2247
- }
2248
- }
2249
- else {
2250
- console.log(chalk.red(` Unknown: ${requested}. Options: ${models.join(', ')}`));
2251
- }
2252
- rl.prompt();
2253
- return;
2254
- }
2255
- case '/mode': {
2256
- const modeArg = arg.toLowerCase();
2257
- if (!modeArg) {
2258
- // Spell out what each mode PERMITS, from the same descriptions the policy
2259
- // uses. Listing bare names ("ask · edit · plan · auto") told the user
2260
- // nothing about the actual trade-off, which is the whole decision being
2261
- // made here — and while the modes were still decorative, it was worse than
2262
- // nothing, since three of the four names described behaviour that did not
2263
- // exist.
2264
- console.log(chalk.dim(` Current: ${chalk.cyan(agentMode)}`));
2265
- for (const m of ['ask', 'edit', 'auto', 'plan']) {
2266
- const marker = m === agentMode ? chalk.cyan('❯') : ' ';
2267
- console.log(` ${marker} ${chalk.bold(m.padEnd(5))} ${chalk.dim(describeMode(m).replace(`${m} — `, ''))}`);
2268
- }
2269
- }
2270
- else if (['ask', 'edit', 'plan', 'auto'].includes(modeArg)) {
2271
- agentMode = modeArg;
2272
- updateFooter();
2273
- console.log(chalk.green(` Mode → ${chalk.bold(modeArg)}`) + chalk.dim(` ${describeMode(modeArg).replace(`${modeArg} — `, '')}`));
2274
- }
2275
- else {
2276
- console.log(chalk.red(` Unknown: ${modeArg}`));
2277
- }
2278
- rl.prompt();
2279
- return;
2280
- }
2281
- case '/effort': {
2282
- // Options are THIS model's real scale, not a fixed list shared by
2283
- // every provider — see "Per-model effort scale" above. A DeepSeek
2284
- // session shows low/high/max (its 3 real buckets); a Qwen session
2285
- // shows Thinking Off/On (its boolean toggle); gpt-4.1 shows nothing
2286
- // meaningful to change at all.
2287
- const effortArg = arg.toLowerCase();
2288
- const config = effortConfigFor(modelAlias);
2289
- const optionsLine = config.levels.map((v, i) => `${v} (${config.names[i]})`).join(' · ');
2290
- if (!effortArg) {
2291
- console.log(chalk.dim(` Current: ${chalk.cyan(effortLevel)} Options: ${optionsLine}`));
2292
- }
2293
- else if (config.levels.includes(effortArg)) {
2294
- effortLevel = effortArg;
2295
- console.log(chalk.green(` Effort → ${chalk.bold(effortArg)}`));
2296
- }
2297
- else {
2298
- console.log(chalk.red(` Unknown: ${effortArg}. Options: ${optionsLine}`));
2299
- }
2300
- rl.prompt();
2301
- return;
2302
- }
2303
- case '/yolo':
2304
- setAutoApprove('all');
2305
- updateFooter();
2306
- console.log(chalk.yellow(' Yolo mode: all permissions auto-approved.'));
2307
- rl.prompt();
2308
- return;
2309
- case '/balance': {
2310
- rl.pause();
2311
- try {
2312
- const bal = await getBalance();
2313
- console.log(chalk.dim(' Balance: ') + chalk.cyan(`$${bal.toFixed(4)}`));
2314
- }
2315
- catch (err) {
2316
- console.error(chalk.red(' Failed: ') + String(err.message));
2317
- }
2318
- rl.resume();
2319
- rl.prompt();
2320
- return;
2321
- }
2322
- case '/add': {
2323
- if (!arg) {
2324
- console.log(chalk.red(' Usage: /add <filepath>'));
2325
- rl.prompt();
2326
- return;
2327
- }
2328
- const filePath = path.isAbsolute(arg) ? arg : path.join(workDir, arg);
2329
- try {
2330
- const content = fs.readFileSync(filePath, 'utf-8');
2331
- const rel = path.relative(workDir, filePath);
2332
- messages.push({ role: 'user', content: [{ type: 'text', text: `Content of \`${rel}\`:\n\`\`\`\n${content}\n\`\`\`` }] }, { role: 'assistant', content: [{ type: 'text', text: `Got it, I've read \`${rel}\`.` }] });
2333
- console.log(chalk.green(` Added ${rel} (${content.split('\n').length} lines)`));
2334
- }
2335
- catch (err) {
2336
- console.error(chalk.red(' Failed: ') + String(err.message));
2337
- }
2338
- rl.prompt();
2339
- return;
2340
- }
2341
- // Attach an image (or PDF) and send it as a REAL turn to the model —
2342
- // unlike /add above, this is NOT a fake Q&A stub: the model must
2343
- // actually see the pixels/pages to reason about a UI bug screenshot,
2344
- // a diagram, or a scanned document, so it goes through the same
2345
- // runTurn() path as a normal typed message. Mirrors the VS Code
2346
- // extension's attachment handling in ChatPanel.ts's
2347
- // _handleUserMessage (leading text-label block naming the file, then
2348
- // the image/document block, since the block itself can't carry a
2349
- // `name` field the Anthropic API would recognize).
2350
- case '/image': {
2351
- if (agentRunning) {
2352
- console.log(chalk.red(' Cannot attach while agent is running.'));
2353
- rl.prompt();
2354
- return;
2355
- }
2356
- if (!arg) {
2357
- console.log(chalk.red(' Usage: /image <filepath> [caption]'));
2358
- rl.prompt();
2359
- return;
2360
- }
2361
- const [rawPath, ...captionParts] = arg.split(/\s+/);
2362
- const caption = captionParts.join(' ').trim();
2363
- const filePath = path.isAbsolute(rawPath) ? rawPath : path.join(workDir, rawPath);
2364
- const ext = path.extname(filePath).toLowerCase();
2365
- const IMAGE_MIME = {
2366
- '.png': 'image/png', '.jpg': 'image/jpeg', '.jpeg': 'image/jpeg',
2367
- '.gif': 'image/gif', '.webp': 'image/webp',
2368
- };
2369
- const isPdf = ext === '.pdf';
2370
- const mimeType = IMAGE_MIME[ext];
2371
- if (!mimeType && !isPdf) {
2372
- console.log(chalk.red(` Unsupported file type: ${ext || '(none)'}. Supported: png, jpg, jpeg, gif, webp, pdf.`));
2373
- rl.prompt();
2374
- return;
2375
- }
2376
- const model = normaliseModelId(modelAlias);
2377
- let buf;
2378
- try {
2379
- buf = fs.readFileSync(filePath);
2380
- }
2381
- catch (err) {
2382
- console.error(chalk.red(' Failed: ') + String(err.message));
2383
- rl.prompt();
2384
- return;
2385
- }
2386
- // 10MB raw-file soft cap: base64 inflates size ~33%, and very large
2387
- // images/scans routinely exceed the Anthropic API's per-request payload
2388
- // limit — fail fast locally with a clear reason instead of a confusing
2389
- // 413/400 from the backend several seconds into the turn.
2390
- const MAX_BYTES = 10 * 1024 * 1024;
2391
- if (buf.length > MAX_BYTES) {
2392
- console.log(chalk.red(` File too large (${(buf.length / 1024 / 1024).toFixed(1)}MB) — max 10MB.`));
2393
- rl.prompt();
2394
- return;
2395
- }
2396
- const rel = path.relative(workDir, filePath);
2397
- const data = buf.toString('base64');
2398
- const label = caption ? `[Attached: ${rel}]\n${caption}` : `[Attached: ${rel}]`;
2399
- const content = [{ type: 'text', text: label }];
2400
- // Blind model (no vision / no PDF support): transcribe via the
2401
- // sidecar ONCE, right now, and push TEXT into `messages` instead of
2402
- // the raw block — see describeAttachment's header for why this must
2403
- // happen before the attachment ever reaches history. The turn still
2404
- // runs on the model the user chose; it is never silently switched.
2405
- const needsSidecar = isPdf ? needsPdfSidecar(model) : needsVisionSidecar(model);
2406
- if (needsSidecar) {
2407
- console.log(chalk.dim(` ${resolveModelLabel(modelAlias)} can't read ${isPdf ? 'PDFs' : 'images'} directly — describing ${rel} with Claude Sonnet 5 first…`));
2408
- try {
2409
- const { text } = await describeAttachment(isPdf ? 'pdf' : 'image', data, mimeType, rel);
2410
- content.push({ type: 'text', text: `<attachment_description>\n${text}\n</attachment_description>` });
2411
- }
2412
- catch (err) {
2413
- console.log(chalk.red(` Could not describe ${rel}: ${err.message}`));
2414
- rl.prompt();
2415
- return;
2416
- }
2417
- }
2418
- else {
2419
- content.push(isPdf
2420
- ? { type: 'document', source: { type: 'base64', media_type: 'application/pdf', data } }
2421
- : { type: 'image', source: { type: 'base64', media_type: mimeType, data } });
2422
- }
2423
- console.log(chalk.dim(` Attached ${rel} (${(buf.length / 1024).toFixed(0)}KB) — sending…`));
2424
- checkpoints.beginTurn(`/image ${rel}${caption ? ' ' + caption : ''}`, messages.length);
2425
- messages.push({ role: 'user', content: content });
2426
- rl.pause();
2427
- abortSignal.aborted = false;
2428
- agentRunning = true;
2429
- await mcpReady(); // see the normal-message path below
2430
- try {
2431
- const result = await runTurn(messages, modelAlias, workDir, abortSignal, env, nexrallMd, agentMode, effortLevel, checkpoints, saveProgress, takePendingInput, worktreeState, peerCtx);
2432
- messages = result.messages;
2433
- updateTitle();
2434
- saveSession(sessionId, sessionTitle, workDir, messages);
2435
- }
2436
- catch (err) {
2437
- messages = recoverTurn(err, messages);
2438
- updateTitle();
2439
- saveSession(sessionId, sessionTitle || 'Untitled', workDir, messages);
2440
- }
2441
- finally {
2442
- agentRunning = false;
2443
- checkpoints.commitTurn();
2444
- }
2445
- rl.resume();
2446
- rl.prompt();
2447
- return;
2448
- }
2449
- case '/update': {
2450
- rl.close();
2451
- await updateCommand({ yes: false });
2452
- process.exit(0);
2453
- }
2454
- case '/plugins': {
2455
- const plugins = loadPlugins(workDir);
2456
- if (!plugins.length) {
2457
- console.log(chalk.dim(' No plugins installed. Install one: nex plugin install owner/repo'));
2458
- }
2459
- else {
2460
- console.log();
2461
- console.log(chalk.bold(' Installed plugins:'));
2462
- for (const p of plugins) {
2463
- const ver = p.version ? chalk.dim(` v${p.version}`) : '';
2464
- const scope = chalk.dim(` (${p.scope})`);
2465
- console.log(' ' + chalk.cyan(p.name.padEnd(20)) + (p.description ?? '') + ver + scope);
2466
- }
2467
- console.log();
2468
- }
2469
- // Plugins may add skills/commands — refresh the palette.
2470
- skills = loadSkills(workDir);
2471
- rl.prompt();
2472
- return;
2473
- }
2474
- case '/mcp': {
2475
- // Reporting "not connected" purely because the background connect
2476
- // hasn't finished would be misleading, so wait for the real answer.
2477
- await mcpReady();
2478
- console.log();
2479
- console.log(chalk.bold(' MCP servers:'));
2480
- console.log(formatMcpStatus());
2481
- console.log();
2482
- rl.prompt();
2483
- return;
2484
- }
2485
- case '/trust': {
2486
- // Trust is per-session and never persisted, so there is no stored
2487
- // grant to inspect or revoke — the only thing worth showing is what
2488
- // in THIS folder can configure the agent, which is exactly what the
2489
- // startup prompt listed.
2490
- console.log();
2491
- console.log(chalk.bold(' Workspace trust:'));
2492
- console.log(chalk.dim(` ${workDir}`));
2493
- const signals = detectTrustSignals(workDir);
2494
- console.log();
2495
- if (signals.length) {
2496
- console.log(chalk.dim(' This folder configures the agent:'));
2497
- for (const s of signals)
2498
- console.log(chalk.dim(` • ${s}`));
2499
- }
2500
- else {
2501
- console.log(chalk.dim(' This folder contains no agent configuration files.'));
2502
- }
2503
- console.log();
2504
- console.log(chalk.dim(' You confirmed this folder when the session started. ') +
2505
- chalk.dim('Trust is not saved — you will be asked again next time, so a\n ') +
2506
- chalk.dim('newly added .nexrall/mcp.json can never run un-announced.'));
2507
- console.log();
2508
- rl.prompt();
2509
- return;
2510
- }
2511
- case '/terminal-setup': {
2512
- // Shift+Enter cannot be fixed from inside this process: most
2513
- // terminals send a bare `\r` for it, byte-identical to Enter (see
2514
- // terminal/terminalSetup.ts's own header, and decideEnterKey's
2515
- // byte table). This reconfigures the TERMINAL to send a distinct
2516
- // sequence; the handler already understands it.
2517
- console.log();
2518
- console.log(chalk.bold(' Terminal setup — Shift+Enter for a newline:'));
2519
- console.log();
2520
- const target = decideTerminalSetup({
2521
- env: process.env,
2522
- platform: process.platform,
2523
- macOSMajor: detectMacOSMajor(),
2524
- });
2525
- if (target.kind === 'already-native') {
2526
- console.log(chalk.green(` ✓ ${target.terminal} already sends Shift+Enter as a newline.`));
2527
- console.log(chalk.dim(' Nothing to install.'));
2528
- }
2529
- else if (target.kind === 'editor') {
2530
- const outcome = await installEditorKeybinding(target.editor, {
2531
- readFile: (p) => fs.promises.readFile(p, 'utf-8'),
2532
- writeFile: (p, body) => fs.promises.writeFile(p, body, 'utf-8'),
2533
- copyFile: (from, to) => fs.promises.copyFile(from, to),
2534
- mkdir: async (p) => { await fs.promises.mkdir(p, { recursive: true }); },
2535
- randomSuffix: () => crypto.randomBytes(4).toString('hex'),
2536
- });
2537
- switch (outcome.kind) {
2538
- case 'installed':
2539
- console.log(chalk.green(` ✓ Installed the ${target.editor} Shift+Enter keybinding.`));
2540
- console.log(chalk.dim(` ${outcome.filePath}`));
2541
- console.log(chalk.dim(' Takes effect immediately — no reload needed.'));
2542
- break;
2543
- case 'already-configured':
2544
- console.log(chalk.green(` ✓ ${target.editor} already has this keybinding.`));
2545
- console.log(chalk.dim(` ${outcome.filePath}`));
2546
- break;
2547
- case 'conflict':
2548
- // Deliberately not overwritten: the user, or claude's own
2549
- // /terminal-setup, chose that sequence. Report it instead.
2550
- console.log(chalk.yellow(` ! ${target.editor} already binds shift+enter to a different sequence.`));
2551
- console.log(chalk.dim(` Existing: ${JSON.stringify(outcome.existingText)}`));
2552
- console.log(chalk.dim(` Left as-is. Change it to ${JSON.stringify(SHIFT_ENTER_SEQUENCE)} to use it here.`));
2553
- console.log(chalk.dim(` ${outcome.filePath}`));
2554
- break;
2555
- case 'failed':
2556
- console.log(chalk.red(` ✗ Couldn't update ${target.editor}'s keybindings: ${outcome.reason}`));
2557
- console.log(chalk.dim(` ${outcome.filePath}`));
2558
- break;
2559
- }
2560
- }
2561
- else if (target.kind === 'apple-terminal') {
2562
- // Below macOS 27 there is no Shift+Enter to install here — see
2563
- // enableAppleTerminalOptionAsMeta's doc for why Option+Enter is
2564
- // the supported route on those versions.
2565
- const res = await enableAppleTerminalOptionAsMeta({
2566
- run: async (cmd, args) => {
2567
- try {
2568
- const stdout = execFileSync(cmd, args, { encoding: 'utf-8', stdio: ['ignore', 'pipe', 'ignore'] });
2569
- return { code: 0, stdout };
2570
- }
2571
- catch (err) {
2572
- const e = err;
2573
- return { code: e.status ?? 1, stdout: e.stdout ?? '' };
2574
- }
2575
- },
2576
- plistPath: path.join(os.homedir(), 'Library', 'Preferences', 'com.apple.Terminal.plist'),
2577
- });
2578
- if (res.ok) {
2579
- console.log(chalk.green(' ✓ Enabled "Use Option as Meta Key" in Apple Terminal.'));
2580
- console.log(chalk.dim(` Profiles: ${res.profiles.join(', ')}`));
2581
- console.log(chalk.dim(' Option+Enter now inserts a newline. Restart Terminal to apply.'));
2582
- console.log(chalk.dim(` Apple Terminal only gained Shift+Return in macOS ${MACOS_NATIVE_SHIFT_RETURN_MAJOR}; ` +
2583
- 'until then Option+Enter is the equivalent.'));
2584
- }
2585
- else {
2586
- console.log(chalk.red(` ✗ Couldn't configure Apple Terminal: ${res.reason}`));
2587
- }
2588
- }
2589
- else {
2590
- console.log(chalk.yellow(` ! Not sure how to configure this terminal${target.terminal ? ` (${target.terminal})` : ''}.`));
2591
- console.log(chalk.dim(' Bind Shift+Enter to send ESC then CR, if your terminal supports it.'));
2592
- }
2593
- // Always shown: these need no configuration anywhere, so a failed or
2594
- // skipped install never leaves the user without a way to type a
2595
- // newline.
2596
- console.log();
2597
- console.log(chalk.dim(' Works in every terminal, no setup required:'));
2598
- console.log(chalk.dim(' • Ctrl+J — insert a newline'));
2599
- console.log(chalk.dim(' • \\ then Enter — continue on the next line'));
2600
- console.log();
2601
- rl.prompt();
2602
- return;
2603
- }
2604
- case '/agents': {
2605
- console.log();
2606
- console.log(chalk.bold(' Sub-agent types (used by the `task` tool):'));
2607
- console.log(formatAgentsList(workDir));
2608
- console.log();
2609
- console.log(chalk.dim(' Define your own in .nexrall/agents/<name>.md or ~/.nexrall/agents/<name>.md'));
2610
- console.log();
2611
- rl.prompt();
2612
- return;
2613
- }
2614
- case '/context': {
2615
- console.log(formatContextUsage(messages, modelAlias));
2616
- rl.prompt();
2617
- return;
2618
- }
2619
- case '/skills':
2620
- case '/commands': {
2621
- skills = loadSkills(workDir);
2622
- const invocable = userInvokableSkills(skills);
2623
- if (!invocable.length) {
2624
- console.log(chalk.dim(' No skills. Add .nexrall/skills/<name>/SKILL.md (or .nexrall/commands/<name>.md) to create one.'));
2625
- }
2626
- else {
2627
- console.log();
2628
- console.log(chalk.bold(' Skills:'));
2629
- for (const s of invocable) {
2630
- const scope = s.source === 'project' ? '' : chalk.dim(` (${s.source})`);
2631
- const auto = s.disableModelInvocation ? chalk.dim(' [manual only]') : '';
2632
- console.log(' ' + chalk.cyan(`/${s.name}`.padEnd(20)) + chalk.dim(s.description) + scope + auto);
2633
- }
2634
- console.log();
2635
- }
2636
- rl.prompt();
2637
- return;
2638
- }
2639
- default: {
2640
- // Custom skill/command?
2641
- const custom = findSkill(skills, cmd);
2642
- if (!custom) {
2643
- console.log(chalk.red(` Unknown command: ${cmd}`));
2644
- console.log(chalk.dim(' Type /help for built-ins or /skills for custom ones.'));
2645
- rl.prompt();
2646
- return;
2647
- }
2648
- const expanded = expandSkill(custom, arg, workDir);
2649
- const turnModel = custom.model ?? modelAlias;
2650
- const turnMode = custom.mode ?? agentMode;
2651
- console.log(chalk.dim(` Running /${custom.name}…`));
2652
- checkpoints.beginTurn(`/${custom.name} ${arg}`.trim(), messages.length);
2653
- messages.push({ role: 'user', content: [{ type: 'text', text: expanded }] });
2654
- rl.pause();
2655
- abortSignal.aborted = false;
2656
- agentRunning = true;
2657
- await mcpReady(); // see the normal-message path below
2658
- try {
2659
- const result = await runTurn(messages, turnModel, workDir, abortSignal, env, nexrallMd, turnMode, effortLevel, checkpoints, saveProgress, takePendingInput, worktreeState, peerCtx);
2660
- messages = result.messages;
2661
- updateTitle();
2662
- saveSession(sessionId, sessionTitle, workDir, messages);
2663
- }
2664
- catch (err) {
2665
- messages = recoverTurn(err, messages);
2666
- updateTitle();
2667
- saveSession(sessionId, sessionTitle || 'Untitled', workDir, messages);
2668
- }
2669
- finally {
2670
- agentRunning = false;
2671
- checkpoints.commitTurn();
2672
- }
2673
- rl.resume();
2674
- rl.prompt();
2675
- return;
2676
- }
2677
- }
2678
- }
2679
- // ── Normal message ────────────────────────────────────────────────────
2680
- checkpoints.beginTurn(userInput, messages.length);
2681
- messages.push({ role: 'user', content: [{ type: 'text', text: userInput }] });
2682
- rl.pause();
2683
- abortSignal.aborted = false;
2684
- agentRunning = true;
2685
- // MCP connects in the background so it can't delay the UI (see mcpStarted).
2686
- // Gate the turn on it here: the model must see the COMPLETE tool list, and
2687
- // by the time a user has typed a message this has almost always finished
2688
- // already, so in practice it waits for nothing.
2689
- await mcpReady();
2690
- try {
2691
- const result = await runTurn(messages, modelAlias, workDir, abortSignal, env, nexrallMd, agentMode, effortLevel, checkpoints, saveProgress, takePendingInput, worktreeState, peerCtx);
2692
- messages = result.messages;
2693
- updateTitle();
2694
- saveSession(sessionId, sessionTitle, workDir, messages);
2695
- }
2696
- catch (err) {
2697
- messages = recoverTurn(err, messages);
2698
- updateTitle();
2699
- saveSession(sessionId, sessionTitle || 'Untitled', workDir, messages);
2700
- }
2701
- finally {
2702
- agentRunning = false;
2703
- checkpoints.commitTurn();
2704
- }
2705
- rl.resume();
2706
- rl.prompt();
2707
- });
2708
- rl.on('close', () => {
2709
- updateTitle();
2710
- if (messages.length)
2711
- saveSession(sessionId, sessionTitle || 'Untitled', workDir, messages);
2712
- console.log('\n' + chalk.dim(' Goodbye.'));
2713
- process.exit(0);
2714
- });
2715
- }