aether-code 0.31.0 → 0.32.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/agent.js CHANGED
@@ -1,332 +1,332 @@
1
- // Agent loop. Streams each turn from /api/v1/agent/stream, prints text deltas
2
- // in real-time, executes any tool calls, loops until the model returns no
3
- // tool calls (task done) or max-turns is reached.
4
-
5
- import os from "node:os";
6
- import path from "node:path";
7
- import { agentTurnStream, AetherError } from "./api.js";
8
- import { TOOL_DEFINITIONS, executeTool } from "./tools.js";
9
- import { unnamespaceToolName } from "./mcp.js";
10
- import { loadAllSkills, selectSkills, renderSkillsBlock } from "./skills.js";
11
- import { c, divider, turn, toolLabel, toolSummary, makeTokenStripper, errorLine, startSpinner } from "./render.js";
12
-
13
- const DEFAULT_MAX_TURNS = 25;
14
-
15
- // Environment block prepended to the first user message so the model can
16
- // resolve named locations to real absolute paths.
17
- function envContext(cwd) {
18
- const home = os.homedir();
19
- const desktop = path.join(home, "Desktop");
20
- const documents = path.join(home, "Documents");
21
- const win = process.platform === "win32";
22
- // OS-specific shell guidance: gemma tends to emit Unix commands (mkdir -p,
23
- // touch, &&-chains) which FAIL in Windows cmd.exe ("The syntax of the command
24
- // is incorrect"). Steer it to the file tools (which are cross-platform and
25
- // auto-create parent dirs) and OS-appropriate shell usage.
26
- const shellNote = win
27
- ? `shell: Windows cmd.exe. To create files OR directories, use the write_file tool — ` +
28
- `it creates any missing parent folders automatically. Do NOT use shell "mkdir -p", ` +
29
- `"touch", "ls", "rm", "cp", "mv", or "&&"-chained Unix commands; they FAIL in cmd.exe. ` +
30
- `Run programs/tests with their interpreter (python, node, java, etc.).`
31
- : `shell: POSIX sh. Standard Unix commands are available. write_file still auto-creates parent dirs.`;
32
- return (
33
- `[environment]\n` +
34
- `os: ${process.platform}\n` +
35
- `cwd: ${cwd}\n` +
36
- `home: ${home}\n` +
37
- `desktop: ${desktop}\n` +
38
- `documents: ${documents}\n` +
39
- `${shellNote}\n` +
40
- `When the user names a location ("my desktop", "home", "documents"), write to ` +
41
- `the matching ABSOLUTE path above (e.g. desktop -> ${desktop}). Otherwise ` +
42
- `work under the cwd. Use absolute paths when a specific location is named.\n` +
43
- `[/environment]\n\n`
44
- );
45
- }
46
-
47
- export async function runAgent({
48
- initialPrompt,
49
- priorMessages,
50
- cwd,
51
- autoYes = false,
52
- unsafePaths = false,
53
- maxTurns = DEFAULT_MAX_TURNS,
54
- model = null, // null = server default (gemma). Premium models gated server-side.
55
- onTokens = () => {},
56
- // Optional MCPManager. When provided, its tools are merged into the agent's
57
- // toolset and tool calls prefixed `mcp__` are routed to it instead of the
58
- // built-in executeTool.
59
- mcpManager = null,
60
- }) {
61
- // Merge built-in tools with MCP-provided tools. MCP tools come second so
62
- // any name collision (unlikely given namespacing, but defense in depth)
63
- // resolves to the built-in.
64
- const tools = mcpManager
65
- ? [...TOOL_DEFINITIONS, ...mcpManager.getToolDefinitions()]
66
- : TOOL_DEFINITIONS;
67
-
68
- // Load skills once per runAgent call (bundled + user-installed). They
69
- // get selected per-turn against the current prompt + any file paths the
70
- // model has read so far. Loading errors are non-fatal — a bad skill file
71
- // shouldn't kill the agent.
72
- let allSkills = [];
73
- try {
74
- allSkills = loadAllSkills();
75
- } catch (e) {
76
- process.stderr.write(c.yellow(`(skill load failed: ${e.message})\n`));
77
- }
78
- const referencedPaths = [];
79
- // Two callers: one-shot (initialPrompt only, fresh conversation) and REPL
80
- // (priorMessages + initialPrompt to continue an ongoing chat).
81
- // On the FIRST message of a session, prepend an environment block so the
82
- // model knows real absolute paths (cwd / home / desktop). Without it, "build
83
- // X on my desktop" became `mkdir X` in whatever dir aether was launched from
84
- // (e.g. C:\WINDOWS\system32). Only prepended once — later turns carry it in
85
- // history.
86
- const messages = priorMessages
87
- ? [...priorMessages, { role: "user", content: initialPrompt }]
88
- : [{ role: "user", content: envContext(cwd) + initialPrompt }];
89
- let totalCredits = 0;
90
- let totalIn = 0;
91
- let totalOut = 0;
92
- let lastBalance = null;
93
- // Loop guard: count identical (name+args) tool calls across the whole run so a
94
- // confused model can't burn turns re-running the same call (e.g. glob *.md x9).
95
- const callCounts = new Map();
96
- // Files the run created/edited, shown in a summary at the end so the user
97
- // knows exactly what was produced and where to find it.
98
- const filesTouched = new Map();
99
- // Files the run read — used by the completeness guard to spot files that were
100
- // read (e.g. callers in a refactor) but never edited.
101
- const filesRead = new Set();
102
- // Completeness guard: gemma sometimes reads files then narrates the fix as
103
- // text, or refactors one file but forgets the others. If a MODIFICATION task
104
- // finishes having read files it never edited, we nudge it once to finish.
105
- const MOD_RE =
106
- /\b(refactor|fix|add|adds?|change|update|implement|creat|writ|remov|delet|rename|extract|move|replace|build|generate|insert|append|convert|migrat|wire|hook|edit|modif|patch|integrat)/i;
107
- const looksLikeModification = MOD_RE.test(initialPrompt);
108
- let appliedNothingNudges = 0;
109
-
110
- for (let i = 0; i < maxTurns; i++) {
111
- // No turn header and no leading blank here — each step (assistant text and
112
- // each tool label) begins with its own "\n● ", so spacing stays exactly one
113
- // blank line per step instead of stacking up.
114
-
115
- // Stream the assistant's response. Print text deltas as they arrive,
116
- // along with tool-call announcements as soon as the model commits to
117
- // calling a particular tool (i.e. the `name` arrives in the stream).
118
- const announced = new Set();
119
- let lastWasText = false;
120
- const stripper = makeTokenStripper();
121
-
122
- // Select skills for this turn against the current user prompt + any
123
- // paths the model has read so far. Prepend the matching skills' bodies
124
- // to the last user message of a shallow-cloned messages array — we
125
- // don't want skill text accumulating in the persisted history, only
126
- // being available to the model for the turn where it's relevant.
127
- const turnMessages = buildTurnMessages(messages, allSkills, referencedPaths);
128
-
129
- let res;
130
- // "Thinking" spinner: shown from request-send until the first token or tool
131
- // call arrives, so the wait doesn't look dead. Stopped exactly once.
132
- const spinner = startSpinner("thinking");
133
- let spinStopped = false;
134
- const stopSpin = () => { if (!spinStopped) { spinStopped = true; spinner.stop(); } };
135
- try {
136
- res = await agentTurnStream({
137
- messages: turnMessages,
138
- tools,
139
- model,
140
- onDelta: (text) => {
141
- // Buffered strip of leaked model channel/control tokens (which can
142
- // be split across stream chunks) before display.
143
- const clean = stripper.push(text);
144
- if (!clean) return;
145
- // Don't open a "● " bullet for leading whitespace (e.g. when a whole
146
- // turn's text was suppressed as a leak, leaving only a stray newline).
147
- if (!lastWasText && !clean.trim()) return;
148
- stopSpin();
149
- if (!lastWasText) {
150
- process.stdout.write("\n" + c.cyan("● "));
151
- lastWasText = true;
152
- }
153
- process.stdout.write(clean);
154
- },
155
- onToolCallDelta: (delta) => {
156
- // Just close the streamed text line when the model starts a tool
157
- // call — the clean label is printed at execution time, so no noisy
158
- // "preparing args" placeholder here.
159
- if (delta.name && !announced.has(delta.index)) {
160
- stopSpin();
161
- announced.add(delta.index);
162
- if (lastWasText) process.stdout.write("\n");
163
- lastWasText = false;
164
- }
165
- },
166
- });
167
- } catch (err) {
168
- stopSpin();
169
- if (err instanceof AetherError) {
170
- return { ok: false, error: err, totalCredits, totalIn, totalOut, balance: lastBalance, messages };
171
- }
172
- throw err;
173
- }
174
- stopSpin(); // ensure it's cleared even if the turn produced no output
175
-
176
- // Flush any held-back partial token, then close the line.
177
- const tail = stripper.flush();
178
- if (tail && (lastWasText || tail.trim())) {
179
- if (!lastWasText) { process.stdout.write("\n" + c.cyan("● ")); lastWasText = true; }
180
- process.stdout.write(tail);
181
- }
182
- if (lastWasText) process.stdout.write("\n");
183
- totalCredits += res.creditsCharged ?? 0;
184
- totalIn += res.usage?.prompt_tokens ?? 0;
185
- totalOut += res.usage?.completion_tokens ?? 0;
186
- if (typeof res.balanceAfter === "number") lastBalance = res.balanceAfter;
187
- onTokens({ totalCredits, totalIn, totalOut, balance: lastBalance });
188
- // Per-turn cost line removed for a cleaner look — the session summary at the
189
- // end carries the totals.
190
-
191
- // Push assistant message into history
192
- messages.push({
193
- role: "assistant",
194
- content: res.message.content,
195
- tool_calls: res.message.tool_calls,
196
- });
197
-
198
- const toolCalls = res.message.tool_calls ?? [];
199
- if (toolCalls.length === 0) {
200
- // Completeness guard: on a modification task, if the model read files it
201
- // never edited, it likely either described the fix instead of applying it
202
- // (zero edits) or refactored some files but forgot others (partial). Nudge
203
- // once (at most) to finish. The "if it needs no change, just finish" escape
204
- // keeps a legitimately read-for-context file from forcing a wrong edit.
205
- const unedited = [...filesRead].filter((p) => !filesTouched.has(p));
206
- if (looksLikeModification && appliedNothingNudges < 1 && (filesTouched.size === 0 ? referencedPaths.length > 0 : unedited.length > 0)) {
207
- appliedNothingNudges++;
208
- const msg =
209
- filesTouched.size === 0
210
- ? "You read the file(s) but applied NO changes (no write_file / edit_file). This task asks you to " +
211
- "modify code — ACTUALLY APPLY the change now with edit_file/write_file, don't just describe it."
212
- : `You edited ${[...filesTouched.keys()].join(", ")} but read ${unedited.join(", ")} without editing ${unedited.length === 1 ? "it" : "them"}. ` +
213
- `If the task needs ${unedited.length === 1 ? "that file" : "those files"} changed too (e.g. updating callers/imports after a refactor), do it NOW. ` +
214
- `If ${unedited.length === 1 ? "it genuinely needs" : "they genuinely need"} no change, then finish.`;
215
- messages.push({ role: "user", content: msg });
216
- continue; // give the model another turn to finish
217
- }
218
- if (filesTouched.size) {
219
- console.log("\n" + c.bold("Files") + c.gray(` in ${cwd}`));
220
- for (const [p, action] of filesTouched) {
221
- console.log(" " + c.green("●") + " " + p + c.gray(` — ${action}`));
222
- }
223
- }
224
- return { ok: true, totalCredits, totalIn, totalOut, turns: i + 1, balance: lastBalance, messages };
225
- }
226
-
227
- // Execute each tool call. Show the actual args (now that we have them
228
- // fully assembled) and run.
229
- for (const call of toolCalls) {
230
- let args = {};
231
- try { args = JSON.parse(call.function.arguments || "{}"); } catch { /* leave empty */ }
232
- // todo_write renders its own Plan box, so it returns an empty label.
233
- const label = toolLabel(call.function.name, args);
234
- if (label) console.log("\n" + c.cyan("●") + " " + label);
235
-
236
- // Loop guard — short-circuit a tool call we've already run 3+ times with
237
- // the same args, and tell the model to change approach instead of looping.
238
- const sig = `${call.function.name}:${call.function.arguments || ""}`;
239
- const seen = (callCounts.get(sig) || 0) + 1;
240
- callCounts.set(sig, seen);
241
- if (seen > 3) {
242
- console.log(" " + c.red("└─") + " " + c.gray("skipped (repeated call)"));
243
- messages.push({
244
- role: "tool",
245
- tool_call_id: call.id,
246
- content:
247
- "STOP: you have already run this exact tool call 3 times and the result will not change. Do NOT call it again. " +
248
- "If you were looking for a file that doesn't exist, create it with write_file. Otherwise change approach or finish and summarize.",
249
- });
250
- continue;
251
- }
252
-
253
- // Route to MCP if the tool name is namespaced (mcp__server__tool);
254
- // otherwise execute the built-in tool. unnamespaceToolName returns
255
- // null for non-MCP names, which is our cheap dispatch test.
256
- const slow = call.function.name === "web_search" || call.function.name === "web_fetch";
257
- const tspin = slow ? startSpinner(call.function.name === "web_search" ? "searching the web" : "fetching page") : null;
258
- let result;
259
- try {
260
- if (mcpManager && unnamespaceToolName(call.function.name)) {
261
- result = await mcpManager.callTool(call.function.name, args);
262
- } else {
263
- result = await executeTool(call, { cwd, autoYes, unsafePaths });
264
- }
265
- } finally {
266
- if (tspin) tspin.stop();
267
- }
268
-
269
- // Track paths the model has touched. Skills with path-pattern triggers
270
- // (e.g. RE skill on `*.exe`) match against this list, so reading a
271
- // binary in turn 3 can activate the RE skill in turn 4.
272
- if (call.function.name === "read_file" || call.function.name === "edit_file" || call.function.name === "write_file") {
273
- if (typeof args.path === "string") referencedPaths.push(args.path);
274
- }
275
- if (result.ok && typeof args.path === "string") {
276
- const rel = path.relative(cwd, path.resolve(cwd, args.path)) || args.path;
277
- if (call.function.name === "write_file" || call.function.name === "edit_file") {
278
- filesTouched.set(rel, call.function.name === "write_file" ? "created" : "edited");
279
- } else if (call.function.name === "read_file") {
280
- filesRead.add(rel);
281
- }
282
- }
283
- const summary = toolSummary(call.function.name, result);
284
- if (summary) console.log(summary);
285
-
286
- messages.push({
287
- role: "tool",
288
- tool_call_id: call.id,
289
- content: result.output ?? (result.ok ? "(no output)" : "Failed."),
290
- });
291
- }
292
- }
293
-
294
- console.log(c.yellow(`\nReached max turns (${maxTurns}). Stopping.`));
295
- return { ok: false, error: new Error("Max turns reached"), totalCredits, totalIn, totalOut, balance: lastBalance, messages };
296
- }
297
-
298
- /**
299
- * Per-turn skill injection. Selects skills against the latest user message
300
- * + paths the model has touched, then prepends matching bodies onto the
301
- * final user message of a shallow-cloned messages array. Returns the
302
- * original array unchanged when no skills match — zero overhead on the
303
- * no-skills path.
304
- *
305
- * Why prepend to user message instead of inserting a system message:
306
- * the server's AGENT_SYSTEM check skips its own system prompt when ANY
307
- * system message is present in the request. Adding a skills system
308
- * message would silently delete the server's discipline — which is
309
- * worse than no skills at all. Prepending into the user message keeps
310
- * both layers active.
311
- */
312
- function buildTurnMessages(messages, allSkills, referencedPaths) {
313
- if (allSkills.length === 0) return messages;
314
- // Find the latest user message — that's where the current task lives.
315
- let lastUserIdx = -1;
316
- for (let i = messages.length - 1; i >= 0; i--) {
317
- if (messages[i].role === "user") { lastUserIdx = i; break; }
318
- }
319
- if (lastUserIdx === -1) return messages;
320
- const prompt = typeof messages[lastUserIdx].content === "string"
321
- ? messages[lastUserIdx].content
322
- : "";
323
- const active = selectSkills({ skills: allSkills, prompt, referencedPaths });
324
- if (active.length === 0) return messages;
325
- const block = renderSkillsBlock(active);
326
- const cloned = [...messages];
327
- cloned[lastUserIdx] = {
328
- ...cloned[lastUserIdx],
329
- content: `${block}\n\n---\n\n${prompt}`,
330
- };
331
- return cloned;
332
- }
1
+ // Agent loop. Streams each turn from /api/v1/agent/stream, prints text deltas
2
+ // in real-time, executes any tool calls, loops until the model returns no
3
+ // tool calls (task done) or max-turns is reached.
4
+
5
+ import os from "node:os";
6
+ import path from "node:path";
7
+ import { agentTurnStream, AetherError } from "./api.js";
8
+ import { TOOL_DEFINITIONS, executeTool } from "./tools.js";
9
+ import { unnamespaceToolName } from "./mcp.js";
10
+ import { loadAllSkills, selectSkills, renderSkillsBlock } from "./skills.js";
11
+ import { c, divider, turn, toolLabel, toolSummary, makeTokenStripper, errorLine, startSpinner } from "./render.js";
12
+
13
+ const DEFAULT_MAX_TURNS = 25;
14
+
15
+ // Environment block prepended to the first user message so the model can
16
+ // resolve named locations to real absolute paths.
17
+ function envContext(cwd) {
18
+ const home = os.homedir();
19
+ const desktop = path.join(home, "Desktop");
20
+ const documents = path.join(home, "Documents");
21
+ const win = process.platform === "win32";
22
+ // OS-specific shell guidance: gemma tends to emit Unix commands (mkdir -p,
23
+ // touch, &&-chains) which FAIL in Windows cmd.exe ("The syntax of the command
24
+ // is incorrect"). Steer it to the file tools (which are cross-platform and
25
+ // auto-create parent dirs) and OS-appropriate shell usage.
26
+ const shellNote = win
27
+ ? `shell: Windows cmd.exe. To create files OR directories, use the write_file tool — ` +
28
+ `it creates any missing parent folders automatically. Do NOT use shell "mkdir -p", ` +
29
+ `"touch", "ls", "rm", "cp", "mv", or "&&"-chained Unix commands; they FAIL in cmd.exe. ` +
30
+ `Run programs/tests with their interpreter (python, node, java, etc.).`
31
+ : `shell: POSIX sh. Standard Unix commands are available. write_file still auto-creates parent dirs.`;
32
+ return (
33
+ `[environment]\n` +
34
+ `os: ${process.platform}\n` +
35
+ `cwd: ${cwd}\n` +
36
+ `home: ${home}\n` +
37
+ `desktop: ${desktop}\n` +
38
+ `documents: ${documents}\n` +
39
+ `${shellNote}\n` +
40
+ `When the user names a location ("my desktop", "home", "documents"), write to ` +
41
+ `the matching ABSOLUTE path above (e.g. desktop -> ${desktop}). Otherwise ` +
42
+ `work under the cwd. Use absolute paths when a specific location is named.\n` +
43
+ `[/environment]\n\n`
44
+ );
45
+ }
46
+
47
+ export async function runAgent({
48
+ initialPrompt,
49
+ priorMessages,
50
+ cwd,
51
+ autoYes = false,
52
+ unsafePaths = false,
53
+ maxTurns = DEFAULT_MAX_TURNS,
54
+ model = null, // null = server default (gemma). Premium models gated server-side.
55
+ onTokens = () => {},
56
+ // Optional MCPManager. When provided, its tools are merged into the agent's
57
+ // toolset and tool calls prefixed `mcp__` are routed to it instead of the
58
+ // built-in executeTool.
59
+ mcpManager = null,
60
+ }) {
61
+ // Merge built-in tools with MCP-provided tools. MCP tools come second so
62
+ // any name collision (unlikely given namespacing, but defense in depth)
63
+ // resolves to the built-in.
64
+ const tools = mcpManager
65
+ ? [...TOOL_DEFINITIONS, ...mcpManager.getToolDefinitions()]
66
+ : TOOL_DEFINITIONS;
67
+
68
+ // Load skills once per runAgent call (bundled + user-installed). They
69
+ // get selected per-turn against the current prompt + any file paths the
70
+ // model has read so far. Loading errors are non-fatal — a bad skill file
71
+ // shouldn't kill the agent.
72
+ let allSkills = [];
73
+ try {
74
+ allSkills = loadAllSkills();
75
+ } catch (e) {
76
+ process.stderr.write(c.yellow(`(skill load failed: ${e.message})\n`));
77
+ }
78
+ const referencedPaths = [];
79
+ // Two callers: one-shot (initialPrompt only, fresh conversation) and REPL
80
+ // (priorMessages + initialPrompt to continue an ongoing chat).
81
+ // On the FIRST message of a session, prepend an environment block so the
82
+ // model knows real absolute paths (cwd / home / desktop). Without it, "build
83
+ // X on my desktop" became `mkdir X` in whatever dir aether was launched from
84
+ // (e.g. C:\WINDOWS\system32). Only prepended once — later turns carry it in
85
+ // history.
86
+ const messages = priorMessages
87
+ ? [...priorMessages, { role: "user", content: initialPrompt }]
88
+ : [{ role: "user", content: envContext(cwd) + initialPrompt }];
89
+ let totalCredits = 0;
90
+ let totalIn = 0;
91
+ let totalOut = 0;
92
+ let lastBalance = null;
93
+ // Loop guard: count identical (name+args) tool calls across the whole run so a
94
+ // confused model can't burn turns re-running the same call (e.g. glob *.md x9).
95
+ const callCounts = new Map();
96
+ // Files the run created/edited, shown in a summary at the end so the user
97
+ // knows exactly what was produced and where to find it.
98
+ const filesTouched = new Map();
99
+ // Files the run read — used by the completeness guard to spot files that were
100
+ // read (e.g. callers in a refactor) but never edited.
101
+ const filesRead = new Set();
102
+ // Completeness guard: gemma sometimes reads files then narrates the fix as
103
+ // text, or refactors one file but forgets the others. If a MODIFICATION task
104
+ // finishes having read files it never edited, we nudge it once to finish.
105
+ const MOD_RE =
106
+ /\b(refactor|fix|add|adds?|change|update|implement|creat|writ|remov|delet|rename|extract|move|replace|build|generate|insert|append|convert|migrat|wire|hook|edit|modif|patch|integrat)/i;
107
+ const looksLikeModification = MOD_RE.test(initialPrompt);
108
+ let appliedNothingNudges = 0;
109
+
110
+ for (let i = 0; i < maxTurns; i++) {
111
+ // No turn header and no leading blank here — each step (assistant text and
112
+ // each tool label) begins with its own "\n● ", so spacing stays exactly one
113
+ // blank line per step instead of stacking up.
114
+
115
+ // Stream the assistant's response. Print text deltas as they arrive,
116
+ // along with tool-call announcements as soon as the model commits to
117
+ // calling a particular tool (i.e. the `name` arrives in the stream).
118
+ const announced = new Set();
119
+ let lastWasText = false;
120
+ const stripper = makeTokenStripper();
121
+
122
+ // Select skills for this turn against the current user prompt + any
123
+ // paths the model has read so far. Prepend the matching skills' bodies
124
+ // to the last user message of a shallow-cloned messages array — we
125
+ // don't want skill text accumulating in the persisted history, only
126
+ // being available to the model for the turn where it's relevant.
127
+ const turnMessages = buildTurnMessages(messages, allSkills, referencedPaths);
128
+
129
+ let res;
130
+ // "Thinking" spinner: shown from request-send until the first token or tool
131
+ // call arrives, so the wait doesn't look dead. Stopped exactly once.
132
+ const spinner = startSpinner("thinking");
133
+ let spinStopped = false;
134
+ const stopSpin = () => { if (!spinStopped) { spinStopped = true; spinner.stop(); } };
135
+ try {
136
+ res = await agentTurnStream({
137
+ messages: turnMessages,
138
+ tools,
139
+ model,
140
+ onDelta: (text) => {
141
+ // Buffered strip of leaked model channel/control tokens (which can
142
+ // be split across stream chunks) before display.
143
+ const clean = stripper.push(text);
144
+ if (!clean) return;
145
+ // Don't open a "● " bullet for leading whitespace (e.g. when a whole
146
+ // turn's text was suppressed as a leak, leaving only a stray newline).
147
+ if (!lastWasText && !clean.trim()) return;
148
+ stopSpin();
149
+ if (!lastWasText) {
150
+ process.stdout.write("\n" + c.cyan("● "));
151
+ lastWasText = true;
152
+ }
153
+ process.stdout.write(clean);
154
+ },
155
+ onToolCallDelta: (delta) => {
156
+ // Just close the streamed text line when the model starts a tool
157
+ // call — the clean label is printed at execution time, so no noisy
158
+ // "preparing args" placeholder here.
159
+ if (delta.name && !announced.has(delta.index)) {
160
+ stopSpin();
161
+ announced.add(delta.index);
162
+ if (lastWasText) process.stdout.write("\n");
163
+ lastWasText = false;
164
+ }
165
+ },
166
+ });
167
+ } catch (err) {
168
+ stopSpin();
169
+ if (err instanceof AetherError) {
170
+ return { ok: false, error: err, totalCredits, totalIn, totalOut, balance: lastBalance, messages };
171
+ }
172
+ throw err;
173
+ }
174
+ stopSpin(); // ensure it's cleared even if the turn produced no output
175
+
176
+ // Flush any held-back partial token, then close the line.
177
+ const tail = stripper.flush();
178
+ if (tail && (lastWasText || tail.trim())) {
179
+ if (!lastWasText) { process.stdout.write("\n" + c.cyan("● ")); lastWasText = true; }
180
+ process.stdout.write(tail);
181
+ }
182
+ if (lastWasText) process.stdout.write("\n");
183
+ totalCredits += res.creditsCharged ?? 0;
184
+ totalIn += res.usage?.prompt_tokens ?? 0;
185
+ totalOut += res.usage?.completion_tokens ?? 0;
186
+ if (typeof res.balanceAfter === "number") lastBalance = res.balanceAfter;
187
+ onTokens({ totalCredits, totalIn, totalOut, balance: lastBalance });
188
+ // Per-turn cost line removed for a cleaner look — the session summary at the
189
+ // end carries the totals.
190
+
191
+ // Push assistant message into history
192
+ messages.push({
193
+ role: "assistant",
194
+ content: res.message.content,
195
+ tool_calls: res.message.tool_calls,
196
+ });
197
+
198
+ const toolCalls = res.message.tool_calls ?? [];
199
+ if (toolCalls.length === 0) {
200
+ // Completeness guard: on a modification task, if the model read files it
201
+ // never edited, it likely either described the fix instead of applying it
202
+ // (zero edits) or refactored some files but forgot others (partial). Nudge
203
+ // once (at most) to finish. The "if it needs no change, just finish" escape
204
+ // keeps a legitimately read-for-context file from forcing a wrong edit.
205
+ const unedited = [...filesRead].filter((p) => !filesTouched.has(p));
206
+ if (looksLikeModification && appliedNothingNudges < 1 && (filesTouched.size === 0 ? referencedPaths.length > 0 : unedited.length > 0)) {
207
+ appliedNothingNudges++;
208
+ const msg =
209
+ filesTouched.size === 0
210
+ ? "You read the file(s) but applied NO changes (no write_file / edit_file). This task asks you to " +
211
+ "modify code — ACTUALLY APPLY the change now with edit_file/write_file, don't just describe it."
212
+ : `You edited ${[...filesTouched.keys()].join(", ")} but read ${unedited.join(", ")} without editing ${unedited.length === 1 ? "it" : "them"}. ` +
213
+ `If the task needs ${unedited.length === 1 ? "that file" : "those files"} changed too (e.g. updating callers/imports after a refactor), do it NOW. ` +
214
+ `If ${unedited.length === 1 ? "it genuinely needs" : "they genuinely need"} no change, then finish.`;
215
+ messages.push({ role: "user", content: msg });
216
+ continue; // give the model another turn to finish
217
+ }
218
+ if (filesTouched.size) {
219
+ console.log("\n" + c.bold("Files") + c.gray(` in ${cwd}`));
220
+ for (const [p, action] of filesTouched) {
221
+ console.log(" " + c.green("●") + " " + p + c.gray(` — ${action}`));
222
+ }
223
+ }
224
+ return { ok: true, totalCredits, totalIn, totalOut, turns: i + 1, balance: lastBalance, messages };
225
+ }
226
+
227
+ // Execute each tool call. Show the actual args (now that we have them
228
+ // fully assembled) and run.
229
+ for (const call of toolCalls) {
230
+ let args = {};
231
+ try { args = JSON.parse(call.function.arguments || "{}"); } catch { /* leave empty */ }
232
+ // todo_write renders its own Plan box, so it returns an empty label.
233
+ const label = toolLabel(call.function.name, args);
234
+ if (label) console.log("\n" + c.cyan("●") + " " + label);
235
+
236
+ // Loop guard — short-circuit a tool call we've already run 3+ times with
237
+ // the same args, and tell the model to change approach instead of looping.
238
+ const sig = `${call.function.name}:${call.function.arguments || ""}`;
239
+ const seen = (callCounts.get(sig) || 0) + 1;
240
+ callCounts.set(sig, seen);
241
+ if (seen > 3) {
242
+ console.log(" " + c.red("└─") + " " + c.gray("skipped (repeated call)"));
243
+ messages.push({
244
+ role: "tool",
245
+ tool_call_id: call.id,
246
+ content:
247
+ "STOP: you have already run this exact tool call 3 times and the result will not change. Do NOT call it again. " +
248
+ "If you were looking for a file that doesn't exist, create it with write_file. Otherwise change approach or finish and summarize.",
249
+ });
250
+ continue;
251
+ }
252
+
253
+ // Route to MCP if the tool name is namespaced (mcp__server__tool);
254
+ // otherwise execute the built-in tool. unnamespaceToolName returns
255
+ // null for non-MCP names, which is our cheap dispatch test.
256
+ const slow = call.function.name === "web_search" || call.function.name === "web_fetch";
257
+ const tspin = slow ? startSpinner(call.function.name === "web_search" ? "searching the web" : "fetching page") : null;
258
+ let result;
259
+ try {
260
+ if (mcpManager && unnamespaceToolName(call.function.name)) {
261
+ result = await mcpManager.callTool(call.function.name, args);
262
+ } else {
263
+ result = await executeTool(call, { cwd, autoYes, unsafePaths });
264
+ }
265
+ } finally {
266
+ if (tspin) tspin.stop();
267
+ }
268
+
269
+ // Track paths the model has touched. Skills with path-pattern triggers
270
+ // (e.g. RE skill on `*.exe`) match against this list, so reading a
271
+ // binary in turn 3 can activate the RE skill in turn 4.
272
+ if (call.function.name === "read_file" || call.function.name === "edit_file" || call.function.name === "write_file") {
273
+ if (typeof args.path === "string") referencedPaths.push(args.path);
274
+ }
275
+ if (result.ok && typeof args.path === "string") {
276
+ const rel = path.relative(cwd, path.resolve(cwd, args.path)) || args.path;
277
+ if (call.function.name === "write_file" || call.function.name === "edit_file") {
278
+ filesTouched.set(rel, call.function.name === "write_file" ? "created" : "edited");
279
+ } else if (call.function.name === "read_file") {
280
+ filesRead.add(rel);
281
+ }
282
+ }
283
+ const summary = toolSummary(call.function.name, result);
284
+ if (summary) console.log(summary);
285
+
286
+ messages.push({
287
+ role: "tool",
288
+ tool_call_id: call.id,
289
+ content: result.output ?? (result.ok ? "(no output)" : "Failed."),
290
+ });
291
+ }
292
+ }
293
+
294
+ console.log(c.yellow(`\nReached max turns (${maxTurns}). Stopping.`));
295
+ return { ok: false, error: new Error("Max turns reached"), totalCredits, totalIn, totalOut, balance: lastBalance, messages };
296
+ }
297
+
298
+ /**
299
+ * Per-turn skill injection. Selects skills against the latest user message
300
+ * + paths the model has touched, then prepends matching bodies onto the
301
+ * final user message of a shallow-cloned messages array. Returns the
302
+ * original array unchanged when no skills match — zero overhead on the
303
+ * no-skills path.
304
+ *
305
+ * Why prepend to user message instead of inserting a system message:
306
+ * the server's AGENT_SYSTEM check skips its own system prompt when ANY
307
+ * system message is present in the request. Adding a skills system
308
+ * message would silently delete the server's discipline — which is
309
+ * worse than no skills at all. Prepending into the user message keeps
310
+ * both layers active.
311
+ */
312
+ function buildTurnMessages(messages, allSkills, referencedPaths) {
313
+ if (allSkills.length === 0) return messages;
314
+ // Find the latest user message — that's where the current task lives.
315
+ let lastUserIdx = -1;
316
+ for (let i = messages.length - 1; i >= 0; i--) {
317
+ if (messages[i].role === "user") { lastUserIdx = i; break; }
318
+ }
319
+ if (lastUserIdx === -1) return messages;
320
+ const prompt = typeof messages[lastUserIdx].content === "string"
321
+ ? messages[lastUserIdx].content
322
+ : "";
323
+ const active = selectSkills({ skills: allSkills, prompt, referencedPaths });
324
+ if (active.length === 0) return messages;
325
+ const block = renderSkillsBlock(active);
326
+ const cloned = [...messages];
327
+ cloned[lastUserIdx] = {
328
+ ...cloned[lastUserIdx],
329
+ content: `${block}\n\n---\n\n${prompt}`,
330
+ };
331
+ return cloned;
332
+ }