aether-code 0.43.1 → 0.43.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/agent.js CHANGED
@@ -3,98 +3,69 @@
3
3
  // tool calls (task done) or max-turns is reached.
4
4
 
5
5
  import os from "node:os";
6
- import fs from "node:fs";
7
6
  import path from "node:path";
7
+ import { spawn } from "node:child_process";
8
8
  import { agentTurnStream, AetherError } from "./api.js";
9
9
  import { TOOL_DEFINITIONS, executeTool } from "./tools.js";
10
10
  import { unnamespaceToolName } from "./mcp.js";
11
11
  import { loadAllSkills, selectSkills, renderSkillsBlock } from "./skills.js";
12
- import { c, divider, turn, toolLabel, toolSummary, makeTokenStripper, errorLine, startSpinner, turnStats, fmtTokens } from "./render.js";
13
-
14
- /* ─────────────────────── Image attachment parsing ─────────────────────── */
15
-
16
- const IMAGE_EXTS = new Set([".png", ".jpg", ".jpeg", ".gif", ".webp", ".bmp"]);
17
- const MIME_MAP = {
18
- ".png": "image/png",
19
- ".jpg": "image/jpeg",
20
- ".jpeg": "image/jpeg",
21
- ".gif": "image/gif",
22
- ".webp": "image/webp",
23
- ".bmp": "image/bmp",
24
- };
25
-
26
- // Parse @path/to/image references from user input. Supports quoted paths for
27
- // spaces: @"C:\Users\me\my screenshot.png". Returns { text, images }.
28
- function parseImages(prompt, cwd) {
29
- const re = /@("(?:[^"\\]|\\.)*"|[^\s]+)/g;
30
- const images = [];
31
- let cleaned = prompt;
32
-
33
- for (const m of [...prompt.matchAll(re)]) {
34
- const raw = m[1].replace(/^"|"$/g, "");
35
- const ext = path.extname(raw).toLowerCase();
36
- if (!IMAGE_EXTS.has(ext)) continue;
37
-
38
- const abs = path.isAbsolute(raw) ? raw : path.resolve(cwd, raw);
39
- try {
40
- const buf = fs.readFileSync(abs);
41
- if (buf.length > 20 * 1024 * 1024) {
42
- console.log(c.yellow(` ⚠ Image too large (>20MB), skipping: ${raw}`));
43
- continue;
44
- }
45
- const mime = MIME_MAP[ext] || "image/png";
46
- images.push({
47
- url: `data:${mime};base64,${buf.toString("base64")}`,
48
- name: path.basename(raw),
49
- size: buf.length,
50
- });
51
- cleaned = cleaned.replace(m[0], "");
52
- } catch (e) {
53
- console.log(c.yellow(` ⚠ Could not read image: ${raw} — ${e.code === "ENOENT" ? "file not found" : e.message}`));
54
- }
55
- }
56
-
57
- return { text: cleaned.replace(/\s{2,}/g, " ").trim() || prompt.trim(), images };
58
- }
12
+ import { scanProject, detectVerifyCommand, detectBuildCommand } from "./project-context.js";
13
+ import { c, divider, turn, toolLabel, toolSummary, makeTokenStripper, errorLine, startSpinner } from "./render.js";
59
14
 
60
- // Build OpenAI-compatible content array with text + images, or plain string
61
- // when no images are attached.
62
- function buildContent(text, images) {
63
- if (images.length === 0) return text;
64
- const parts = [];
65
- for (const img of images) {
66
- parts.push({ type: "image_url", image_url: { url: img.url } });
67
- }
68
- parts.push({ type: "text", text });
69
- return parts;
70
- }
15
+ const DEFAULT_MAX_TURNS = 25;
16
+ const DEFAULT_MAX_CREDITS = 500;
17
+ const MAX_TOTAL_TOOL_CALLS = 40;
18
+ const MAX_CONSECUTIVE_TOOL_ONLY = 6;
19
+
20
+ // Rolling history window (messages) sent upstream per turn. The session keeps
21
+ // the full transcript locally, but we only forward the anchor (first message,
22
+ // which carries the [environment] block) plus the most recent N messages. This
23
+ // bounds per-turn token cost AND keeps us safely under the server's own cap —
24
+ // a long session used to grow unbounded and, past the server's limit, every
25
+ // send failed with "Error: Invalid input." with no recovery but /clear.
26
+ const HISTORY_WINDOW = 100;
71
27
 
72
- // Friendly file size: 1.2MB, 450KB, 128B
73
- function fmtSize(bytes) {
74
- if (bytes >= 1_048_576) return (bytes / 1_048_576).toFixed(1) + "MB";
75
- if (bytes >= 1024) return (bytes / 1024).toFixed(0) + "KB";
76
- return bytes + "B";
28
+ /**
29
+ * Trim a message array to `maxMessages`, preserving:
30
+ * - the anchor message at index 0 (the first user turn carries the
31
+ * [environment] block with real absolute paths — dropping it makes the
32
+ * model lose cwd/home/desktop);
33
+ * - tool-call/tool-result pairing at the window boundary: a `tool` message
34
+ * must immediately follow the assistant message whose `tool_calls` produced
35
+ * it, so the window is never allowed to BEGIN on an orphaned `tool` result.
36
+ * Returns the original array untouched when it's already within the window.
37
+ */
38
+ export function trimHistory(messages, maxMessages = HISTORY_WINDOW) {
39
+ if (messages.length <= maxMessages) return messages;
40
+ const head = [messages[0]];
41
+ const windowSize = Math.max(1, maxMessages - 1);
42
+ let window = messages.slice(1).slice(-windowSize);
43
+ let start = 0;
44
+ while (start < window.length && window[start]?.role === "tool") start++;
45
+ window = window.slice(start);
46
+ return [...head, ...window];
77
47
  }
78
48
 
79
- const DEFAULT_MAX_TURNS = 25;
49
+ export const MOD_RE =
50
+ /\b(refactor|fix|add|adds?|change|update|implement|creat|writ|remov|delet|rename|extract|move|replace|build|generate|insert|append|convert|migrat|wire|hook|edit|modif|patch|integrat)/i;
80
51
 
81
- // Environment block prepended to the first user message so the model can
82
- // resolve named locations to real absolute paths.
83
- function envContext(cwd) {
52
+ export function envContext(cwd) {
84
53
  const home = os.homedir();
85
54
  const desktop = path.join(home, "Desktop");
86
55
  const documents = path.join(home, "Documents");
87
56
  const win = process.platform === "win32";
88
- // OS-specific shell guidance: gemma tends to emit Unix commands (mkdir -p,
89
- // touch, &&-chains) which FAIL in Windows cmd.exe ("The syntax of the command
90
- // is incorrect"). Steer it to the file tools (which are cross-platform and
91
- // auto-create parent dirs) and OS-appropriate shell usage.
92
57
  const shellNote = win
93
58
  ? `shell: Windows cmd.exe. To create files OR directories, use the write_file tool — ` +
94
59
  `it creates any missing parent folders automatically. Do NOT use shell "mkdir -p", ` +
95
60
  `"touch", "ls", "rm", "cp", "mv", or "&&"-chained Unix commands; they FAIL in cmd.exe. ` +
96
61
  `Run programs/tests with their interpreter (python, node, java, etc.).`
97
62
  : `shell: POSIX sh. Standard Unix commands are available. write_file still auto-creates parent dirs.`;
63
+
64
+ let projectBlock = "";
65
+ try {
66
+ projectBlock = scanProject(cwd);
67
+ } catch { /* non-fatal */ }
68
+
98
69
  return (
99
70
  `[environment]\n` +
100
71
  `os: ${process.platform}\n` +
@@ -106,7 +77,8 @@ function envContext(cwd) {
106
77
  `When the user names a location ("my desktop", "home", "documents"), write to ` +
107
78
  `the matching ABSOLUTE path above (e.g. desktop -> ${desktop}). Otherwise ` +
108
79
  `work under the cwd. Use absolute paths when a specific location is named.\n` +
109
- `[/environment]\n\n`
80
+ `[/environment]\n\n` +
81
+ (projectBlock ? projectBlock + "\n" : "")
110
82
  );
111
83
  }
112
84
 
@@ -117,6 +89,7 @@ export async function runAgent({
117
89
  autoYes = false,
118
90
  unsafePaths = false,
119
91
  maxTurns = DEFAULT_MAX_TURNS,
92
+ maxCredits = DEFAULT_MAX_CREDITS,
120
93
  model = null, // null = server default (gemma). Premium models gated server-side.
121
94
  onTokens = () => {},
122
95
  // Optional MCPManager. When provided, its tools are merged into the agent's
@@ -142,15 +115,6 @@ export async function runAgent({
142
115
  process.stderr.write(c.yellow(`(skill load failed: ${e.message})\n`));
143
116
  }
144
117
  const referencedPaths = [];
145
-
146
- // Parse image attachments (@path/to/image.png) from the user's prompt.
147
- const { text: cleanPrompt, images: attachedImages } = parseImages(initialPrompt, cwd);
148
- if (attachedImages.length > 0) {
149
- for (const img of attachedImages) {
150
- console.log(`\n${c.cyan("●")} ${c.cyan(c.bold("image"))} ${c.gray(img.name)} ${c.gray(`(${fmtSize(img.size)})`)}`);
151
- }
152
- }
153
-
154
118
  // Two callers: one-shot (initialPrompt only, fresh conversation) and REPL
155
119
  // (priorMessages + initialPrompt to continue an ongoing chat).
156
120
  // On the FIRST message of a session, prepend an environment block so the
@@ -158,11 +122,9 @@ export async function runAgent({
158
122
  // X on my desktop" became `mkdir X` in whatever dir aether was launched from
159
123
  // (e.g. C:\WINDOWS\system32). Only prepended once — later turns carry it in
160
124
  // history.
161
- const firstText = priorMessages ? cleanPrompt : envContext(cwd) + cleanPrompt;
162
- const firstContent = buildContent(firstText, attachedImages);
163
125
  const messages = priorMessages
164
- ? [...priorMessages, { role: "user", content: firstContent }]
165
- : [{ role: "user", content: firstContent }];
126
+ ? [...priorMessages, { role: "user", content: initialPrompt }]
127
+ : [{ role: "user", content: envContext(cwd) + initialPrompt }];
166
128
  let totalCredits = 0;
167
129
  let totalIn = 0;
168
130
  let totalOut = 0;
@@ -179,30 +141,31 @@ export async function runAgent({
179
141
  // Completeness guard: gemma sometimes reads files then narrates the fix as
180
142
  // text, or refactors one file but forgets the others. If a MODIFICATION task
181
143
  // finishes having read files it never edited, we nudge it once to finish.
182
- const MOD_RE =
183
- /\b(refactor|fix|add|adds?|change|update|implement|creat|writ|remov|delet|rename|extract|move|replace|build|generate|insert|append|convert|migrat|wire|hook|edit|modif|patch|integrat)/i;
184
144
  const looksLikeModification = MOD_RE.test(initialPrompt);
185
145
  let appliedNothingNudges = 0;
146
+ let verifyRounds = 0;
147
+ let buildRounds = 0;
148
+ let consecutiveToolOnlyTurns = 0;
149
+ let totalToolCalls = 0;
186
150
 
187
151
  for (let i = 0; i < maxTurns; i++) {
188
- const turnStart = Date.now();
152
+ // No turn header and no leading blank here — each step (assistant text and
153
+ // each tool label) begins with its own "\n● ", so spacing stays exactly one
154
+ // blank line per step instead of stacking up.
189
155
 
190
156
  // Stream the assistant's response. Print text deltas as they arrive,
191
157
  // along with tool-call announcements as soon as the model commits to
192
158
  // calling a particular tool (i.e. the `name` arrives in the stream).
193
159
  const announced = new Set();
194
160
  let lastWasText = false;
195
- let streamedChars = 0;
196
161
  const stripper = makeTokenStripper();
197
- let leakBuf = "";
198
- let leakSuppressing = false;
199
162
 
200
163
  // Select skills for this turn against the current user prompt + any
201
164
  // paths the model has read so far. Prepend the matching skills' bodies
202
165
  // to the last user message of a shallow-cloned messages array — we
203
166
  // don't want skill text accumulating in the persisted history, only
204
167
  // being available to the model for the turn where it's relevant.
205
- const turnMessages = buildTurnMessages(messages, allSkills, referencedPaths);
168
+ const turnMessages = trimHistory(buildTurnMessages(messages, allSkills, referencedPaths));
206
169
 
207
170
  let res;
208
171
  // "Thinking" spinner: shown from request-send until the first token or tool
@@ -216,34 +179,19 @@ export async function runAgent({
216
179
  tools,
217
180
  model,
218
181
  onDelta: (text) => {
219
- streamedChars += text.length;
220
- spinner.update({ tokens: Math.ceil(streamedChars / 3.5) });
182
+ // Buffered strip of leaked model channel/control tokens (which can
183
+ // be split across stream chunks) before display.
221
184
  const clean = stripper.push(text);
222
185
  if (!clean) return;
223
-
224
- // Suppress model echoing tool results (response:web_search, [{snippet:, etc.)
225
- leakBuf += clean;
226
- if (leakBuf.length < 40 && /^[\s\n]*(?:response:|[\[{]"?(?:snippet|title|url))/.test(leakBuf)) return;
227
- if (leakSuppressing) {
228
- if (clean.includes("\n\n")) { leakSuppressing = false; leakBuf = ""; }
229
- return;
230
- }
231
- const checkBuf = leakBuf.trim();
232
- if (checkBuf && /^response:\w|^\[\{(?:"?snippet|"?title)/.test(checkBuf)) {
233
- leakSuppressing = true;
234
- leakBuf = "";
235
- return;
236
- }
237
- const emit = leakBuf;
238
- leakBuf = "";
239
-
240
- if (!lastWasText && !emit.trim()) return;
186
+ // Don't open a "● " bullet for leading whitespace (e.g. when a whole
187
+ // turn's text was suppressed as a leak, leaving only a stray newline).
188
+ if (!lastWasText && !clean.trim()) return;
241
189
  stopSpin();
242
190
  if (!lastWasText) {
243
191
  process.stdout.write("\n" + c.cyan("● "));
244
192
  lastWasText = true;
245
193
  }
246
- process.stdout.write(emit);
194
+ process.stdout.write(clean);
247
195
  },
248
196
  onToolCallDelta: (delta) => {
249
197
  // Just close the streamed text line when the model starts a tool
@@ -273,14 +221,18 @@ export async function runAgent({
273
221
  process.stdout.write(tail);
274
222
  }
275
223
  if (lastWasText) process.stdout.write("\n");
276
- const turnIn = res.usage?.prompt_tokens ?? 0;
277
- const turnOut = res.usage?.completion_tokens ?? 0;
278
- totalCredits += res.creditsCharged ?? 0;
279
- totalIn += turnIn;
280
- totalOut += turnOut;
224
+ const turnCredits = res.creditsCharged ?? 0;
225
+ totalCredits += turnCredits;
226
+ totalIn += res.usage?.prompt_tokens ?? 0;
227
+ totalOut += res.usage?.completion_tokens ?? 0;
281
228
  if (typeof res.balanceAfter === "number") lastBalance = res.balanceAfter;
282
229
  onTokens({ totalCredits, totalIn, totalOut, balance: lastBalance });
283
230
 
231
+ // Per-turn cost — visible so users catch runaway loops early
232
+ const inK = ((res.usage?.prompt_tokens ?? 0) / 1000).toFixed(1);
233
+ const outTok = res.usage?.completion_tokens ?? 0;
234
+ console.log(c.gray(` ${inK}k→${outTok} tokens · ${turnCredits} credits (total: ${totalCredits})`));
235
+
284
236
  // Push assistant message into history
285
237
  messages.push({
286
238
  role: "assistant",
@@ -289,6 +241,31 @@ export async function runAgent({
289
241
  });
290
242
 
291
243
  const toolCalls = res.message.tool_calls ?? [];
244
+ const hasTextContent = !!(res.message.content && res.message.content.trim());
245
+
246
+ if (toolCalls.length > 0 && !hasTextContent) {
247
+ consecutiveToolOnlyTurns++;
248
+ } else if (hasTextContent) {
249
+ consecutiveToolOnlyTurns = 0;
250
+ }
251
+
252
+ // Credit cap — hard stop to prevent runaway cost
253
+ if (maxCredits > 0 && totalCredits >= maxCredits) {
254
+ console.log(c.red(`\n[Credit budget exhausted: ${totalCredits} credits used, limit was ${maxCredits}. Stopping.]`));
255
+ return { ok: true, totalCredits, totalIn, totalOut, balance: lastBalance, messages };
256
+ }
257
+
258
+ // Tool call count cap
259
+ if (toolCalls.length > 0 && totalToolCalls + toolCalls.length > MAX_TOTAL_TOOL_CALLS) {
260
+ console.log(c.red(`\n[Tool call limit reached (${MAX_TOTAL_TOOL_CALLS}). Stopping.]`));
261
+ return { ok: true, totalCredits, totalIn, totalOut, balance: lastBalance, messages };
262
+ }
263
+
264
+ // Consecutive tool-only turns cap (model stuck in a tool loop with no text)
265
+ if (consecutiveToolOnlyTurns >= MAX_CONSECUTIVE_TOOL_ONLY) {
266
+ console.log(c.red(`\n[Too many consecutive tool-only turns (${MAX_CONSECUTIVE_TOOL_ONLY}). Stopping.]`));
267
+ return { ok: true, totalCredits, totalIn, totalOut, balance: lastBalance, messages };
268
+ }
292
269
  if (toolCalls.length === 0) {
293
270
  // Completeness guard: on a modification task, if the model read files it
294
271
  // never edited, it likely either described the fix instead of applying it
@@ -308,6 +285,88 @@ export async function runAgent({
308
285
  messages.push({ role: "user", content: msg });
309
286
  continue; // give the model another turn to finish
310
287
  }
288
+ // Auto-build: if we wrote source files, detect the build toolchain and
289
+ // compile automatically. If build fails, feed errors back to the model
290
+ // so it can fix and recompile. Max 5 rounds to avoid infinite loops.
291
+ if (filesTouched.size > 0 && buildRounds < 5) {
292
+ const buildResult = await runBuild(cwd, [...filesTouched.keys()]);
293
+ if (buildResult && !buildResult.ok) {
294
+ buildRounds++;
295
+ console.log("\n" + c.yellow("●") + " " + c.bold("auto-build") + c.gray(` (${buildResult.label}) — round ${buildRounds}/5`));
296
+ console.log(c.red(" └─") + " " + c.gray("build failed — feeding errors back to fix"));
297
+ const errPreview = buildResult.output.length > 6000
298
+ ? buildResult.output.slice(0, 6000) + "\n…(truncated)"
299
+ : buildResult.output;
300
+ messages.push({
301
+ role: "user",
302
+ content:
303
+ `AUTO-BUILD FAILED (${buildResult.label}):\n\`\`\`\n${errPreview}\n\`\`\`\n` +
304
+ `Fix the build errors NOW using edit_file. Do not explain — just fix the code and move on.`,
305
+ });
306
+ continue;
307
+ }
308
+ if (buildResult && buildResult.ok) {
309
+ console.log("\n" + c.green("●") + " " + c.bold("auto-build") + c.gray(` (${buildResult.label}) — success`));
310
+ // If there's a run command and we haven't already run it, execute the result
311
+ if (buildResult.runCmd && buildRounds === 0) {
312
+ const runResult = await runBuiltProgram(cwd, buildResult.runCmd, buildResult.label);
313
+ if (runResult && !runResult.ok && buildRounds < 5) {
314
+ buildRounds++;
315
+ console.log(c.yellow(" ├─") + " " + c.bold("auto-run") + c.gray(` — crashed (exit ${runResult.code})`));
316
+ const errPreview = runResult.output.length > 6000
317
+ ? runResult.output.slice(0, 6000) + "\n…(truncated)"
318
+ : runResult.output;
319
+ messages.push({
320
+ role: "user",
321
+ content:
322
+ `BUILD SUCCEEDED but the program CRASHED when run:\n\`\`\`\n${errPreview}\n\`\`\`\n` +
323
+ `Fix the runtime error NOW using edit_file. Do not explain — just fix and move on.`,
324
+ });
325
+ continue;
326
+ }
327
+ if (runResult && runResult.ok) {
328
+ const preview = runResult.output.length > 2000
329
+ ? runResult.output.slice(0, 2000) + "\n…(truncated)"
330
+ : runResult.output;
331
+ if (runResult.timedOut) {
332
+ console.log(c.green(" ├─") + " " + c.bold("auto-run") + c.gray(" — running (still alive after 10s)"));
333
+ } else {
334
+ console.log(c.green(" ├─") + " " + c.bold("auto-run") + c.gray(` — exit 0`));
335
+ }
336
+ if (preview.trim()) {
337
+ const lines = preview.trim().split("\n").slice(0, 15);
338
+ for (const line of lines) {
339
+ console.log(c.gray(" │ " + line));
340
+ }
341
+ }
342
+ }
343
+ }
344
+ }
345
+ }
346
+ // Auto-verify: if we touched files, run the project's typecheck/lint
347
+ // and feed errors back so the model can self-correct without the user
348
+ // having to say "fix that." At most 2 verify rounds to avoid infinite loops.
349
+ if (filesTouched.size > 0 && verifyRounds < 2) {
350
+ const verifyResult = await runVerify(cwd);
351
+ if (verifyResult && !verifyResult.ok) {
352
+ verifyRounds++;
353
+ console.log("\n" + c.yellow("●") + " " + c.bold("auto-verify") + c.gray(` (${verifyResult.label})`));
354
+ console.log(c.red(" └─") + " " + c.gray("errors found — feeding back to fix"));
355
+ const errPreview = verifyResult.output.length > 4000
356
+ ? verifyResult.output.slice(0, 4000) + "\n…(truncated)"
357
+ : verifyResult.output;
358
+ messages.push({
359
+ role: "user",
360
+ content:
361
+ `AUTO-VERIFY FAILED (${verifyResult.label}):\n\`\`\`\n${errPreview}\n\`\`\`\n` +
362
+ `Fix these errors NOW using edit_file. Do not explain — just fix and move on.`,
363
+ });
364
+ continue;
365
+ }
366
+ if (verifyResult && verifyResult.ok) {
367
+ console.log("\n" + c.green("●") + " " + c.bold("auto-verify") + c.gray(` (${verifyResult.label}) — passed`));
368
+ }
369
+ }
311
370
  if (filesTouched.size) {
312
371
  console.log("\n" + c.bold("Files") + c.gray(` in ${cwd}`));
313
372
  for (const [p, action] of filesTouched) {
@@ -319,6 +378,7 @@ export async function runAgent({
319
378
 
320
379
  // Execute each tool call. Show the actual args (now that we have them
321
380
  // fully assembled) and run.
381
+ totalToolCalls += toolCalls.length;
322
382
  for (const call of toolCalls) {
323
383
  let args = {};
324
384
  try { args = JSON.parse(call.function.arguments || "{}"); } catch { /* leave empty */ }
@@ -376,46 +436,135 @@ export async function runAgent({
376
436
  const summary = toolSummary(call.function.name, result);
377
437
  if (summary) console.log(summary);
378
438
 
379
- // Cap tool result content sent to the model to prevent it from
380
- // echoing raw data (snippets, HTML dumps) back as text output.
381
- let toolContent = result.output ?? (result.ok ? "(no output)" : "Failed.");
382
- if (call.function.name === "web_search") {
383
- try {
384
- const arr = JSON.parse(toolContent);
385
- if (Array.isArray(arr)) {
386
- toolContent = JSON.stringify(arr.map((r) => ({
387
- title: r.title, url: r.url,
388
- snippet: (r.snippet || "").slice(0, 120),
389
- })));
390
- }
391
- } catch { /* leave as-is */ }
392
- } else if (call.function.name === "web_fetch") {
393
- if (toolContent.length > 12000) toolContent = toolContent.slice(0, 12000) + "\n...(truncated)";
394
- } else if (call.function.name === "read_file") {
395
- if (toolContent.length > 30000) toolContent = toolContent.slice(0, 30000) + "\n...(truncated)";
396
- }
397
439
  messages.push({
398
440
  role: "tool",
399
441
  tool_call_id: call.id,
400
- content: toolContent,
442
+ content: result.output ?? (result.ok ? "(no output)" : "Failed."),
401
443
  });
402
444
  }
403
-
404
- // Per-turn stats line — shows elapsed time, token flow, and tool count.
405
- const stats = turnStats({
406
- elapsed: Date.now() - turnStart,
407
- tokensIn: turnIn,
408
- tokensOut: turnOut,
409
- tools: toolCalls.length,
410
- credits: res.creditsCharged ?? 0,
411
- });
412
- if (stats) console.log(stats);
413
445
  }
414
446
 
415
447
  console.log(c.yellow(`\nReached max turns (${maxTurns}). Stopping.`));
416
448
  return { ok: false, error: new Error("Max turns reached"), totalCredits, totalIn, totalOut, balance: lastBalance, messages };
417
449
  }
418
450
 
451
+ // A build/verify command failing because the TOOLCHAIN ITSELF is missing (e.g.
452
+ // g++/cargo/dotnet not installed) is not a code bug — it must NOT be reported as
453
+ // "build failed" and fed back to the model, which would derail the run into
454
+ // 5 rounds of "fixing" code that's actually fine. These are the shell's
455
+ // command-not-found messages (Windows cmd.exe + POSIX sh); real compiler
456
+ // diagnostics look like `main.cpp:5:10: error: ...` and never match these.
457
+ // Matches ONLY the shell's "I can't find this executable" messages — Windows
458
+ // cmd.exe ("'g++' is not recognized…") and POSIX sh ("g++: command not found",
459
+ // "sh: 1: g++: not found"). Real compiler diagnostics (`main.cpp:5: error:`,
460
+ // `fatal error: foo.h: No such file or directory`) never match these, so a
461
+ // genuine build error is still reported and fixed.
462
+ const TOOLCHAIN_MISSING_RE =
463
+ /is not recognized as an internal or external command|: command not found|: not found\b/i;
464
+
465
+ // Auto-verify: run the project's typecheck/lint command silently and return
466
+ // { ok, output, label }. Returns null if no verify command is detected OR the
467
+ // toolchain isn't installed (skip, don't report a false failure).
468
+ let cachedVerify = undefined;
469
+ async function runVerify(cwd) {
470
+ if (cachedVerify === undefined) cachedVerify = detectVerifyCommand(cwd);
471
+ if (!cachedVerify) return null;
472
+ const { cmd, label } = cachedVerify;
473
+ return new Promise((resolve) => {
474
+ const child = spawn(cmd, [], { cwd, shell: true, stdio: ["ignore", "pipe", "pipe"] });
475
+ let stdout = "";
476
+ let stderr = "";
477
+ const timer = setTimeout(() => { child.kill("SIGTERM"); }, 60_000);
478
+ child.stdout.on("data", (d) => { stdout += d.toString(); });
479
+ child.stderr.on("data", (d) => { stderr += d.toString(); });
480
+ child.on("close", (code) => {
481
+ clearTimeout(timer);
482
+ const output = (stdout + "\n" + stderr).trim();
483
+ // Toolchain not installed → skip silently rather than nag the model.
484
+ if (code !== 0 && TOOLCHAIN_MISSING_RE.test(output)) { resolve(null); return; }
485
+ resolve({ ok: code === 0, output, label });
486
+ });
487
+ child.on("error", () => {
488
+ clearTimeout(timer);
489
+ resolve(null);
490
+ });
491
+ });
492
+ }
493
+
494
+ // Auto-build: detect the build toolchain from written files and compile.
495
+ // Returns { ok, output, label, runCmd } or null if no buildable project detected.
496
+ let cachedBuild = undefined;
497
+ async function runBuild(cwd, touchedFiles) {
498
+ // Re-detect every time because the model may create new files (e.g. Makefile)
499
+ // or change the project structure between rounds.
500
+ const detected = detectBuildCommand(cwd, touchedFiles);
501
+ if (!detected) {
502
+ // Fallback: if we previously detected a build command, reuse it (the model
503
+ // may have only edited files this round, not created new ones).
504
+ if (cachedBuild) return runBuildCmd(cwd, cachedBuild);
505
+ return null;
506
+ }
507
+ cachedBuild = detected;
508
+ return runBuildCmd(cwd, detected);
509
+ }
510
+
511
+ async function runBuildCmd(cwd, { cmd, label, type, runCmd }) {
512
+ return new Promise((resolve) => {
513
+ const child = spawn(cmd, [], { cwd, shell: true, stdio: ["ignore", "pipe", "pipe"] });
514
+ let stdout = "";
515
+ let stderr = "";
516
+ const timer = setTimeout(() => { child.kill("SIGTERM"); }, 120_000);
517
+ child.stdout.on("data", (d) => { stdout += d.toString(); });
518
+ child.stderr.on("data", (d) => { stderr += d.toString(); });
519
+ child.on("close", (code) => {
520
+ clearTimeout(timer);
521
+ const output = (stdout + "\n" + stderr).trim();
522
+ // Compiler/toolchain not installed → skip auto-build silently instead of
523
+ // reporting a build failure and derailing the run.
524
+ if (code !== 0 && TOOLCHAIN_MISSING_RE.test(output)) { resolve(null); return; }
525
+ resolve({ ok: code === 0, output, label, type, runCmd });
526
+ });
527
+ child.on("error", () => {
528
+ clearTimeout(timer);
529
+ // ENOENT (command not found) → toolchain missing → skip.
530
+ resolve(null);
531
+ });
532
+ });
533
+ }
534
+
535
+ // Run the compiled/interpreted program after a successful build.
536
+ // Returns { ok, output, code, timedOut } or null on spawn error.
537
+ async function runBuiltProgram(cwd, runCmd, label) {
538
+ return new Promise((resolve) => {
539
+ const child = spawn(runCmd, [], { cwd, shell: true, stdio: ["ignore", "pipe", "pipe"] });
540
+ let stdout = "";
541
+ let stderr = "";
542
+ let timedOut = false;
543
+ const timer = setTimeout(() => { timedOut = true; child.kill("SIGTERM"); }, 10_000);
544
+ child.stdout.on("data", (d) => {
545
+ stdout += d.toString();
546
+ if (stdout.length > 50_000) { child.kill("SIGTERM"); }
547
+ });
548
+ child.stderr.on("data", (d) => {
549
+ stderr += d.toString();
550
+ if (stderr.length > 50_000) { child.kill("SIGTERM"); }
551
+ });
552
+ child.on("close", (code) => {
553
+ clearTimeout(timer);
554
+ const output = (stdout + "\n" + stderr).trim();
555
+ if (timedOut) {
556
+ resolve({ ok: true, output, code: null, timedOut: true });
557
+ } else {
558
+ resolve({ ok: code === 0, output, code, timedOut: false });
559
+ }
560
+ });
561
+ child.on("error", (err) => {
562
+ clearTimeout(timer);
563
+ resolve({ ok: false, output: err.message, code: null, timedOut: false });
564
+ });
565
+ });
566
+ }
567
+
419
568
  /**
420
569
  * Per-turn skill injection. Selects skills against the latest user message
421
570
  * + paths the model has touched, then prepends matching bodies onto the
@@ -430,7 +579,7 @@ export async function runAgent({
430
579
  * worse than no skills at all. Prepending into the user message keeps
431
580
  * both layers active.
432
581
  */
433
- function buildTurnMessages(messages, allSkills, referencedPaths) {
582
+ export function buildTurnMessages(messages, allSkills, referencedPaths) {
434
583
  if (allSkills.length === 0) return messages;
435
584
  // Find the latest user message — that's where the current task lives.
436
585
  let lastUserIdx = -1;
@@ -438,30 +587,16 @@ function buildTurnMessages(messages, allSkills, referencedPaths) {
438
587
  if (messages[i].role === "user") { lastUserIdx = i; break; }
439
588
  }
440
589
  if (lastUserIdx === -1) return messages;
441
- // Content can be a string or a multimodal array (when images are attached).
442
- // Extract the text portion for skill selection; prepend skills to text only.
443
- const rawContent = messages[lastUserIdx].content;
444
- let prompt;
445
- if (typeof rawContent === "string") {
446
- prompt = rawContent;
447
- } else if (Array.isArray(rawContent)) {
448
- const textPart = rawContent.find((p) => p.type === "text");
449
- prompt = textPart?.text ?? "";
450
- } else {
451
- prompt = "";
452
- }
590
+ const prompt = typeof messages[lastUserIdx].content === "string"
591
+ ? messages[lastUserIdx].content
592
+ : "";
453
593
  const active = selectSkills({ skills: allSkills, prompt, referencedPaths });
454
594
  if (active.length === 0) return messages;
455
595
  const block = renderSkillsBlock(active);
456
596
  const cloned = [...messages];
457
- // Inject skill block into the text portion of the content.
458
- if (typeof rawContent === "string") {
459
- cloned[lastUserIdx] = { ...cloned[lastUserIdx], content: `${block}\n\n---\n\n${prompt}` };
460
- } else if (Array.isArray(rawContent)) {
461
- const newParts = rawContent.map((p) =>
462
- p.type === "text" ? { ...p, text: `${block}\n\n---\n\n${p.text}` } : p,
463
- );
464
- cloned[lastUserIdx] = { ...cloned[lastUserIdx], content: newParts };
465
- }
597
+ cloned[lastUserIdx] = {
598
+ ...cloned[lastUserIdx],
599
+ content: `${block}\n\n---\n\n${prompt}`,
600
+ };
466
601
  return cloned;
467
602
  }