aether-code 0.43.1 → 0.43.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -21
- package/bin/aether-code.js +23 -18
- package/package.json +2 -4
- package/skills/adult-creative-writing.md +60 -60
- package/skills/debugging.md +51 -51
- package/skills/game-modding.md +73 -73
- package/skills/reverse-engineering.md +41 -41
- package/skills/scraping-automation.md +77 -77
- package/skills/security-research.md +67 -67
- package/src/agent.js +311 -175
- package/src/api.js +32 -22
- package/src/box-input.js +36 -141
- package/src/config.js +38 -38
- package/src/diff.js +49 -49
- package/src/ink-input.js +91 -91
- package/src/mcp-cli.js +94 -94
- package/src/mcp-registry.js +266 -266
- package/src/mcp.js +260 -259
- package/src/menu.js +83 -83
- package/src/project-context.js +712 -0
- package/src/render.js +254 -288
- package/src/repl.js +441 -408
- package/src/sessions.js +167 -0
- package/src/setup.js +8 -6
- package/src/tools.js +393 -16
- package/src/update-check.js +7 -64
- package/src/version.js +15 -0
- package/scripts/postinstall.js +0 -182
package/src/agent.js
CHANGED
|
@@ -3,98 +3,69 @@
|
|
|
3
3
|
// tool calls (task done) or max-turns is reached.
|
|
4
4
|
|
|
5
5
|
import os from "node:os";
|
|
6
|
-
import fs from "node:fs";
|
|
7
6
|
import path from "node:path";
|
|
7
|
+
import { spawn } from "node:child_process";
|
|
8
8
|
import { agentTurnStream, AetherError } from "./api.js";
|
|
9
9
|
import { TOOL_DEFINITIONS, executeTool } from "./tools.js";
|
|
10
10
|
import { unnamespaceToolName } from "./mcp.js";
|
|
11
11
|
import { loadAllSkills, selectSkills, renderSkillsBlock } from "./skills.js";
|
|
12
|
-
import {
|
|
13
|
-
|
|
14
|
-
/* ─────────────────────── Image attachment parsing ─────────────────────── */
|
|
15
|
-
|
|
16
|
-
const IMAGE_EXTS = new Set([".png", ".jpg", ".jpeg", ".gif", ".webp", ".bmp"]);
|
|
17
|
-
const MIME_MAP = {
|
|
18
|
-
".png": "image/png",
|
|
19
|
-
".jpg": "image/jpeg",
|
|
20
|
-
".jpeg": "image/jpeg",
|
|
21
|
-
".gif": "image/gif",
|
|
22
|
-
".webp": "image/webp",
|
|
23
|
-
".bmp": "image/bmp",
|
|
24
|
-
};
|
|
25
|
-
|
|
26
|
-
// Parse @path/to/image references from user input. Supports quoted paths for
|
|
27
|
-
// spaces: @"C:\Users\me\my screenshot.png". Returns { text, images }.
|
|
28
|
-
function parseImages(prompt, cwd) {
|
|
29
|
-
const re = /@("(?:[^"\\]|\\.)*"|[^\s]+)/g;
|
|
30
|
-
const images = [];
|
|
31
|
-
let cleaned = prompt;
|
|
32
|
-
|
|
33
|
-
for (const m of [...prompt.matchAll(re)]) {
|
|
34
|
-
const raw = m[1].replace(/^"|"$/g, "");
|
|
35
|
-
const ext = path.extname(raw).toLowerCase();
|
|
36
|
-
if (!IMAGE_EXTS.has(ext)) continue;
|
|
37
|
-
|
|
38
|
-
const abs = path.isAbsolute(raw) ? raw : path.resolve(cwd, raw);
|
|
39
|
-
try {
|
|
40
|
-
const buf = fs.readFileSync(abs);
|
|
41
|
-
if (buf.length > 20 * 1024 * 1024) {
|
|
42
|
-
console.log(c.yellow(` ⚠ Image too large (>20MB), skipping: ${raw}`));
|
|
43
|
-
continue;
|
|
44
|
-
}
|
|
45
|
-
const mime = MIME_MAP[ext] || "image/png";
|
|
46
|
-
images.push({
|
|
47
|
-
url: `data:${mime};base64,${buf.toString("base64")}`,
|
|
48
|
-
name: path.basename(raw),
|
|
49
|
-
size: buf.length,
|
|
50
|
-
});
|
|
51
|
-
cleaned = cleaned.replace(m[0], "");
|
|
52
|
-
} catch (e) {
|
|
53
|
-
console.log(c.yellow(` ⚠ Could not read image: ${raw} — ${e.code === "ENOENT" ? "file not found" : e.message}`));
|
|
54
|
-
}
|
|
55
|
-
}
|
|
56
|
-
|
|
57
|
-
return { text: cleaned.replace(/\s{2,}/g, " ").trim() || prompt.trim(), images };
|
|
58
|
-
}
|
|
12
|
+
import { scanProject, detectVerifyCommand, detectBuildCommand } from "./project-context.js";
|
|
13
|
+
import { c, divider, turn, toolLabel, toolSummary, makeTokenStripper, errorLine, startSpinner } from "./render.js";
|
|
59
14
|
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
15
|
+
const DEFAULT_MAX_TURNS = 25;
|
|
16
|
+
const DEFAULT_MAX_CREDITS = 500;
|
|
17
|
+
const MAX_TOTAL_TOOL_CALLS = 40;
|
|
18
|
+
const MAX_CONSECUTIVE_TOOL_ONLY = 6;
|
|
19
|
+
|
|
20
|
+
// Rolling history window (messages) sent upstream per turn. The session keeps
|
|
21
|
+
// the full transcript locally, but we only forward the anchor (first message,
|
|
22
|
+
// which carries the [environment] block) plus the most recent N messages. This
|
|
23
|
+
// bounds per-turn token cost AND keeps us safely under the server's own cap —
|
|
24
|
+
// a long session used to grow unbounded and, past the server's limit, every
|
|
25
|
+
// send failed with "Error: Invalid input." with no recovery but /clear.
|
|
26
|
+
const HISTORY_WINDOW = 100;
|
|
71
27
|
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
28
|
+
/**
|
|
29
|
+
* Trim a message array to `maxMessages`, preserving:
|
|
30
|
+
* - the anchor message at index 0 (the first user turn carries the
|
|
31
|
+
* [environment] block with real absolute paths — dropping it makes the
|
|
32
|
+
* model lose cwd/home/desktop);
|
|
33
|
+
* - tool-call/tool-result pairing at the window boundary: a `tool` message
|
|
34
|
+
* must immediately follow the assistant message whose `tool_calls` produced
|
|
35
|
+
* it, so the window is never allowed to BEGIN on an orphaned `tool` result.
|
|
36
|
+
* Returns the original array untouched when it's already within the window.
|
|
37
|
+
*/
|
|
38
|
+
export function trimHistory(messages, maxMessages = HISTORY_WINDOW) {
|
|
39
|
+
if (messages.length <= maxMessages) return messages;
|
|
40
|
+
const head = [messages[0]];
|
|
41
|
+
const windowSize = Math.max(1, maxMessages - 1);
|
|
42
|
+
let window = messages.slice(1).slice(-windowSize);
|
|
43
|
+
let start = 0;
|
|
44
|
+
while (start < window.length && window[start]?.role === "tool") start++;
|
|
45
|
+
window = window.slice(start);
|
|
46
|
+
return [...head, ...window];
|
|
77
47
|
}
|
|
78
48
|
|
|
79
|
-
const
|
|
49
|
+
export const MOD_RE =
|
|
50
|
+
/\b(refactor|fix|add|adds?|change|update|implement|creat|writ|remov|delet|rename|extract|move|replace|build|generate|insert|append|convert|migrat|wire|hook|edit|modif|patch|integrat)/i;
|
|
80
51
|
|
|
81
|
-
|
|
82
|
-
// resolve named locations to real absolute paths.
|
|
83
|
-
function envContext(cwd) {
|
|
52
|
+
export function envContext(cwd) {
|
|
84
53
|
const home = os.homedir();
|
|
85
54
|
const desktop = path.join(home, "Desktop");
|
|
86
55
|
const documents = path.join(home, "Documents");
|
|
87
56
|
const win = process.platform === "win32";
|
|
88
|
-
// OS-specific shell guidance: gemma tends to emit Unix commands (mkdir -p,
|
|
89
|
-
// touch, &&-chains) which FAIL in Windows cmd.exe ("The syntax of the command
|
|
90
|
-
// is incorrect"). Steer it to the file tools (which are cross-platform and
|
|
91
|
-
// auto-create parent dirs) and OS-appropriate shell usage.
|
|
92
57
|
const shellNote = win
|
|
93
58
|
? `shell: Windows cmd.exe. To create files OR directories, use the write_file tool — ` +
|
|
94
59
|
`it creates any missing parent folders automatically. Do NOT use shell "mkdir -p", ` +
|
|
95
60
|
`"touch", "ls", "rm", "cp", "mv", or "&&"-chained Unix commands; they FAIL in cmd.exe. ` +
|
|
96
61
|
`Run programs/tests with their interpreter (python, node, java, etc.).`
|
|
97
62
|
: `shell: POSIX sh. Standard Unix commands are available. write_file still auto-creates parent dirs.`;
|
|
63
|
+
|
|
64
|
+
let projectBlock = "";
|
|
65
|
+
try {
|
|
66
|
+
projectBlock = scanProject(cwd);
|
|
67
|
+
} catch { /* non-fatal */ }
|
|
68
|
+
|
|
98
69
|
return (
|
|
99
70
|
`[environment]\n` +
|
|
100
71
|
`os: ${process.platform}\n` +
|
|
@@ -106,7 +77,8 @@ function envContext(cwd) {
|
|
|
106
77
|
`When the user names a location ("my desktop", "home", "documents"), write to ` +
|
|
107
78
|
`the matching ABSOLUTE path above (e.g. desktop -> ${desktop}). Otherwise ` +
|
|
108
79
|
`work under the cwd. Use absolute paths when a specific location is named.\n` +
|
|
109
|
-
`[/environment]\n\n`
|
|
80
|
+
`[/environment]\n\n` +
|
|
81
|
+
(projectBlock ? projectBlock + "\n" : "")
|
|
110
82
|
);
|
|
111
83
|
}
|
|
112
84
|
|
|
@@ -117,6 +89,7 @@ export async function runAgent({
|
|
|
117
89
|
autoYes = false,
|
|
118
90
|
unsafePaths = false,
|
|
119
91
|
maxTurns = DEFAULT_MAX_TURNS,
|
|
92
|
+
maxCredits = DEFAULT_MAX_CREDITS,
|
|
120
93
|
model = null, // null = server default (gemma). Premium models gated server-side.
|
|
121
94
|
onTokens = () => {},
|
|
122
95
|
// Optional MCPManager. When provided, its tools are merged into the agent's
|
|
@@ -142,15 +115,6 @@ export async function runAgent({
|
|
|
142
115
|
process.stderr.write(c.yellow(`(skill load failed: ${e.message})\n`));
|
|
143
116
|
}
|
|
144
117
|
const referencedPaths = [];
|
|
145
|
-
|
|
146
|
-
// Parse image attachments (@path/to/image.png) from the user's prompt.
|
|
147
|
-
const { text: cleanPrompt, images: attachedImages } = parseImages(initialPrompt, cwd);
|
|
148
|
-
if (attachedImages.length > 0) {
|
|
149
|
-
for (const img of attachedImages) {
|
|
150
|
-
console.log(`\n${c.cyan("●")} ${c.cyan(c.bold("image"))} ${c.gray(img.name)} ${c.gray(`(${fmtSize(img.size)})`)}`);
|
|
151
|
-
}
|
|
152
|
-
}
|
|
153
|
-
|
|
154
118
|
// Two callers: one-shot (initialPrompt only, fresh conversation) and REPL
|
|
155
119
|
// (priorMessages + initialPrompt to continue an ongoing chat).
|
|
156
120
|
// On the FIRST message of a session, prepend an environment block so the
|
|
@@ -158,15 +122,14 @@ export async function runAgent({
|
|
|
158
122
|
// X on my desktop" became `mkdir X` in whatever dir aether was launched from
|
|
159
123
|
// (e.g. C:\WINDOWS\system32). Only prepended once — later turns carry it in
|
|
160
124
|
// history.
|
|
161
|
-
const firstText = priorMessages ? cleanPrompt : envContext(cwd) + cleanPrompt;
|
|
162
|
-
const firstContent = buildContent(firstText, attachedImages);
|
|
163
125
|
const messages = priorMessages
|
|
164
|
-
? [...priorMessages, { role: "user", content:
|
|
165
|
-
: [{ role: "user", content:
|
|
126
|
+
? [...priorMessages, { role: "user", content: initialPrompt }]
|
|
127
|
+
: [{ role: "user", content: envContext(cwd) + initialPrompt }];
|
|
166
128
|
let totalCredits = 0;
|
|
167
129
|
let totalIn = 0;
|
|
168
130
|
let totalOut = 0;
|
|
169
131
|
let lastBalance = null;
|
|
132
|
+
const t0 = Date.now();
|
|
170
133
|
// Loop guard: count identical (name+args) tool calls across the whole run so a
|
|
171
134
|
// confused model can't burn turns re-running the same call (e.g. glob *.md x9).
|
|
172
135
|
const callCounts = new Map();
|
|
@@ -179,30 +142,31 @@ export async function runAgent({
|
|
|
179
142
|
// Completeness guard: gemma sometimes reads files then narrates the fix as
|
|
180
143
|
// text, or refactors one file but forgets the others. If a MODIFICATION task
|
|
181
144
|
// finishes having read files it never edited, we nudge it once to finish.
|
|
182
|
-
const MOD_RE =
|
|
183
|
-
/\b(refactor|fix|add|adds?|change|update|implement|creat|writ|remov|delet|rename|extract|move|replace|build|generate|insert|append|convert|migrat|wire|hook|edit|modif|patch|integrat)/i;
|
|
184
145
|
const looksLikeModification = MOD_RE.test(initialPrompt);
|
|
185
146
|
let appliedNothingNudges = 0;
|
|
147
|
+
let verifyRounds = 0;
|
|
148
|
+
let buildRounds = 0;
|
|
149
|
+
let consecutiveToolOnlyTurns = 0;
|
|
150
|
+
let totalToolCalls = 0;
|
|
186
151
|
|
|
187
152
|
for (let i = 0; i < maxTurns; i++) {
|
|
188
|
-
|
|
153
|
+
// No turn header and no leading blank here — each step (assistant text and
|
|
154
|
+
// each tool label) begins with its own "\n● ", so spacing stays exactly one
|
|
155
|
+
// blank line per step instead of stacking up.
|
|
189
156
|
|
|
190
157
|
// Stream the assistant's response. Print text deltas as they arrive,
|
|
191
158
|
// along with tool-call announcements as soon as the model commits to
|
|
192
159
|
// calling a particular tool (i.e. the `name` arrives in the stream).
|
|
193
160
|
const announced = new Set();
|
|
194
161
|
let lastWasText = false;
|
|
195
|
-
let streamedChars = 0;
|
|
196
162
|
const stripper = makeTokenStripper();
|
|
197
|
-
let leakBuf = "";
|
|
198
|
-
let leakSuppressing = false;
|
|
199
163
|
|
|
200
164
|
// Select skills for this turn against the current user prompt + any
|
|
201
165
|
// paths the model has read so far. Prepend the matching skills' bodies
|
|
202
166
|
// to the last user message of a shallow-cloned messages array — we
|
|
203
167
|
// don't want skill text accumulating in the persisted history, only
|
|
204
168
|
// being available to the model for the turn where it's relevant.
|
|
205
|
-
const turnMessages = buildTurnMessages(messages, allSkills, referencedPaths);
|
|
169
|
+
const turnMessages = trimHistory(buildTurnMessages(messages, allSkills, referencedPaths));
|
|
206
170
|
|
|
207
171
|
let res;
|
|
208
172
|
// "Thinking" spinner: shown from request-send until the first token or tool
|
|
@@ -216,34 +180,19 @@ export async function runAgent({
|
|
|
216
180
|
tools,
|
|
217
181
|
model,
|
|
218
182
|
onDelta: (text) => {
|
|
219
|
-
|
|
220
|
-
|
|
183
|
+
// Buffered strip of leaked model channel/control tokens (which can
|
|
184
|
+
// be split across stream chunks) before display.
|
|
221
185
|
const clean = stripper.push(text);
|
|
222
186
|
if (!clean) return;
|
|
223
|
-
|
|
224
|
-
//
|
|
225
|
-
|
|
226
|
-
if (leakBuf.length < 40 && /^[\s\n]*(?:response:|[\[{]"?(?:snippet|title|url))/.test(leakBuf)) return;
|
|
227
|
-
if (leakSuppressing) {
|
|
228
|
-
if (clean.includes("\n\n")) { leakSuppressing = false; leakBuf = ""; }
|
|
229
|
-
return;
|
|
230
|
-
}
|
|
231
|
-
const checkBuf = leakBuf.trim();
|
|
232
|
-
if (checkBuf && /^response:\w|^\[\{(?:"?snippet|"?title)/.test(checkBuf)) {
|
|
233
|
-
leakSuppressing = true;
|
|
234
|
-
leakBuf = "";
|
|
235
|
-
return;
|
|
236
|
-
}
|
|
237
|
-
const emit = leakBuf;
|
|
238
|
-
leakBuf = "";
|
|
239
|
-
|
|
240
|
-
if (!lastWasText && !emit.trim()) return;
|
|
187
|
+
// Don't open a "● " bullet for leading whitespace (e.g. when a whole
|
|
188
|
+
// turn's text was suppressed as a leak, leaving only a stray newline).
|
|
189
|
+
if (!lastWasText && !clean.trim()) return;
|
|
241
190
|
stopSpin();
|
|
242
191
|
if (!lastWasText) {
|
|
243
|
-
process.stdout.write("\n"
|
|
192
|
+
process.stdout.write("\n");
|
|
244
193
|
lastWasText = true;
|
|
245
194
|
}
|
|
246
|
-
process.stdout.write(
|
|
195
|
+
process.stdout.write(clean);
|
|
247
196
|
},
|
|
248
197
|
onToolCallDelta: (delta) => {
|
|
249
198
|
// Just close the streamed text line when the model starts a tool
|
|
@@ -260,7 +209,7 @@ export async function runAgent({
|
|
|
260
209
|
} catch (err) {
|
|
261
210
|
stopSpin();
|
|
262
211
|
if (err instanceof AetherError) {
|
|
263
|
-
return { ok: false, error: err, totalCredits, totalIn, totalOut, balance: lastBalance, messages };
|
|
212
|
+
return { ok: false, error: err, totalCredits, totalIn, totalOut, balance: lastBalance, messages, elapsed: Date.now() - t0 };
|
|
264
213
|
}
|
|
265
214
|
throw err;
|
|
266
215
|
}
|
|
@@ -269,18 +218,22 @@ export async function runAgent({
|
|
|
269
218
|
// Flush any held-back partial token, then close the line.
|
|
270
219
|
const tail = stripper.flush();
|
|
271
220
|
if (tail && (lastWasText || tail.trim())) {
|
|
272
|
-
if (!lastWasText) { process.stdout.write("\n"
|
|
221
|
+
if (!lastWasText) { process.stdout.write("\n"); lastWasText = true; }
|
|
273
222
|
process.stdout.write(tail);
|
|
274
223
|
}
|
|
275
224
|
if (lastWasText) process.stdout.write("\n");
|
|
276
|
-
const
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
totalOut += turnOut;
|
|
225
|
+
const turnCredits = res.creditsCharged ?? 0;
|
|
226
|
+
totalCredits += turnCredits;
|
|
227
|
+
totalIn += res.usage?.prompt_tokens ?? 0;
|
|
228
|
+
totalOut += res.usage?.completion_tokens ?? 0;
|
|
281
229
|
if (typeof res.balanceAfter === "number") lastBalance = res.balanceAfter;
|
|
282
230
|
onTokens({ totalCredits, totalIn, totalOut, balance: lastBalance });
|
|
283
231
|
|
|
232
|
+
// Per-turn cost — only shown when credits are high enough to warrant attention
|
|
233
|
+
if (turnCredits > 20) {
|
|
234
|
+
console.log(c.gray(` ${turnCredits} credits this turn (${totalCredits} total)`));
|
|
235
|
+
}
|
|
236
|
+
|
|
284
237
|
// Push assistant message into history
|
|
285
238
|
messages.push({
|
|
286
239
|
role: "assistant",
|
|
@@ -289,6 +242,31 @@ export async function runAgent({
|
|
|
289
242
|
});
|
|
290
243
|
|
|
291
244
|
const toolCalls = res.message.tool_calls ?? [];
|
|
245
|
+
const hasTextContent = !!(res.message.content && res.message.content.trim());
|
|
246
|
+
|
|
247
|
+
if (toolCalls.length > 0 && !hasTextContent) {
|
|
248
|
+
consecutiveToolOnlyTurns++;
|
|
249
|
+
} else if (hasTextContent) {
|
|
250
|
+
consecutiveToolOnlyTurns = 0;
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
// Credit cap — hard stop to prevent runaway cost
|
|
254
|
+
if (maxCredits > 0 && totalCredits >= maxCredits) {
|
|
255
|
+
console.log(c.red(`\n[Credit budget exhausted: ${totalCredits} credits used, limit was ${maxCredits}. Stopping.]`));
|
|
256
|
+
return { ok: true, totalCredits, totalIn, totalOut, balance: lastBalance, messages, elapsed: Date.now() - t0 };
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
// Tool call count cap
|
|
260
|
+
if (toolCalls.length > 0 && totalToolCalls + toolCalls.length > MAX_TOTAL_TOOL_CALLS) {
|
|
261
|
+
console.log(c.red(`\n[Tool call limit reached (${MAX_TOTAL_TOOL_CALLS}). Stopping.]`));
|
|
262
|
+
return { ok: true, totalCredits, totalIn, totalOut, balance: lastBalance, messages, elapsed: Date.now() - t0 };
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
// Consecutive tool-only turns cap (model stuck in a tool loop with no text)
|
|
266
|
+
if (consecutiveToolOnlyTurns >= MAX_CONSECUTIVE_TOOL_ONLY) {
|
|
267
|
+
console.log(c.red(`\n[Too many consecutive tool-only turns (${MAX_CONSECUTIVE_TOOL_ONLY}). Stopping.]`));
|
|
268
|
+
return { ok: true, totalCredits, totalIn, totalOut, balance: lastBalance, messages, elapsed: Date.now() - t0 };
|
|
269
|
+
}
|
|
292
270
|
if (toolCalls.length === 0) {
|
|
293
271
|
// Completeness guard: on a modification task, if the model read files it
|
|
294
272
|
// never edited, it likely either described the fix instead of applying it
|
|
@@ -308,23 +286,106 @@ export async function runAgent({
|
|
|
308
286
|
messages.push({ role: "user", content: msg });
|
|
309
287
|
continue; // give the model another turn to finish
|
|
310
288
|
}
|
|
289
|
+
// Auto-build: if we wrote source files, detect the build toolchain and
|
|
290
|
+
// compile automatically. If build fails, feed errors back to the model
|
|
291
|
+
// so it can fix and recompile. Max 5 rounds to avoid infinite loops.
|
|
292
|
+
if (filesTouched.size > 0 && buildRounds < 5) {
|
|
293
|
+
const buildResult = await runBuild(cwd, [...filesTouched.keys()]);
|
|
294
|
+
if (buildResult && !buildResult.ok) {
|
|
295
|
+
buildRounds++;
|
|
296
|
+
console.log("\n" + c.yellow("●") + " " + c.bold("auto-build") + c.gray(` (${buildResult.label}) — round ${buildRounds}/5`));
|
|
297
|
+
console.log(c.red(" └─") + " " + c.gray("build failed — feeding errors back to fix"));
|
|
298
|
+
const errPreview = buildResult.output.length > 6000
|
|
299
|
+
? buildResult.output.slice(0, 6000) + "\n…(truncated)"
|
|
300
|
+
: buildResult.output;
|
|
301
|
+
messages.push({
|
|
302
|
+
role: "user",
|
|
303
|
+
content:
|
|
304
|
+
`AUTO-BUILD FAILED (${buildResult.label}):\n\`\`\`\n${errPreview}\n\`\`\`\n` +
|
|
305
|
+
`Fix the build errors NOW using edit_file. Do not explain — just fix the code and move on.`,
|
|
306
|
+
});
|
|
307
|
+
continue;
|
|
308
|
+
}
|
|
309
|
+
if (buildResult && buildResult.ok) {
|
|
310
|
+
console.log("\n" + c.green("●") + " " + c.bold("auto-build") + c.gray(` (${buildResult.label}) — success`));
|
|
311
|
+
// If there's a run command and we haven't already run it, execute the result
|
|
312
|
+
if (buildResult.runCmd && buildRounds === 0) {
|
|
313
|
+
const runResult = await runBuiltProgram(cwd, buildResult.runCmd, buildResult.label);
|
|
314
|
+
if (runResult && !runResult.ok && buildRounds < 5) {
|
|
315
|
+
buildRounds++;
|
|
316
|
+
console.log(c.yellow(" ├─") + " " + c.bold("auto-run") + c.gray(` — crashed (exit ${runResult.code})`));
|
|
317
|
+
const errPreview = runResult.output.length > 6000
|
|
318
|
+
? runResult.output.slice(0, 6000) + "\n…(truncated)"
|
|
319
|
+
: runResult.output;
|
|
320
|
+
messages.push({
|
|
321
|
+
role: "user",
|
|
322
|
+
content:
|
|
323
|
+
`BUILD SUCCEEDED but the program CRASHED when run:\n\`\`\`\n${errPreview}\n\`\`\`\n` +
|
|
324
|
+
`Fix the runtime error NOW using edit_file. Do not explain — just fix and move on.`,
|
|
325
|
+
});
|
|
326
|
+
continue;
|
|
327
|
+
}
|
|
328
|
+
if (runResult && runResult.ok) {
|
|
329
|
+
const preview = runResult.output.length > 2000
|
|
330
|
+
? runResult.output.slice(0, 2000) + "\n…(truncated)"
|
|
331
|
+
: runResult.output;
|
|
332
|
+
if (runResult.timedOut) {
|
|
333
|
+
console.log(c.green(" ├─") + " " + c.bold("auto-run") + c.gray(" — running (still alive after 10s)"));
|
|
334
|
+
} else {
|
|
335
|
+
console.log(c.green(" ├─") + " " + c.bold("auto-run") + c.gray(` — exit 0`));
|
|
336
|
+
}
|
|
337
|
+
if (preview.trim()) {
|
|
338
|
+
const lines = preview.trim().split("\n").slice(0, 15);
|
|
339
|
+
for (const line of lines) {
|
|
340
|
+
console.log(c.gray(" │ " + line));
|
|
341
|
+
}
|
|
342
|
+
}
|
|
343
|
+
}
|
|
344
|
+
}
|
|
345
|
+
}
|
|
346
|
+
}
|
|
347
|
+
// Auto-verify: if we touched files, run the project's typecheck/lint
|
|
348
|
+
// and feed errors back so the model can self-correct without the user
|
|
349
|
+
// having to say "fix that." At most 2 verify rounds to avoid infinite loops.
|
|
350
|
+
if (filesTouched.size > 0 && verifyRounds < 2) {
|
|
351
|
+
const verifyResult = await runVerify(cwd);
|
|
352
|
+
if (verifyResult && !verifyResult.ok) {
|
|
353
|
+
verifyRounds++;
|
|
354
|
+
console.log("\n" + c.yellow("●") + " " + c.bold("auto-verify") + c.gray(` (${verifyResult.label})`));
|
|
355
|
+
console.log(c.red(" └─") + " " + c.gray("errors found — feeding back to fix"));
|
|
356
|
+
const errPreview = verifyResult.output.length > 4000
|
|
357
|
+
? verifyResult.output.slice(0, 4000) + "\n…(truncated)"
|
|
358
|
+
: verifyResult.output;
|
|
359
|
+
messages.push({
|
|
360
|
+
role: "user",
|
|
361
|
+
content:
|
|
362
|
+
`AUTO-VERIFY FAILED (${verifyResult.label}):\n\`\`\`\n${errPreview}\n\`\`\`\n` +
|
|
363
|
+
`Fix these errors NOW using edit_file. Do not explain — just fix and move on.`,
|
|
364
|
+
});
|
|
365
|
+
continue;
|
|
366
|
+
}
|
|
367
|
+
if (verifyResult && verifyResult.ok) {
|
|
368
|
+
console.log("\n" + c.green("●") + " " + c.bold("auto-verify") + c.gray(` (${verifyResult.label}) — passed`));
|
|
369
|
+
}
|
|
370
|
+
}
|
|
311
371
|
if (filesTouched.size) {
|
|
312
372
|
console.log("\n" + c.bold("Files") + c.gray(` in ${cwd}`));
|
|
313
373
|
for (const [p, action] of filesTouched) {
|
|
314
374
|
console.log(" " + c.green("●") + " " + p + c.gray(` — ${action}`));
|
|
315
375
|
}
|
|
316
376
|
}
|
|
317
|
-
return { ok: true, totalCredits, totalIn, totalOut, turns: i + 1, balance: lastBalance, messages };
|
|
377
|
+
return { ok: true, totalCredits, totalIn, totalOut, turns: i + 1, balance: lastBalance, messages, elapsed: Date.now() - t0 };
|
|
318
378
|
}
|
|
319
379
|
|
|
320
380
|
// Execute each tool call. Show the actual args (now that we have them
|
|
321
381
|
// fully assembled) and run.
|
|
382
|
+
totalToolCalls += toolCalls.length;
|
|
322
383
|
for (const call of toolCalls) {
|
|
323
384
|
let args = {};
|
|
324
385
|
try { args = JSON.parse(call.function.arguments || "{}"); } catch { /* leave empty */ }
|
|
325
386
|
// todo_write renders its own Plan box, so it returns an empty label.
|
|
326
387
|
const label = toolLabel(call.function.name, args);
|
|
327
|
-
if (label) console.log("\n" + c.
|
|
388
|
+
if (label) console.log("\n" + c.green("●") + " " + label);
|
|
328
389
|
|
|
329
390
|
// Loop guard — short-circuit a tool call we've already run 3+ times with
|
|
330
391
|
// the same args, and tell the model to change approach instead of looping.
|
|
@@ -376,44 +437,133 @@ export async function runAgent({
|
|
|
376
437
|
const summary = toolSummary(call.function.name, result);
|
|
377
438
|
if (summary) console.log(summary);
|
|
378
439
|
|
|
379
|
-
// Cap tool result content sent to the model to prevent it from
|
|
380
|
-
// echoing raw data (snippets, HTML dumps) back as text output.
|
|
381
|
-
let toolContent = result.output ?? (result.ok ? "(no output)" : "Failed.");
|
|
382
|
-
if (call.function.name === "web_search") {
|
|
383
|
-
try {
|
|
384
|
-
const arr = JSON.parse(toolContent);
|
|
385
|
-
if (Array.isArray(arr)) {
|
|
386
|
-
toolContent = JSON.stringify(arr.map((r) => ({
|
|
387
|
-
title: r.title, url: r.url,
|
|
388
|
-
snippet: (r.snippet || "").slice(0, 120),
|
|
389
|
-
})));
|
|
390
|
-
}
|
|
391
|
-
} catch { /* leave as-is */ }
|
|
392
|
-
} else if (call.function.name === "web_fetch") {
|
|
393
|
-
if (toolContent.length > 12000) toolContent = toolContent.slice(0, 12000) + "\n...(truncated)";
|
|
394
|
-
} else if (call.function.name === "read_file") {
|
|
395
|
-
if (toolContent.length > 30000) toolContent = toolContent.slice(0, 30000) + "\n...(truncated)";
|
|
396
|
-
}
|
|
397
440
|
messages.push({
|
|
398
441
|
role: "tool",
|
|
399
442
|
tool_call_id: call.id,
|
|
400
|
-
content:
|
|
443
|
+
content: result.output ?? (result.ok ? "(no output)" : "Failed."),
|
|
401
444
|
});
|
|
402
445
|
}
|
|
446
|
+
}
|
|
447
|
+
|
|
448
|
+
console.log(c.yellow(`\nReached max turns (${maxTurns}). Stopping.`));
|
|
449
|
+
return { ok: false, error: new Error("Max turns reached"), totalCredits, totalIn, totalOut, balance: lastBalance, messages, elapsed: Date.now() - t0 };
|
|
450
|
+
}
|
|
403
451
|
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
452
|
+
// A build/verify command failing because the TOOLCHAIN ITSELF is missing (e.g.
|
|
453
|
+
// g++/cargo/dotnet not installed) is not a code bug — it must NOT be reported as
|
|
454
|
+
// "build failed" and fed back to the model, which would derail the run into
|
|
455
|
+
// 5 rounds of "fixing" code that's actually fine. These are the shell's
|
|
456
|
+
// command-not-found messages (Windows cmd.exe + POSIX sh); real compiler
|
|
457
|
+
// diagnostics look like `main.cpp:5:10: error: ...` and never match these.
|
|
458
|
+
// Matches ONLY the shell's "I can't find this executable" messages — Windows
|
|
459
|
+
// cmd.exe ("'g++' is not recognized…") and POSIX sh ("g++: command not found",
|
|
460
|
+
// "sh: 1: g++: not found"). Real compiler diagnostics (`main.cpp:5: error:`,
|
|
461
|
+
// `fatal error: foo.h: No such file or directory`) never match these, so a
|
|
462
|
+
// genuine build error is still reported and fixed.
|
|
463
|
+
const TOOLCHAIN_MISSING_RE =
|
|
464
|
+
/is not recognized as an internal or external command|: command not found|: not found\b/i;
|
|
465
|
+
|
|
466
|
+
// Auto-verify: run the project's typecheck/lint command silently and return
|
|
467
|
+
// { ok, output, label }. Returns null if no verify command is detected OR the
|
|
468
|
+
// toolchain isn't installed (skip, don't report a false failure).
|
|
469
|
+
let cachedVerify = undefined;
|
|
470
|
+
async function runVerify(cwd) {
|
|
471
|
+
if (cachedVerify === undefined) cachedVerify = detectVerifyCommand(cwd);
|
|
472
|
+
if (!cachedVerify) return null;
|
|
473
|
+
const { cmd, label } = cachedVerify;
|
|
474
|
+
return new Promise((resolve) => {
|
|
475
|
+
const child = spawn(cmd, [], { cwd, shell: true, stdio: ["ignore", "pipe", "pipe"] });
|
|
476
|
+
let stdout = "";
|
|
477
|
+
let stderr = "";
|
|
478
|
+
const timer = setTimeout(() => { child.kill("SIGTERM"); }, 60_000);
|
|
479
|
+
child.stdout.on("data", (d) => { stdout += d.toString(); });
|
|
480
|
+
child.stderr.on("data", (d) => { stderr += d.toString(); });
|
|
481
|
+
child.on("close", (code) => {
|
|
482
|
+
clearTimeout(timer);
|
|
483
|
+
const output = (stdout + "\n" + stderr).trim();
|
|
484
|
+
// Toolchain not installed → skip silently rather than nag the model.
|
|
485
|
+
if (code !== 0 && TOOLCHAIN_MISSING_RE.test(output)) { resolve(null); return; }
|
|
486
|
+
resolve({ ok: code === 0, output, label });
|
|
411
487
|
});
|
|
412
|
-
|
|
488
|
+
child.on("error", () => {
|
|
489
|
+
clearTimeout(timer);
|
|
490
|
+
resolve(null);
|
|
491
|
+
});
|
|
492
|
+
});
|
|
493
|
+
}
|
|
494
|
+
|
|
495
|
+
// Auto-build: detect the build toolchain from written files and compile.
|
|
496
|
+
// Returns { ok, output, label, runCmd } or null if no buildable project detected.
|
|
497
|
+
let cachedBuild = undefined;
|
|
498
|
+
async function runBuild(cwd, touchedFiles) {
|
|
499
|
+
// Re-detect every time because the model may create new files (e.g. Makefile)
|
|
500
|
+
// or change the project structure between rounds.
|
|
501
|
+
const detected = detectBuildCommand(cwd, touchedFiles);
|
|
502
|
+
if (!detected) {
|
|
503
|
+
// Fallback: if we previously detected a build command, reuse it (the model
|
|
504
|
+
// may have only edited files this round, not created new ones).
|
|
505
|
+
if (cachedBuild) return runBuildCmd(cwd, cachedBuild);
|
|
506
|
+
return null;
|
|
413
507
|
}
|
|
508
|
+
cachedBuild = detected;
|
|
509
|
+
return runBuildCmd(cwd, detected);
|
|
510
|
+
}
|
|
414
511
|
|
|
415
|
-
|
|
416
|
-
return
|
|
512
|
+
async function runBuildCmd(cwd, { cmd, label, type, runCmd }) {
|
|
513
|
+
return new Promise((resolve) => {
|
|
514
|
+
const child = spawn(cmd, [], { cwd, shell: true, stdio: ["ignore", "pipe", "pipe"] });
|
|
515
|
+
let stdout = "";
|
|
516
|
+
let stderr = "";
|
|
517
|
+
const timer = setTimeout(() => { child.kill("SIGTERM"); }, 120_000);
|
|
518
|
+
child.stdout.on("data", (d) => { stdout += d.toString(); });
|
|
519
|
+
child.stderr.on("data", (d) => { stderr += d.toString(); });
|
|
520
|
+
child.on("close", (code) => {
|
|
521
|
+
clearTimeout(timer);
|
|
522
|
+
const output = (stdout + "\n" + stderr).trim();
|
|
523
|
+
// Compiler/toolchain not installed → skip auto-build silently instead of
|
|
524
|
+
// reporting a build failure and derailing the run.
|
|
525
|
+
if (code !== 0 && TOOLCHAIN_MISSING_RE.test(output)) { resolve(null); return; }
|
|
526
|
+
resolve({ ok: code === 0, output, label, type, runCmd });
|
|
527
|
+
});
|
|
528
|
+
child.on("error", () => {
|
|
529
|
+
clearTimeout(timer);
|
|
530
|
+
// ENOENT (command not found) → toolchain missing → skip.
|
|
531
|
+
resolve(null);
|
|
532
|
+
});
|
|
533
|
+
});
|
|
534
|
+
}
|
|
535
|
+
|
|
536
|
+
// Run the compiled/interpreted program after a successful build.
|
|
537
|
+
// Returns { ok, output, code, timedOut } or null on spawn error.
|
|
538
|
+
async function runBuiltProgram(cwd, runCmd, label) {
|
|
539
|
+
return new Promise((resolve) => {
|
|
540
|
+
const child = spawn(runCmd, [], { cwd, shell: true, stdio: ["ignore", "pipe", "pipe"] });
|
|
541
|
+
let stdout = "";
|
|
542
|
+
let stderr = "";
|
|
543
|
+
let timedOut = false;
|
|
544
|
+
const timer = setTimeout(() => { timedOut = true; child.kill("SIGTERM"); }, 10_000);
|
|
545
|
+
child.stdout.on("data", (d) => {
|
|
546
|
+
stdout += d.toString();
|
|
547
|
+
if (stdout.length > 50_000) { child.kill("SIGTERM"); }
|
|
548
|
+
});
|
|
549
|
+
child.stderr.on("data", (d) => {
|
|
550
|
+
stderr += d.toString();
|
|
551
|
+
if (stderr.length > 50_000) { child.kill("SIGTERM"); }
|
|
552
|
+
});
|
|
553
|
+
child.on("close", (code) => {
|
|
554
|
+
clearTimeout(timer);
|
|
555
|
+
const output = (stdout + "\n" + stderr).trim();
|
|
556
|
+
if (timedOut) {
|
|
557
|
+
resolve({ ok: true, output, code: null, timedOut: true });
|
|
558
|
+
} else {
|
|
559
|
+
resolve({ ok: code === 0, output, code, timedOut: false });
|
|
560
|
+
}
|
|
561
|
+
});
|
|
562
|
+
child.on("error", (err) => {
|
|
563
|
+
clearTimeout(timer);
|
|
564
|
+
resolve({ ok: false, output: err.message, code: null, timedOut: false });
|
|
565
|
+
});
|
|
566
|
+
});
|
|
417
567
|
}
|
|
418
568
|
|
|
419
569
|
/**
|
|
@@ -430,7 +580,7 @@ export async function runAgent({
|
|
|
430
580
|
* worse than no skills at all. Prepending into the user message keeps
|
|
431
581
|
* both layers active.
|
|
432
582
|
*/
|
|
433
|
-
function buildTurnMessages(messages, allSkills, referencedPaths) {
|
|
583
|
+
export function buildTurnMessages(messages, allSkills, referencedPaths) {
|
|
434
584
|
if (allSkills.length === 0) return messages;
|
|
435
585
|
// Find the latest user message — that's where the current task lives.
|
|
436
586
|
let lastUserIdx = -1;
|
|
@@ -438,30 +588,16 @@ function buildTurnMessages(messages, allSkills, referencedPaths) {
|
|
|
438
588
|
if (messages[i].role === "user") { lastUserIdx = i; break; }
|
|
439
589
|
}
|
|
440
590
|
if (lastUserIdx === -1) return messages;
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
let prompt;
|
|
445
|
-
if (typeof rawContent === "string") {
|
|
446
|
-
prompt = rawContent;
|
|
447
|
-
} else if (Array.isArray(rawContent)) {
|
|
448
|
-
const textPart = rawContent.find((p) => p.type === "text");
|
|
449
|
-
prompt = textPart?.text ?? "";
|
|
450
|
-
} else {
|
|
451
|
-
prompt = "";
|
|
452
|
-
}
|
|
591
|
+
const prompt = typeof messages[lastUserIdx].content === "string"
|
|
592
|
+
? messages[lastUserIdx].content
|
|
593
|
+
: "";
|
|
453
594
|
const active = selectSkills({ skills: allSkills, prompt, referencedPaths });
|
|
454
595
|
if (active.length === 0) return messages;
|
|
455
596
|
const block = renderSkillsBlock(active);
|
|
456
597
|
const cloned = [...messages];
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
}
|
|
461
|
-
const newParts = rawContent.map((p) =>
|
|
462
|
-
p.type === "text" ? { ...p, text: `${block}\n\n---\n\n${p.text}` } : p,
|
|
463
|
-
);
|
|
464
|
-
cloned[lastUserIdx] = { ...cloned[lastUserIdx], content: newParts };
|
|
465
|
-
}
|
|
598
|
+
cloned[lastUserIdx] = {
|
|
599
|
+
...cloned[lastUserIdx],
|
|
600
|
+
content: `${block}\n\n---\n\n${prompt}`,
|
|
601
|
+
};
|
|
466
602
|
return cloned;
|
|
467
603
|
}
|