car-runtime 0.52.0 → 0.53.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +30 -0
- package/agent-loop.d.ts +30 -0
- package/agent-loop.js +578 -0
- package/agent-loop.mjs +4 -0
- package/docs/ASSISTANT.md +1 -0
- package/docs/CLI.md +257 -50
- package/docs/agent-ir-spec.md +40 -2
- package/index.d.ts +267 -11
- package/package.json +20 -2
package/README.md
CHANGED
|
@@ -74,6 +74,36 @@ const result = await executeProposal(rt, proposal, async (callJson) => {
|
|
|
74
74
|
});
|
|
75
75
|
```
|
|
76
76
|
|
|
77
|
+
## Packaged agent loop
|
|
78
|
+
|
|
79
|
+
Do not copy a harness into each agent. Import the versioned loop and keep the
|
|
80
|
+
project entry file declarative:
|
|
81
|
+
|
|
82
|
+
```javascript
|
|
83
|
+
import { main } from 'car-runtime/agent-loop';
|
|
84
|
+
|
|
85
|
+
main({
|
|
86
|
+
agentName: 'Lookup Agent',
|
|
87
|
+
identity: 'Use lookup before answering; never guess.',
|
|
88
|
+
toolSchemas: [{
|
|
89
|
+
name: 'lookup', description: 'Look up one key',
|
|
90
|
+
parameters: { type: 'object', properties: { key: { type: 'string' } }, required: ['key'] },
|
|
91
|
+
}],
|
|
92
|
+
tools: { lookup: async ({ key }) => ({ key, value: await lookup(key) }) },
|
|
93
|
+
policies: [],
|
|
94
|
+
maxTurns: 8,
|
|
95
|
+
});
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
`runAgent(config, goal)` is also exported for embedding and tests. `main(config)`
|
|
99
|
+
provides `--task`, `--json`, and `--serve`. The loop adds `finish`, statically
|
|
100
|
+
verifies every proposal before execution, preserves tool-call/result IDs,
|
|
101
|
+
honors per-schema `timeoutMs`, traces task and chat runs, and uses request-shaped
|
|
102
|
+
inference with prompt-cache breakpoints. It loads the native binding only when a
|
|
103
|
+
loop runs, so `require('car-runtime/agent-loop')` is safe for package discovery.
|
|
104
|
+
Type declarations ship as `agent-loop.d.ts` and shared config/outcome types in
|
|
105
|
+
`index.d.ts`.
|
|
106
|
+
|
|
77
107
|
Full API reference lives in [`index.d.ts`](./index.d.ts). The package also
|
|
78
108
|
ships a `docs/` directory (`node_modules/car-runtime/docs/`) with prose
|
|
79
109
|
reference docs — `SPEC.md`, `GUIDE.md`, `CLI.md`, `ASSISTANT.md`, `MCP.md`,
|
package/agent-loop.d.ts
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
import type {
|
|
2
|
+
AgentLoopConfig,
|
|
3
|
+
AgentLoopOptions,
|
|
4
|
+
AgentOutcome,
|
|
5
|
+
AgentToolSchema,
|
|
6
|
+
AgentTool,
|
|
7
|
+
CarRuntime,
|
|
8
|
+
} from './index';
|
|
9
|
+
|
|
10
|
+
export function runAgent(
|
|
11
|
+
config: AgentLoopConfig,
|
|
12
|
+
goal: string,
|
|
13
|
+
options?: AgentLoopOptions,
|
|
14
|
+
): Promise<AgentOutcome>;
|
|
15
|
+
|
|
16
|
+
export function runChatTurn(
|
|
17
|
+
runtime: CarRuntime,
|
|
18
|
+
config: AgentLoopConfig,
|
|
19
|
+
toolSchemas: AgentToolSchema[],
|
|
20
|
+
tools: Record<string, AgentTool>,
|
|
21
|
+
toolTimeouts: Record<string, number>,
|
|
22
|
+
sessionId: string,
|
|
23
|
+
messages: Record<string, unknown>[],
|
|
24
|
+
requestedModel?: string | null,
|
|
25
|
+
): Promise<string>;
|
|
26
|
+
|
|
27
|
+
export function main(config: AgentLoopConfig): Promise<void>;
|
|
28
|
+
|
|
29
|
+
declare const agentLoop: { main: typeof main; runAgent: typeof runAgent; runChatTurn: typeof runChatTurn };
|
|
30
|
+
export default agentLoop;
|
package/agent-loop.js
ADDED
|
@@ -0,0 +1,578 @@
|
|
|
1
|
+
'use strict';
|
|
2
|
+
|
|
3
|
+
// Versioned generic CAR agent loop. Agent projects import this package export;
|
|
4
|
+
// they do not copy or fork the propose -> verify -> execute -> observe cycle.
|
|
5
|
+
// The native binding is loaded lazily so tooling can inspect/require this
|
|
6
|
+
// subpath without first installing a platform binary.
|
|
7
|
+
let nativeApi = null;
|
|
8
|
+
let cancelHandlerInstalled = false;
|
|
9
|
+
function runtimeApi() {
|
|
10
|
+
if (nativeApi === null) nativeApi = require('./index.js');
|
|
11
|
+
if (!cancelHandlerInstalled && typeof nativeApi.registerToolCancelHandler === 'function') {
|
|
12
|
+
nativeApi.registerToolCancelHandler((requestId) => {
|
|
13
|
+
const ctl = ABORTS.get(requestId);
|
|
14
|
+
if (ctl) {
|
|
15
|
+
ABORTS.delete(requestId);
|
|
16
|
+
try { ctl.abort(new Error('tool callback reaped by daemon (budget exceeded)')); } catch { /* already aborted */ }
|
|
17
|
+
}
|
|
18
|
+
});
|
|
19
|
+
cancelHandlerInstalled = true;
|
|
20
|
+
}
|
|
21
|
+
return nativeApi;
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
// Process-wide abort registry keyed on the daemon's per-call `request_id`
|
|
25
|
+
// (Parslee-ai/car#264). When the daemon reaps a tool callback (the call
|
|
26
|
+
// exceeded its budget) it emits `tools.cancel` with the `request_id`; the
|
|
27
|
+
// handler below aborts the matching controller so a long child (e.g. a
|
|
28
|
+
// `claude -p` / `codex exec` driven by drive_cli) is killed instead of
|
|
29
|
+
// orphaned. Each tool callback registers its controller under its request_id
|
|
30
|
+
// and removes it on completion. Registered once at module load; a daemon
|
|
31
|
+
// without the cancel surface simply never fires it.
|
|
32
|
+
const ABORTS = new Map();
|
|
33
|
+
|
|
34
|
+
// A built-in sentinel tool every agent gets for free, so the loop always has a
|
|
35
|
+
// clean way to terminate with a final answer. Your agent calls finish(answer).
|
|
36
|
+
const FINISH_SCHEMA = {
|
|
37
|
+
name: 'finish',
|
|
38
|
+
description: 'Return the final answer to the user and stop. Call this exactly once, when the task is complete or cannot proceed.',
|
|
39
|
+
parameters: {
|
|
40
|
+
type: 'object',
|
|
41
|
+
properties: { answer: { type: 'string', description: 'The final answer or status for the user.' } },
|
|
42
|
+
required: ['answer'],
|
|
43
|
+
},
|
|
44
|
+
};
|
|
45
|
+
|
|
46
|
+
function buildProposal(modelUsed, toolCalls, toolTimeouts = {}) {
|
|
47
|
+
// Glue between the inference IR (ToolCall {id,name,arguments}) and the
|
|
48
|
+
// action IR (Action {id,type,tool,parameters,timeout_ms}). The only IR
|
|
49
|
+
// plumbing you owe.
|
|
50
|
+
//
|
|
51
|
+
// `timeout_ms` (Parslee-ai/car#259): a tool that shells out to a build,
|
|
52
|
+
// drives another CLI, or calls a slow API needs more than the daemon's
|
|
53
|
+
// default callback budget. Declare `timeoutMs` on the tool's schema and
|
|
54
|
+
// it flows here as the action's per-call budget — without it the call is
|
|
55
|
+
// reaped at the default. Omitted (undefined) when the schema sets none,
|
|
56
|
+
// so the action falls back to the daemon default.
|
|
57
|
+
return JSON.stringify({
|
|
58
|
+
source: modelUsed || 'model',
|
|
59
|
+
actions: toolCalls.map((tc, i) => ({
|
|
60
|
+
id: tc.id || `a${i}`,
|
|
61
|
+
type: 'tool_call',
|
|
62
|
+
tool: tc.name,
|
|
63
|
+
parameters: tc.arguments || {},
|
|
64
|
+
dependencies: [],
|
|
65
|
+
timeout_ms: toolTimeouts[tc.name],
|
|
66
|
+
})),
|
|
67
|
+
});
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* Build the `executeProposal` tool callback. ONE implementation, used by BOTH
|
|
72
|
+
* the one-shot `--task` loop (`runAgent`) and the `--serve` chat path
|
|
73
|
+
* (`runChatTurn`). They used to carry two copies, and the chat copy silently
|
|
74
|
+
* missed `timeout_ms` and the `request_id`/AbortController wiring — the path a
|
|
75
|
+
* registered CarHost agent actually runs. Do not re-fork it.
|
|
76
|
+
*
|
|
77
|
+
* The callback receives { tool, params, action_id, request_id, timeout_ms }.
|
|
78
|
+
* The `request_id` keys this call's abort controller (Parslee-ai/car#264) so
|
|
79
|
+
* the daemon's `tools.cancel` can kill a reaped child. The tool fn receives the
|
|
80
|
+
* budget (timeoutMs) and an AbortSignal as a second arg; tools that shell out
|
|
81
|
+
* should honor the signal (e.g. pass it to child_process / fetch) so a reap
|
|
82
|
+
* actually terminates their child. The callback's `timeout_ms` is authoritative
|
|
83
|
+
* — it is the budget of the action actually executing (Parslee-ai/car#259) — so
|
|
84
|
+
* it wins over the locally derived `toolTimeouts` map, which is the fallback.
|
|
85
|
+
* `onFinish` (optional) is called with the `finish` tool's answer.
|
|
86
|
+
*/
|
|
87
|
+
function makeToolCallback(tools, toolTimeouts = {}, onFinish = null) {
|
|
88
|
+
return async (callJson) => {
|
|
89
|
+
const { tool, params, request_id: requestId, timeout_ms: timeoutMs } = JSON.parse(callJson);
|
|
90
|
+
const fn = tools[tool];
|
|
91
|
+
if (!fn) throw new Error(`unknown tool: ${tool}`);
|
|
92
|
+
const ctl = new AbortController();
|
|
93
|
+
if (requestId) ABORTS.set(requestId, ctl);
|
|
94
|
+
try {
|
|
95
|
+
const out = await fn(params || {}, { signal: ctl.signal, timeoutMs: timeoutMs ?? toolTimeouts[tool] });
|
|
96
|
+
if (tool === 'finish' && onFinish) onFinish((out && out.answer) ?? '');
|
|
97
|
+
return JSON.stringify(out ?? {});
|
|
98
|
+
} finally {
|
|
99
|
+
// In a `finally` so a throwing tool can never leak its entry.
|
|
100
|
+
if (requestId) ABORTS.delete(requestId);
|
|
101
|
+
}
|
|
102
|
+
};
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
// ---- Prompt caching (Anthropic) --------------------------------------------
|
|
106
|
+
//
|
|
107
|
+
// The loop resends the WHOLE growing thread every turn, so an N-turn run bills
|
|
108
|
+
// the stable prefix N times at the full input rate — cost that grows with the
|
|
109
|
+
// square of the turn count. CAR's protocol layer already knows how to mark
|
|
110
|
+
// Anthropic cache breakpoints; it only needs `cache_control: true`, and the
|
|
111
|
+
// 9-positional-argument `inferTracked` has no slot to carry it. That is the
|
|
112
|
+
// whole reason every caller has been paying full price: the flag exists, the
|
|
113
|
+
// call shape could not reach it. `inferTrackedWithRequest` takes the options
|
|
114
|
+
// object (a JSON `GenerateRequest`) and can.
|
|
115
|
+
//
|
|
116
|
+
// BREAKPOINT PLACEMENT — we do not pick it; CAR does, and its choice is already
|
|
117
|
+
// the max-reuse one (car-inference/src/protocol.rs, AnthropicHandler::
|
|
118
|
+
// build_request_body). With `cache_control: true` it emits three of Anthropic's
|
|
119
|
+
// four allowed breakpoints:
|
|
120
|
+
// 1. the system block — identity plus the finish instruction, byte-identical
|
|
121
|
+
// for the entire run;
|
|
122
|
+
// 2. the LAST tool definition — so identity + every tool schema forms ONE
|
|
123
|
+
// cached prefix, the largest genuinely stable block this loop has;
|
|
124
|
+
// 3. the LAST message — the moving breakpoint. Anthropic serves the longest
|
|
125
|
+
// cached prefix it can find, so each turn writes only its own delta and
|
|
126
|
+
// READS everything the previous turn wrote. That is what turns quadratic
|
|
127
|
+
// re-billing into one write plus N cheap reads.
|
|
128
|
+
// The 4th breakpoint is deliberately left unused. It would buy a
|
|
129
|
+
// `context_stable_prefix` split of the system prompt, which helps only when the
|
|
130
|
+
// system prompt has a volatile tail; ours has none, so splitting it would
|
|
131
|
+
// shrink the cached block rather than grow it.
|
|
132
|
+
//
|
|
133
|
+
// TTL — `one_hour`, not the 5-minute default. Turns in an agentic loop are
|
|
134
|
+
// separated by real tool execution (browser drives, FMS reads, verification),
|
|
135
|
+
// which routinely exceeds five minutes; a 5-minute entry would expire mid-run
|
|
136
|
+
// and re-bill the entire prefix as a fresh write. A 1h write costs ~2x base
|
|
137
|
+
// input against ~1.25x for 5m, but that one-time 0.75x is far cheaper than a
|
|
138
|
+
// single full re-write of the prefix, and every surviving turn then reads at
|
|
139
|
+
// ~0.1x.
|
|
140
|
+
//
|
|
141
|
+
/** One request-shaped, cache-aware inference path for task and chat loops. */
|
|
142
|
+
async function inferTurn(rt, { model, maxTokens, toolSchemas, messages, toolChoice = 'auto' }) {
|
|
143
|
+
return JSON.parse(await rt.inferTrackedWithRequest(JSON.stringify({
|
|
144
|
+
prompt: '',
|
|
145
|
+
model: model ?? null,
|
|
146
|
+
params: {
|
|
147
|
+
max_tokens: maxTokens,
|
|
148
|
+
tool_choice: toolChoice,
|
|
149
|
+
strict_model: model != null,
|
|
150
|
+
cache_ttl: 'one_hour',
|
|
151
|
+
},
|
|
152
|
+
tools: toolSchemas,
|
|
153
|
+
messages,
|
|
154
|
+
cache_control: true,
|
|
155
|
+
})));
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
/**
|
|
159
|
+
* Run the agent once toward `goal`. Returns an AgentOutcome-shaped object:
|
|
160
|
+
* { status, summary, evidence[], metrics{}, tools_called[] }
|
|
161
|
+
* `status` is one of the six OutcomeStatus values (success|partial_success|
|
|
162
|
+
* done|give_up|timeout|failure). AgentOutcome is caller-built — CAR does not
|
|
163
|
+
* return it; we assemble it from what the loop observed.
|
|
164
|
+
*
|
|
165
|
+
* Run-trace lifecycle: each `runAgent` invocation is one run. Before the first
|
|
166
|
+
* proposal we bracket the run open with `rt.runsStart` (daemon mints a durable
|
|
167
|
+
* `run_id`); after the terminal AgentOutcome is assembled we close it with
|
|
168
|
+
* `rt.runsComplete`. Both are best-effort and never print, so the daemon traces
|
|
169
|
+
* the run for CarHost while older daemons (no runs.*) behave exactly as before.
|
|
170
|
+
*/
|
|
171
|
+
async function runAgent(config, goal, { maxTurns } = {}) {
|
|
172
|
+
const { CarRuntime, executeProposal } = runtimeApi();
|
|
173
|
+
const turnsCap = maxTurns ?? config.maxTurns ?? 8;
|
|
174
|
+
const toolSchemas = [...(config.toolSchemas || []), FINISH_SCHEMA];
|
|
175
|
+
const tools = { finish: ({ answer }) => ({ answer }), ...(config.tools || {}) };
|
|
176
|
+
// Per-tool execution budget (Parslee-ai/car#259): a tool schema may set
|
|
177
|
+
// `timeoutMs` (e.g. a CLI driver that runs for 180s); it flows onto each
|
|
178
|
+
// action so the daemon honors it instead of reaping at the default.
|
|
179
|
+
const toolTimeouts = Object.fromEntries(
|
|
180
|
+
toolSchemas.filter((s) => s && s.timeoutMs != null).map((s) => [s.name, s.timeoutMs]),
|
|
181
|
+
);
|
|
182
|
+
|
|
183
|
+
const rt = new CarRuntime();
|
|
184
|
+
// Best-effort: some daemon versions don't expose agents.register_basics. It
|
|
185
|
+
// only adds CAR's built-in utility tools, which a tool-declaring agent doesn't
|
|
186
|
+
// depend on, so a missing method must not abort the run.
|
|
187
|
+
try { await rt.registerAgentBasics(); } catch { /* unsupported on this daemon — fine */ }
|
|
188
|
+
for (const s of toolSchemas) await rt.registerTool(s.name);
|
|
189
|
+
// Guardrails are declarative policies, enforced in Rust BEFORE the tool fires
|
|
190
|
+
// — not prompt rules. Each entry is the argument list for registerPolicy.
|
|
191
|
+
for (const p of (config.policies || [])) await rt.registerPolicy(...p);
|
|
192
|
+
|
|
193
|
+
// Run-trace bracket (open). Tell the daemon a run is starting so CarHost can
|
|
194
|
+
// trace it: prompt -> CLI outcome -> verifier verdict -> AgentOutcome. The
|
|
195
|
+
// daemon mints a durable run_id and tags it as this session's current run
|
|
196
|
+
// BEFORE replying, so the per-turn recorder reads the right id; we await that
|
|
197
|
+
// ack before submitting any proposal. The owning agent_id resolves from
|
|
198
|
+
// CAR_AGENT_ID (the supervisor injects it) when supervised, else falls back to
|
|
199
|
+
// config.agentName for the unsupervised one-shot / run_scenarios path.
|
|
200
|
+
// Best-effort, exactly like registerAgentBasics: a daemon without runs.* (an
|
|
201
|
+
// older build) makes this throw, and the run must continue unchanged. NEVER
|
|
202
|
+
// print here — the last stdout line in --json mode must stay the AgentOutcome
|
|
203
|
+
// (run_scenarios.py parses it).
|
|
204
|
+
let runId = null;
|
|
205
|
+
try {
|
|
206
|
+
const started = JSON.parse(await rt.runsStart(JSON.stringify({
|
|
207
|
+
intent: goal,
|
|
208
|
+
agent_id: process.env.CAR_AGENT_ID || config.agentName,
|
|
209
|
+
agent_name: config.agentName,
|
|
210
|
+
outcome_description: config.targetOutcome ?? '',
|
|
211
|
+
})));
|
|
212
|
+
runId = started.run_id ?? null;
|
|
213
|
+
} catch { /* daemon lacks runs.* — degrade to untraced behavior */ }
|
|
214
|
+
|
|
215
|
+
// Local Qwen3 models default to "thinking" (they emit a long <think> block
|
|
216
|
+
// before acting), which on a multi-tool agent prompt can exceed the daemon's
|
|
217
|
+
// per-call read timeout — the agent loop then fails with a timeout instead of
|
|
218
|
+
// calling a tool. When pinned to a local model, append Qwen3's `/no_think`
|
|
219
|
+
// soft switch so it acts via tools directly. Only applied for clearly-local
|
|
220
|
+
// model ids (a null/router model may resolve to a cloud model, where thinking
|
|
221
|
+
// is fine and fast); cloud models simply ignore the token.
|
|
222
|
+
const isLocalModel = typeof config.defaultModel === 'string'
|
|
223
|
+
&& /^(mlx|qwen)\//.test(config.defaultModel);
|
|
224
|
+
const noThink = isLocalModel ? '\n\n/no_think' : '';
|
|
225
|
+
const messages = [
|
|
226
|
+
{ role: 'system', content: `${config.identity}\n\nWhen the task is complete or you cannot proceed, call the \`finish\` tool with a concise answer. Do not narrate; act via tools.${noThink}` },
|
|
227
|
+
{ role: 'user', content: goal },
|
|
228
|
+
];
|
|
229
|
+
|
|
230
|
+
const metrics = { turns: 0, tool_calls: 0, actions_succeeded: 0, actions_failed: 0 };
|
|
231
|
+
const toolsCalled = new Set();
|
|
232
|
+
let outcome = null; // null === no terminal AgentOutcome yet
|
|
233
|
+
let turns = 0;
|
|
234
|
+
|
|
235
|
+
while (outcome === null) {
|
|
236
|
+
if (++turns > turnsCap) {
|
|
237
|
+
outcome = mkOutcome('timeout', `hit ${turnsCap}-turn cap without finishing`,
|
|
238
|
+
[{ kind: 'stop_reason', description: 'max turns', data: { turnsCap } }], metrics, toolsCalled);
|
|
239
|
+
break;
|
|
240
|
+
}
|
|
241
|
+
metrics.turns = turns;
|
|
242
|
+
|
|
243
|
+
// 1. PROPOSE — multi-turn, tool-aware inference (NOT plain `infer`).
|
|
244
|
+
let tracked;
|
|
245
|
+
try {
|
|
246
|
+
tracked = await inferTurn(rt, {
|
|
247
|
+
model: config.defaultModel ?? null,
|
|
248
|
+
maxTokens: config.maxTokens ?? 1024,
|
|
249
|
+
toolSchemas,
|
|
250
|
+
messages,
|
|
251
|
+
});
|
|
252
|
+
} catch (e) {
|
|
253
|
+
outcome = mkOutcome('failure', `inference failed: ${e.message || e}`,
|
|
254
|
+
[{ kind: 'stop_reason', description: 'infer_tracked error', data: null }], metrics, toolsCalled);
|
|
255
|
+
break;
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
const calls = tracked.tool_calls || [];
|
|
259
|
+
if (calls.length === 0) {
|
|
260
|
+
// Model answered in prose with no tool call — treat as a neutral Done.
|
|
261
|
+
outcome = mkOutcome('done', tracked.text || 'no further actions',
|
|
262
|
+
[{ kind: 'self_assessment', description: tracked.text || '', data: null }], metrics, toolsCalled);
|
|
263
|
+
break;
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
// Normalize ids so assistant tool_calls and tool_results correlate across
|
|
267
|
+
// turns (local models often omit ids).
|
|
268
|
+
calls.forEach((c, i) => { c.id = c.id || `a${i}`; });
|
|
269
|
+
messages.push({ role: 'assistant', content: tracked.text || '', tool_calls: calls });
|
|
270
|
+
const idToTool = Object.fromEntries(calls.map((c) => [c.id, c.name]));
|
|
271
|
+
|
|
272
|
+
const proposal = buildProposal(tracked.model_used, calls, toolTimeouts);
|
|
273
|
+
|
|
274
|
+
// 2. VERIFY — static gate. Never execute an unverified proposal.
|
|
275
|
+
let check;
|
|
276
|
+
try {
|
|
277
|
+
check = JSON.parse(await rt.verifyProposal(proposal));
|
|
278
|
+
} catch (e) {
|
|
279
|
+
outcome = mkOutcome('failure', `verification failed: ${e.message || e}`,
|
|
280
|
+
[{ kind: 'stop_reason', description: 'verifyProposal error', data: null }], metrics, toolsCalled);
|
|
281
|
+
break;
|
|
282
|
+
}
|
|
283
|
+
if (!check.valid) {
|
|
284
|
+
// Feed the rejection back as tool_results so tool_use/tool_result stay
|
|
285
|
+
// paired, then let the model repair on the next turn.
|
|
286
|
+
for (const c of calls) {
|
|
287
|
+
messages.push({ role: 'tool_result', tool_use_id: c.id,
|
|
288
|
+
content: JSON.stringify({ error: `runtime rejected proposal: ${JSON.stringify(check.issues)}` }) });
|
|
289
|
+
}
|
|
290
|
+
continue;
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
// 3. EXECUTE — CAR owns the DAG, retries, timeouts, rollback. The callback
|
|
294
|
+
// receives { tool, params } (note: `params`).
|
|
295
|
+
let finishAnswer = null;
|
|
296
|
+
let result;
|
|
297
|
+
try {
|
|
298
|
+
result = JSON.parse(await executeProposal(rt, proposal,
|
|
299
|
+
makeToolCallback(tools, toolTimeouts, (a) => { finishAnswer = a; })));
|
|
300
|
+
} catch (e) {
|
|
301
|
+
outcome = mkOutcome('failure', `execution failed: ${e.message || e}`,
|
|
302
|
+
[{ kind: 'stop_reason', description: 'executeProposal error', data: null }], metrics, toolsCalled);
|
|
303
|
+
break;
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
// 4. OBSERVE — feed each ActionResult back as a tool_result turn.
|
|
307
|
+
// tools_called records tools that SUCCESSFULLY executed, so a policy-denied
|
|
308
|
+
// /failed tool is correctly absent (guardrail scenarios can assert
|
|
309
|
+
// tool_not_called against it).
|
|
310
|
+
for (const r of (result.results || [])) {
|
|
311
|
+
const ok = r.status === 'succeeded';
|
|
312
|
+
if (ok && idToTool[r.action_id]) toolsCalled.add(idToTool[r.action_id]);
|
|
313
|
+
messages.push({ role: 'tool_result', tool_use_id: r.action_id,
|
|
314
|
+
content: JSON.stringify(ok ? (r.output ?? {}) : { error: r.error }) });
|
|
315
|
+
metrics.tool_calls += 1;
|
|
316
|
+
if (ok) metrics.actions_succeeded += 1; else metrics.actions_failed += 1;
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
// 5. Terminal? finish() succeeded -> success.
|
|
320
|
+
if (finishAnswer !== null) {
|
|
321
|
+
outcome = mkOutcome('success', finishAnswer,
|
|
322
|
+
[{ kind: 'tool_result', description: 'finish called', data: { answer: finishAnswer } }], metrics, toolsCalled);
|
|
323
|
+
}
|
|
324
|
+
// else: loop for the next proposal.
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
// Run-trace bracket (close). Report the terminal AgentOutcome to the daemon so
|
|
328
|
+
// CarHost shows the run's final status and stops streaming it. Await the ack
|
|
329
|
+
// before returning (the connection may close right after) so a healthy run is
|
|
330
|
+
// never raced into `Incomplete`. Best-effort + never prints, mirroring the
|
|
331
|
+
// open bracket: an older daemon without runs.* (or one that never acked the
|
|
332
|
+
// start, leaving runId null) just skips this and behaves as before.
|
|
333
|
+
if (runId !== null) {
|
|
334
|
+
try {
|
|
335
|
+
await rt.runsComplete(JSON.stringify({ run_id: runId, outcome }));
|
|
336
|
+
} catch { /* daemon lacks runs.* — nothing to report to */ }
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
return outcome;
|
|
340
|
+
}
|
|
341
|
+
|
|
342
|
+
function mkOutcome(status, summary, evidence, metrics, toolsCalled) {
|
|
343
|
+
return { status, summary, evidence, metrics, tools_called: [...toolsCalled].sort(), timestamp: new Date().toISOString() };
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
// ---- CLI entrypoint -------------------------------------------------------
|
|
347
|
+
//
|
|
348
|
+
// node agent.mjs --task "<goal>" [--json] one-shot; prints outcome
|
|
349
|
+
// node agent.mjs --serve supervised/standing mode
|
|
350
|
+
//
|
|
351
|
+
// run_scenarios.py invokes the --task --json form.
|
|
352
|
+
|
|
353
|
+
function parseArgs(argv) {
|
|
354
|
+
const a = { task: null, json: false, serve: false };
|
|
355
|
+
for (let i = 0; i < argv.length; i++) {
|
|
356
|
+
if (argv[i] === '--task') a.task = argv[++i];
|
|
357
|
+
else if (argv[i] === '--json') a.json = true;
|
|
358
|
+
else if (argv[i] === '--serve') a.serve = true;
|
|
359
|
+
}
|
|
360
|
+
return a;
|
|
361
|
+
}
|
|
362
|
+
|
|
363
|
+
// ---- Chat serving (agent.chat surface) --------------------------------------
|
|
364
|
+
//
|
|
365
|
+
// In `--serve` mode the agent attaches to the daemon (the binding sends
|
|
366
|
+
// CAR_AGENT_ID + CAR_AGENT_TOKEN on session.auth) and registers an `agent.chat`
|
|
367
|
+
// handler. The daemon reverse-calls `agent.chat { session_id, prompt }` for
|
|
368
|
+
// every host `agents.chat`; we keep a per-session message THREAD and run the
|
|
369
|
+
// same propose→verify→execute loop per turn, streaming the reply back via
|
|
370
|
+
// `agent.chat.event`. Threads are ephemeral (process lifetime). The agent's
|
|
371
|
+
// declared policies still gate tool execution, so guardrails (draft-only, etc.)
|
|
372
|
+
// carry into the conversation.
|
|
373
|
+
|
|
374
|
+
/** Run one chat turn against a persistent `messages` thread; stream via chatEvent. */
|
|
375
|
+
async function runChatTurn(
|
|
376
|
+
rt, config, toolSchemas, tools, toolTimeouts, sessionId, messages, requestedModel = null,
|
|
377
|
+
) {
|
|
378
|
+
const { executeProposal } = runtimeApi();
|
|
379
|
+
const turnsCap = config.maxTurns ?? 8;
|
|
380
|
+
const selectedModel = typeof requestedModel === 'string' && requestedModel.trim()
|
|
381
|
+
? requestedModel
|
|
382
|
+
: null;
|
|
383
|
+
|
|
384
|
+
// Run-trace bracket (open) — the SAME bracket `runAgent` opens, on the chat
|
|
385
|
+
// path. Without it a chat-driven turn does real tool work that never appears
|
|
386
|
+
// in `runs.list` / `runs.get_trace`, so a host that dispatches through
|
|
387
|
+
// `agents.chat` has no daemon-side record of what it ran: CarHost shows the
|
|
388
|
+
// agent as merely "running", and an outer loop cannot read back the tool
|
|
389
|
+
// returns or the terminal outcome. Same best-effort contract as `runAgent`'s
|
|
390
|
+
// — never throws, never prints (the last stdout line in --json mode must stay
|
|
391
|
+
// the AgentOutcome), and an older daemon without runs.* behaves exactly as it
|
|
392
|
+
// did before.
|
|
393
|
+
const intent = [...messages].reverse().find((m) => m.role === 'user')?.content ?? 'chat turn';
|
|
394
|
+
let runId = null;
|
|
395
|
+
try {
|
|
396
|
+
const started = JSON.parse(await rt.runsStart(JSON.stringify({
|
|
397
|
+
intent,
|
|
398
|
+
agent_id: process.env.CAR_AGENT_ID || config.agentName,
|
|
399
|
+
agent_name: config.agentName,
|
|
400
|
+
outcome_description: config.targetOutcome ?? '',
|
|
401
|
+
})));
|
|
402
|
+
runId = started.run_id ?? null;
|
|
403
|
+
} catch { /* daemon lacks runs.* — degrade to untraced behavior */ }
|
|
404
|
+
|
|
405
|
+
const metrics = { turns: 0, tool_calls: 0, actions_succeeded: 0, actions_failed: 0 };
|
|
406
|
+
const toolsCalled = new Set();
|
|
407
|
+
let terminal = null;
|
|
408
|
+
let finalText = '';
|
|
409
|
+
for (let turns = 0; turns < turnsCap; turns++) {
|
|
410
|
+
metrics.turns = turns + 1;
|
|
411
|
+
// Same cached request form as the one-shot loop. A chat thread grows for
|
|
412
|
+
// the whole session, so it is the path that benefits most from the moving
|
|
413
|
+
// conversation breakpoint.
|
|
414
|
+
const tracked = await inferTurn(rt, {
|
|
415
|
+
model: selectedModel ?? config.defaultModel ?? null,
|
|
416
|
+
maxTokens: config.maxTokens ?? 1024,
|
|
417
|
+
toolSchemas,
|
|
418
|
+
messages,
|
|
419
|
+
});
|
|
420
|
+
const calls = tracked.tool_calls || [];
|
|
421
|
+
if (calls.length === 0) {
|
|
422
|
+
finalText = tracked.text || '';
|
|
423
|
+
messages.push({ role: 'assistant', content: finalText });
|
|
424
|
+
terminal = 'done';
|
|
425
|
+
break;
|
|
426
|
+
}
|
|
427
|
+
calls.forEach((c, i) => { c.id = c.id || `a${i}`; });
|
|
428
|
+
messages.push({ role: 'assistant', content: tracked.text || '', tool_calls: calls });
|
|
429
|
+
const idToTool = Object.fromEntries(calls.map((c) => [c.id, c.name]));
|
|
430
|
+
// Surface non-finish tool calls as progress so the host UI can show them.
|
|
431
|
+
for (const c of calls) {
|
|
432
|
+
if (c.name !== 'finish') await rt.chatEvent(sessionId, 'tool_call', c.name).catch(() => {});
|
|
433
|
+
}
|
|
434
|
+
const proposal = buildProposal(tracked.model_used, calls, toolTimeouts);
|
|
435
|
+
const check = JSON.parse(await rt.verifyProposal(proposal));
|
|
436
|
+
if (!check.valid) {
|
|
437
|
+
for (const c of calls) {
|
|
438
|
+
messages.push({ role: 'tool_result', tool_use_id: c.id,
|
|
439
|
+
content: JSON.stringify({ error: `runtime rejected proposal: ${JSON.stringify(check.issues)}` }) });
|
|
440
|
+
}
|
|
441
|
+
continue;
|
|
442
|
+
}
|
|
443
|
+
// Same callback the one-shot loop uses, so the declared per-action budget
|
|
444
|
+
// and the request_id/AbortController cancel wiring reach tools on the
|
|
445
|
+
// --serve path too.
|
|
446
|
+
let finishAnswer = null;
|
|
447
|
+
const result = JSON.parse(await executeProposal(rt, proposal,
|
|
448
|
+
makeToolCallback(tools, toolTimeouts, (a) => { finishAnswer = a; })));
|
|
449
|
+
for (const r of (result.results || [])) {
|
|
450
|
+
const ok = r.status === 'succeeded';
|
|
451
|
+
if (ok && idToTool[r.action_id]) toolsCalled.add(idToTool[r.action_id]);
|
|
452
|
+
messages.push({ role: 'tool_result', tool_use_id: r.action_id,
|
|
453
|
+
content: JSON.stringify(ok ? (r.output ?? {}) : { error: r.error }) });
|
|
454
|
+
metrics.tool_calls += 1;
|
|
455
|
+
if (ok) metrics.actions_succeeded += 1; else metrics.actions_failed += 1;
|
|
456
|
+
}
|
|
457
|
+
if (finishAnswer !== null) { finalText = finishAnswer; terminal = 'success'; break; }
|
|
458
|
+
}
|
|
459
|
+
if (finalText) await rt.chatEvent(sessionId, 'token', finalText).catch(() => {});
|
|
460
|
+
await rt.chatEvent(sessionId, 'done', finalText).catch(() => {});
|
|
461
|
+
|
|
462
|
+
// Run-trace bracket (close). The outcome is assembled exactly as `runAgent`
|
|
463
|
+
// assembles it, so a chat-driven run and a --task run are the same shape in
|
|
464
|
+
// the trace and a host reads one code path, not two.
|
|
465
|
+
if (runId !== null) {
|
|
466
|
+
const outcome = terminal === 'success'
|
|
467
|
+
? mkOutcome('success', finalText,
|
|
468
|
+
[{ kind: 'tool_result', description: 'finish called', data: { answer: finalText } }], metrics, toolsCalled)
|
|
469
|
+
: (terminal === 'done'
|
|
470
|
+
? mkOutcome('done', finalText || 'no further actions',
|
|
471
|
+
[{ kind: 'self_assessment', description: finalText || '', data: null }], metrics, toolsCalled)
|
|
472
|
+
: mkOutcome('timeout', `hit ${turnsCap}-turn cap without finishing`,
|
|
473
|
+
[{ kind: 'stop_reason', description: 'max turns', data: { turnsCap } }], metrics, toolsCalled));
|
|
474
|
+
try {
|
|
475
|
+
await rt.runsComplete(JSON.stringify({ run_id: runId, outcome }));
|
|
476
|
+
} catch { /* daemon lacks runs.* — nothing to report to */ }
|
|
477
|
+
}
|
|
478
|
+
return finalText;
|
|
479
|
+
}
|
|
480
|
+
|
|
481
|
+
/** Set up the long-lived chat runtime: register tools/policies + the agent.chat handler. */
|
|
482
|
+
async function serveChat(config) {
|
|
483
|
+
const { CarRuntime, registerChatHandler } = runtimeApi();
|
|
484
|
+
const rt = new CarRuntime();
|
|
485
|
+
const toolSchemas = [...(config.toolSchemas || []), FINISH_SCHEMA];
|
|
486
|
+
const tools = { finish: ({ answer }) => ({ answer }), ...(config.tools || {}) };
|
|
487
|
+
const toolTimeouts = Object.fromEntries(
|
|
488
|
+
toolSchemas.filter((s) => s && s.timeoutMs != null).map((s) => [s.name, s.timeoutMs]),
|
|
489
|
+
);
|
|
490
|
+
try { await rt.registerAgentBasics(); } catch { /* unsupported — fine */ }
|
|
491
|
+
for (const s of toolSchemas) await rt.registerTool(s.name);
|
|
492
|
+
for (const p of (config.policies || [])) await rt.registerPolicy(...p);
|
|
493
|
+
|
|
494
|
+
if (typeof registerChatHandler !== 'function') {
|
|
495
|
+
console.error(`[${config.agentName}] car-runtime has no agent.chat support — chat disabled (update car-runtime).`);
|
|
496
|
+
return rt;
|
|
497
|
+
}
|
|
498
|
+
|
|
499
|
+
const threads = new Map(); // session_id -> messages[]
|
|
500
|
+
const isLocalModel = typeof config.defaultModel === 'string' && /^(mlx|qwen)\//.test(config.defaultModel);
|
|
501
|
+
const noThink = isLocalModel ? '\n\n/no_think' : '';
|
|
502
|
+
|
|
503
|
+
registerChatHandler((paramsJson) => {
|
|
504
|
+
// Fire-and-forget — the daemon already got its {accepted:true} ack. Run the
|
|
505
|
+
// turn on its own microtask and stream results back via chatEvent.
|
|
506
|
+
let params;
|
|
507
|
+
try { params = JSON.parse(paramsJson); } catch { return; }
|
|
508
|
+
const sessionId = params.session_id;
|
|
509
|
+
if (!sessionId) return;
|
|
510
|
+
let messages = threads.get(sessionId);
|
|
511
|
+
if (!messages) {
|
|
512
|
+
messages = [{ role: 'system', content: `${config.identity}\n\nWhen the task is complete or you cannot proceed, call the \`finish\` tool with a concise answer. Do not narrate; act via tools.${noThink}` }];
|
|
513
|
+
threads.set(sessionId, messages);
|
|
514
|
+
}
|
|
515
|
+
messages.push({ role: 'user', content: params.prompt ?? '' });
|
|
516
|
+
runChatTurn(
|
|
517
|
+
rt, config, toolSchemas, tools, toolTimeouts, sessionId, messages, params.model,
|
|
518
|
+
)
|
|
519
|
+
.catch((e) => rt.chatEvent(sessionId, 'error', String(e && e.message || e)).catch(() => {}));
|
|
520
|
+
});
|
|
521
|
+
console.error(`[${config.agentName}] chat ready (agent.chat) — drive via CarHost or an agents.chat host client`);
|
|
522
|
+
return rt;
|
|
523
|
+
}
|
|
524
|
+
|
|
525
|
+
async function main(config) {
|
|
526
|
+
const args = parseArgs(process.argv.slice(2));
|
|
527
|
+
|
|
528
|
+
if (args.serve) {
|
|
529
|
+
// Standing mode keeps the supervised process alive so CarHost shows it
|
|
530
|
+
// "running". Always serve chat (agent.chat); if the agent has a standing
|
|
531
|
+
// goal + interval, ALSO run it on a loop.
|
|
532
|
+
// car_register.py can set these via CAR_STANDING_GOAL / CAR_INTERVAL_SECS.
|
|
533
|
+
try { await serveChat(config); } catch (e) { console.error(`[${config.agentName}] chat setup failed:`, e); }
|
|
534
|
+
const goal = config.standingGoal ?? process.env.CAR_STANDING_GOAL ?? null;
|
|
535
|
+
const everyMs = (config.intervalSecs ?? Number(process.env.CAR_INTERVAL_SECS || 0)) * 1000;
|
|
536
|
+
console.error(`[${config.agentName}] serving${goal ? ` — "${goal}" every ${everyMs / 1000}s` : ' (idle; start via dashboard or --task)'}`);
|
|
537
|
+
if (goal && everyMs > 0) {
|
|
538
|
+
for (;;) {
|
|
539
|
+
try {
|
|
540
|
+
// Each iteration is its own run: runAgent opens and closes one
|
|
541
|
+
// runs.start/runs.complete bracket on its own fresh CarRuntime.
|
|
542
|
+
const o = await runAgent(config, goal);
|
|
543
|
+
console.error(`[${config.agentName}] ${o.status} — ${o.summary}`);
|
|
544
|
+
} catch (e) { console.error(`[${config.agentName}] error:`, e); }
|
|
545
|
+
await new Promise((r) => setTimeout(r, everyMs));
|
|
546
|
+
}
|
|
547
|
+
} else {
|
|
548
|
+
// Keep the event loop alive. A pending promise alone does NOT keep Node
|
|
549
|
+
// running — it exits when the loop is empty — so use a no-op heartbeat.
|
|
550
|
+
setInterval(() => {}, 1 << 30);
|
|
551
|
+
}
|
|
552
|
+
return;
|
|
553
|
+
}
|
|
554
|
+
|
|
555
|
+
if (!args.task) {
|
|
556
|
+
console.error('usage: node agent.mjs --task "<goal>" [--json] | --serve');
|
|
557
|
+
process.exit(2);
|
|
558
|
+
}
|
|
559
|
+
|
|
560
|
+
let outcome;
|
|
561
|
+
try {
|
|
562
|
+
outcome = await runAgent(config, args.task);
|
|
563
|
+
} catch (e) {
|
|
564
|
+
outcome = { status: 'failure', summary: String(e && e.message || e),
|
|
565
|
+
evidence: [{ kind: 'stop_reason', description: 'uncaught', data: null }],
|
|
566
|
+
metrics: { turns: 0, tool_calls: 0, actions_succeeded: 0, actions_failed: 0 }, tools_called: [] };
|
|
567
|
+
}
|
|
568
|
+
|
|
569
|
+
if (args.json) {
|
|
570
|
+
// The LAST stdout line is the machine-readable outcome (run_scenarios reads it).
|
|
571
|
+
console.log(JSON.stringify(outcome));
|
|
572
|
+
} else {
|
|
573
|
+
console.log(`${outcome.status} — ${outcome.summary}`);
|
|
574
|
+
}
|
|
575
|
+
process.exit(outcome.status === 'failure' ? 1 : 0);
|
|
576
|
+
}
|
|
577
|
+
|
|
578
|
+
module.exports = { main, runAgent, runChatTurn };
|
package/agent-loop.mjs
ADDED