car-runtime 0.52.0 → 0.53.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -74,6 +74,36 @@ const result = await executeProposal(rt, proposal, async (callJson) => {
74
74
  });
75
75
  ```
76
76
 
77
+ ## Packaged agent loop
78
+
79
+ Do not copy a harness into each agent. Import the versioned loop and keep the
80
+ project entry file declarative:
81
+
82
+ ```javascript
83
+ import { main } from 'car-runtime/agent-loop';
84
+
85
+ main({
86
+ agentName: 'Lookup Agent',
87
+ identity: 'Use lookup before answering; never guess.',
88
+ toolSchemas: [{
89
+ name: 'lookup', description: 'Look up one key',
90
+ parameters: { type: 'object', properties: { key: { type: 'string' } }, required: ['key'] },
91
+ }],
92
+ tools: { lookup: async ({ key }) => ({ key, value: await lookup(key) }) },
93
+ policies: [],
94
+ maxTurns: 8,
95
+ });
96
+ ```
97
+
98
+ `runAgent(config, goal)` is also exported for embedding and tests. `main(config)`
99
+ provides `--task`, `--json`, and `--serve`. The loop adds `finish`, statically
100
+ verifies every proposal before execution, preserves tool-call/result IDs,
101
+ honors per-schema `timeoutMs`, traces task and chat runs, and uses request-shaped
102
+ inference with prompt-cache breakpoints. It loads the native binding only when a
103
+ loop runs, so `require('car-runtime/agent-loop')` is safe for package discovery.
104
+ Type declarations ship as `agent-loop.d.ts` and shared config/outcome types in
105
+ `index.d.ts`.
106
+
77
107
  Full API reference lives in [`index.d.ts`](./index.d.ts). The package also
78
108
  ships a `docs/` directory (`node_modules/car-runtime/docs/`) with prose
79
109
  reference docs — `SPEC.md`, `GUIDE.md`, `CLI.md`, `ASSISTANT.md`, `MCP.md`,
@@ -0,0 +1,30 @@
1
+ import type {
2
+ AgentLoopConfig,
3
+ AgentLoopOptions,
4
+ AgentOutcome,
5
+ AgentToolSchema,
6
+ AgentTool,
7
+ CarRuntime,
8
+ } from './index';
9
+
10
+ export function runAgent(
11
+ config: AgentLoopConfig,
12
+ goal: string,
13
+ options?: AgentLoopOptions,
14
+ ): Promise<AgentOutcome>;
15
+
16
+ export function runChatTurn(
17
+ runtime: CarRuntime,
18
+ config: AgentLoopConfig,
19
+ toolSchemas: AgentToolSchema[],
20
+ tools: Record<string, AgentTool>,
21
+ toolTimeouts: Record<string, number>,
22
+ sessionId: string,
23
+ messages: Record<string, unknown>[],
24
+ requestedModel?: string | null,
25
+ ): Promise<string>;
26
+
27
+ export function main(config: AgentLoopConfig): Promise<void>;
28
+
29
+ declare const agentLoop: { main: typeof main; runAgent: typeof runAgent; runChatTurn: typeof runChatTurn };
30
+ export default agentLoop;
package/agent-loop.js ADDED
@@ -0,0 +1,578 @@
1
+ 'use strict';
2
+
3
+ // Versioned generic CAR agent loop. Agent projects import this package export;
4
+ // they do not copy or fork the propose -> verify -> execute -> observe cycle.
5
+ // The native binding is loaded lazily so tooling can inspect/require this
6
+ // subpath without first installing a platform binary.
7
+ let nativeApi = null;
8
+ let cancelHandlerInstalled = false;
9
+ function runtimeApi() {
10
+ if (nativeApi === null) nativeApi = require('./index.js');
11
+ if (!cancelHandlerInstalled && typeof nativeApi.registerToolCancelHandler === 'function') {
12
+ nativeApi.registerToolCancelHandler((requestId) => {
13
+ const ctl = ABORTS.get(requestId);
14
+ if (ctl) {
15
+ ABORTS.delete(requestId);
16
+ try { ctl.abort(new Error('tool callback reaped by daemon (budget exceeded)')); } catch { /* already aborted */ }
17
+ }
18
+ });
19
+ cancelHandlerInstalled = true;
20
+ }
21
+ return nativeApi;
22
+ }
23
+
24
+ // Process-wide abort registry keyed on the daemon's per-call `request_id`
25
+ // (Parslee-ai/car#264). When the daemon reaps a tool callback (the call
26
+ // exceeded its budget) it emits `tools.cancel` with the `request_id`; the
27
+ // handler below aborts the matching controller so a long child (e.g. a
28
+ // `claude -p` / `codex exec` driven by drive_cli) is killed instead of
29
+ // orphaned. Each tool callback registers its controller under its request_id
30
+ // and removes it on completion. Registered once at module load; a daemon
31
+ // without the cancel surface simply never fires it.
32
+ const ABORTS = new Map();
33
+
34
+ // A built-in sentinel tool every agent gets for free, so the loop always has a
35
+ // clean way to terminate with a final answer. Your agent calls finish(answer).
36
+ const FINISH_SCHEMA = {
37
+ name: 'finish',
38
+ description: 'Return the final answer to the user and stop. Call this exactly once, when the task is complete or cannot proceed.',
39
+ parameters: {
40
+ type: 'object',
41
+ properties: { answer: { type: 'string', description: 'The final answer or status for the user.' } },
42
+ required: ['answer'],
43
+ },
44
+ };
45
+
46
+ function buildProposal(modelUsed, toolCalls, toolTimeouts = {}) {
47
+ // Glue between the inference IR (ToolCall {id,name,arguments}) and the
48
+ // action IR (Action {id,type,tool,parameters,timeout_ms}). The only IR
49
+ // plumbing you owe.
50
+ //
51
+ // `timeout_ms` (Parslee-ai/car#259): a tool that shells out to a build,
52
+ // drives another CLI, or calls a slow API needs more than the daemon's
53
+ // default callback budget. Declare `timeoutMs` on the tool's schema and
54
+ // it flows here as the action's per-call budget — without it the call is
55
+ // reaped at the default. Omitted (undefined) when the schema sets none,
56
+ // so the action falls back to the daemon default.
57
+ return JSON.stringify({
58
+ source: modelUsed || 'model',
59
+ actions: toolCalls.map((tc, i) => ({
60
+ id: tc.id || `a${i}`,
61
+ type: 'tool_call',
62
+ tool: tc.name,
63
+ parameters: tc.arguments || {},
64
+ dependencies: [],
65
+ timeout_ms: toolTimeouts[tc.name],
66
+ })),
67
+ });
68
+ }
69
+
70
+ /**
71
+ * Build the `executeProposal` tool callback. ONE implementation, used by BOTH
72
+ * the one-shot `--task` loop (`runAgent`) and the `--serve` chat path
73
+ * (`runChatTurn`). They used to carry two copies, and the chat copy silently
74
+ * missed `timeout_ms` and the `request_id`/AbortController wiring — the path a
75
+ * registered CarHost agent actually runs. Do not re-fork it.
76
+ *
77
+ * The callback receives { tool, params, action_id, request_id, timeout_ms }.
78
+ * The `request_id` keys this call's abort controller (Parslee-ai/car#264) so
79
+ * the daemon's `tools.cancel` can kill a reaped child. The tool fn receives the
80
+ * budget (timeoutMs) and an AbortSignal as a second arg; tools that shell out
81
+ * should honor the signal (e.g. pass it to child_process / fetch) so a reap
82
+ * actually terminates their child. The callback's `timeout_ms` is authoritative
83
+ * — it is the budget of the action actually executing (Parslee-ai/car#259) — so
84
+ * it wins over the locally derived `toolTimeouts` map, which is the fallback.
85
+ * `onFinish` (optional) is called with the `finish` tool's answer.
86
+ */
87
+ function makeToolCallback(tools, toolTimeouts = {}, onFinish = null) {
88
+ return async (callJson) => {
89
+ const { tool, params, request_id: requestId, timeout_ms: timeoutMs } = JSON.parse(callJson);
90
+ const fn = tools[tool];
91
+ if (!fn) throw new Error(`unknown tool: ${tool}`);
92
+ const ctl = new AbortController();
93
+ if (requestId) ABORTS.set(requestId, ctl);
94
+ try {
95
+ const out = await fn(params || {}, { signal: ctl.signal, timeoutMs: timeoutMs ?? toolTimeouts[tool] });
96
+ if (tool === 'finish' && onFinish) onFinish((out && out.answer) ?? '');
97
+ return JSON.stringify(out ?? {});
98
+ } finally {
99
+ // In a `finally` so a throwing tool can never leak its entry.
100
+ if (requestId) ABORTS.delete(requestId);
101
+ }
102
+ };
103
+ }
104
+
105
+ // ---- Prompt caching (Anthropic) --------------------------------------------
106
+ //
107
+ // The loop resends the WHOLE growing thread every turn, so an N-turn run bills
108
+ // the stable prefix N times at the full input rate — cost that grows with the
109
+ // square of the turn count. CAR's protocol layer already knows how to mark
110
+ // Anthropic cache breakpoints; it only needs `cache_control: true`, and the
111
+ // 9-positional-argument `inferTracked` has no slot to carry it. That is the
112
+ // whole reason every caller has been paying full price: the flag exists, the
113
+ // call shape could not reach it. `inferTrackedWithRequest` takes the options
114
+ // object (a JSON `GenerateRequest`) and can.
115
+ //
116
+ // BREAKPOINT PLACEMENT — we do not pick it; CAR does, and its choice is already
117
+ // the max-reuse one (car-inference/src/protocol.rs, AnthropicHandler::
118
+ // build_request_body). With `cache_control: true` it emits three of Anthropic's
119
+ // four allowed breakpoints:
120
+ // 1. the system block — identity plus the finish instruction, byte-identical
121
+ // for the entire run;
122
+ // 2. the LAST tool definition — so identity + every tool schema forms ONE
123
+ // cached prefix, the largest genuinely stable block this loop has;
124
+ // 3. the LAST message — the moving breakpoint. Anthropic serves the longest
125
+ // cached prefix it can find, so each turn writes only its own delta and
126
+ // READS everything the previous turn wrote. That is what turns quadratic
127
+ // re-billing into one write plus N cheap reads.
128
+ // The 4th breakpoint is deliberately left unused. It would buy a
129
+ // `context_stable_prefix` split of the system prompt, which helps only when the
130
+ // system prompt has a volatile tail; ours has none, so splitting it would
131
+ // shrink the cached block rather than grow it.
132
+ //
133
+ // TTL — `one_hour`, not the 5-minute default. Turns in an agentic loop are
134
+ // separated by real tool execution (browser drives, FMS reads, verification),
135
+ // which routinely exceeds five minutes; a 5-minute entry would expire mid-run
136
+ // and re-bill the entire prefix as a fresh write. A 1h write costs ~2x base
137
+ // input against ~1.25x for 5m, but that one-time 0.75x is far cheaper than a
138
+ // single full re-write of the prefix, and every surviving turn then reads at
139
+ // ~0.1x.
140
+ //
141
+ /** One request-shaped, cache-aware inference path for task and chat loops. */
142
+ async function inferTurn(rt, { model, maxTokens, toolSchemas, messages, toolChoice = 'auto' }) {
143
+ return JSON.parse(await rt.inferTrackedWithRequest(JSON.stringify({
144
+ prompt: '',
145
+ model: model ?? null,
146
+ params: {
147
+ max_tokens: maxTokens,
148
+ tool_choice: toolChoice,
149
+ strict_model: model != null,
150
+ cache_ttl: 'one_hour',
151
+ },
152
+ tools: toolSchemas,
153
+ messages,
154
+ cache_control: true,
155
+ })));
156
+ }
157
+
158
+ /**
159
+ * Run the agent once toward `goal`. Returns an AgentOutcome-shaped object:
160
+ * { status, summary, evidence[], metrics{}, tools_called[] }
161
+ * `status` is one of the six OutcomeStatus values (success|partial_success|
162
+ * done|give_up|timeout|failure). AgentOutcome is caller-built — CAR does not
163
+ * return it; we assemble it from what the loop observed.
164
+ *
165
+ * Run-trace lifecycle: each `runAgent` invocation is one run. Before the first
166
+ * proposal we bracket the run open with `rt.runsStart` (daemon mints a durable
167
+ * `run_id`); after the terminal AgentOutcome is assembled we close it with
168
+ * `rt.runsComplete`. Both are best-effort and never print, so the daemon traces
169
+ * the run for CarHost while older daemons (no runs.*) behave exactly as before.
170
+ */
171
+ async function runAgent(config, goal, { maxTurns } = {}) {
172
+ const { CarRuntime, executeProposal } = runtimeApi();
173
+ const turnsCap = maxTurns ?? config.maxTurns ?? 8;
174
+ const toolSchemas = [...(config.toolSchemas || []), FINISH_SCHEMA];
175
+ const tools = { finish: ({ answer }) => ({ answer }), ...(config.tools || {}) };
176
+ // Per-tool execution budget (Parslee-ai/car#259): a tool schema may set
177
+ // `timeoutMs` (e.g. a CLI driver that runs for 180s); it flows onto each
178
+ // action so the daemon honors it instead of reaping at the default.
179
+ const toolTimeouts = Object.fromEntries(
180
+ toolSchemas.filter((s) => s && s.timeoutMs != null).map((s) => [s.name, s.timeoutMs]),
181
+ );
182
+
183
+ const rt = new CarRuntime();
184
+ // Best-effort: some daemon versions don't expose agents.register_basics. It
185
+ // only adds CAR's built-in utility tools, which a tool-declaring agent doesn't
186
+ // depend on, so a missing method must not abort the run.
187
+ try { await rt.registerAgentBasics(); } catch { /* unsupported on this daemon — fine */ }
188
+ for (const s of toolSchemas) await rt.registerTool(s.name);
189
+ // Guardrails are declarative policies, enforced in Rust BEFORE the tool fires
190
+ // — not prompt rules. Each entry is the argument list for registerPolicy.
191
+ for (const p of (config.policies || [])) await rt.registerPolicy(...p);
192
+
193
+ // Run-trace bracket (open). Tell the daemon a run is starting so CarHost can
194
+ // trace it: prompt -> CLI outcome -> verifier verdict -> AgentOutcome. The
195
+ // daemon mints a durable run_id and tags it as this session's current run
196
+ // BEFORE replying, so the per-turn recorder reads the right id; we await that
197
+ // ack before submitting any proposal. The owning agent_id resolves from
198
+ // CAR_AGENT_ID (the supervisor injects it) when supervised, else falls back to
199
+ // config.agentName for the unsupervised one-shot / run_scenarios path.
200
+ // Best-effort, exactly like registerAgentBasics: a daemon without runs.* (an
201
+ // older build) makes this throw, and the run must continue unchanged. NEVER
202
+ // print here — the last stdout line in --json mode must stay the AgentOutcome
203
+ // (run_scenarios.py parses it).
204
+ let runId = null;
205
+ try {
206
+ const started = JSON.parse(await rt.runsStart(JSON.stringify({
207
+ intent: goal,
208
+ agent_id: process.env.CAR_AGENT_ID || config.agentName,
209
+ agent_name: config.agentName,
210
+ outcome_description: config.targetOutcome ?? '',
211
+ })));
212
+ runId = started.run_id ?? null;
213
+ } catch { /* daemon lacks runs.* — degrade to untraced behavior */ }
214
+
215
+ // Local Qwen3 models default to "thinking" (they emit a long <think> block
216
+ // before acting), which on a multi-tool agent prompt can exceed the daemon's
217
+ // per-call read timeout — the agent loop then fails with a timeout instead of
218
+ // calling a tool. When pinned to a local model, append Qwen3's `/no_think`
219
+ // soft switch so it acts via tools directly. Only applied for clearly-local
220
+ // model ids (a null/router model may resolve to a cloud model, where thinking
221
+ // is fine and fast); cloud models simply ignore the token.
222
+ const isLocalModel = typeof config.defaultModel === 'string'
223
+ && /^(mlx|qwen)\//.test(config.defaultModel);
224
+ const noThink = isLocalModel ? '\n\n/no_think' : '';
225
+ const messages = [
226
+ { role: 'system', content: `${config.identity}\n\nWhen the task is complete or you cannot proceed, call the \`finish\` tool with a concise answer. Do not narrate; act via tools.${noThink}` },
227
+ { role: 'user', content: goal },
228
+ ];
229
+
230
+ const metrics = { turns: 0, tool_calls: 0, actions_succeeded: 0, actions_failed: 0 };
231
+ const toolsCalled = new Set();
232
+ let outcome = null; // null === no terminal AgentOutcome yet
233
+ let turns = 0;
234
+
235
+ while (outcome === null) {
236
+ if (++turns > turnsCap) {
237
+ outcome = mkOutcome('timeout', `hit ${turnsCap}-turn cap without finishing`,
238
+ [{ kind: 'stop_reason', description: 'max turns', data: { turnsCap } }], metrics, toolsCalled);
239
+ break;
240
+ }
241
+ metrics.turns = turns;
242
+
243
+ // 1. PROPOSE — multi-turn, tool-aware inference (NOT plain `infer`).
244
+ let tracked;
245
+ try {
246
+ tracked = await inferTurn(rt, {
247
+ model: config.defaultModel ?? null,
248
+ maxTokens: config.maxTokens ?? 1024,
249
+ toolSchemas,
250
+ messages,
251
+ });
252
+ } catch (e) {
253
+ outcome = mkOutcome('failure', `inference failed: ${e.message || e}`,
254
+ [{ kind: 'stop_reason', description: 'infer_tracked error', data: null }], metrics, toolsCalled);
255
+ break;
256
+ }
257
+
258
+ const calls = tracked.tool_calls || [];
259
+ if (calls.length === 0) {
260
+ // Model answered in prose with no tool call — treat as a neutral Done.
261
+ outcome = mkOutcome('done', tracked.text || 'no further actions',
262
+ [{ kind: 'self_assessment', description: tracked.text || '', data: null }], metrics, toolsCalled);
263
+ break;
264
+ }
265
+
266
+ // Normalize ids so assistant tool_calls and tool_results correlate across
267
+ // turns (local models often omit ids).
268
+ calls.forEach((c, i) => { c.id = c.id || `a${i}`; });
269
+ messages.push({ role: 'assistant', content: tracked.text || '', tool_calls: calls });
270
+ const idToTool = Object.fromEntries(calls.map((c) => [c.id, c.name]));
271
+
272
+ const proposal = buildProposal(tracked.model_used, calls, toolTimeouts);
273
+
274
+ // 2. VERIFY — static gate. Never execute an unverified proposal.
275
+ let check;
276
+ try {
277
+ check = JSON.parse(await rt.verifyProposal(proposal));
278
+ } catch (e) {
279
+ outcome = mkOutcome('failure', `verification failed: ${e.message || e}`,
280
+ [{ kind: 'stop_reason', description: 'verifyProposal error', data: null }], metrics, toolsCalled);
281
+ break;
282
+ }
283
+ if (!check.valid) {
284
+ // Feed the rejection back as tool_results so tool_use/tool_result stay
285
+ // paired, then let the model repair on the next turn.
286
+ for (const c of calls) {
287
+ messages.push({ role: 'tool_result', tool_use_id: c.id,
288
+ content: JSON.stringify({ error: `runtime rejected proposal: ${JSON.stringify(check.issues)}` }) });
289
+ }
290
+ continue;
291
+ }
292
+
293
+ // 3. EXECUTE — CAR owns the DAG, retries, timeouts, rollback. The callback
294
+ // receives { tool, params } (note: `params`).
295
+ let finishAnswer = null;
296
+ let result;
297
+ try {
298
+ result = JSON.parse(await executeProposal(rt, proposal,
299
+ makeToolCallback(tools, toolTimeouts, (a) => { finishAnswer = a; })));
300
+ } catch (e) {
301
+ outcome = mkOutcome('failure', `execution failed: ${e.message || e}`,
302
+ [{ kind: 'stop_reason', description: 'executeProposal error', data: null }], metrics, toolsCalled);
303
+ break;
304
+ }
305
+
306
+ // 4. OBSERVE — feed each ActionResult back as a tool_result turn.
307
+ // tools_called records tools that SUCCESSFULLY executed, so a policy-denied
308
+ // /failed tool is correctly absent (guardrail scenarios can assert
309
+ // tool_not_called against it).
310
+ for (const r of (result.results || [])) {
311
+ const ok = r.status === 'succeeded';
312
+ if (ok && idToTool[r.action_id]) toolsCalled.add(idToTool[r.action_id]);
313
+ messages.push({ role: 'tool_result', tool_use_id: r.action_id,
314
+ content: JSON.stringify(ok ? (r.output ?? {}) : { error: r.error }) });
315
+ metrics.tool_calls += 1;
316
+ if (ok) metrics.actions_succeeded += 1; else metrics.actions_failed += 1;
317
+ }
318
+
319
+ // 5. Terminal? finish() succeeded -> success.
320
+ if (finishAnswer !== null) {
321
+ outcome = mkOutcome('success', finishAnswer,
322
+ [{ kind: 'tool_result', description: 'finish called', data: { answer: finishAnswer } }], metrics, toolsCalled);
323
+ }
324
+ // else: loop for the next proposal.
325
+ }
326
+
327
+ // Run-trace bracket (close). Report the terminal AgentOutcome to the daemon so
328
+ // CarHost shows the run's final status and stops streaming it. Await the ack
329
+ // before returning (the connection may close right after) so a healthy run is
330
+ // never raced into `Incomplete`. Best-effort + never prints, mirroring the
331
+ // open bracket: an older daemon without runs.* (or one that never acked the
332
+ // start, leaving runId null) just skips this and behaves as before.
333
+ if (runId !== null) {
334
+ try {
335
+ await rt.runsComplete(JSON.stringify({ run_id: runId, outcome }));
336
+ } catch { /* daemon lacks runs.* — nothing to report to */ }
337
+ }
338
+
339
+ return outcome;
340
+ }
341
+
342
+ function mkOutcome(status, summary, evidence, metrics, toolsCalled) {
343
+ return { status, summary, evidence, metrics, tools_called: [...toolsCalled].sort(), timestamp: new Date().toISOString() };
344
+ }
345
+
346
+ // ---- CLI entrypoint -------------------------------------------------------
347
+ //
348
+ // node agent.mjs --task "<goal>" [--json] one-shot; prints outcome
349
+ // node agent.mjs --serve supervised/standing mode
350
+ //
351
+ // run_scenarios.py invokes the --task --json form.
352
+
353
+ function parseArgs(argv) {
354
+ const a = { task: null, json: false, serve: false };
355
+ for (let i = 0; i < argv.length; i++) {
356
+ if (argv[i] === '--task') a.task = argv[++i];
357
+ else if (argv[i] === '--json') a.json = true;
358
+ else if (argv[i] === '--serve') a.serve = true;
359
+ }
360
+ return a;
361
+ }
362
+
363
+ // ---- Chat serving (agent.chat surface) --------------------------------------
364
+ //
365
+ // In `--serve` mode the agent attaches to the daemon (the binding sends
366
+ // CAR_AGENT_ID + CAR_AGENT_TOKEN on session.auth) and registers an `agent.chat`
367
+ // handler. The daemon reverse-calls `agent.chat { session_id, prompt }` for
368
+ // every host `agents.chat`; we keep a per-session message THREAD and run the
369
+ // same propose→verify→execute loop per turn, streaming the reply back via
370
+ // `agent.chat.event`. Threads are ephemeral (process lifetime). The agent's
371
+ // declared policies still gate tool execution, so guardrails (draft-only, etc.)
372
+ // carry into the conversation.
373
+
374
+ /** Run one chat turn against a persistent `messages` thread; stream via chatEvent. */
375
+ async function runChatTurn(
376
+ rt, config, toolSchemas, tools, toolTimeouts, sessionId, messages, requestedModel = null,
377
+ ) {
378
+ const { executeProposal } = runtimeApi();
379
+ const turnsCap = config.maxTurns ?? 8;
380
+ const selectedModel = typeof requestedModel === 'string' && requestedModel.trim()
381
+ ? requestedModel
382
+ : null;
383
+
384
+ // Run-trace bracket (open) — the SAME bracket `runAgent` opens, on the chat
385
+ // path. Without it a chat-driven turn does real tool work that never appears
386
+ // in `runs.list` / `runs.get_trace`, so a host that dispatches through
387
+ // `agents.chat` has no daemon-side record of what it ran: CarHost shows the
388
+ // agent as merely "running", and an outer loop cannot read back the tool
389
+ // returns or the terminal outcome. Same best-effort contract as `runAgent`'s
390
+ // — never throws, never prints (the last stdout line in --json mode must stay
391
+ // the AgentOutcome), and an older daemon without runs.* behaves exactly as it
392
+ // did before.
393
+ const intent = [...messages].reverse().find((m) => m.role === 'user')?.content ?? 'chat turn';
394
+ let runId = null;
395
+ try {
396
+ const started = JSON.parse(await rt.runsStart(JSON.stringify({
397
+ intent,
398
+ agent_id: process.env.CAR_AGENT_ID || config.agentName,
399
+ agent_name: config.agentName,
400
+ outcome_description: config.targetOutcome ?? '',
401
+ })));
402
+ runId = started.run_id ?? null;
403
+ } catch { /* daemon lacks runs.* — degrade to untraced behavior */ }
404
+
405
+ const metrics = { turns: 0, tool_calls: 0, actions_succeeded: 0, actions_failed: 0 };
406
+ const toolsCalled = new Set();
407
+ let terminal = null;
408
+ let finalText = '';
409
+ for (let turns = 0; turns < turnsCap; turns++) {
410
+ metrics.turns = turns + 1;
411
+ // Same cached request form as the one-shot loop. A chat thread grows for
412
+ // the whole session, so it is the path that benefits most from the moving
413
+ // conversation breakpoint.
414
+ const tracked = await inferTurn(rt, {
415
+ model: selectedModel ?? config.defaultModel ?? null,
416
+ maxTokens: config.maxTokens ?? 1024,
417
+ toolSchemas,
418
+ messages,
419
+ });
420
+ const calls = tracked.tool_calls || [];
421
+ if (calls.length === 0) {
422
+ finalText = tracked.text || '';
423
+ messages.push({ role: 'assistant', content: finalText });
424
+ terminal = 'done';
425
+ break;
426
+ }
427
+ calls.forEach((c, i) => { c.id = c.id || `a${i}`; });
428
+ messages.push({ role: 'assistant', content: tracked.text || '', tool_calls: calls });
429
+ const idToTool = Object.fromEntries(calls.map((c) => [c.id, c.name]));
430
+ // Surface non-finish tool calls as progress so the host UI can show them.
431
+ for (const c of calls) {
432
+ if (c.name !== 'finish') await rt.chatEvent(sessionId, 'tool_call', c.name).catch(() => {});
433
+ }
434
+ const proposal = buildProposal(tracked.model_used, calls, toolTimeouts);
435
+ const check = JSON.parse(await rt.verifyProposal(proposal));
436
+ if (!check.valid) {
437
+ for (const c of calls) {
438
+ messages.push({ role: 'tool_result', tool_use_id: c.id,
439
+ content: JSON.stringify({ error: `runtime rejected proposal: ${JSON.stringify(check.issues)}` }) });
440
+ }
441
+ continue;
442
+ }
443
+ // Same callback the one-shot loop uses, so the declared per-action budget
444
+ // and the request_id/AbortController cancel wiring reach tools on the
445
+ // --serve path too.
446
+ let finishAnswer = null;
447
+ const result = JSON.parse(await executeProposal(rt, proposal,
448
+ makeToolCallback(tools, toolTimeouts, (a) => { finishAnswer = a; })));
449
+ for (const r of (result.results || [])) {
450
+ const ok = r.status === 'succeeded';
451
+ if (ok && idToTool[r.action_id]) toolsCalled.add(idToTool[r.action_id]);
452
+ messages.push({ role: 'tool_result', tool_use_id: r.action_id,
453
+ content: JSON.stringify(ok ? (r.output ?? {}) : { error: r.error }) });
454
+ metrics.tool_calls += 1;
455
+ if (ok) metrics.actions_succeeded += 1; else metrics.actions_failed += 1;
456
+ }
457
+ if (finishAnswer !== null) { finalText = finishAnswer; terminal = 'success'; break; }
458
+ }
459
+ if (finalText) await rt.chatEvent(sessionId, 'token', finalText).catch(() => {});
460
+ await rt.chatEvent(sessionId, 'done', finalText).catch(() => {});
461
+
462
+ // Run-trace bracket (close). The outcome is assembled exactly as `runAgent`
463
+ // assembles it, so a chat-driven run and a --task run are the same shape in
464
+ // the trace and a host reads one code path, not two.
465
+ if (runId !== null) {
466
+ const outcome = terminal === 'success'
467
+ ? mkOutcome('success', finalText,
468
+ [{ kind: 'tool_result', description: 'finish called', data: { answer: finalText } }], metrics, toolsCalled)
469
+ : (terminal === 'done'
470
+ ? mkOutcome('done', finalText || 'no further actions',
471
+ [{ kind: 'self_assessment', description: finalText || '', data: null }], metrics, toolsCalled)
472
+ : mkOutcome('timeout', `hit ${turnsCap}-turn cap without finishing`,
473
+ [{ kind: 'stop_reason', description: 'max turns', data: { turnsCap } }], metrics, toolsCalled));
474
+ try {
475
+ await rt.runsComplete(JSON.stringify({ run_id: runId, outcome }));
476
+ } catch { /* daemon lacks runs.* — nothing to report to */ }
477
+ }
478
+ return finalText;
479
+ }
480
+
481
+ /** Set up the long-lived chat runtime: register tools/policies + the agent.chat handler. */
482
+ async function serveChat(config) {
483
+ const { CarRuntime, registerChatHandler } = runtimeApi();
484
+ const rt = new CarRuntime();
485
+ const toolSchemas = [...(config.toolSchemas || []), FINISH_SCHEMA];
486
+ const tools = { finish: ({ answer }) => ({ answer }), ...(config.tools || {}) };
487
+ const toolTimeouts = Object.fromEntries(
488
+ toolSchemas.filter((s) => s && s.timeoutMs != null).map((s) => [s.name, s.timeoutMs]),
489
+ );
490
+ try { await rt.registerAgentBasics(); } catch { /* unsupported — fine */ }
491
+ for (const s of toolSchemas) await rt.registerTool(s.name);
492
+ for (const p of (config.policies || [])) await rt.registerPolicy(...p);
493
+
494
+ if (typeof registerChatHandler !== 'function') {
495
+ console.error(`[${config.agentName}] car-runtime has no agent.chat support — chat disabled (update car-runtime).`);
496
+ return rt;
497
+ }
498
+
499
+ const threads = new Map(); // session_id -> messages[]
500
+ const isLocalModel = typeof config.defaultModel === 'string' && /^(mlx|qwen)\//.test(config.defaultModel);
501
+ const noThink = isLocalModel ? '\n\n/no_think' : '';
502
+
503
+ registerChatHandler((paramsJson) => {
504
+ // Fire-and-forget — the daemon already got its {accepted:true} ack. Run the
505
+ // turn on its own microtask and stream results back via chatEvent.
506
+ let params;
507
+ try { params = JSON.parse(paramsJson); } catch { return; }
508
+ const sessionId = params.session_id;
509
+ if (!sessionId) return;
510
+ let messages = threads.get(sessionId);
511
+ if (!messages) {
512
+ messages = [{ role: 'system', content: `${config.identity}\n\nWhen the task is complete or you cannot proceed, call the \`finish\` tool with a concise answer. Do not narrate; act via tools.${noThink}` }];
513
+ threads.set(sessionId, messages);
514
+ }
515
+ messages.push({ role: 'user', content: params.prompt ?? '' });
516
+ runChatTurn(
517
+ rt, config, toolSchemas, tools, toolTimeouts, sessionId, messages, params.model,
518
+ )
519
+ .catch((e) => rt.chatEvent(sessionId, 'error', String(e && e.message || e)).catch(() => {}));
520
+ });
521
+ console.error(`[${config.agentName}] chat ready (agent.chat) — drive via CarHost or an agents.chat host client`);
522
+ return rt;
523
+ }
524
+
525
+ async function main(config) {
526
+ const args = parseArgs(process.argv.slice(2));
527
+
528
+ if (args.serve) {
529
+ // Standing mode keeps the supervised process alive so CarHost shows it
530
+ // "running". Always serve chat (agent.chat); if the agent has a standing
531
+ // goal + interval, ALSO run it on a loop.
532
+ // car_register.py can set these via CAR_STANDING_GOAL / CAR_INTERVAL_SECS.
533
+ try { await serveChat(config); } catch (e) { console.error(`[${config.agentName}] chat setup failed:`, e); }
534
+ const goal = config.standingGoal ?? process.env.CAR_STANDING_GOAL ?? null;
535
+ const everyMs = (config.intervalSecs ?? Number(process.env.CAR_INTERVAL_SECS || 0)) * 1000;
536
+ console.error(`[${config.agentName}] serving${goal ? ` — "${goal}" every ${everyMs / 1000}s` : ' (idle; start via dashboard or --task)'}`);
537
+ if (goal && everyMs > 0) {
538
+ for (;;) {
539
+ try {
540
+ // Each iteration is its own run: runAgent opens and closes one
541
+ // runs.start/runs.complete bracket on its own fresh CarRuntime.
542
+ const o = await runAgent(config, goal);
543
+ console.error(`[${config.agentName}] ${o.status} — ${o.summary}`);
544
+ } catch (e) { console.error(`[${config.agentName}] error:`, e); }
545
+ await new Promise((r) => setTimeout(r, everyMs));
546
+ }
547
+ } else {
548
+ // Keep the event loop alive. A pending promise alone does NOT keep Node
549
+ // running — it exits when the loop is empty — so use a no-op heartbeat.
550
+ setInterval(() => {}, 1 << 30);
551
+ }
552
+ return;
553
+ }
554
+
555
+ if (!args.task) {
556
+ console.error('usage: node agent.mjs --task "<goal>" [--json] | --serve');
557
+ process.exit(2);
558
+ }
559
+
560
+ let outcome;
561
+ try {
562
+ outcome = await runAgent(config, args.task);
563
+ } catch (e) {
564
+ outcome = { status: 'failure', summary: String(e && e.message || e),
565
+ evidence: [{ kind: 'stop_reason', description: 'uncaught', data: null }],
566
+ metrics: { turns: 0, tool_calls: 0, actions_succeeded: 0, actions_failed: 0 }, tools_called: [] };
567
+ }
568
+
569
+ if (args.json) {
570
+ // The LAST stdout line is the machine-readable outcome (run_scenarios reads it).
571
+ console.log(JSON.stringify(outcome));
572
+ } else {
573
+ console.log(`${outcome.status} — ${outcome.summary}`);
574
+ }
575
+ process.exit(outcome.status === 'failure' ? 1 : 0);
576
+ }
577
+
578
+ module.exports = { main, runAgent, runChatTurn };
package/agent-loop.mjs ADDED
@@ -0,0 +1,4 @@
1
+ import loop from './agent-loop.js';
2
+
3
+ export const { main, runAgent, runChatTurn } = loop;
4
+ export default loop;
package/docs/ASSISTANT.md CHANGED
@@ -422,6 +422,7 @@ emitter that builds it:
422
422
  "network": "none",
423
423
  "tier": "sandbox_edit",
424
424
  "root": "/work",
425
+ "mount": null,
425
426
  "fallback_notice": null
426
427
  },
427
428
  "elapsed_seconds": 12.4