car-runtime 0.52.1 → 0.53.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -74,6 +74,36 @@ const result = await executeProposal(rt, proposal, async (callJson) => {
74
74
  });
75
75
  ```
76
76
 
77
+ ## Packaged agent loop
78
+
79
+ Do not copy a harness into each agent. Import the versioned loop and keep the
80
+ project entry file declarative:
81
+
82
+ ```javascript
83
+ import { main } from 'car-runtime/agent-loop';
84
+
85
+ main({
86
+ agentName: 'Lookup Agent',
87
+ identity: 'Use lookup before answering; never guess.',
88
+ toolSchemas: [{
89
+ name: 'lookup', description: 'Look up one key',
90
+ parameters: { type: 'object', properties: { key: { type: 'string' } }, required: ['key'] },
91
+ }],
92
+ tools: { lookup: async ({ key }) => ({ key, value: await lookup(key) }) },
93
+ policies: [],
94
+ maxTurns: 8,
95
+ });
96
+ ```
97
+
98
+ `runAgent(config, goal)` is also exported for embedding and tests. `main(config)`
99
+ provides `--task`, `--json`, and `--serve`. The loop adds `finish`, statically
100
+ verifies every proposal before execution, preserves tool-call/result IDs,
101
+ honors per-schema `timeoutMs`, traces task and chat runs, and uses request-shaped
102
+ inference with prompt-cache breakpoints. It loads the native binding only when a
103
+ loop runs, so `require('car-runtime/agent-loop')` is safe for package discovery.
104
+ Type declarations ship as `agent-loop.d.ts` and shared config/outcome types in
105
+ `index.d.ts`.
106
+
77
107
  Full API reference lives in [`index.d.ts`](./index.d.ts). The package also
78
108
  ships a `docs/` directory (`node_modules/car-runtime/docs/`) with prose
79
109
  reference docs — `SPEC.md`, `GUIDE.md`, `CLI.md`, `ASSISTANT.md`, `MCP.md`,
@@ -0,0 +1,30 @@
1
+ import type {
2
+ AgentLoopConfig,
3
+ AgentLoopOptions,
4
+ AgentOutcome,
5
+ AgentToolSchema,
6
+ AgentTool,
7
+ CarRuntime,
8
+ } from './index';
9
+
10
+ export function runAgent(
11
+ config: AgentLoopConfig,
12
+ goal: string,
13
+ options?: AgentLoopOptions,
14
+ ): Promise<AgentOutcome>;
15
+
16
+ export function runChatTurn(
17
+ runtime: CarRuntime,
18
+ config: AgentLoopConfig,
19
+ toolSchemas: AgentToolSchema[],
20
+ tools: Record<string, AgentTool>,
21
+ toolTimeouts: Record<string, number>,
22
+ sessionId: string,
23
+ messages: Record<string, unknown>[],
24
+ requestedModel?: string | null,
25
+ ): Promise<string>;
26
+
27
+ export function main(config: AgentLoopConfig): Promise<void>;
28
+
29
+ declare const agentLoop: { main: typeof main; runAgent: typeof runAgent; runChatTurn: typeof runChatTurn };
30
+ export default agentLoop;
package/agent-loop.js ADDED
@@ -0,0 +1,578 @@
1
+ 'use strict';
2
+
3
+ // Versioned generic CAR agent loop. Agent projects import this package export;
4
+ // they do not copy or fork the propose -> verify -> execute -> observe cycle.
5
+ // The native binding is loaded lazily so tooling can inspect/require this
6
+ // subpath without first installing a platform binary.
7
+ let nativeApi = null;
8
+ let cancelHandlerInstalled = false;
9
+ function runtimeApi() {
10
+ if (nativeApi === null) nativeApi = require('./index.js');
11
+ if (!cancelHandlerInstalled && typeof nativeApi.registerToolCancelHandler === 'function') {
12
+ nativeApi.registerToolCancelHandler((requestId) => {
13
+ const ctl = ABORTS.get(requestId);
14
+ if (ctl) {
15
+ ABORTS.delete(requestId);
16
+ try { ctl.abort(new Error('tool callback reaped by daemon (budget exceeded)')); } catch { /* already aborted */ }
17
+ }
18
+ });
19
+ cancelHandlerInstalled = true;
20
+ }
21
+ return nativeApi;
22
+ }
23
+
24
+ // Process-wide abort registry keyed on the daemon's per-call `request_id`
25
+ // (Parslee-ai/car#264). When the daemon reaps a tool callback (the call
26
+ // exceeded its budget) it emits `tools.cancel` with the `request_id`; the
27
+ // handler below aborts the matching controller so a long child (e.g. a
28
+ // `claude -p` / `codex exec` driven by drive_cli) is killed instead of
29
+ // orphaned. Each tool callback registers its controller under its request_id
30
+ // and removes it on completion. Registered once at module load; a daemon
31
+ // without the cancel surface simply never fires it.
32
+ const ABORTS = new Map();
33
+
34
+ // A built-in sentinel tool every agent gets for free, so the loop always has a
35
+ // clean way to terminate with a final answer. Your agent calls finish(answer).
36
+ const FINISH_SCHEMA = {
37
+ name: 'finish',
38
+ description: 'Return the final answer to the user and stop. Call this exactly once, when the task is complete or cannot proceed.',
39
+ parameters: {
40
+ type: 'object',
41
+ properties: { answer: { type: 'string', description: 'The final answer or status for the user.' } },
42
+ required: ['answer'],
43
+ },
44
+ };
45
+
46
+ function buildProposal(modelUsed, toolCalls, toolTimeouts = {}) {
47
+ // Glue between the inference IR (ToolCall {id,name,arguments}) and the
48
+ // action IR (Action {id,type,tool,parameters,timeout_ms}). The only IR
49
+ // plumbing you owe.
50
+ //
51
+ // `timeout_ms` (Parslee-ai/car#259): a tool that shells out to a build,
52
+ // drives another CLI, or calls a slow API needs more than the daemon's
53
+ // default callback budget. Declare `timeoutMs` on the tool's schema and
54
+ // it flows here as the action's per-call budget — without it the call is
55
+ // reaped at the default. Omitted (undefined) when the schema sets none,
56
+ // so the action falls back to the daemon default.
57
+ return JSON.stringify({
58
+ source: modelUsed || 'model',
59
+ actions: toolCalls.map((tc, i) => ({
60
+ id: tc.id || `a${i}`,
61
+ type: 'tool_call',
62
+ tool: tc.name,
63
+ parameters: tc.arguments || {},
64
+ dependencies: [],
65
+ timeout_ms: toolTimeouts[tc.name],
66
+ })),
67
+ });
68
+ }
69
+
70
+ /**
71
+ * Build the `executeProposal` tool callback. ONE implementation, used by BOTH
72
+ * the one-shot `--task` loop (`runAgent`) and the `--serve` chat path
73
+ * (`runChatTurn`). They used to carry two copies, and the chat copy silently
74
+ * missed `timeout_ms` and the `request_id`/AbortController wiring — the path a
75
+ * registered CarHost agent actually runs. Do not re-fork it.
76
+ *
77
+ * The callback receives { tool, params, action_id, request_id, timeout_ms }.
78
+ * The `request_id` keys this call's abort controller (Parslee-ai/car#264) so
79
+ * the daemon's `tools.cancel` can kill a reaped child. The tool fn receives the
80
+ * budget (timeoutMs) and an AbortSignal as a second arg; tools that shell out
81
+ * should honor the signal (e.g. pass it to child_process / fetch) so a reap
82
+ * actually terminates their child. The callback's `timeout_ms` is authoritative
83
+ * — it is the budget of the action actually executing (Parslee-ai/car#259) — so
84
+ * it wins over the locally derived `toolTimeouts` map, which is the fallback.
85
+ * `onFinish` (optional) is called with the `finish` tool's answer.
86
+ */
87
+ function makeToolCallback(tools, toolTimeouts = {}, onFinish = null) {
88
+ return async (callJson) => {
89
+ const { tool, params, request_id: requestId, timeout_ms: timeoutMs } = JSON.parse(callJson);
90
+ const fn = tools[tool];
91
+ if (!fn) throw new Error(`unknown tool: ${tool}`);
92
+ const ctl = new AbortController();
93
+ if (requestId) ABORTS.set(requestId, ctl);
94
+ try {
95
+ const out = await fn(params || {}, { signal: ctl.signal, timeoutMs: timeoutMs ?? toolTimeouts[tool] });
96
+ if (tool === 'finish' && onFinish) onFinish((out && out.answer) ?? '');
97
+ return JSON.stringify(out ?? {});
98
+ } finally {
99
+ // In a `finally` so a throwing tool can never leak its entry.
100
+ if (requestId) ABORTS.delete(requestId);
101
+ }
102
+ };
103
+ }
104
+
105
+ // ---- Prompt caching (Anthropic) --------------------------------------------
106
+ //
107
+ // The loop resends the WHOLE growing thread every turn, so an N-turn run bills
108
+ // the stable prefix N times at the full input rate — cost that grows with the
109
+ // square of the turn count. CAR's protocol layer already knows how to mark
110
+ // Anthropic cache breakpoints; it only needs `cache_control: true`, and the
111
+ // 9-positional-argument `inferTracked` has no slot to carry it. That is the
112
+ // whole reason every caller has been paying full price: the flag exists, the
113
+ // call shape could not reach it. `inferTrackedWithRequest` takes the options
114
+ // object (a JSON `GenerateRequest`) and can.
115
+ //
116
+ // BREAKPOINT PLACEMENT — we do not pick it; CAR does, and its choice is already
117
+ // the max-reuse one (car-inference/src/protocol.rs, AnthropicHandler::
118
+ // build_request_body). With `cache_control: true` it emits three of Anthropic's
119
+ // four allowed breakpoints:
120
+ // 1. the system block — identity plus the finish instruction, byte-identical
121
+ // for the entire run;
122
+ // 2. the LAST tool definition — so identity + every tool schema forms ONE
123
+ // cached prefix, the largest genuinely stable block this loop has;
124
+ // 3. the LAST message — the moving breakpoint. Anthropic serves the longest
125
+ // cached prefix it can find, so each turn writes only its own delta and
126
+ // READS everything the previous turn wrote. That is what turns quadratic
127
+ // re-billing into one write plus N cheap reads.
128
+ // The 4th breakpoint is deliberately left unused. It would buy a
129
+ // `context_stable_prefix` split of the system prompt, which helps only when the
130
+ // system prompt has a volatile tail; ours has none, so splitting it would
131
+ // shrink the cached block rather than grow it.
132
+ //
133
+ // TTL — `one_hour`, not the 5-minute default. Turns in an agentic loop are
134
+ // separated by real tool execution (browser drives, FMS reads, verification),
135
+ // which routinely exceeds five minutes; a 5-minute entry would expire mid-run
136
+ // and re-bill the entire prefix as a fresh write. A 1h write costs ~2x base
137
+ // input against ~1.25x for 5m, but that one-time 0.75x is far cheaper than a
138
+ // single full re-write of the prefix, and every surviving turn then reads at
139
+ // ~0.1x.
140
+ //
141
+ /** One request-shaped, cache-aware inference path for task and chat loops. */
142
+ async function inferTurn(rt, { model, maxTokens, toolSchemas, messages, toolChoice = 'auto' }) {
143
+ return JSON.parse(await rt.inferTrackedWithRequest(JSON.stringify({
144
+ prompt: '',
145
+ model: model ?? null,
146
+ params: {
147
+ max_tokens: maxTokens,
148
+ tool_choice: toolChoice,
149
+ strict_model: model != null,
150
+ cache_ttl: 'one_hour',
151
+ },
152
+ tools: toolSchemas,
153
+ messages,
154
+ cache_control: true,
155
+ })));
156
+ }
157
+
158
+ /**
159
+ * Run the agent once toward `goal`. Returns an AgentOutcome-shaped object:
160
+ * { status, summary, evidence[], metrics{}, tools_called[] }
161
+ * `status` is one of the six OutcomeStatus values (success|partial_success|
162
+ * done|give_up|timeout|failure). AgentOutcome is caller-built — CAR does not
163
+ * return it; we assemble it from what the loop observed.
164
+ *
165
+ * Run-trace lifecycle: each `runAgent` invocation is one run. Before the first
166
+ * proposal we bracket the run open with `rt.runsStart` (daemon mints a durable
167
+ * `run_id`); after the terminal AgentOutcome is assembled we close it with
168
+ * `rt.runsComplete`. Both are best-effort and never print, so the daemon traces
169
+ * the run for CarHost while older daemons (no runs.*) behave exactly as before.
170
+ */
171
+ async function runAgent(config, goal, { maxTurns } = {}) {
172
+ const { CarRuntime, executeProposal } = runtimeApi();
173
+ const turnsCap = maxTurns ?? config.maxTurns ?? 8;
174
+ const toolSchemas = [...(config.toolSchemas || []), FINISH_SCHEMA];
175
+ const tools = { finish: ({ answer }) => ({ answer }), ...(config.tools || {}) };
176
+ // Per-tool execution budget (Parslee-ai/car#259): a tool schema may set
177
+ // `timeoutMs` (e.g. a CLI driver that runs for 180s); it flows onto each
178
+ // action so the daemon honors it instead of reaping at the default.
179
+ const toolTimeouts = Object.fromEntries(
180
+ toolSchemas.filter((s) => s && s.timeoutMs != null).map((s) => [s.name, s.timeoutMs]),
181
+ );
182
+
183
+ const rt = new CarRuntime();
184
+ // Best-effort: some daemon versions don't expose agents.register_basics. It
185
+ // only adds CAR's built-in utility tools, which a tool-declaring agent doesn't
186
+ // depend on, so a missing method must not abort the run.
187
+ try { await rt.registerAgentBasics(); } catch { /* unsupported on this daemon — fine */ }
188
+ for (const s of toolSchemas) await rt.registerTool(s.name);
189
+ // Guardrails are declarative policies, enforced in Rust BEFORE the tool fires
190
+ // — not prompt rules. Each entry is the argument list for registerPolicy.
191
+ for (const p of (config.policies || [])) await rt.registerPolicy(...p);
192
+
193
+ // Run-trace bracket (open). Tell the daemon a run is starting so CarHost can
194
+ // trace it: prompt -> CLI outcome -> verifier verdict -> AgentOutcome. The
195
+ // daemon mints a durable run_id and tags it as this session's current run
196
+ // BEFORE replying, so the per-turn recorder reads the right id; we await that
197
+ // ack before submitting any proposal. The owning agent_id resolves from
198
+ // CAR_AGENT_ID (the supervisor injects it) when supervised, else falls back to
199
+ // config.agentName for the unsupervised one-shot / run_scenarios path.
200
+ // Best-effort, exactly like registerAgentBasics: a daemon without runs.* (an
201
+ // older build) makes this throw, and the run must continue unchanged. NEVER
202
+ // print here — the last stdout line in --json mode must stay the AgentOutcome
203
+ // (run_scenarios.py parses it).
204
+ let runId = null;
205
+ try {
206
+ const started = JSON.parse(await rt.runsStart(JSON.stringify({
207
+ intent: goal,
208
+ agent_id: process.env.CAR_AGENT_ID || config.agentName,
209
+ agent_name: config.agentName,
210
+ outcome_description: config.targetOutcome ?? '',
211
+ })));
212
+ runId = started.run_id ?? null;
213
+ } catch { /* daemon lacks runs.* — degrade to untraced behavior */ }
214
+
215
+ // Local Qwen3 models default to "thinking" (they emit a long <think> block
216
+ // before acting), which on a multi-tool agent prompt can exceed the daemon's
217
+ // per-call read timeout — the agent loop then fails with a timeout instead of
218
+ // calling a tool. When pinned to a local model, append Qwen3's `/no_think`
219
+ // soft switch so it acts via tools directly. Only applied for clearly-local
220
+ // model ids (a null/router model may resolve to a cloud model, where thinking
221
+ // is fine and fast); cloud models simply ignore the token.
222
+ const isLocalModel = typeof config.defaultModel === 'string'
223
+ && /^(mlx|qwen)\//.test(config.defaultModel);
224
+ const noThink = isLocalModel ? '\n\n/no_think' : '';
225
+ const messages = [
226
+ { role: 'system', content: `${config.identity}\n\nWhen the task is complete or you cannot proceed, call the \`finish\` tool with a concise answer. Do not narrate; act via tools.${noThink}` },
227
+ { role: 'user', content: goal },
228
+ ];
229
+
230
+ const metrics = { turns: 0, tool_calls: 0, actions_succeeded: 0, actions_failed: 0 };
231
+ const toolsCalled = new Set();
232
+ let outcome = null; // null === no terminal AgentOutcome yet
233
+ let turns = 0;
234
+
235
+ while (outcome === null) {
236
+ if (++turns > turnsCap) {
237
+ outcome = mkOutcome('timeout', `hit ${turnsCap}-turn cap without finishing`,
238
+ [{ kind: 'stop_reason', description: 'max turns', data: { turnsCap } }], metrics, toolsCalled);
239
+ break;
240
+ }
241
+ metrics.turns = turns;
242
+
243
+ // 1. PROPOSE — multi-turn, tool-aware inference (NOT plain `infer`).
244
+ let tracked;
245
+ try {
246
+ tracked = await inferTurn(rt, {
247
+ model: config.defaultModel ?? null,
248
+ maxTokens: config.maxTokens ?? 1024,
249
+ toolSchemas,
250
+ messages,
251
+ });
252
+ } catch (e) {
253
+ outcome = mkOutcome('failure', `inference failed: ${e.message || e}`,
254
+ [{ kind: 'stop_reason', description: 'infer_tracked error', data: null }], metrics, toolsCalled);
255
+ break;
256
+ }
257
+
258
+ const calls = tracked.tool_calls || [];
259
+ if (calls.length === 0) {
260
+ // Model answered in prose with no tool call — treat as a neutral Done.
261
+ outcome = mkOutcome('done', tracked.text || 'no further actions',
262
+ [{ kind: 'self_assessment', description: tracked.text || '', data: null }], metrics, toolsCalled);
263
+ break;
264
+ }
265
+
266
+ // Normalize ids so assistant tool_calls and tool_results correlate across
267
+ // turns (local models often omit ids).
268
+ calls.forEach((c, i) => { c.id = c.id || `a${i}`; });
269
+ messages.push({ role: 'assistant', content: tracked.text || '', tool_calls: calls });
270
+ const idToTool = Object.fromEntries(calls.map((c) => [c.id, c.name]));
271
+
272
+ const proposal = buildProposal(tracked.model_used, calls, toolTimeouts);
273
+
274
+ // 2. VERIFY — static gate. Never execute an unverified proposal.
275
+ let check;
276
+ try {
277
+ check = JSON.parse(await rt.verifyProposal(proposal));
278
+ } catch (e) {
279
+ outcome = mkOutcome('failure', `verification failed: ${e.message || e}`,
280
+ [{ kind: 'stop_reason', description: 'verifyProposal error', data: null }], metrics, toolsCalled);
281
+ break;
282
+ }
283
+ if (!check.valid) {
284
+ // Feed the rejection back as tool_results so tool_use/tool_result stay
285
+ // paired, then let the model repair on the next turn.
286
+ for (const c of calls) {
287
+ messages.push({ role: 'tool_result', tool_use_id: c.id,
288
+ content: JSON.stringify({ error: `runtime rejected proposal: ${JSON.stringify(check.issues)}` }) });
289
+ }
290
+ continue;
291
+ }
292
+
293
+ // 3. EXECUTE — CAR owns the DAG, retries, timeouts, rollback. The callback
294
+ // receives { tool, params } (note: `params`).
295
+ let finishAnswer = null;
296
+ let result;
297
+ try {
298
+ result = JSON.parse(await executeProposal(rt, proposal,
299
+ makeToolCallback(tools, toolTimeouts, (a) => { finishAnswer = a; })));
300
+ } catch (e) {
301
+ outcome = mkOutcome('failure', `execution failed: ${e.message || e}`,
302
+ [{ kind: 'stop_reason', description: 'executeProposal error', data: null }], metrics, toolsCalled);
303
+ break;
304
+ }
305
+
306
+ // 4. OBSERVE — feed each ActionResult back as a tool_result turn.
307
+ // tools_called records tools that SUCCESSFULLY executed, so a policy-denied
308
+ // /failed tool is correctly absent (guardrail scenarios can assert
309
+ // tool_not_called against it).
310
+ for (const r of (result.results || [])) {
311
+ const ok = r.status === 'succeeded';
312
+ if (ok && idToTool[r.action_id]) toolsCalled.add(idToTool[r.action_id]);
313
+ messages.push({ role: 'tool_result', tool_use_id: r.action_id,
314
+ content: JSON.stringify(ok ? (r.output ?? {}) : { error: r.error }) });
315
+ metrics.tool_calls += 1;
316
+ if (ok) metrics.actions_succeeded += 1; else metrics.actions_failed += 1;
317
+ }
318
+
319
+ // 5. Terminal? finish() succeeded -> success.
320
+ if (finishAnswer !== null) {
321
+ outcome = mkOutcome('success', finishAnswer,
322
+ [{ kind: 'tool_result', description: 'finish called', data: { answer: finishAnswer } }], metrics, toolsCalled);
323
+ }
324
+ // else: loop for the next proposal.
325
+ }
326
+
327
+ // Run-trace bracket (close). Report the terminal AgentOutcome to the daemon so
328
+ // CarHost shows the run's final status and stops streaming it. Await the ack
329
+ // before returning (the connection may close right after) so a healthy run is
330
+ // never raced into `Incomplete`. Best-effort + never prints, mirroring the
331
+ // open bracket: an older daemon without runs.* (or one that never acked the
332
+ // start, leaving runId null) just skips this and behaves as before.
333
+ if (runId !== null) {
334
+ try {
335
+ await rt.runsComplete(JSON.stringify({ run_id: runId, outcome }));
336
+ } catch { /* daemon lacks runs.* — nothing to report to */ }
337
+ }
338
+
339
+ return outcome;
340
+ }
341
+
342
+ function mkOutcome(status, summary, evidence, metrics, toolsCalled) {
343
+ return { status, summary, evidence, metrics, tools_called: [...toolsCalled].sort(), timestamp: new Date().toISOString() };
344
+ }
345
+
346
+ // ---- CLI entrypoint -------------------------------------------------------
347
+ //
348
+ // node agent.mjs --task "<goal>" [--json] one-shot; prints outcome
349
+ // node agent.mjs --serve supervised/standing mode
350
+ //
351
+ // run_scenarios.py invokes the --task --json form.
352
+
353
+ function parseArgs(argv) {
354
+ const a = { task: null, json: false, serve: false };
355
+ for (let i = 0; i < argv.length; i++) {
356
+ if (argv[i] === '--task') a.task = argv[++i];
357
+ else if (argv[i] === '--json') a.json = true;
358
+ else if (argv[i] === '--serve') a.serve = true;
359
+ }
360
+ return a;
361
+ }
362
+
363
+ // ---- Chat serving (agent.chat surface) --------------------------------------
364
+ //
365
+ // In `--serve` mode the agent attaches to the daemon (the binding sends
366
+ // CAR_AGENT_ID + CAR_AGENT_TOKEN on session.auth) and registers an `agent.chat`
367
+ // handler. The daemon reverse-calls `agent.chat { session_id, prompt }` for
368
+ // every host `agents.chat`; we keep a per-session message THREAD and run the
369
+ // same propose→verify→execute loop per turn, streaming the reply back via
370
+ // `agent.chat.event`. Threads are ephemeral (process lifetime). The agent's
371
+ // declared policies still gate tool execution, so guardrails (draft-only, etc.)
372
+ // carry into the conversation.
373
+
374
+ /** Run one chat turn against a persistent `messages` thread; stream via chatEvent. */
375
+ async function runChatTurn(
376
+ rt, config, toolSchemas, tools, toolTimeouts, sessionId, messages, requestedModel = null,
377
+ ) {
378
+ const { executeProposal } = runtimeApi();
379
+ const turnsCap = config.maxTurns ?? 8;
380
+ const selectedModel = typeof requestedModel === 'string' && requestedModel.trim()
381
+ ? requestedModel
382
+ : null;
383
+
384
+ // Run-trace bracket (open) — the SAME bracket `runAgent` opens, on the chat
385
+ // path. Without it a chat-driven turn does real tool work that never appears
386
+ // in `runs.list` / `runs.get_trace`, so a host that dispatches through
387
+ // `agents.chat` has no daemon-side record of what it ran: CarHost shows the
388
+ // agent as merely "running", and an outer loop cannot read back the tool
389
+ // returns or the terminal outcome. Same best-effort contract as `runAgent`'s
390
+ // — never throws, never prints (the last stdout line in --json mode must stay
391
+ // the AgentOutcome), and an older daemon without runs.* behaves exactly as it
392
+ // did before.
393
+ const intent = [...messages].reverse().find((m) => m.role === 'user')?.content ?? 'chat turn';
394
+ let runId = null;
395
+ try {
396
+ const started = JSON.parse(await rt.runsStart(JSON.stringify({
397
+ intent,
398
+ agent_id: process.env.CAR_AGENT_ID || config.agentName,
399
+ agent_name: config.agentName,
400
+ outcome_description: config.targetOutcome ?? '',
401
+ })));
402
+ runId = started.run_id ?? null;
403
+ } catch { /* daemon lacks runs.* — degrade to untraced behavior */ }
404
+
405
+ const metrics = { turns: 0, tool_calls: 0, actions_succeeded: 0, actions_failed: 0 };
406
+ const toolsCalled = new Set();
407
+ let terminal = null;
408
+ let finalText = '';
409
+ for (let turns = 0; turns < turnsCap; turns++) {
410
+ metrics.turns = turns + 1;
411
+ // Same cached request form as the one-shot loop. A chat thread grows for
412
+ // the whole session, so it is the path that benefits most from the moving
413
+ // conversation breakpoint.
414
+ const tracked = await inferTurn(rt, {
415
+ model: selectedModel ?? config.defaultModel ?? null,
416
+ maxTokens: config.maxTokens ?? 1024,
417
+ toolSchemas,
418
+ messages,
419
+ });
420
+ const calls = tracked.tool_calls || [];
421
+ if (calls.length === 0) {
422
+ finalText = tracked.text || '';
423
+ messages.push({ role: 'assistant', content: finalText });
424
+ terminal = 'done';
425
+ break;
426
+ }
427
+ calls.forEach((c, i) => { c.id = c.id || `a${i}`; });
428
+ messages.push({ role: 'assistant', content: tracked.text || '', tool_calls: calls });
429
+ const idToTool = Object.fromEntries(calls.map((c) => [c.id, c.name]));
430
+ // Surface non-finish tool calls as progress so the host UI can show them.
431
+ for (const c of calls) {
432
+ if (c.name !== 'finish') await rt.chatEvent(sessionId, 'tool_call', c.name).catch(() => {});
433
+ }
434
+ const proposal = buildProposal(tracked.model_used, calls, toolTimeouts);
435
+ const check = JSON.parse(await rt.verifyProposal(proposal));
436
+ if (!check.valid) {
437
+ for (const c of calls) {
438
+ messages.push({ role: 'tool_result', tool_use_id: c.id,
439
+ content: JSON.stringify({ error: `runtime rejected proposal: ${JSON.stringify(check.issues)}` }) });
440
+ }
441
+ continue;
442
+ }
443
+ // Same callback the one-shot loop uses, so the declared per-action budget
444
+ // and the request_id/AbortController cancel wiring reach tools on the
445
+ // --serve path too.
446
+ let finishAnswer = null;
447
+ const result = JSON.parse(await executeProposal(rt, proposal,
448
+ makeToolCallback(tools, toolTimeouts, (a) => { finishAnswer = a; })));
449
+ for (const r of (result.results || [])) {
450
+ const ok = r.status === 'succeeded';
451
+ if (ok && idToTool[r.action_id]) toolsCalled.add(idToTool[r.action_id]);
452
+ messages.push({ role: 'tool_result', tool_use_id: r.action_id,
453
+ content: JSON.stringify(ok ? (r.output ?? {}) : { error: r.error }) });
454
+ metrics.tool_calls += 1;
455
+ if (ok) metrics.actions_succeeded += 1; else metrics.actions_failed += 1;
456
+ }
457
+ if (finishAnswer !== null) { finalText = finishAnswer; terminal = 'success'; break; }
458
+ }
459
+ if (finalText) await rt.chatEvent(sessionId, 'token', finalText).catch(() => {});
460
+ await rt.chatEvent(sessionId, 'done', finalText).catch(() => {});
461
+
462
+ // Run-trace bracket (close). The outcome is assembled exactly as `runAgent`
463
+ // assembles it, so a chat-driven run and a --task run are the same shape in
464
+ // the trace and a host reads one code path, not two.
465
+ if (runId !== null) {
466
+ const outcome = terminal === 'success'
467
+ ? mkOutcome('success', finalText,
468
+ [{ kind: 'tool_result', description: 'finish called', data: { answer: finalText } }], metrics, toolsCalled)
469
+ : (terminal === 'done'
470
+ ? mkOutcome('done', finalText || 'no further actions',
471
+ [{ kind: 'self_assessment', description: finalText || '', data: null }], metrics, toolsCalled)
472
+ : mkOutcome('timeout', `hit ${turnsCap}-turn cap without finishing`,
473
+ [{ kind: 'stop_reason', description: 'max turns', data: { turnsCap } }], metrics, toolsCalled));
474
+ try {
475
+ await rt.runsComplete(JSON.stringify({ run_id: runId, outcome }));
476
+ } catch { /* daemon lacks runs.* — nothing to report to */ }
477
+ }
478
+ return finalText;
479
+ }
480
+
481
+ /** Set up the long-lived chat runtime: register tools/policies + the agent.chat handler. */
482
+ async function serveChat(config) {
483
+ const { CarRuntime, registerChatHandler } = runtimeApi();
484
+ const rt = new CarRuntime();
485
+ const toolSchemas = [...(config.toolSchemas || []), FINISH_SCHEMA];
486
+ const tools = { finish: ({ answer }) => ({ answer }), ...(config.tools || {}) };
487
+ const toolTimeouts = Object.fromEntries(
488
+ toolSchemas.filter((s) => s && s.timeoutMs != null).map((s) => [s.name, s.timeoutMs]),
489
+ );
490
+ try { await rt.registerAgentBasics(); } catch { /* unsupported — fine */ }
491
+ for (const s of toolSchemas) await rt.registerTool(s.name);
492
+ for (const p of (config.policies || [])) await rt.registerPolicy(...p);
493
+
494
+ if (typeof registerChatHandler !== 'function') {
495
+ console.error(`[${config.agentName}] car-runtime has no agent.chat support — chat disabled (update car-runtime).`);
496
+ return rt;
497
+ }
498
+
499
+ const threads = new Map(); // session_id -> messages[]
500
+ const isLocalModel = typeof config.defaultModel === 'string' && /^(mlx|qwen)\//.test(config.defaultModel);
501
+ const noThink = isLocalModel ? '\n\n/no_think' : '';
502
+
503
+ registerChatHandler((paramsJson) => {
504
+ // Fire-and-forget — the daemon already got its {accepted:true} ack. Run the
505
+ // turn on its own microtask and stream results back via chatEvent.
506
+ let params;
507
+ try { params = JSON.parse(paramsJson); } catch { return; }
508
+ const sessionId = params.session_id;
509
+ if (!sessionId) return;
510
+ let messages = threads.get(sessionId);
511
+ if (!messages) {
512
+ messages = [{ role: 'system', content: `${config.identity}\n\nWhen the task is complete or you cannot proceed, call the \`finish\` tool with a concise answer. Do not narrate; act via tools.${noThink}` }];
513
+ threads.set(sessionId, messages);
514
+ }
515
+ messages.push({ role: 'user', content: params.prompt ?? '' });
516
+ runChatTurn(
517
+ rt, config, toolSchemas, tools, toolTimeouts, sessionId, messages, params.model,
518
+ )
519
+ .catch((e) => rt.chatEvent(sessionId, 'error', String(e && e.message || e)).catch(() => {}));
520
+ });
521
+ console.error(`[${config.agentName}] chat ready (agent.chat) — drive via CarHost or an agents.chat host client`);
522
+ return rt;
523
+ }
524
+
525
+ async function main(config) {
526
+ const args = parseArgs(process.argv.slice(2));
527
+
528
+ if (args.serve) {
529
+ // Standing mode keeps the supervised process alive so CarHost shows it
530
+ // "running". Always serve chat (agent.chat); if the agent has a standing
531
+ // goal + interval, ALSO run it on a loop.
532
+ // car_register.py can set these via CAR_STANDING_GOAL / CAR_INTERVAL_SECS.
533
+ try { await serveChat(config); } catch (e) { console.error(`[${config.agentName}] chat setup failed:`, e); }
534
+ const goal = config.standingGoal ?? process.env.CAR_STANDING_GOAL ?? null;
535
+ const everyMs = (config.intervalSecs ?? Number(process.env.CAR_INTERVAL_SECS || 0)) * 1000;
536
+ console.error(`[${config.agentName}] serving${goal ? ` — "${goal}" every ${everyMs / 1000}s` : ' (idle; start via dashboard or --task)'}`);
537
+ if (goal && everyMs > 0) {
538
+ for (;;) {
539
+ try {
540
+ // Each iteration is its own run: runAgent opens and closes one
541
+ // runs.start/runs.complete bracket on its own fresh CarRuntime.
542
+ const o = await runAgent(config, goal);
543
+ console.error(`[${config.agentName}] ${o.status} — ${o.summary}`);
544
+ } catch (e) { console.error(`[${config.agentName}] error:`, e); }
545
+ await new Promise((r) => setTimeout(r, everyMs));
546
+ }
547
+ } else {
548
+ // Keep the event loop alive. A pending promise alone does NOT keep Node
549
+ // running — it exits when the loop is empty — so use a no-op heartbeat.
550
+ setInterval(() => {}, 1 << 30);
551
+ }
552
+ return;
553
+ }
554
+
555
+ if (!args.task) {
556
+ console.error('usage: node agent.mjs --task "<goal>" [--json] | --serve');
557
+ process.exit(2);
558
+ }
559
+
560
+ let outcome;
561
+ try {
562
+ outcome = await runAgent(config, args.task);
563
+ } catch (e) {
564
+ outcome = { status: 'failure', summary: String(e && e.message || e),
565
+ evidence: [{ kind: 'stop_reason', description: 'uncaught', data: null }],
566
+ metrics: { turns: 0, tool_calls: 0, actions_succeeded: 0, actions_failed: 0 }, tools_called: [] };
567
+ }
568
+
569
+ if (args.json) {
570
+ // The LAST stdout line is the machine-readable outcome (run_scenarios reads it).
571
+ console.log(JSON.stringify(outcome));
572
+ } else {
573
+ console.log(`${outcome.status} — ${outcome.summary}`);
574
+ }
575
+ process.exit(outcome.status === 'failure' ? 1 : 0);
576
+ }
577
+
578
+ module.exports = { main, runAgent, runChatTurn };
package/agent-loop.mjs ADDED
@@ -0,0 +1,4 @@
1
+ import loop from './agent-loop.js';
2
+
3
+ export const { main, runAgent, runChatTurn } = loop;
4
+ export default loop;
package/docs/ASSISTANT.md CHANGED
@@ -422,6 +422,7 @@ emitter that builds it:
422
422
  "network": "none",
423
423
  "tier": "sandbox_edit",
424
424
  "root": "/work",
425
+ "mount": null,
425
426
  "fallback_notice": null
426
427
  },
427
428
  "elapsed_seconds": 12.4
package/docs/CLI.md CHANGED
@@ -2,17 +2,17 @@
2
2
 
3
3
  > **Generated file — do not hand-edit below the task map.** Produced by
4
4
  > `scripts/gen-cli-docs.sh` from `car --help` / `car help <command>` on car
5
- > 0.52.1 (2026-09-04). Every subcommand the installed binary reports is
5
+ > 0.52.1 (2026-09-06). Every subcommand the installed binary reports is
6
6
  > below; a new subcommand cannot ship without appearing here the next time
7
7
  > this script runs. To regenerate: `bash scripts/gen-cli-docs.sh`.
8
8
  >
9
- > 68 top-level commands, 101 nested subcommands
9
+ > 69 top-level commands, 106 nested subcommands
10
10
  > (one level deep) — counted from the live binary at generation time, not
11
11
  > typed by hand.
12
12
 
13
13
  ## Finding your way around
14
14
 
15
- `car` is one binary with 68 subcommands spanning several different jobs:
15
+ `car` is one binary with 69 subcommands spanning several different jobs:
16
16
  running the built-in agent, coding, local model management, OS integrations,
17
17
  and installing other people's agents on your machine. This map groups the
18
18
  commands people actually reach for; the full alphabetical reference with every
@@ -174,6 +174,7 @@ commands on a cadence via launchd / cron / schtasks).
174
174
  | [`car code-task`](#car-code-task) | Run a coder session headlessly and IN THIS PROCESS: derive or accept an outcome contract, work in a git worktree until the runtime's own re-run of that contract is green, then deliver the result as a pull request |
175
175
  | [`car coder-ab`](#car-coder-ab) | A/B-test CAR's coder against an external agent (Codex / Claude Code) over a corpus, and grow that corpus from git history — the productionized dogfooding loop (docs/proposals/coder-ab-dogfood.md) |
176
176
  | [`car keys`](#car-keys) | Store cloud-provider API keys in the OS keychain, so a native-app user never sets an environment variable (docs/proposals/native-secrets-no-env.md). The key is read env-first, keychain-fallback by the runtime |
177
+ | [`car selfheal`](#car-selfheal) | Inspect and operate the daemon's deterministic self-healing detector |
177
178
  | [`car daemon`](#car-daemon) | Start the daemon server (delegates to car-server binary) |
178
179
  | [`car models`](#car-models) | Manage local inference models |
179
180
  | [`car setup`](#car-setup) | Set up the right model for this machine — detect hardware, recommend, and install. Run with no flags for an interactive walkthrough |
@@ -647,6 +648,90 @@ Options:
647
648
  -h, --help Print help
648
649
  ```
649
650
 
651
+ ### car selfheal
652
+
653
+ ```text
654
+ Inspect and operate the daemon's deterministic self-healing detector
655
+
656
+ Usage: car selfheal <COMMAND>
657
+
658
+ Commands:
659
+ status Show cadence, active counts, and the route resolved by the last tick
660
+ list List active detections, optionally filtered by kind, severity, or time
661
+ show Print the trusted local issue document for one detection
662
+ dismiss Dismiss one active detection by dedup key
663
+ run Run one deterministic detection tick now
664
+ help Print this message or the help of the given subcommand(s)
665
+
666
+ Options:
667
+ -h, --help Print help
668
+ ```
669
+
670
+ #### car selfheal status
671
+
672
+ ```text
673
+ Show cadence, active counts, and the route resolved by the last tick
674
+
675
+ Usage: car selfheal status
676
+
677
+ Options:
678
+ -h, --help Print help
679
+ ```
680
+
681
+ #### car selfheal list
682
+
683
+ ```text
684
+ List active detections, optionally filtered by kind, severity, or time
685
+
686
+ Usage: car selfheal list [OPTIONS]
687
+
688
+ Options:
689
+ --kind <KIND> Detection kind (metrics_alert, agent_gave_up, agent_log_error,
690
+ agent_silently_idle, recurring_tool_failure, capability_miss)
691
+ --severity <SEVERITY> Severity (warning or critical)
692
+ --since <SINCE> Only detections observed at or after this RFC3339 timestamp
693
+ -h, --help Print help
694
+ ```
695
+
696
+ #### car selfheal show
697
+
698
+ ```text
699
+ Print the trusted local issue document for one detection
700
+
701
+ Usage: car selfheal show <DEDUP_KEY>
702
+
703
+ Arguments:
704
+ <DEDUP_KEY> Detection SHA-256 dedup key
705
+
706
+ Options:
707
+ -h, --help Print help
708
+ ```
709
+
710
+ #### car selfheal dismiss
711
+
712
+ ```text
713
+ Dismiss one active detection by dedup key
714
+
715
+ Usage: car selfheal dismiss <DEDUP_KEY>
716
+
717
+ Arguments:
718
+ <DEDUP_KEY> Detection SHA-256 dedup key
719
+
720
+ Options:
721
+ -h, --help Print help
722
+ ```
723
+
724
+ #### car selfheal run
725
+
726
+ ```text
727
+ Run one deterministic detection tick now
728
+
729
+ Usage: car selfheal run
730
+
731
+ Options:
732
+ -h, --help Print help
733
+ ```
734
+
650
735
  ### car daemon
651
736
 
652
737
  ```text
@@ -181,6 +181,35 @@ Proposed → Validated → Executing → Succeeded
181
181
 
182
182
  `ActionStatus` is observable through the event log, not part of the input contract.
183
183
 
184
+ #### Execution outcome event data
185
+
186
+ Every per-action `ActionFailed` event carries:
187
+
188
+ - `params_digest`: lowercase SHA-256 of the RFC 8785/JCS-canonicalized action
189
+ `parameters` object. The event never copies raw parameters; consumers join
190
+ through `proposal_id` + `action_id` to the authoritative `ProposalReceived`
191
+ record and can use the digest to detect a mismatch.
192
+ - `expected_effects`: the action's declared expected-effects object, unchanged.
193
+ - `error_class`: one of `timeout`, `rejected_by_policy`, `tool_error`,
194
+ `validation`, or `unknown`.
195
+
196
+ `ActionSucceeded` carries `params_digest` and `expected_effects` too, making the
197
+ success/failure join symmetric without adding an error classification to a
198
+ successful call.
199
+
200
+ The normalized error mapping is intentionally low-cardinality:
201
+
202
+ | `error_class` | Mapping |
203
+ |---|---|
204
+ | `timeout` | the engine's action deadline expired, or the daemon-to-host tool callback reported its own timeout |
205
+ | `rejected_by_policy` | a dispatch-time tool guard returned the stable `denied by policy:` or `rejected by policy:` prefix |
206
+ | `validation` | post-dispatch output or callback-state JCS/I-JSON, state-key-set, or serialization validation failed |
207
+ | `tool_error` | any other error returned while dispatching a tool action |
208
+ | `unknown` | a post-dispatch failure on an action with no tool |
209
+
210
+ Normal action/schema/policy admission failures happen before execution and are
211
+ `ActionRejected`, not `ActionFailed`, so this mapping does not reclassify them.
212
+
184
213
  ---
185
214
 
186
215
  ## Precondition
@@ -220,6 +249,7 @@ Registered when a tool is added to the runtime. Carries everything the runtime n
220
249
  ```jsonc
221
250
  {
222
251
  "name": "deploy",
252
+ "source": "user_defined",
223
253
  "description": "Deploys an artifact to a target environment.",
224
254
  "parameters": {
225
255
  "type": "object",
@@ -238,6 +268,7 @@ Registered when a tool is added to the runtime. Carries everything the runtime n
238
268
  | Field | Type | Required | Default | Notes |
239
269
  |-------|------|----------|---------|-------|
240
270
  | `name` | string | **yes** | — | unique within a runtime |
271
+ | `source` | `builtin \| user_defined \| subprocess \| mcp` | no | `user_defined` | stable origin category assigned by the runtime; MCP server detail remains registry-private |
241
272
  | `description` | string | no | `""` | human-readable; included in tool catalog |
242
273
  | `parameters` | JSON Schema | no | `{}` | validated by the runtime before dispatch |
243
274
  | `returns` | JSON Schema | no | none | validated against tool return value when set |
@@ -562,15 +593,21 @@ allow = ["staging", "preview"] # any other target — or none at all — is de
562
593
  | `allow` | no | permitted values; defaults to empty, which denies every call |
563
594
 
564
595
  #### `deny_tool_param_matching`
565
- The content counterpart to `deny_tool_param`, for prohibitions no fixed substring expresses — credential shapes, account numbers, an address family. `matches` is a regex over the string-coerced parameter value. The match is **unanchored**, so the pattern fires anywhere in the value; anchor it with `^`/`$` when that matters. Like `deny_tool_param`, an absent parameter is not a violation — use `allow_tool_param` when absence itself must be refused.
596
+ The content counterpart to `deny_tool_param`, for conditions no fixed substring expresses — credential shapes, account numbers, an address family, or an open-ended trusted prefix. `matches` is a regex over the string-coerced parameter value. The match is **unanchored**, so the pattern fires anywhere in the value; anchor it with `^`/`$` when that matters.
566
597
 
567
- The pattern is compiled once when the rule set is applied, not per action. A pattern that fails to compile **denies every call to that tool** rather than disappearing, matching the loader's loud-error posture.
598
+ By default a regex match denies and an absent parameter does not. Set `negate = true` for the "unless" form: a mismatch denies, and an absent parameter also denies because nothing proves the required pattern. The pattern is compiled once when the rule set is applied, not per action. A pattern that fails to compile **denies every call to that tool** rather than disappearing, matching the loader's loud-error posture.
568
599
 
569
600
  ```toml
570
601
  [[deny_tool_param_matching]]
571
602
  tool = "http_request"
572
603
  param = "body"
573
604
  matches = "sk-[A-Za-z0-9]{20,}" # never let an API-key-shaped string leave in a body
605
+
606
+ [[deny_tool_param_matching]]
607
+ tool = "docker.rm"
608
+ param = "name"
609
+ matches = "^parslee-"
610
+ negate = true # deny unless the name has the trusted prefix
574
611
  ```
575
612
 
576
613
  | Param | Required | Notes |
@@ -578,6 +615,7 @@ matches = "sk-[A-Za-z0-9]{20,}" # never let an API-key-shaped string leave in
578
615
  | `tool` | yes | tool name the rule applies to |
579
616
  | `param` | yes | parameter key inspected on the action |
580
617
  | `matches` | yes | regex source; unanchored; an uncompilable pattern denies the tool outright |
618
+ | `negate` | no | defaults to `false`; when `true`, deny mismatch or absence instead of match |
581
619
 
582
620
  #### `rate_limit_tool`
583
621
  A sliding-window cap on how often `tool` may be called. The call is denied when admitting it would make it the `max_calls + 1`-th call to `tool` within the trailing `interval_secs`. `max_calls = 0` denies every call. This bounds how much of a side effect an agent can produce in a stretch of wall-clock time, independently of whether any single call is legitimate.
package/index.d.ts CHANGED
@@ -32,6 +32,61 @@
32
32
  * `ws://127.0.0.1:9100`).
33
33
  */
34
34
 
35
+ /** Agent-loop tool declaration. `timeoutMs` becomes the action budget. */
36
+ export interface AgentToolSchema {
37
+ name: string;
38
+ description: string;
39
+ parameters: Record<string, unknown>;
40
+ timeoutMs?: number;
41
+ }
42
+
43
+ export interface AgentToolContext {
44
+ signal: AbortSignal;
45
+ timeoutMs?: number;
46
+ }
47
+
48
+ export type AgentTool = (
49
+ params: Record<string, unknown>,
50
+ context: AgentToolContext,
51
+ ) => unknown | Promise<unknown>;
52
+
53
+ export type AgentOutcomeStatus =
54
+ | 'success' | 'partial_success' | 'done' | 'give_up' | 'timeout' | 'failure';
55
+
56
+ export interface AgentOutcome {
57
+ status: AgentOutcomeStatus;
58
+ summary: string;
59
+ evidence: Array<{ kind: string; description: string; data: unknown }>;
60
+ metrics: {
61
+ turns: number;
62
+ tool_calls: number;
63
+ actions_succeeded: number;
64
+ actions_failed: number;
65
+ };
66
+ tools_called: string[];
67
+ timestamp: string;
68
+ }
69
+
70
+ /** Declarative input consumed by `car-runtime/agent-loop`. */
71
+ export interface AgentLoopConfig {
72
+ agentId?: string;
73
+ agentName: string;
74
+ identity: string;
75
+ toolSchemas?: AgentToolSchema[];
76
+ tools?: Record<string, AgentTool>;
77
+ policies?: Array<[string, string, string?, string?, string?, string?]>;
78
+ defaultModel?: string | null;
79
+ maxTokens?: number;
80
+ maxTurns?: number;
81
+ targetOutcome?: string;
82
+ standingGoal?: string | null;
83
+ intervalSecs?: number;
84
+ }
85
+
86
+ export interface AgentLoopOptions {
87
+ maxTurns?: number;
88
+ }
89
+
35
90
  /** Persistent runtime instance with state, memory, tools, and policies. */
36
91
  /**
37
92
  * Optional settings for `coderStart`. Every field is independently omittable;
@@ -58,6 +113,24 @@ export interface CoderStartOptions {
58
113
  * hypothesis, the other buys a retry.
59
114
  */
60
115
  transientRetries?: number | undefined | null;
116
+ /**
117
+ * Farm a **foreman** session's subtasks across every reachable CAR instance
118
+ * that can serve this repository, instead of this machine alone. The
119
+ * merge-verify gate and delivery stay on the orchestrating host — a peer
120
+ * returns a patch and this host gates it — so a distributed run still
121
+ * produces a gated pull request.
122
+ *
123
+ * Only the foreman engine decomposes a goal into subtasks, so any other
124
+ * engine runs locally and says so. Off by default: it spends agent quota on
125
+ * other people's machines.
126
+ */
127
+ distributed?: boolean | undefined | null;
128
+ /**
129
+ * Restrict placement to these instances by name. Empty or omitted means
130
+ * every instance that reports it can serve the repository. Ignored unless
131
+ * `distributed` is set.
132
+ */
133
+ workers?: Array<string> | undefined | null;
61
134
  /**
62
135
  * A `coder.discuss` conversation this run was distilled from. Its agreed
63
136
  * constraints ride into contract derivation, so a rule stated once in the
@@ -68,9 +141,41 @@ export interface CoderStartOptions {
68
141
  discussionId?: string | undefined | null;
69
142
  }
70
143
 
144
+ export interface DaemonRpcError extends Error {
145
+ /** Numeric JSON-RPC error code returned by the daemon. */
146
+ code: number;
147
+ /** Daemon-provided diagnostic text. */
148
+ message: string;
149
+ /** Optional JSON-RPC error data returned by the daemon. */
150
+ data?: unknown;
151
+ }
152
+
71
153
  export class CarRuntime {
72
154
  constructor();
73
155
 
156
+ /**
157
+ * Invoke any daemon JSON-RPC method with a JSON-encoded params value.
158
+ * `daemonCall` is the call-by-name escape hatch; use the typed wrappers as the primary API.
159
+ * The result is returned as JSON. Daemon rejections are `DaemonRpcError`;
160
+ * transport failures reject without a synthetic numeric code.
161
+ */
162
+ daemonCall(method: string, paramsJson: string): Promise<string>;
163
+
164
+ /** Host-management-token twin of `daemonCall`; the method allowlist remains enforced. */
165
+ daemonCallHostManagement(method: string, paramsJson: string): Promise<string>;
166
+
167
+ /** Register a server-initiated JSON-RPC request handler. */
168
+ registerDaemonHandler(
169
+ method: string,
170
+ handler: (paramsJson: string) => Promise<string>,
171
+ ): void;
172
+
173
+ /** Register a server-initiated JSON-RPC notification handler. */
174
+ registerDaemonNotificationHandler(
175
+ method: string,
176
+ handler: (paramsJson: string) => void,
177
+ ): void;
178
+
74
179
  // --- Memory persistence ---
75
180
 
76
181
  /**
@@ -142,7 +247,54 @@ export class CarRuntime {
142
247
  adapter?: string,
143
248
  verifyCommand?: Array<string>,
144
249
  unionVerifyCommand?: Array<string>,
145
- maxAttempts?: number
250
+ maxAttempts?: number,
251
+ distributed?: boolean,
252
+ workers?: Array<string>
253
+ ): Promise<string>;
254
+
255
+ // --- Fleet ---
256
+
257
+ /**
258
+ * This instance's agents, capabilities, and models — one `InstanceInventory`
259
+ * JSON object. The same report peers receive over A2A, plus this session's
260
+ * own registered tools and learned skills.
261
+ */
262
+ fleetInventory(): Promise<string>;
263
+
264
+ /**
265
+ * Every agent, capability, and model across this daemon and every reachable
266
+ * CAR instance, folded so one row names every instance that offers it.
267
+ * `includeRemote` defaults to true. `timeoutMs` bounds each peer
268
+ * individually: a sleeping machine appears as an unreachable row carrying the
269
+ * reason, never a missing one. Returns `FleetComposite` JSON.
270
+ */
271
+ fleetComposite(includeRemote?: boolean, timeoutMs?: number): Promise<string>;
272
+
273
+ /** Whether this instance takes farmed-out coding work. `{ config, profile }` JSON. */
274
+ fleetWorkerGet(): Promise<string>;
275
+
276
+ /**
277
+ * Enroll (or withdraw) this instance as a fleet worker. **Operator-only, and
278
+ * a real grant**: enrolling lets a trusted peer run a coding CLI against the
279
+ * checkouts named in `repos`. Only the fields supplied change.
280
+ *
281
+ * The limits belong to this machine, not the caller: `dispatchesPerHour`
282
+ * budgets one peer's spend (concurrency is not a spend bound),
283
+ * `maxSubtaskSecs` caps the timeout a sender asks for, and `allowedTools` is
284
+ * intersected with whatever the dispatch requests. `fetchMissingBase` makes
285
+ * this machine a **runner**: rather than decline a base commit it lacks, it
286
+ * fetches from `fetchRemote` (its own, default `origin`).
287
+ */
288
+ fleetWorkerSet(
289
+ acceptsWork?: boolean,
290
+ repos?: Array<string>,
291
+ maxParallel?: number,
292
+ localParallel?: number,
293
+ dispatchesPerHour?: number,
294
+ maxSubtaskSecs?: number,
295
+ allowedTools?: Array<string>,
296
+ fetchMissingBase?: boolean,
297
+ fetchRemote?: string
146
298
  ): Promise<string>;
147
299
 
148
300
  // --- Tools & policies ---
@@ -152,7 +304,8 @@ export class CarRuntime {
152
304
 
153
305
  /**
154
306
  * The tools currently registered on this runtime, as a JSON array of full
155
- * `ToolSchema` objects sorted by name.
307
+ * `ToolSchema` objects sorted by name. Every schema includes its runtime-
308
+ * assigned `source` (`builtin|user_defined|subprocess|mcp`).
156
309
  *
157
310
  * Counterpart to `registerTool` / `registerToolSchema`, which had none: a
158
311
  * caller could add tools but never ask what was actually in effect, so a
@@ -296,7 +449,8 @@ export class CarRuntime {
296
449
  // --- Memory / Facts (graph-backed) ---
297
450
 
298
451
  /**
299
- * Add a fact. `kind` is typically "pattern" or "constraint".
452
+ * Add a fact. `kind` is typically "pattern" or "constraint". Optional
453
+ * `factId`, ordered `tags`, and `source` are preserved by the daemon.
300
454
  *
301
455
  * In Daemon mode, rejects with the daemon-unreachable error
302
456
  * instead of silently returning 0 (#146).
@@ -306,9 +460,16 @@ export class CarRuntime {
306
460
  body: string,
307
461
  kind: string,
308
462
  confidence?: number | null,
463
+ factId?: string | null,
464
+ tags?: string[] | null,
465
+ source?: string | null,
309
466
  ): Promise<number>;
310
467
 
311
- /** Query facts via graph spreading activation. Returns a JSON array. */
468
+ /**
469
+ * Query facts via graph spreading activation. Returns a JSON array whose
470
+ * rows include `fact_id` (null for graph nodes without one), `subject`,
471
+ * `body`, `kind`, `confidence`, `tags`, and `source`.
472
+ */
312
473
  queryFacts(query: string, k?: number | null): string;
313
474
 
314
475
  /**
@@ -462,7 +623,37 @@ export class CarRuntime {
462
623
  syncStatus(requestJson: string): Promise<string>;
463
624
  /** `sync.append` — record an op on any surface: `{ surface, payload, scope? }` (B6). */
464
625
  syncAppend(requestJson: string): Promise<string>;
465
- /** `agents.peers` — the agents this runtime can message, from the daemon's live connection table. */
626
+ /** `host.agents` — current host agent registry snapshot. */
627
+ hostAgents(): Promise<string>;
628
+ /** `host.events` — recent host events, newest last; omit `limit` for the daemon default. */
629
+ hostEvents(limit?: number): Promise<string>;
630
+ /** `host.approvals` — pending host approvals. */
631
+ hostApprovals(): Promise<string>;
632
+ /** `host.register_agent` — register an agent on this connection. */
633
+ hostRegisterAgent(requestJson: string): Promise<string>;
634
+ /** `host.unregister_agent` — unregister an agent owned by this connection. */
635
+ hostUnregisterAgent(requestJson: string): Promise<string>;
636
+ /** `host.set_status` — publish status for an agent owned by this connection. */
637
+ hostSetStatus(requestJson: string): Promise<string>;
638
+ /** `host.register_device` — register a device on this connection. */
639
+ hostRegisterDevice(requestJson: string): Promise<string>;
640
+ /** `host.update_device` — update a device owned by this connection. */
641
+ hostUpdateDevice(requestJson: string): Promise<string>;
642
+ /** `host.devices` — current host device registry snapshot. */
643
+ hostDevices(): Promise<string>;
644
+ /** `host.notify` — emit a user-facing host notification. */
645
+ hostNotify(requestJson: string): Promise<string>;
646
+ /** `host.request_approval` — request approval for a gated action. */
647
+ hostRequestApproval(requestJson: string): Promise<string>;
648
+ /** `host.resolve_approval` — resolve one pending host approval. */
649
+ hostResolveApproval(requestJson: string): Promise<string>;
650
+ // `host.subscribe` event delivery is deferred to the callback-aware
651
+ // daemon-session API; subscribing without a consumer would drop the stream.
652
+ /**
653
+ * `agents.peers` — visible peers as JSON. Each row distinguishes the
654
+ * kind-level `can_receive` capability from the current `reachable` delivery
655
+ * preflight; the send remains authoritative.
656
+ */
466
657
  agentsPeers(requestJson: string): Promise<string>;
467
658
  /** `agents.message` — send text to one peer: `{ to, body, summary? }`. The sender is derived server-side. */
468
659
  agentsMessage(requestJson: string): Promise<string>;
@@ -795,6 +986,21 @@ export class CarRuntime {
795
986
  * of silently serving a different model (Parslee-ai/car#888). Absent on
796
987
  * the common path.
797
988
  *
989
+ * `fallback_from` is an ARRAY of every candidate the chain moved past,
990
+ * in the order it tried them: `[{ candidate, reason }, ...]`, where
991
+ * `reason` is one of `"credential_rejected"`, `"credential_absent"`,
992
+ * `"rate_limited"`, `"quota_exhausted"`, `"timed_out"` or `"failed"`.
993
+ * Absent when the first candidate served. Before this, a run whose
994
+ * backbone changed because of a rate limit or a timeout recorded no
995
+ * cause anywhere, so a surprising result got attributed to the code
996
+ * rather than to the model swap (Parslee-ai/car#1351).
997
+ *
998
+ * `reason` is classified from the runtime's typed error, not from error
999
+ * prose. `"credential_rejected"` is deliberately BROADER than
1000
+ * `auth_fallback_from`: it covers a provider refusing an API key, whose
1001
+ * remedy is to fix the key, not to sign in. Do not derive one field
1002
+ * from the other.
1003
+ *
798
1004
  * **Note:** intent is not exposed on the tracked path until the
799
1005
  * positional argument list is converted to an options object —
800
1006
  * this method already takes 9 positional parameters and adding
@@ -1462,7 +1668,10 @@ export class CarRuntime {
1462
1668
 
1463
1669
  /** Structured audit query over the event log (G2). `queryJson` is an
1464
1670
  * EventQuery object (kinds/actionId/proposalId/since/until/dataMatches/limit);
1465
- * returns `{count, events}` as a JSON string, most-recent-first. */
1671
+ * returns `{count, events}` as a JSON string, most-recent-first.
1672
+ * `ActionFailed.data` includes `params_digest`, `expected_effects`, and
1673
+ * `error_class` (`timeout|rejected_by_policy|tool_error|validation|unknown`),
1674
+ * never raw parameters. `ActionSucceeded.data` includes the first two. */
1466
1675
  eventQuery(queryJson: string): Promise<string>;
1467
1676
 
1468
1677
  /** Get/set the event-log retention policy (G2). Pass a
@@ -1498,21 +1707,39 @@ export class CarRuntime {
1498
1707
  * `cost_overage` alert. */
1499
1708
  metricsAlerts(thresholdsJson?: string): Promise<string>;
1500
1709
 
1501
- /** Watch-only self-heal detector status as JSON: cadence, last tick, source
1502
- * `route`, validated `source_checkout` or `refusal_reason`, detector IDs,
1503
- * active/dismissed counts, and `filing_mode: "watch-only"`. */
1710
+ /** Self-healing repair loop status: enabled/why-not, cadence, targets,
1711
+ * rejected targets, review panel, engine. */
1712
+ healStatus(): Promise<string>;
1713
+ /** Run one self-healing repair sweep now. May open a pull request; never merges. */
1714
+ healRun(): Promise<string>;
1715
+ /** Self-heal status as JSON: cadence, `auto_fix_enabled`, `max_concurrent`,
1716
+ * `max_per_day`, `max_rounds_per_key`, optional `auto_fix_refusal_reason`,
1717
+ * last tick, source route/refusal,
1718
+ * detector counts, and `filing_mode` (`watch-only` or `pr-only`). */
1719
+
1504
1720
  selfhealStatus(): Promise<string>;
1505
1721
 
1506
1722
  /** List active (not dismissed) self-heal detections as JSON. Each includes
1507
- * `route` and an optional `local_issue_path`. `queryJson` optionally carries
1508
- * `kind`, `severity`, `since`, `offset`, and `limit` (bounded to 500). */
1723
+ * `route` and an optional `local_issue_path`. Recurring tool failures add
1724
+ * `eligible`, a secret-safe `reconstructed_call` (`tool` plus exact `params`),
1725
+ * optional owner-private `reconstructed_call_path`, `auto_fix_attempts`,
1726
+ * `auto_fix_exhausted`, `auto_fix_in_progress`, and
1727
+ * `last_auto_fix_attempt` (including `exit_code` and `failure_class`). Remote
1728
+ * deduplication adds `auto_fix_awaiting_review`, `auto_fix_parked`,
1729
+ * `remote_pr_number`, and `remote_pr_url`.
1730
+ * `queryJson` carries optional `kind`, `severity`,
1731
+ * `since`, `offset`, and `limit` (max 500). */
1509
1732
  selfhealDetections(queryJson?: string): Promise<string>;
1510
1733
 
1511
- /** Append a dismissal marker for a stable detection dedup key. This does not
1512
- * delete history, file an issue, use network, or remediate anything. */
1734
+ /** Append a dismissal marker for a stable detection dedup key without
1735
+ * deleting history. */
1513
1736
  selfhealDismiss(dedupKey: string): Promise<string>;
1514
1737
 
1515
- /** Run one non-overlapping watch-only detection tick immediately. */
1738
+ /** Start one bounded template-owned coder round for an eligible recurring
1739
+ * tool failure. Returns the durable attempt result as JSON. */
1740
+ selfhealFix(dedupKey: string): Promise<string>;
1741
+
1742
+ /** Run one non-overlapping detection tick and default-on auto-fix hook. */
1516
1743
  selfhealRun(): Promise<string>;
1517
1744
 
1518
1745
  /** Execution log counts and approximate retained native bytes. Returns JSON. */
@@ -1611,9 +1838,10 @@ export class CarRuntime {
1611
1838
  * automatically.
1612
1839
  *
1613
1840
  * Tools registered via the schemaless `registerTool(name)` bypass type
1614
- * validation; this is the opt-in upgrade path.
1841
+ * validation; this is the opt-in upgrade path. The daemon assigns
1842
+ * `source = "user_defined"`; callers cannot claim another origin.
1615
1843
  *
1616
- * `schemaJson` matches:
1844
+ * `schemaJson` carries the caller-settable fields:
1617
1845
  * ```json
1618
1846
  * {
1619
1847
  * "name": "read_file",
@@ -2596,10 +2824,21 @@ export function reapStaleAgents(
2596
2824
  * register an `agent.chat` handler to serve conversational turns, and/or
2597
2825
  * a `registerToolHandler` for explicit tool `data` parts.
2598
2826
  *
2827
+ * `allow_non_loopback_bind` (boolean, default `false`) is required to bind
2828
+ * anything but loopback. This listener serves NO authentication — there
2829
+ * is no auth parameter, and its router is `NoAuth` — and without
2830
+ * `share_session_runtime` its runtime registers the agent-basics
2831
+ * filesystem tools, so a reachable bind publishes `write_file` /
2832
+ * `edit_file` to anyone who can route to the port. A wildcard bind
2833
+ * (`0.0.0.0:...`) is refused too. To be reachable by other CAR daemons
2834
+ * you want the peer-authenticated, messaging-only listener `car-server`
2835
+ * already runs by default, not this.
2836
+ *
2599
2837
  * Returns `'{"bound":"127.0.0.1:8731"}'` on success. Errors if a
2600
- * server is already running, the bind fails, `share_session_runtime`
2601
- * is set but no session runtime is available (e.g. invoked from a
2602
- * non-WS path), or `paramsJson` is malformed.
2838
+ * server is already running, the bind fails, the bind is non-loopback
2839
+ * without `allow_non_loopback_bind`, `share_session_runtime` is set but no
2840
+ * session runtime is available (e.g. invoked from a non-WS path), or
2841
+ * `paramsJson` is malformed.
2603
2842
  */
2604
2843
  export function startA2AServer(rt: CarRuntime, paramsJson: string): Promise<string>;
2605
2844
 
package/package.json CHANGED
@@ -1,9 +1,23 @@
1
1
  {
2
2
  "name": "car-runtime",
3
- "version": "0.52.1",
3
+ "version": "0.53.0",
4
4
  "description": "Common Agent Runtime — a deterministic execution layer for AI agents",
5
5
  "main": "index.js",
6
6
  "types": "index.d.ts",
7
+ "exports": {
8
+ ".": {
9
+ "types": "./index.d.ts",
10
+ "require": "./index.js",
11
+ "default": "./index.js"
12
+ },
13
+ "./agent-loop": {
14
+ "types": "./agent-loop.d.ts",
15
+ "import": "./agent-loop.mjs",
16
+ "require": "./agent-loop.js",
17
+ "default": "./agent-loop.js"
18
+ },
19
+ "./package.json": "./package.json"
20
+ },
7
21
  "bin": {
8
22
  "car-server": "bin/car-server"
9
23
  },
@@ -36,6 +50,9 @@
36
50
  "files": [
37
51
  "index.js",
38
52
  "index.d.ts",
53
+ "agent-loop.js",
54
+ "agent-loop.mjs",
55
+ "agent-loop.d.ts",
39
56
  "install.js",
40
57
  "assets.json",
41
58
  "bin/car-server",
@@ -45,7 +62,8 @@
45
62
  ],
46
63
  "scripts": {
47
64
  "install": "node install.js",
48
- "prepack": "node sync-docs.js"
65
+ "prepack": "node sync-docs.js",
66
+ "test": "node --test test/install.test.js"
49
67
  },
50
68
  "license": "SEE LICENSE IN LICENSE",
51
69
  "private": false