car-runtime 0.52.1 → 0.53.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +30 -0
- package/agent-loop.d.ts +30 -0
- package/agent-loop.js +578 -0
- package/agent-loop.mjs +4 -0
- package/docs/ASSISTANT.md +1 -0
- package/docs/CLI.md +88 -3
- package/docs/agent-ir-spec.md +40 -2
- package/index.d.ts +258 -19
- package/package.json +20 -2
package/README.md
CHANGED
|
@@ -74,6 +74,36 @@ const result = await executeProposal(rt, proposal, async (callJson) => {
|
|
|
74
74
|
});
|
|
75
75
|
```
|
|
76
76
|
|
|
77
|
+
## Packaged agent loop
|
|
78
|
+
|
|
79
|
+
Do not copy a harness into each agent. Import the versioned loop and keep the
|
|
80
|
+
project entry file declarative:
|
|
81
|
+
|
|
82
|
+
```javascript
|
|
83
|
+
import { main } from 'car-runtime/agent-loop';
|
|
84
|
+
|
|
85
|
+
main({
|
|
86
|
+
agentName: 'Lookup Agent',
|
|
87
|
+
identity: 'Use lookup before answering; never guess.',
|
|
88
|
+
toolSchemas: [{
|
|
89
|
+
name: 'lookup', description: 'Look up one key',
|
|
90
|
+
parameters: { type: 'object', properties: { key: { type: 'string' } }, required: ['key'] },
|
|
91
|
+
}],
|
|
92
|
+
tools: { lookup: async ({ key }) => ({ key, value: await lookup(key) }) },
|
|
93
|
+
policies: [],
|
|
94
|
+
maxTurns: 8,
|
|
95
|
+
});
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
`runAgent(config, goal)` is also exported for embedding and tests. `main(config)`
|
|
99
|
+
provides `--task`, `--json`, and `--serve`. The loop adds `finish`, statically
|
|
100
|
+
verifies every proposal before execution, preserves tool-call/result IDs,
|
|
101
|
+
honors per-schema `timeoutMs`, traces task and chat runs, and uses request-shaped
|
|
102
|
+
inference with prompt-cache breakpoints. It loads the native binding only when a
|
|
103
|
+
loop runs, so `require('car-runtime/agent-loop')` is safe for package discovery.
|
|
104
|
+
Type declarations ship as `agent-loop.d.ts` and shared config/outcome types in
|
|
105
|
+
`index.d.ts`.
|
|
106
|
+
|
|
77
107
|
Full API reference lives in [`index.d.ts`](./index.d.ts). The package also
|
|
78
108
|
ships a `docs/` directory (`node_modules/car-runtime/docs/`) with prose
|
|
79
109
|
reference docs — `SPEC.md`, `GUIDE.md`, `CLI.md`, `ASSISTANT.md`, `MCP.md`,
|
package/agent-loop.d.ts
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
import type {
|
|
2
|
+
AgentLoopConfig,
|
|
3
|
+
AgentLoopOptions,
|
|
4
|
+
AgentOutcome,
|
|
5
|
+
AgentToolSchema,
|
|
6
|
+
AgentTool,
|
|
7
|
+
CarRuntime,
|
|
8
|
+
} from './index';
|
|
9
|
+
|
|
10
|
+
export function runAgent(
|
|
11
|
+
config: AgentLoopConfig,
|
|
12
|
+
goal: string,
|
|
13
|
+
options?: AgentLoopOptions,
|
|
14
|
+
): Promise<AgentOutcome>;
|
|
15
|
+
|
|
16
|
+
export function runChatTurn(
|
|
17
|
+
runtime: CarRuntime,
|
|
18
|
+
config: AgentLoopConfig,
|
|
19
|
+
toolSchemas: AgentToolSchema[],
|
|
20
|
+
tools: Record<string, AgentTool>,
|
|
21
|
+
toolTimeouts: Record<string, number>,
|
|
22
|
+
sessionId: string,
|
|
23
|
+
messages: Record<string, unknown>[],
|
|
24
|
+
requestedModel?: string | null,
|
|
25
|
+
): Promise<string>;
|
|
26
|
+
|
|
27
|
+
export function main(config: AgentLoopConfig): Promise<void>;
|
|
28
|
+
|
|
29
|
+
declare const agentLoop: { main: typeof main; runAgent: typeof runAgent; runChatTurn: typeof runChatTurn };
|
|
30
|
+
export default agentLoop;
|
package/agent-loop.js
ADDED
|
@@ -0,0 +1,578 @@
|
|
|
1
|
+
'use strict';
|
|
2
|
+
|
|
3
|
+
// Versioned generic CAR agent loop. Agent projects import this package export;
|
|
4
|
+
// they do not copy or fork the propose -> verify -> execute -> observe cycle.
|
|
5
|
+
// The native binding is loaded lazily so tooling can inspect/require this
|
|
6
|
+
// subpath without first installing a platform binary.
|
|
7
|
+
let nativeApi = null;
|
|
8
|
+
let cancelHandlerInstalled = false;
|
|
9
|
+
function runtimeApi() {
|
|
10
|
+
if (nativeApi === null) nativeApi = require('./index.js');
|
|
11
|
+
if (!cancelHandlerInstalled && typeof nativeApi.registerToolCancelHandler === 'function') {
|
|
12
|
+
nativeApi.registerToolCancelHandler((requestId) => {
|
|
13
|
+
const ctl = ABORTS.get(requestId);
|
|
14
|
+
if (ctl) {
|
|
15
|
+
ABORTS.delete(requestId);
|
|
16
|
+
try { ctl.abort(new Error('tool callback reaped by daemon (budget exceeded)')); } catch { /* already aborted */ }
|
|
17
|
+
}
|
|
18
|
+
});
|
|
19
|
+
cancelHandlerInstalled = true;
|
|
20
|
+
}
|
|
21
|
+
return nativeApi;
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
// Process-wide abort registry keyed on the daemon's per-call `request_id`
|
|
25
|
+
// (Parslee-ai/car#264). When the daemon reaps a tool callback (the call
|
|
26
|
+
// exceeded its budget) it emits `tools.cancel` with the `request_id`; the
|
|
27
|
+
// handler below aborts the matching controller so a long child (e.g. a
|
|
28
|
+
// `claude -p` / `codex exec` driven by drive_cli) is killed instead of
|
|
29
|
+
// orphaned. Each tool callback registers its controller under its request_id
|
|
30
|
+
// and removes it on completion. Registered once at module load; a daemon
|
|
31
|
+
// without the cancel surface simply never fires it.
|
|
32
|
+
const ABORTS = new Map();
|
|
33
|
+
|
|
34
|
+
// A built-in sentinel tool every agent gets for free, so the loop always has a
|
|
35
|
+
// clean way to terminate with a final answer. Your agent calls finish(answer).
|
|
36
|
+
const FINISH_SCHEMA = {
|
|
37
|
+
name: 'finish',
|
|
38
|
+
description: 'Return the final answer to the user and stop. Call this exactly once, when the task is complete or cannot proceed.',
|
|
39
|
+
parameters: {
|
|
40
|
+
type: 'object',
|
|
41
|
+
properties: { answer: { type: 'string', description: 'The final answer or status for the user.' } },
|
|
42
|
+
required: ['answer'],
|
|
43
|
+
},
|
|
44
|
+
};
|
|
45
|
+
|
|
46
|
+
function buildProposal(modelUsed, toolCalls, toolTimeouts = {}) {
|
|
47
|
+
// Glue between the inference IR (ToolCall {id,name,arguments}) and the
|
|
48
|
+
// action IR (Action {id,type,tool,parameters,timeout_ms}). The only IR
|
|
49
|
+
// plumbing you owe.
|
|
50
|
+
//
|
|
51
|
+
// `timeout_ms` (Parslee-ai/car#259): a tool that shells out to a build,
|
|
52
|
+
// drives another CLI, or calls a slow API needs more than the daemon's
|
|
53
|
+
// default callback budget. Declare `timeoutMs` on the tool's schema and
|
|
54
|
+
// it flows here as the action's per-call budget — without it the call is
|
|
55
|
+
// reaped at the default. Omitted (undefined) when the schema sets none,
|
|
56
|
+
// so the action falls back to the daemon default.
|
|
57
|
+
return JSON.stringify({
|
|
58
|
+
source: modelUsed || 'model',
|
|
59
|
+
actions: toolCalls.map((tc, i) => ({
|
|
60
|
+
id: tc.id || `a${i}`,
|
|
61
|
+
type: 'tool_call',
|
|
62
|
+
tool: tc.name,
|
|
63
|
+
parameters: tc.arguments || {},
|
|
64
|
+
dependencies: [],
|
|
65
|
+
timeout_ms: toolTimeouts[tc.name],
|
|
66
|
+
})),
|
|
67
|
+
});
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* Build the `executeProposal` tool callback. ONE implementation, used by BOTH
|
|
72
|
+
* the one-shot `--task` loop (`runAgent`) and the `--serve` chat path
|
|
73
|
+
* (`runChatTurn`). They used to carry two copies, and the chat copy silently
|
|
74
|
+
* missed `timeout_ms` and the `request_id`/AbortController wiring — the path a
|
|
75
|
+
* registered CarHost agent actually runs. Do not re-fork it.
|
|
76
|
+
*
|
|
77
|
+
* The callback receives { tool, params, action_id, request_id, timeout_ms }.
|
|
78
|
+
* The `request_id` keys this call's abort controller (Parslee-ai/car#264) so
|
|
79
|
+
* the daemon's `tools.cancel` can kill a reaped child. The tool fn receives the
|
|
80
|
+
* budget (timeoutMs) and an AbortSignal as a second arg; tools that shell out
|
|
81
|
+
* should honor the signal (e.g. pass it to child_process / fetch) so a reap
|
|
82
|
+
* actually terminates their child. The callback's `timeout_ms` is authoritative
|
|
83
|
+
* — it is the budget of the action actually executing (Parslee-ai/car#259) — so
|
|
84
|
+
* it wins over the locally derived `toolTimeouts` map, which is the fallback.
|
|
85
|
+
* `onFinish` (optional) is called with the `finish` tool's answer.
|
|
86
|
+
*/
|
|
87
|
+
function makeToolCallback(tools, toolTimeouts = {}, onFinish = null) {
|
|
88
|
+
return async (callJson) => {
|
|
89
|
+
const { tool, params, request_id: requestId, timeout_ms: timeoutMs } = JSON.parse(callJson);
|
|
90
|
+
const fn = tools[tool];
|
|
91
|
+
if (!fn) throw new Error(`unknown tool: ${tool}`);
|
|
92
|
+
const ctl = new AbortController();
|
|
93
|
+
if (requestId) ABORTS.set(requestId, ctl);
|
|
94
|
+
try {
|
|
95
|
+
const out = await fn(params || {}, { signal: ctl.signal, timeoutMs: timeoutMs ?? toolTimeouts[tool] });
|
|
96
|
+
if (tool === 'finish' && onFinish) onFinish((out && out.answer) ?? '');
|
|
97
|
+
return JSON.stringify(out ?? {});
|
|
98
|
+
} finally {
|
|
99
|
+
// In a `finally` so a throwing tool can never leak its entry.
|
|
100
|
+
if (requestId) ABORTS.delete(requestId);
|
|
101
|
+
}
|
|
102
|
+
};
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
// ---- Prompt caching (Anthropic) --------------------------------------------
|
|
106
|
+
//
|
|
107
|
+
// The loop resends the WHOLE growing thread every turn, so an N-turn run bills
|
|
108
|
+
// the stable prefix N times at the full input rate — cost that grows with the
|
|
109
|
+
// square of the turn count. CAR's protocol layer already knows how to mark
|
|
110
|
+
// Anthropic cache breakpoints; it only needs `cache_control: true`, and the
|
|
111
|
+
// 9-positional-argument `inferTracked` has no slot to carry it. That is the
|
|
112
|
+
// whole reason every caller has been paying full price: the flag exists, the
|
|
113
|
+
// call shape could not reach it. `inferTrackedWithRequest` takes the options
|
|
114
|
+
// object (a JSON `GenerateRequest`) and can.
|
|
115
|
+
//
|
|
116
|
+
// BREAKPOINT PLACEMENT — we do not pick it; CAR does, and its choice is already
|
|
117
|
+
// the max-reuse one (car-inference/src/protocol.rs, AnthropicHandler::
|
|
118
|
+
// build_request_body). With `cache_control: true` it emits three of Anthropic's
|
|
119
|
+
// four allowed breakpoints:
|
|
120
|
+
// 1. the system block — identity plus the finish instruction, byte-identical
|
|
121
|
+
// for the entire run;
|
|
122
|
+
// 2. the LAST tool definition — so identity + every tool schema forms ONE
|
|
123
|
+
// cached prefix, the largest genuinely stable block this loop has;
|
|
124
|
+
// 3. the LAST message — the moving breakpoint. Anthropic serves the longest
|
|
125
|
+
// cached prefix it can find, so each turn writes only its own delta and
|
|
126
|
+
// READS everything the previous turn wrote. That is what turns quadratic
|
|
127
|
+
// re-billing into one write plus N cheap reads.
|
|
128
|
+
// The 4th breakpoint is deliberately left unused. It would buy a
|
|
129
|
+
// `context_stable_prefix` split of the system prompt, which helps only when the
|
|
130
|
+
// system prompt has a volatile tail; ours has none, so splitting it would
|
|
131
|
+
// shrink the cached block rather than grow it.
|
|
132
|
+
//
|
|
133
|
+
// TTL — `one_hour`, not the 5-minute default. Turns in an agentic loop are
|
|
134
|
+
// separated by real tool execution (browser drives, FMS reads, verification),
|
|
135
|
+
// which routinely exceeds five minutes; a 5-minute entry would expire mid-run
|
|
136
|
+
// and re-bill the entire prefix as a fresh write. A 1h write costs ~2x base
|
|
137
|
+
// input against ~1.25x for 5m, but that one-time 0.75x is far cheaper than a
|
|
138
|
+
// single full re-write of the prefix, and every surviving turn then reads at
|
|
139
|
+
// ~0.1x.
|
|
140
|
+
//
|
|
141
|
+
/** One request-shaped, cache-aware inference path for task and chat loops. */
|
|
142
|
+
async function inferTurn(rt, { model, maxTokens, toolSchemas, messages, toolChoice = 'auto' }) {
|
|
143
|
+
return JSON.parse(await rt.inferTrackedWithRequest(JSON.stringify({
|
|
144
|
+
prompt: '',
|
|
145
|
+
model: model ?? null,
|
|
146
|
+
params: {
|
|
147
|
+
max_tokens: maxTokens,
|
|
148
|
+
tool_choice: toolChoice,
|
|
149
|
+
strict_model: model != null,
|
|
150
|
+
cache_ttl: 'one_hour',
|
|
151
|
+
},
|
|
152
|
+
tools: toolSchemas,
|
|
153
|
+
messages,
|
|
154
|
+
cache_control: true,
|
|
155
|
+
})));
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
/**
|
|
159
|
+
* Run the agent once toward `goal`. Returns an AgentOutcome-shaped object:
|
|
160
|
+
* { status, summary, evidence[], metrics{}, tools_called[] }
|
|
161
|
+
* `status` is one of the six OutcomeStatus values (success|partial_success|
|
|
162
|
+
* done|give_up|timeout|failure). AgentOutcome is caller-built — CAR does not
|
|
163
|
+
* return it; we assemble it from what the loop observed.
|
|
164
|
+
*
|
|
165
|
+
* Run-trace lifecycle: each `runAgent` invocation is one run. Before the first
|
|
166
|
+
* proposal we bracket the run open with `rt.runsStart` (daemon mints a durable
|
|
167
|
+
* `run_id`); after the terminal AgentOutcome is assembled we close it with
|
|
168
|
+
* `rt.runsComplete`. Both are best-effort and never print, so the daemon traces
|
|
169
|
+
* the run for CarHost while older daemons (no runs.*) behave exactly as before.
|
|
170
|
+
*/
|
|
171
|
+
async function runAgent(config, goal, { maxTurns } = {}) {
|
|
172
|
+
const { CarRuntime, executeProposal } = runtimeApi();
|
|
173
|
+
const turnsCap = maxTurns ?? config.maxTurns ?? 8;
|
|
174
|
+
const toolSchemas = [...(config.toolSchemas || []), FINISH_SCHEMA];
|
|
175
|
+
const tools = { finish: ({ answer }) => ({ answer }), ...(config.tools || {}) };
|
|
176
|
+
// Per-tool execution budget (Parslee-ai/car#259): a tool schema may set
|
|
177
|
+
// `timeoutMs` (e.g. a CLI driver that runs for 180s); it flows onto each
|
|
178
|
+
// action so the daemon honors it instead of reaping at the default.
|
|
179
|
+
const toolTimeouts = Object.fromEntries(
|
|
180
|
+
toolSchemas.filter((s) => s && s.timeoutMs != null).map((s) => [s.name, s.timeoutMs]),
|
|
181
|
+
);
|
|
182
|
+
|
|
183
|
+
const rt = new CarRuntime();
|
|
184
|
+
// Best-effort: some daemon versions don't expose agents.register_basics. It
|
|
185
|
+
// only adds CAR's built-in utility tools, which a tool-declaring agent doesn't
|
|
186
|
+
// depend on, so a missing method must not abort the run.
|
|
187
|
+
try { await rt.registerAgentBasics(); } catch { /* unsupported on this daemon — fine */ }
|
|
188
|
+
for (const s of toolSchemas) await rt.registerTool(s.name);
|
|
189
|
+
// Guardrails are declarative policies, enforced in Rust BEFORE the tool fires
|
|
190
|
+
// — not prompt rules. Each entry is the argument list for registerPolicy.
|
|
191
|
+
for (const p of (config.policies || [])) await rt.registerPolicy(...p);
|
|
192
|
+
|
|
193
|
+
// Run-trace bracket (open). Tell the daemon a run is starting so CarHost can
|
|
194
|
+
// trace it: prompt -> CLI outcome -> verifier verdict -> AgentOutcome. The
|
|
195
|
+
// daemon mints a durable run_id and tags it as this session's current run
|
|
196
|
+
// BEFORE replying, so the per-turn recorder reads the right id; we await that
|
|
197
|
+
// ack before submitting any proposal. The owning agent_id resolves from
|
|
198
|
+
// CAR_AGENT_ID (the supervisor injects it) when supervised, else falls back to
|
|
199
|
+
// config.agentName for the unsupervised one-shot / run_scenarios path.
|
|
200
|
+
// Best-effort, exactly like registerAgentBasics: a daemon without runs.* (an
|
|
201
|
+
// older build) makes this throw, and the run must continue unchanged. NEVER
|
|
202
|
+
// print here — the last stdout line in --json mode must stay the AgentOutcome
|
|
203
|
+
// (run_scenarios.py parses it).
|
|
204
|
+
let runId = null;
|
|
205
|
+
try {
|
|
206
|
+
const started = JSON.parse(await rt.runsStart(JSON.stringify({
|
|
207
|
+
intent: goal,
|
|
208
|
+
agent_id: process.env.CAR_AGENT_ID || config.agentName,
|
|
209
|
+
agent_name: config.agentName,
|
|
210
|
+
outcome_description: config.targetOutcome ?? '',
|
|
211
|
+
})));
|
|
212
|
+
runId = started.run_id ?? null;
|
|
213
|
+
} catch { /* daemon lacks runs.* — degrade to untraced behavior */ }
|
|
214
|
+
|
|
215
|
+
// Local Qwen3 models default to "thinking" (they emit a long <think> block
|
|
216
|
+
// before acting), which on a multi-tool agent prompt can exceed the daemon's
|
|
217
|
+
// per-call read timeout — the agent loop then fails with a timeout instead of
|
|
218
|
+
// calling a tool. When pinned to a local model, append Qwen3's `/no_think`
|
|
219
|
+
// soft switch so it acts via tools directly. Only applied for clearly-local
|
|
220
|
+
// model ids (a null/router model may resolve to a cloud model, where thinking
|
|
221
|
+
// is fine and fast); cloud models simply ignore the token.
|
|
222
|
+
const isLocalModel = typeof config.defaultModel === 'string'
|
|
223
|
+
&& /^(mlx|qwen)\//.test(config.defaultModel);
|
|
224
|
+
const noThink = isLocalModel ? '\n\n/no_think' : '';
|
|
225
|
+
const messages = [
|
|
226
|
+
{ role: 'system', content: `${config.identity}\n\nWhen the task is complete or you cannot proceed, call the \`finish\` tool with a concise answer. Do not narrate; act via tools.${noThink}` },
|
|
227
|
+
{ role: 'user', content: goal },
|
|
228
|
+
];
|
|
229
|
+
|
|
230
|
+
const metrics = { turns: 0, tool_calls: 0, actions_succeeded: 0, actions_failed: 0 };
|
|
231
|
+
const toolsCalled = new Set();
|
|
232
|
+
let outcome = null; // null === no terminal AgentOutcome yet
|
|
233
|
+
let turns = 0;
|
|
234
|
+
|
|
235
|
+
while (outcome === null) {
|
|
236
|
+
if (++turns > turnsCap) {
|
|
237
|
+
outcome = mkOutcome('timeout', `hit ${turnsCap}-turn cap without finishing`,
|
|
238
|
+
[{ kind: 'stop_reason', description: 'max turns', data: { turnsCap } }], metrics, toolsCalled);
|
|
239
|
+
break;
|
|
240
|
+
}
|
|
241
|
+
metrics.turns = turns;
|
|
242
|
+
|
|
243
|
+
// 1. PROPOSE — multi-turn, tool-aware inference (NOT plain `infer`).
|
|
244
|
+
let tracked;
|
|
245
|
+
try {
|
|
246
|
+
tracked = await inferTurn(rt, {
|
|
247
|
+
model: config.defaultModel ?? null,
|
|
248
|
+
maxTokens: config.maxTokens ?? 1024,
|
|
249
|
+
toolSchemas,
|
|
250
|
+
messages,
|
|
251
|
+
});
|
|
252
|
+
} catch (e) {
|
|
253
|
+
outcome = mkOutcome('failure', `inference failed: ${e.message || e}`,
|
|
254
|
+
[{ kind: 'stop_reason', description: 'infer_tracked error', data: null }], metrics, toolsCalled);
|
|
255
|
+
break;
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
const calls = tracked.tool_calls || [];
|
|
259
|
+
if (calls.length === 0) {
|
|
260
|
+
// Model answered in prose with no tool call — treat as a neutral Done.
|
|
261
|
+
outcome = mkOutcome('done', tracked.text || 'no further actions',
|
|
262
|
+
[{ kind: 'self_assessment', description: tracked.text || '', data: null }], metrics, toolsCalled);
|
|
263
|
+
break;
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
// Normalize ids so assistant tool_calls and tool_results correlate across
|
|
267
|
+
// turns (local models often omit ids).
|
|
268
|
+
calls.forEach((c, i) => { c.id = c.id || `a${i}`; });
|
|
269
|
+
messages.push({ role: 'assistant', content: tracked.text || '', tool_calls: calls });
|
|
270
|
+
const idToTool = Object.fromEntries(calls.map((c) => [c.id, c.name]));
|
|
271
|
+
|
|
272
|
+
const proposal = buildProposal(tracked.model_used, calls, toolTimeouts);
|
|
273
|
+
|
|
274
|
+
// 2. VERIFY — static gate. Never execute an unverified proposal.
|
|
275
|
+
let check;
|
|
276
|
+
try {
|
|
277
|
+
check = JSON.parse(await rt.verifyProposal(proposal));
|
|
278
|
+
} catch (e) {
|
|
279
|
+
outcome = mkOutcome('failure', `verification failed: ${e.message || e}`,
|
|
280
|
+
[{ kind: 'stop_reason', description: 'verifyProposal error', data: null }], metrics, toolsCalled);
|
|
281
|
+
break;
|
|
282
|
+
}
|
|
283
|
+
if (!check.valid) {
|
|
284
|
+
// Feed the rejection back as tool_results so tool_use/tool_result stay
|
|
285
|
+
// paired, then let the model repair on the next turn.
|
|
286
|
+
for (const c of calls) {
|
|
287
|
+
messages.push({ role: 'tool_result', tool_use_id: c.id,
|
|
288
|
+
content: JSON.stringify({ error: `runtime rejected proposal: ${JSON.stringify(check.issues)}` }) });
|
|
289
|
+
}
|
|
290
|
+
continue;
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
// 3. EXECUTE — CAR owns the DAG, retries, timeouts, rollback. The callback
|
|
294
|
+
// receives { tool, params } (note: `params`).
|
|
295
|
+
let finishAnswer = null;
|
|
296
|
+
let result;
|
|
297
|
+
try {
|
|
298
|
+
result = JSON.parse(await executeProposal(rt, proposal,
|
|
299
|
+
makeToolCallback(tools, toolTimeouts, (a) => { finishAnswer = a; })));
|
|
300
|
+
} catch (e) {
|
|
301
|
+
outcome = mkOutcome('failure', `execution failed: ${e.message || e}`,
|
|
302
|
+
[{ kind: 'stop_reason', description: 'executeProposal error', data: null }], metrics, toolsCalled);
|
|
303
|
+
break;
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
// 4. OBSERVE — feed each ActionResult back as a tool_result turn.
|
|
307
|
+
// tools_called records tools that SUCCESSFULLY executed, so a policy-denied
|
|
308
|
+
// /failed tool is correctly absent (guardrail scenarios can assert
|
|
309
|
+
// tool_not_called against it).
|
|
310
|
+
for (const r of (result.results || [])) {
|
|
311
|
+
const ok = r.status === 'succeeded';
|
|
312
|
+
if (ok && idToTool[r.action_id]) toolsCalled.add(idToTool[r.action_id]);
|
|
313
|
+
messages.push({ role: 'tool_result', tool_use_id: r.action_id,
|
|
314
|
+
content: JSON.stringify(ok ? (r.output ?? {}) : { error: r.error }) });
|
|
315
|
+
metrics.tool_calls += 1;
|
|
316
|
+
if (ok) metrics.actions_succeeded += 1; else metrics.actions_failed += 1;
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
// 5. Terminal? finish() succeeded -> success.
|
|
320
|
+
if (finishAnswer !== null) {
|
|
321
|
+
outcome = mkOutcome('success', finishAnswer,
|
|
322
|
+
[{ kind: 'tool_result', description: 'finish called', data: { answer: finishAnswer } }], metrics, toolsCalled);
|
|
323
|
+
}
|
|
324
|
+
// else: loop for the next proposal.
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
// Run-trace bracket (close). Report the terminal AgentOutcome to the daemon so
|
|
328
|
+
// CarHost shows the run's final status and stops streaming it. Await the ack
|
|
329
|
+
// before returning (the connection may close right after) so a healthy run is
|
|
330
|
+
// never raced into `Incomplete`. Best-effort + never prints, mirroring the
|
|
331
|
+
// open bracket: an older daemon without runs.* (or one that never acked the
|
|
332
|
+
// start, leaving runId null) just skips this and behaves as before.
|
|
333
|
+
if (runId !== null) {
|
|
334
|
+
try {
|
|
335
|
+
await rt.runsComplete(JSON.stringify({ run_id: runId, outcome }));
|
|
336
|
+
} catch { /* daemon lacks runs.* — nothing to report to */ }
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
return outcome;
|
|
340
|
+
}
|
|
341
|
+
|
|
342
|
+
function mkOutcome(status, summary, evidence, metrics, toolsCalled) {
|
|
343
|
+
return { status, summary, evidence, metrics, tools_called: [...toolsCalled].sort(), timestamp: new Date().toISOString() };
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
// ---- CLI entrypoint -------------------------------------------------------
|
|
347
|
+
//
|
|
348
|
+
// node agent.mjs --task "<goal>" [--json] one-shot; prints outcome
|
|
349
|
+
// node agent.mjs --serve supervised/standing mode
|
|
350
|
+
//
|
|
351
|
+
// run_scenarios.py invokes the --task --json form.
|
|
352
|
+
|
|
353
|
+
function parseArgs(argv) {
|
|
354
|
+
const a = { task: null, json: false, serve: false };
|
|
355
|
+
for (let i = 0; i < argv.length; i++) {
|
|
356
|
+
if (argv[i] === '--task') a.task = argv[++i];
|
|
357
|
+
else if (argv[i] === '--json') a.json = true;
|
|
358
|
+
else if (argv[i] === '--serve') a.serve = true;
|
|
359
|
+
}
|
|
360
|
+
return a;
|
|
361
|
+
}
|
|
362
|
+
|
|
363
|
+
// ---- Chat serving (agent.chat surface) --------------------------------------
|
|
364
|
+
//
|
|
365
|
+
// In `--serve` mode the agent attaches to the daemon (the binding sends
|
|
366
|
+
// CAR_AGENT_ID + CAR_AGENT_TOKEN on session.auth) and registers an `agent.chat`
|
|
367
|
+
// handler. The daemon reverse-calls `agent.chat { session_id, prompt }` for
|
|
368
|
+
// every host `agents.chat`; we keep a per-session message THREAD and run the
|
|
369
|
+
// same propose→verify→execute loop per turn, streaming the reply back via
|
|
370
|
+
// `agent.chat.event`. Threads are ephemeral (process lifetime). The agent's
|
|
371
|
+
// declared policies still gate tool execution, so guardrails (draft-only, etc.)
|
|
372
|
+
// carry into the conversation.
|
|
373
|
+
|
|
374
|
+
/** Run one chat turn against a persistent `messages` thread; stream via chatEvent. */
|
|
375
|
+
async function runChatTurn(
|
|
376
|
+
rt, config, toolSchemas, tools, toolTimeouts, sessionId, messages, requestedModel = null,
|
|
377
|
+
) {
|
|
378
|
+
const { executeProposal } = runtimeApi();
|
|
379
|
+
const turnsCap = config.maxTurns ?? 8;
|
|
380
|
+
const selectedModel = typeof requestedModel === 'string' && requestedModel.trim()
|
|
381
|
+
? requestedModel
|
|
382
|
+
: null;
|
|
383
|
+
|
|
384
|
+
// Run-trace bracket (open) — the SAME bracket `runAgent` opens, on the chat
|
|
385
|
+
// path. Without it a chat-driven turn does real tool work that never appears
|
|
386
|
+
// in `runs.list` / `runs.get_trace`, so a host that dispatches through
|
|
387
|
+
// `agents.chat` has no daemon-side record of what it ran: CarHost shows the
|
|
388
|
+
// agent as merely "running", and an outer loop cannot read back the tool
|
|
389
|
+
// returns or the terminal outcome. Same best-effort contract as `runAgent`'s
|
|
390
|
+
// — never throws, never prints (the last stdout line in --json mode must stay
|
|
391
|
+
// the AgentOutcome), and an older daemon without runs.* behaves exactly as it
|
|
392
|
+
// did before.
|
|
393
|
+
const intent = [...messages].reverse().find((m) => m.role === 'user')?.content ?? 'chat turn';
|
|
394
|
+
let runId = null;
|
|
395
|
+
try {
|
|
396
|
+
const started = JSON.parse(await rt.runsStart(JSON.stringify({
|
|
397
|
+
intent,
|
|
398
|
+
agent_id: process.env.CAR_AGENT_ID || config.agentName,
|
|
399
|
+
agent_name: config.agentName,
|
|
400
|
+
outcome_description: config.targetOutcome ?? '',
|
|
401
|
+
})));
|
|
402
|
+
runId = started.run_id ?? null;
|
|
403
|
+
} catch { /* daemon lacks runs.* — degrade to untraced behavior */ }
|
|
404
|
+
|
|
405
|
+
const metrics = { turns: 0, tool_calls: 0, actions_succeeded: 0, actions_failed: 0 };
|
|
406
|
+
const toolsCalled = new Set();
|
|
407
|
+
let terminal = null;
|
|
408
|
+
let finalText = '';
|
|
409
|
+
for (let turns = 0; turns < turnsCap; turns++) {
|
|
410
|
+
metrics.turns = turns + 1;
|
|
411
|
+
// Same cached request form as the one-shot loop. A chat thread grows for
|
|
412
|
+
// the whole session, so it is the path that benefits most from the moving
|
|
413
|
+
// conversation breakpoint.
|
|
414
|
+
const tracked = await inferTurn(rt, {
|
|
415
|
+
model: selectedModel ?? config.defaultModel ?? null,
|
|
416
|
+
maxTokens: config.maxTokens ?? 1024,
|
|
417
|
+
toolSchemas,
|
|
418
|
+
messages,
|
|
419
|
+
});
|
|
420
|
+
const calls = tracked.tool_calls || [];
|
|
421
|
+
if (calls.length === 0) {
|
|
422
|
+
finalText = tracked.text || '';
|
|
423
|
+
messages.push({ role: 'assistant', content: finalText });
|
|
424
|
+
terminal = 'done';
|
|
425
|
+
break;
|
|
426
|
+
}
|
|
427
|
+
calls.forEach((c, i) => { c.id = c.id || `a${i}`; });
|
|
428
|
+
messages.push({ role: 'assistant', content: tracked.text || '', tool_calls: calls });
|
|
429
|
+
const idToTool = Object.fromEntries(calls.map((c) => [c.id, c.name]));
|
|
430
|
+
// Surface non-finish tool calls as progress so the host UI can show them.
|
|
431
|
+
for (const c of calls) {
|
|
432
|
+
if (c.name !== 'finish') await rt.chatEvent(sessionId, 'tool_call', c.name).catch(() => {});
|
|
433
|
+
}
|
|
434
|
+
const proposal = buildProposal(tracked.model_used, calls, toolTimeouts);
|
|
435
|
+
const check = JSON.parse(await rt.verifyProposal(proposal));
|
|
436
|
+
if (!check.valid) {
|
|
437
|
+
for (const c of calls) {
|
|
438
|
+
messages.push({ role: 'tool_result', tool_use_id: c.id,
|
|
439
|
+
content: JSON.stringify({ error: `runtime rejected proposal: ${JSON.stringify(check.issues)}` }) });
|
|
440
|
+
}
|
|
441
|
+
continue;
|
|
442
|
+
}
|
|
443
|
+
// Same callback the one-shot loop uses, so the declared per-action budget
|
|
444
|
+
// and the request_id/AbortController cancel wiring reach tools on the
|
|
445
|
+
// --serve path too.
|
|
446
|
+
let finishAnswer = null;
|
|
447
|
+
const result = JSON.parse(await executeProposal(rt, proposal,
|
|
448
|
+
makeToolCallback(tools, toolTimeouts, (a) => { finishAnswer = a; })));
|
|
449
|
+
for (const r of (result.results || [])) {
|
|
450
|
+
const ok = r.status === 'succeeded';
|
|
451
|
+
if (ok && idToTool[r.action_id]) toolsCalled.add(idToTool[r.action_id]);
|
|
452
|
+
messages.push({ role: 'tool_result', tool_use_id: r.action_id,
|
|
453
|
+
content: JSON.stringify(ok ? (r.output ?? {}) : { error: r.error }) });
|
|
454
|
+
metrics.tool_calls += 1;
|
|
455
|
+
if (ok) metrics.actions_succeeded += 1; else metrics.actions_failed += 1;
|
|
456
|
+
}
|
|
457
|
+
if (finishAnswer !== null) { finalText = finishAnswer; terminal = 'success'; break; }
|
|
458
|
+
}
|
|
459
|
+
if (finalText) await rt.chatEvent(sessionId, 'token', finalText).catch(() => {});
|
|
460
|
+
await rt.chatEvent(sessionId, 'done', finalText).catch(() => {});
|
|
461
|
+
|
|
462
|
+
// Run-trace bracket (close). The outcome is assembled exactly as `runAgent`
|
|
463
|
+
// assembles it, so a chat-driven run and a --task run are the same shape in
|
|
464
|
+
// the trace and a host reads one code path, not two.
|
|
465
|
+
if (runId !== null) {
|
|
466
|
+
const outcome = terminal === 'success'
|
|
467
|
+
? mkOutcome('success', finalText,
|
|
468
|
+
[{ kind: 'tool_result', description: 'finish called', data: { answer: finalText } }], metrics, toolsCalled)
|
|
469
|
+
: (terminal === 'done'
|
|
470
|
+
? mkOutcome('done', finalText || 'no further actions',
|
|
471
|
+
[{ kind: 'self_assessment', description: finalText || '', data: null }], metrics, toolsCalled)
|
|
472
|
+
: mkOutcome('timeout', `hit ${turnsCap}-turn cap without finishing`,
|
|
473
|
+
[{ kind: 'stop_reason', description: 'max turns', data: { turnsCap } }], metrics, toolsCalled));
|
|
474
|
+
try {
|
|
475
|
+
await rt.runsComplete(JSON.stringify({ run_id: runId, outcome }));
|
|
476
|
+
} catch { /* daemon lacks runs.* — nothing to report to */ }
|
|
477
|
+
}
|
|
478
|
+
return finalText;
|
|
479
|
+
}
|
|
480
|
+
|
|
481
|
+
/** Set up the long-lived chat runtime: register tools/policies + the agent.chat handler. */
|
|
482
|
+
async function serveChat(config) {
|
|
483
|
+
const { CarRuntime, registerChatHandler } = runtimeApi();
|
|
484
|
+
const rt = new CarRuntime();
|
|
485
|
+
const toolSchemas = [...(config.toolSchemas || []), FINISH_SCHEMA];
|
|
486
|
+
const tools = { finish: ({ answer }) => ({ answer }), ...(config.tools || {}) };
|
|
487
|
+
const toolTimeouts = Object.fromEntries(
|
|
488
|
+
toolSchemas.filter((s) => s && s.timeoutMs != null).map((s) => [s.name, s.timeoutMs]),
|
|
489
|
+
);
|
|
490
|
+
try { await rt.registerAgentBasics(); } catch { /* unsupported — fine */ }
|
|
491
|
+
for (const s of toolSchemas) await rt.registerTool(s.name);
|
|
492
|
+
for (const p of (config.policies || [])) await rt.registerPolicy(...p);
|
|
493
|
+
|
|
494
|
+
if (typeof registerChatHandler !== 'function') {
|
|
495
|
+
console.error(`[${config.agentName}] car-runtime has no agent.chat support — chat disabled (update car-runtime).`);
|
|
496
|
+
return rt;
|
|
497
|
+
}
|
|
498
|
+
|
|
499
|
+
const threads = new Map(); // session_id -> messages[]
|
|
500
|
+
const isLocalModel = typeof config.defaultModel === 'string' && /^(mlx|qwen)\//.test(config.defaultModel);
|
|
501
|
+
const noThink = isLocalModel ? '\n\n/no_think' : '';
|
|
502
|
+
|
|
503
|
+
registerChatHandler((paramsJson) => {
|
|
504
|
+
// Fire-and-forget — the daemon already got its {accepted:true} ack. Run the
|
|
505
|
+
// turn on its own microtask and stream results back via chatEvent.
|
|
506
|
+
let params;
|
|
507
|
+
try { params = JSON.parse(paramsJson); } catch { return; }
|
|
508
|
+
const sessionId = params.session_id;
|
|
509
|
+
if (!sessionId) return;
|
|
510
|
+
let messages = threads.get(sessionId);
|
|
511
|
+
if (!messages) {
|
|
512
|
+
messages = [{ role: 'system', content: `${config.identity}\n\nWhen the task is complete or you cannot proceed, call the \`finish\` tool with a concise answer. Do not narrate; act via tools.${noThink}` }];
|
|
513
|
+
threads.set(sessionId, messages);
|
|
514
|
+
}
|
|
515
|
+
messages.push({ role: 'user', content: params.prompt ?? '' });
|
|
516
|
+
runChatTurn(
|
|
517
|
+
rt, config, toolSchemas, tools, toolTimeouts, sessionId, messages, params.model,
|
|
518
|
+
)
|
|
519
|
+
.catch((e) => rt.chatEvent(sessionId, 'error', String(e && e.message || e)).catch(() => {}));
|
|
520
|
+
});
|
|
521
|
+
console.error(`[${config.agentName}] chat ready (agent.chat) — drive via CarHost or an agents.chat host client`);
|
|
522
|
+
return rt;
|
|
523
|
+
}
|
|
524
|
+
|
|
525
|
+
async function main(config) {
|
|
526
|
+
const args = parseArgs(process.argv.slice(2));
|
|
527
|
+
|
|
528
|
+
if (args.serve) {
|
|
529
|
+
// Standing mode keeps the supervised process alive so CarHost shows it
|
|
530
|
+
// "running". Always serve chat (agent.chat); if the agent has a standing
|
|
531
|
+
// goal + interval, ALSO run it on a loop.
|
|
532
|
+
// car_register.py can set these via CAR_STANDING_GOAL / CAR_INTERVAL_SECS.
|
|
533
|
+
try { await serveChat(config); } catch (e) { console.error(`[${config.agentName}] chat setup failed:`, e); }
|
|
534
|
+
const goal = config.standingGoal ?? process.env.CAR_STANDING_GOAL ?? null;
|
|
535
|
+
const everyMs = (config.intervalSecs ?? Number(process.env.CAR_INTERVAL_SECS || 0)) * 1000;
|
|
536
|
+
console.error(`[${config.agentName}] serving${goal ? ` — "${goal}" every ${everyMs / 1000}s` : ' (idle; start via dashboard or --task)'}`);
|
|
537
|
+
if (goal && everyMs > 0) {
|
|
538
|
+
for (;;) {
|
|
539
|
+
try {
|
|
540
|
+
// Each iteration is its own run: runAgent opens and closes one
|
|
541
|
+
// runs.start/runs.complete bracket on its own fresh CarRuntime.
|
|
542
|
+
const o = await runAgent(config, goal);
|
|
543
|
+
console.error(`[${config.agentName}] ${o.status} — ${o.summary}`);
|
|
544
|
+
} catch (e) { console.error(`[${config.agentName}] error:`, e); }
|
|
545
|
+
await new Promise((r) => setTimeout(r, everyMs));
|
|
546
|
+
}
|
|
547
|
+
} else {
|
|
548
|
+
// Keep the event loop alive. A pending promise alone does NOT keep Node
|
|
549
|
+
// running — it exits when the loop is empty — so use a no-op heartbeat.
|
|
550
|
+
setInterval(() => {}, 1 << 30);
|
|
551
|
+
}
|
|
552
|
+
return;
|
|
553
|
+
}
|
|
554
|
+
|
|
555
|
+
if (!args.task) {
|
|
556
|
+
console.error('usage: node agent.mjs --task "<goal>" [--json] | --serve');
|
|
557
|
+
process.exit(2);
|
|
558
|
+
}
|
|
559
|
+
|
|
560
|
+
let outcome;
|
|
561
|
+
try {
|
|
562
|
+
outcome = await runAgent(config, args.task);
|
|
563
|
+
} catch (e) {
|
|
564
|
+
outcome = { status: 'failure', summary: String(e && e.message || e),
|
|
565
|
+
evidence: [{ kind: 'stop_reason', description: 'uncaught', data: null }],
|
|
566
|
+
metrics: { turns: 0, tool_calls: 0, actions_succeeded: 0, actions_failed: 0 }, tools_called: [] };
|
|
567
|
+
}
|
|
568
|
+
|
|
569
|
+
if (args.json) {
|
|
570
|
+
// The LAST stdout line is the machine-readable outcome (run_scenarios reads it).
|
|
571
|
+
console.log(JSON.stringify(outcome));
|
|
572
|
+
} else {
|
|
573
|
+
console.log(`${outcome.status} — ${outcome.summary}`);
|
|
574
|
+
}
|
|
575
|
+
process.exit(outcome.status === 'failure' ? 1 : 0);
|
|
576
|
+
}
|
|
577
|
+
|
|
578
|
+
module.exports = { main, runAgent, runChatTurn };
|
package/agent-loop.mjs
ADDED
package/docs/ASSISTANT.md
CHANGED
package/docs/CLI.md
CHANGED
|
@@ -2,17 +2,17 @@
|
|
|
2
2
|
|
|
3
3
|
> **Generated file — do not hand-edit below the task map.** Produced by
|
|
4
4
|
> `scripts/gen-cli-docs.sh` from `car --help` / `car help <command>` on car
|
|
5
|
-
> 0.52.1 (2026-09-
|
|
5
|
+
> 0.52.1 (2026-09-06). Every subcommand the installed binary reports is
|
|
6
6
|
> below; a new subcommand cannot ship without appearing here the next time
|
|
7
7
|
> this script runs. To regenerate: `bash scripts/gen-cli-docs.sh`.
|
|
8
8
|
>
|
|
9
|
-
>
|
|
9
|
+
> 69 top-level commands, 106 nested subcommands
|
|
10
10
|
> (one level deep) — counted from the live binary at generation time, not
|
|
11
11
|
> typed by hand.
|
|
12
12
|
|
|
13
13
|
## Finding your way around
|
|
14
14
|
|
|
15
|
-
`car` is one binary with
|
|
15
|
+
`car` is one binary with 69 subcommands spanning several different jobs:
|
|
16
16
|
running the built-in agent, coding, local model management, OS integrations,
|
|
17
17
|
and installing other people's agents on your machine. This map groups the
|
|
18
18
|
commands people actually reach for; the full alphabetical reference with every
|
|
@@ -174,6 +174,7 @@ commands on a cadence via launchd / cron / schtasks).
|
|
|
174
174
|
| [`car code-task`](#car-code-task) | Run a coder session headlessly and IN THIS PROCESS: derive or accept an outcome contract, work in a git worktree until the runtime's own re-run of that contract is green, then deliver the result as a pull request |
|
|
175
175
|
| [`car coder-ab`](#car-coder-ab) | A/B-test CAR's coder against an external agent (Codex / Claude Code) over a corpus, and grow that corpus from git history — the productionized dogfooding loop (docs/proposals/coder-ab-dogfood.md) |
|
|
176
176
|
| [`car keys`](#car-keys) | Store cloud-provider API keys in the OS keychain, so a native-app user never sets an environment variable (docs/proposals/native-secrets-no-env.md). The key is read env-first, keychain-fallback by the runtime |
|
|
177
|
+
| [`car selfheal`](#car-selfheal) | Inspect and operate the daemon's deterministic self-healing detector |
|
|
177
178
|
| [`car daemon`](#car-daemon) | Start the daemon server (delegates to car-server binary) |
|
|
178
179
|
| [`car models`](#car-models) | Manage local inference models |
|
|
179
180
|
| [`car setup`](#car-setup) | Set up the right model for this machine — detect hardware, recommend, and install. Run with no flags for an interactive walkthrough |
|
|
@@ -647,6 +648,90 @@ Options:
|
|
|
647
648
|
-h, --help Print help
|
|
648
649
|
```
|
|
649
650
|
|
|
651
|
+
### car selfheal
|
|
652
|
+
|
|
653
|
+
```text
|
|
654
|
+
Inspect and operate the daemon's deterministic self-healing detector
|
|
655
|
+
|
|
656
|
+
Usage: car selfheal <COMMAND>
|
|
657
|
+
|
|
658
|
+
Commands:
|
|
659
|
+
status Show cadence, active counts, and the route resolved by the last tick
|
|
660
|
+
list List active detections, optionally filtered by kind, severity, or time
|
|
661
|
+
show Print the trusted local issue document for one detection
|
|
662
|
+
dismiss Dismiss one active detection by dedup key
|
|
663
|
+
run Run one deterministic detection tick now
|
|
664
|
+
help Print this message or the help of the given subcommand(s)
|
|
665
|
+
|
|
666
|
+
Options:
|
|
667
|
+
-h, --help Print help
|
|
668
|
+
```
|
|
669
|
+
|
|
670
|
+
#### car selfheal status
|
|
671
|
+
|
|
672
|
+
```text
|
|
673
|
+
Show cadence, active counts, and the route resolved by the last tick
|
|
674
|
+
|
|
675
|
+
Usage: car selfheal status
|
|
676
|
+
|
|
677
|
+
Options:
|
|
678
|
+
-h, --help Print help
|
|
679
|
+
```
|
|
680
|
+
|
|
681
|
+
#### car selfheal list
|
|
682
|
+
|
|
683
|
+
```text
|
|
684
|
+
List active detections, optionally filtered by kind, severity, or time
|
|
685
|
+
|
|
686
|
+
Usage: car selfheal list [OPTIONS]
|
|
687
|
+
|
|
688
|
+
Options:
|
|
689
|
+
--kind <KIND> Detection kind (metrics_alert, agent_gave_up, agent_log_error,
|
|
690
|
+
agent_silently_idle, recurring_tool_failure, capability_miss)
|
|
691
|
+
--severity <SEVERITY> Severity (warning or critical)
|
|
692
|
+
--since <SINCE> Only detections observed at or after this RFC3339 timestamp
|
|
693
|
+
-h, --help Print help
|
|
694
|
+
```
|
|
695
|
+
|
|
696
|
+
#### car selfheal show
|
|
697
|
+
|
|
698
|
+
```text
|
|
699
|
+
Print the trusted local issue document for one detection
|
|
700
|
+
|
|
701
|
+
Usage: car selfheal show <DEDUP_KEY>
|
|
702
|
+
|
|
703
|
+
Arguments:
|
|
704
|
+
<DEDUP_KEY> Detection SHA-256 dedup key
|
|
705
|
+
|
|
706
|
+
Options:
|
|
707
|
+
-h, --help Print help
|
|
708
|
+
```
|
|
709
|
+
|
|
710
|
+
#### car selfheal dismiss
|
|
711
|
+
|
|
712
|
+
```text
|
|
713
|
+
Dismiss one active detection by dedup key
|
|
714
|
+
|
|
715
|
+
Usage: car selfheal dismiss <DEDUP_KEY>
|
|
716
|
+
|
|
717
|
+
Arguments:
|
|
718
|
+
<DEDUP_KEY> Detection SHA-256 dedup key
|
|
719
|
+
|
|
720
|
+
Options:
|
|
721
|
+
-h, --help Print help
|
|
722
|
+
```
|
|
723
|
+
|
|
724
|
+
#### car selfheal run
|
|
725
|
+
|
|
726
|
+
```text
|
|
727
|
+
Run one deterministic detection tick now
|
|
728
|
+
|
|
729
|
+
Usage: car selfheal run
|
|
730
|
+
|
|
731
|
+
Options:
|
|
732
|
+
-h, --help Print help
|
|
733
|
+
```
|
|
734
|
+
|
|
650
735
|
### car daemon
|
|
651
736
|
|
|
652
737
|
```text
|
package/docs/agent-ir-spec.md
CHANGED
|
@@ -181,6 +181,35 @@ Proposed → Validated → Executing → Succeeded
|
|
|
181
181
|
|
|
182
182
|
`ActionStatus` is observable through the event log, not part of the input contract.
|
|
183
183
|
|
|
184
|
+
#### Execution outcome event data
|
|
185
|
+
|
|
186
|
+
Every per-action `ActionFailed` event carries:
|
|
187
|
+
|
|
188
|
+
- `params_digest`: lowercase SHA-256 of the RFC 8785/JCS-canonicalized action
|
|
189
|
+
`parameters` object. The event never copies raw parameters; consumers join
|
|
190
|
+
through `proposal_id` + `action_id` to the authoritative `ProposalReceived`
|
|
191
|
+
record and can use the digest to detect a mismatch.
|
|
192
|
+
- `expected_effects`: the action's declared expected-effects object, unchanged.
|
|
193
|
+
- `error_class`: one of `timeout`, `rejected_by_policy`, `tool_error`,
|
|
194
|
+
`validation`, or `unknown`.
|
|
195
|
+
|
|
196
|
+
`ActionSucceeded` carries `params_digest` and `expected_effects` too, making the
|
|
197
|
+
success/failure join symmetric without adding an error classification to a
|
|
198
|
+
successful call.
|
|
199
|
+
|
|
200
|
+
The normalized error mapping is intentionally low-cardinality:
|
|
201
|
+
|
|
202
|
+
| `error_class` | Mapping |
|
|
203
|
+
|---|---|
|
|
204
|
+
| `timeout` | the engine's action deadline expired, or the daemon-to-host tool callback reported its own timeout |
|
|
205
|
+
| `rejected_by_policy` | a dispatch-time tool guard returned the stable `denied by policy:` or `rejected by policy:` prefix |
|
|
206
|
+
| `validation` | post-dispatch output or callback-state JCS/I-JSON, state-key-set, or serialization validation failed |
|
|
207
|
+
| `tool_error` | any other error returned while dispatching a tool action |
|
|
208
|
+
| `unknown` | a post-dispatch failure on an action with no tool |
|
|
209
|
+
|
|
210
|
+
Normal action/schema/policy admission failures happen before execution and are
|
|
211
|
+
`ActionRejected`, not `ActionFailed`, so this mapping does not reclassify them.
|
|
212
|
+
|
|
184
213
|
---
|
|
185
214
|
|
|
186
215
|
## Precondition
|
|
@@ -220,6 +249,7 @@ Registered when a tool is added to the runtime. Carries everything the runtime n
|
|
|
220
249
|
```jsonc
|
|
221
250
|
{
|
|
222
251
|
"name": "deploy",
|
|
252
|
+
"source": "user_defined",
|
|
223
253
|
"description": "Deploys an artifact to a target environment.",
|
|
224
254
|
"parameters": {
|
|
225
255
|
"type": "object",
|
|
@@ -238,6 +268,7 @@ Registered when a tool is added to the runtime. Carries everything the runtime n
|
|
|
238
268
|
| Field | Type | Required | Default | Notes |
|
|
239
269
|
|-------|------|----------|---------|-------|
|
|
240
270
|
| `name` | string | **yes** | — | unique within a runtime |
|
|
271
|
+
| `source` | `builtin \| user_defined \| subprocess \| mcp` | no | `user_defined` | stable origin category assigned by the runtime; MCP server detail remains registry-private |
|
|
241
272
|
| `description` | string | no | `""` | human-readable; included in tool catalog |
|
|
242
273
|
| `parameters` | JSON Schema | no | `{}` | validated by the runtime before dispatch |
|
|
243
274
|
| `returns` | JSON Schema | no | none | validated against tool return value when set |
|
|
@@ -562,15 +593,21 @@ allow = ["staging", "preview"] # any other target — or none at all — is de
|
|
|
562
593
|
| `allow` | no | permitted values; defaults to empty, which denies every call |
|
|
563
594
|
|
|
564
595
|
#### `deny_tool_param_matching`
|
|
565
|
-
The content counterpart to `deny_tool_param`, for
|
|
596
|
+
The content counterpart to `deny_tool_param`, for conditions no fixed substring expresses — credential shapes, account numbers, an address family, or an open-ended trusted prefix. `matches` is a regex over the string-coerced parameter value. The match is **unanchored**, so the pattern fires anywhere in the value; anchor it with `^`/`$` when that matters.
|
|
566
597
|
|
|
567
|
-
The pattern is compiled once when the rule set is applied, not per action. A pattern that fails to compile **denies every call to that tool** rather than disappearing, matching the loader's loud-error posture.
|
|
598
|
+
By default a regex match denies and an absent parameter does not. Set `negate = true` for the "unless" form: a mismatch denies, and an absent parameter also denies because nothing proves the required pattern. The pattern is compiled once when the rule set is applied, not per action. A pattern that fails to compile **denies every call to that tool** rather than disappearing, matching the loader's loud-error posture.
|
|
568
599
|
|
|
569
600
|
```toml
|
|
570
601
|
[[deny_tool_param_matching]]
|
|
571
602
|
tool = "http_request"
|
|
572
603
|
param = "body"
|
|
573
604
|
matches = "sk-[A-Za-z0-9]{20,}" # never let an API-key-shaped string leave in a body
|
|
605
|
+
|
|
606
|
+
[[deny_tool_param_matching]]
|
|
607
|
+
tool = "docker.rm"
|
|
608
|
+
param = "name"
|
|
609
|
+
matches = "^parslee-"
|
|
610
|
+
negate = true # deny unless the name has the trusted prefix
|
|
574
611
|
```
|
|
575
612
|
|
|
576
613
|
| Param | Required | Notes |
|
|
@@ -578,6 +615,7 @@ matches = "sk-[A-Za-z0-9]{20,}" # never let an API-key-shaped string leave in
|
|
|
578
615
|
| `tool` | yes | tool name the rule applies to |
|
|
579
616
|
| `param` | yes | parameter key inspected on the action |
|
|
580
617
|
| `matches` | yes | regex source; unanchored; an uncompilable pattern denies the tool outright |
|
|
618
|
+
| `negate` | no | defaults to `false`; when `true`, deny mismatch or absence instead of match |
|
|
581
619
|
|
|
582
620
|
#### `rate_limit_tool`
|
|
583
621
|
A sliding-window cap on how often `tool` may be called. The call is denied when admitting it would make it the `max_calls + 1`-th call to `tool` within the trailing `interval_secs`. `max_calls = 0` denies every call. This bounds how much of a side effect an agent can produce in a stretch of wall-clock time, independently of whether any single call is legitimate.
|
package/index.d.ts
CHANGED
|
@@ -32,6 +32,61 @@
|
|
|
32
32
|
* `ws://127.0.0.1:9100`).
|
|
33
33
|
*/
|
|
34
34
|
|
|
35
|
+
/** Agent-loop tool declaration. `timeoutMs` becomes the action budget. */
|
|
36
|
+
export interface AgentToolSchema {
|
|
37
|
+
name: string;
|
|
38
|
+
description: string;
|
|
39
|
+
parameters: Record<string, unknown>;
|
|
40
|
+
timeoutMs?: number;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
export interface AgentToolContext {
|
|
44
|
+
signal: AbortSignal;
|
|
45
|
+
timeoutMs?: number;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export type AgentTool = (
|
|
49
|
+
params: Record<string, unknown>,
|
|
50
|
+
context: AgentToolContext,
|
|
51
|
+
) => unknown | Promise<unknown>;
|
|
52
|
+
|
|
53
|
+
export type AgentOutcomeStatus =
|
|
54
|
+
| 'success' | 'partial_success' | 'done' | 'give_up' | 'timeout' | 'failure';
|
|
55
|
+
|
|
56
|
+
export interface AgentOutcome {
|
|
57
|
+
status: AgentOutcomeStatus;
|
|
58
|
+
summary: string;
|
|
59
|
+
evidence: Array<{ kind: string; description: string; data: unknown }>;
|
|
60
|
+
metrics: {
|
|
61
|
+
turns: number;
|
|
62
|
+
tool_calls: number;
|
|
63
|
+
actions_succeeded: number;
|
|
64
|
+
actions_failed: number;
|
|
65
|
+
};
|
|
66
|
+
tools_called: string[];
|
|
67
|
+
timestamp: string;
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/** Declarative input consumed by `car-runtime/agent-loop`. */
|
|
71
|
+
export interface AgentLoopConfig {
|
|
72
|
+
agentId?: string;
|
|
73
|
+
agentName: string;
|
|
74
|
+
identity: string;
|
|
75
|
+
toolSchemas?: AgentToolSchema[];
|
|
76
|
+
tools?: Record<string, AgentTool>;
|
|
77
|
+
policies?: Array<[string, string, string?, string?, string?, string?]>;
|
|
78
|
+
defaultModel?: string | null;
|
|
79
|
+
maxTokens?: number;
|
|
80
|
+
maxTurns?: number;
|
|
81
|
+
targetOutcome?: string;
|
|
82
|
+
standingGoal?: string | null;
|
|
83
|
+
intervalSecs?: number;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
export interface AgentLoopOptions {
|
|
87
|
+
maxTurns?: number;
|
|
88
|
+
}
|
|
89
|
+
|
|
35
90
|
/** Persistent runtime instance with state, memory, tools, and policies. */
|
|
36
91
|
/**
|
|
37
92
|
* Optional settings for `coderStart`. Every field is independently omittable;
|
|
@@ -58,6 +113,24 @@ export interface CoderStartOptions {
|
|
|
58
113
|
* hypothesis, the other buys a retry.
|
|
59
114
|
*/
|
|
60
115
|
transientRetries?: number | undefined | null;
|
|
116
|
+
/**
|
|
117
|
+
* Farm a **foreman** session's subtasks across every reachable CAR instance
|
|
118
|
+
* that can serve this repository, instead of this machine alone. The
|
|
119
|
+
* merge-verify gate and delivery stay on the orchestrating host — a peer
|
|
120
|
+
* returns a patch and this host gates it — so a distributed run still
|
|
121
|
+
* produces a gated pull request.
|
|
122
|
+
*
|
|
123
|
+
* Only the foreman engine decomposes a goal into subtasks, so any other
|
|
124
|
+
* engine runs locally and says so. Off by default: it spends agent quota on
|
|
125
|
+
* other people's machines.
|
|
126
|
+
*/
|
|
127
|
+
distributed?: boolean | undefined | null;
|
|
128
|
+
/**
|
|
129
|
+
* Restrict placement to these instances by name. Empty or omitted means
|
|
130
|
+
* every instance that reports it can serve the repository. Ignored unless
|
|
131
|
+
* `distributed` is set.
|
|
132
|
+
*/
|
|
133
|
+
workers?: Array<string> | undefined | null;
|
|
61
134
|
/**
|
|
62
135
|
* A `coder.discuss` conversation this run was distilled from. Its agreed
|
|
63
136
|
* constraints ride into contract derivation, so a rule stated once in the
|
|
@@ -68,9 +141,41 @@ export interface CoderStartOptions {
|
|
|
68
141
|
discussionId?: string | undefined | null;
|
|
69
142
|
}
|
|
70
143
|
|
|
144
|
+
export interface DaemonRpcError extends Error {
|
|
145
|
+
/** Numeric JSON-RPC error code returned by the daemon. */
|
|
146
|
+
code: number;
|
|
147
|
+
/** Daemon-provided diagnostic text. */
|
|
148
|
+
message: string;
|
|
149
|
+
/** Optional JSON-RPC error data returned by the daemon. */
|
|
150
|
+
data?: unknown;
|
|
151
|
+
}
|
|
152
|
+
|
|
71
153
|
export class CarRuntime {
|
|
72
154
|
constructor();
|
|
73
155
|
|
|
156
|
+
/**
|
|
157
|
+
* Invoke any daemon JSON-RPC method with a JSON-encoded params value.
|
|
158
|
+
* `daemonCall` is the call-by-name escape hatch; use the typed wrappers as the primary API.
|
|
159
|
+
* The result is returned as JSON. Daemon rejections are `DaemonRpcError`;
|
|
160
|
+
* transport failures reject without a synthetic numeric code.
|
|
161
|
+
*/
|
|
162
|
+
daemonCall(method: string, paramsJson: string): Promise<string>;
|
|
163
|
+
|
|
164
|
+
/** Host-management-token twin of `daemonCall`; the method allowlist remains enforced. */
|
|
165
|
+
daemonCallHostManagement(method: string, paramsJson: string): Promise<string>;
|
|
166
|
+
|
|
167
|
+
/** Register a server-initiated JSON-RPC request handler. */
|
|
168
|
+
registerDaemonHandler(
|
|
169
|
+
method: string,
|
|
170
|
+
handler: (paramsJson: string) => Promise<string>,
|
|
171
|
+
): void;
|
|
172
|
+
|
|
173
|
+
/** Register a server-initiated JSON-RPC notification handler. */
|
|
174
|
+
registerDaemonNotificationHandler(
|
|
175
|
+
method: string,
|
|
176
|
+
handler: (paramsJson: string) => void,
|
|
177
|
+
): void;
|
|
178
|
+
|
|
74
179
|
// --- Memory persistence ---
|
|
75
180
|
|
|
76
181
|
/**
|
|
@@ -142,7 +247,54 @@ export class CarRuntime {
|
|
|
142
247
|
adapter?: string,
|
|
143
248
|
verifyCommand?: Array<string>,
|
|
144
249
|
unionVerifyCommand?: Array<string>,
|
|
145
|
-
maxAttempts?: number
|
|
250
|
+
maxAttempts?: number,
|
|
251
|
+
distributed?: boolean,
|
|
252
|
+
workers?: Array<string>
|
|
253
|
+
): Promise<string>;
|
|
254
|
+
|
|
255
|
+
// --- Fleet ---
|
|
256
|
+
|
|
257
|
+
/**
|
|
258
|
+
* This instance's agents, capabilities, and models — one `InstanceInventory`
|
|
259
|
+
* JSON object. The same report peers receive over A2A, plus this session's
|
|
260
|
+
* own registered tools and learned skills.
|
|
261
|
+
*/
|
|
262
|
+
fleetInventory(): Promise<string>;
|
|
263
|
+
|
|
264
|
+
/**
|
|
265
|
+
* Every agent, capability, and model across this daemon and every reachable
|
|
266
|
+
* CAR instance, folded so one row names every instance that offers it.
|
|
267
|
+
* `includeRemote` defaults to true. `timeoutMs` bounds each peer
|
|
268
|
+
* individually: a sleeping machine appears as an unreachable row carrying the
|
|
269
|
+
* reason, never a missing one. Returns `FleetComposite` JSON.
|
|
270
|
+
*/
|
|
271
|
+
fleetComposite(includeRemote?: boolean, timeoutMs?: number): Promise<string>;
|
|
272
|
+
|
|
273
|
+
/** Whether this instance takes farmed-out coding work. `{ config, profile }` JSON. */
|
|
274
|
+
fleetWorkerGet(): Promise<string>;
|
|
275
|
+
|
|
276
|
+
/**
|
|
277
|
+
* Enroll (or withdraw) this instance as a fleet worker. **Operator-only, and
|
|
278
|
+
* a real grant**: enrolling lets a trusted peer run a coding CLI against the
|
|
279
|
+
* checkouts named in `repos`. Only the fields supplied change.
|
|
280
|
+
*
|
|
281
|
+
* The limits belong to this machine, not the caller: `dispatchesPerHour`
|
|
282
|
+
* budgets one peer's spend (concurrency is not a spend bound),
|
|
283
|
+
* `maxSubtaskSecs` caps the timeout a sender asks for, and `allowedTools` is
|
|
284
|
+
* intersected with whatever the dispatch requests. `fetchMissingBase` makes
|
|
285
|
+
* this machine a **runner**: rather than decline a base commit it lacks, it
|
|
286
|
+
* fetches from `fetchRemote` (its own, default `origin`).
|
|
287
|
+
*/
|
|
288
|
+
fleetWorkerSet(
|
|
289
|
+
acceptsWork?: boolean,
|
|
290
|
+
repos?: Array<string>,
|
|
291
|
+
maxParallel?: number,
|
|
292
|
+
localParallel?: number,
|
|
293
|
+
dispatchesPerHour?: number,
|
|
294
|
+
maxSubtaskSecs?: number,
|
|
295
|
+
allowedTools?: Array<string>,
|
|
296
|
+
fetchMissingBase?: boolean,
|
|
297
|
+
fetchRemote?: string
|
|
146
298
|
): Promise<string>;
|
|
147
299
|
|
|
148
300
|
// --- Tools & policies ---
|
|
@@ -152,7 +304,8 @@ export class CarRuntime {
|
|
|
152
304
|
|
|
153
305
|
/**
|
|
154
306
|
* The tools currently registered on this runtime, as a JSON array of full
|
|
155
|
-
* `ToolSchema` objects sorted by name.
|
|
307
|
+
* `ToolSchema` objects sorted by name. Every schema includes its runtime-
|
|
308
|
+
* assigned `source` (`builtin|user_defined|subprocess|mcp`).
|
|
156
309
|
*
|
|
157
310
|
* Counterpart to `registerTool` / `registerToolSchema`, which had none: a
|
|
158
311
|
* caller could add tools but never ask what was actually in effect, so a
|
|
@@ -296,7 +449,8 @@ export class CarRuntime {
|
|
|
296
449
|
// --- Memory / Facts (graph-backed) ---
|
|
297
450
|
|
|
298
451
|
/**
|
|
299
|
-
* Add a fact. `kind` is typically "pattern" or "constraint".
|
|
452
|
+
* Add a fact. `kind` is typically "pattern" or "constraint". Optional
|
|
453
|
+
* `factId`, ordered `tags`, and `source` are preserved by the daemon.
|
|
300
454
|
*
|
|
301
455
|
* In Daemon mode, rejects with the daemon-unreachable error
|
|
302
456
|
* instead of silently returning 0 (#146).
|
|
@@ -306,9 +460,16 @@ export class CarRuntime {
|
|
|
306
460
|
body: string,
|
|
307
461
|
kind: string,
|
|
308
462
|
confidence?: number | null,
|
|
463
|
+
factId?: string | null,
|
|
464
|
+
tags?: string[] | null,
|
|
465
|
+
source?: string | null,
|
|
309
466
|
): Promise<number>;
|
|
310
467
|
|
|
311
|
-
/**
|
|
468
|
+
/**
|
|
469
|
+
* Query facts via graph spreading activation. Returns a JSON array whose
|
|
470
|
+
* rows include `fact_id` (null for graph nodes without one), `subject`,
|
|
471
|
+
* `body`, `kind`, `confidence`, `tags`, and `source`.
|
|
472
|
+
*/
|
|
312
473
|
queryFacts(query: string, k?: number | null): string;
|
|
313
474
|
|
|
314
475
|
/**
|
|
@@ -462,7 +623,37 @@ export class CarRuntime {
|
|
|
462
623
|
syncStatus(requestJson: string): Promise<string>;
|
|
463
624
|
/** `sync.append` — record an op on any surface: `{ surface, payload, scope? }` (B6). */
|
|
464
625
|
syncAppend(requestJson: string): Promise<string>;
|
|
465
|
-
/** `agents
|
|
626
|
+
/** `host.agents` — current host agent registry snapshot. */
|
|
627
|
+
hostAgents(): Promise<string>;
|
|
628
|
+
/** `host.events` — recent host events, newest last; omit `limit` for the daemon default. */
|
|
629
|
+
hostEvents(limit?: number): Promise<string>;
|
|
630
|
+
/** `host.approvals` — pending host approvals. */
|
|
631
|
+
hostApprovals(): Promise<string>;
|
|
632
|
+
/** `host.register_agent` — register an agent on this connection. */
|
|
633
|
+
hostRegisterAgent(requestJson: string): Promise<string>;
|
|
634
|
+
/** `host.unregister_agent` — unregister an agent owned by this connection. */
|
|
635
|
+
hostUnregisterAgent(requestJson: string): Promise<string>;
|
|
636
|
+
/** `host.set_status` — publish status for an agent owned by this connection. */
|
|
637
|
+
hostSetStatus(requestJson: string): Promise<string>;
|
|
638
|
+
/** `host.register_device` — register a device on this connection. */
|
|
639
|
+
hostRegisterDevice(requestJson: string): Promise<string>;
|
|
640
|
+
/** `host.update_device` — update a device owned by this connection. */
|
|
641
|
+
hostUpdateDevice(requestJson: string): Promise<string>;
|
|
642
|
+
/** `host.devices` — current host device registry snapshot. */
|
|
643
|
+
hostDevices(): Promise<string>;
|
|
644
|
+
/** `host.notify` — emit a user-facing host notification. */
|
|
645
|
+
hostNotify(requestJson: string): Promise<string>;
|
|
646
|
+
/** `host.request_approval` — request approval for a gated action. */
|
|
647
|
+
hostRequestApproval(requestJson: string): Promise<string>;
|
|
648
|
+
/** `host.resolve_approval` — resolve one pending host approval. */
|
|
649
|
+
hostResolveApproval(requestJson: string): Promise<string>;
|
|
650
|
+
// `host.subscribe` event delivery is deferred to the callback-aware
|
|
651
|
+
// daemon-session API; subscribing without a consumer would drop the stream.
|
|
652
|
+
/**
|
|
653
|
+
* `agents.peers` — visible peers as JSON. Each row distinguishes the
|
|
654
|
+
* kind-level `can_receive` capability from the current `reachable` delivery
|
|
655
|
+
* preflight; the send remains authoritative.
|
|
656
|
+
*/
|
|
466
657
|
agentsPeers(requestJson: string): Promise<string>;
|
|
467
658
|
/** `agents.message` — send text to one peer: `{ to, body, summary? }`. The sender is derived server-side. */
|
|
468
659
|
agentsMessage(requestJson: string): Promise<string>;
|
|
@@ -795,6 +986,21 @@ export class CarRuntime {
|
|
|
795
986
|
* of silently serving a different model (Parslee-ai/car#888). Absent on
|
|
796
987
|
* the common path.
|
|
797
988
|
*
|
|
989
|
+
* `fallback_from` is an ARRAY of every candidate the chain moved past,
|
|
990
|
+
* in the order it tried them: `[{ candidate, reason }, ...]`, where
|
|
991
|
+
* `reason` is one of `"credential_rejected"`, `"credential_absent"`,
|
|
992
|
+
* `"rate_limited"`, `"quota_exhausted"`, `"timed_out"` or `"failed"`.
|
|
993
|
+
* Absent when the first candidate served. Before this, a run whose
|
|
994
|
+
* backbone changed because of a rate limit or a timeout recorded no
|
|
995
|
+
* cause anywhere, so a surprising result got attributed to the code
|
|
996
|
+
* rather than to the model swap (Parslee-ai/car#1351).
|
|
997
|
+
*
|
|
998
|
+
* `reason` is classified from the runtime's typed error, not from error
|
|
999
|
+
* prose. `"credential_rejected"` is deliberately BROADER than
|
|
1000
|
+
* `auth_fallback_from`: it covers a provider refusing an API key, whose
|
|
1001
|
+
* remedy is to fix the key, not to sign in. Do not derive one field
|
|
1002
|
+
* from the other.
|
|
1003
|
+
*
|
|
798
1004
|
* **Note:** intent is not exposed on the tracked path until the
|
|
799
1005
|
* positional argument list is converted to an options object —
|
|
800
1006
|
* this method already takes 9 positional parameters and adding
|
|
@@ -1462,7 +1668,10 @@ export class CarRuntime {
|
|
|
1462
1668
|
|
|
1463
1669
|
/** Structured audit query over the event log (G2). `queryJson` is an
|
|
1464
1670
|
* EventQuery object (kinds/actionId/proposalId/since/until/dataMatches/limit);
|
|
1465
|
-
* returns `{count, events}` as a JSON string, most-recent-first.
|
|
1671
|
+
* returns `{count, events}` as a JSON string, most-recent-first.
|
|
1672
|
+
* `ActionFailed.data` includes `params_digest`, `expected_effects`, and
|
|
1673
|
+
* `error_class` (`timeout|rejected_by_policy|tool_error|validation|unknown`),
|
|
1674
|
+
* never raw parameters. `ActionSucceeded.data` includes the first two. */
|
|
1466
1675
|
eventQuery(queryJson: string): Promise<string>;
|
|
1467
1676
|
|
|
1468
1677
|
/** Get/set the event-log retention policy (G2). Pass a
|
|
@@ -1498,21 +1707,39 @@ export class CarRuntime {
|
|
|
1498
1707
|
* `cost_overage` alert. */
|
|
1499
1708
|
metricsAlerts(thresholdsJson?: string): Promise<string>;
|
|
1500
1709
|
|
|
1501
|
-
|
|
1502
|
-
*
|
|
1503
|
-
|
|
1710
|
+
/** Self-healing repair loop status: enabled/why-not, cadence, targets,
|
|
1711
|
+
* rejected targets, review panel, engine. */
|
|
1712
|
+
healStatus(): Promise<string>;
|
|
1713
|
+
/** Run one self-healing repair sweep now. May open a pull request; never merges. */
|
|
1714
|
+
healRun(): Promise<string>;
|
|
1715
|
+
/** Self-heal status as JSON: cadence, `auto_fix_enabled`, `max_concurrent`,
|
|
1716
|
+
* `max_per_day`, `max_rounds_per_key`, optional `auto_fix_refusal_reason`,
|
|
1717
|
+
* last tick, source route/refusal,
|
|
1718
|
+
* detector counts, and `filing_mode` (`watch-only` or `pr-only`). */
|
|
1719
|
+
|
|
1504
1720
|
selfhealStatus(): Promise<string>;
|
|
1505
1721
|
|
|
1506
1722
|
/** List active (not dismissed) self-heal detections as JSON. Each includes
|
|
1507
|
-
* `route` and an optional `local_issue_path`.
|
|
1508
|
-
* `
|
|
1723
|
+
* `route` and an optional `local_issue_path`. Recurring tool failures add
|
|
1724
|
+
* `eligible`, a secret-safe `reconstructed_call` (`tool` plus exact `params`),
|
|
1725
|
+
* optional owner-private `reconstructed_call_path`, `auto_fix_attempts`,
|
|
1726
|
+
* `auto_fix_exhausted`, `auto_fix_in_progress`, and
|
|
1727
|
+
* `last_auto_fix_attempt` (including `exit_code` and `failure_class`). Remote
|
|
1728
|
+
* deduplication adds `auto_fix_awaiting_review`, `auto_fix_parked`,
|
|
1729
|
+
* `remote_pr_number`, and `remote_pr_url`.
|
|
1730
|
+
* `queryJson` carries optional `kind`, `severity`,
|
|
1731
|
+
* `since`, `offset`, and `limit` (max 500). */
|
|
1509
1732
|
selfhealDetections(queryJson?: string): Promise<string>;
|
|
1510
1733
|
|
|
1511
|
-
/** Append a dismissal marker for a stable detection dedup key
|
|
1512
|
-
*
|
|
1734
|
+
/** Append a dismissal marker for a stable detection dedup key without
|
|
1735
|
+
* deleting history. */
|
|
1513
1736
|
selfhealDismiss(dedupKey: string): Promise<string>;
|
|
1514
1737
|
|
|
1515
|
-
/**
|
|
1738
|
+
/** Start one bounded template-owned coder round for an eligible recurring
|
|
1739
|
+
* tool failure. Returns the durable attempt result as JSON. */
|
|
1740
|
+
selfhealFix(dedupKey: string): Promise<string>;
|
|
1741
|
+
|
|
1742
|
+
/** Run one non-overlapping detection tick and default-on auto-fix hook. */
|
|
1516
1743
|
selfhealRun(): Promise<string>;
|
|
1517
1744
|
|
|
1518
1745
|
/** Execution log counts and approximate retained native bytes. Returns JSON. */
|
|
@@ -1611,9 +1838,10 @@ export class CarRuntime {
|
|
|
1611
1838
|
* automatically.
|
|
1612
1839
|
*
|
|
1613
1840
|
* Tools registered via the schemaless `registerTool(name)` bypass type
|
|
1614
|
-
* validation; this is the opt-in upgrade path.
|
|
1841
|
+
* validation; this is the opt-in upgrade path. The daemon assigns
|
|
1842
|
+
* `source = "user_defined"`; callers cannot claim another origin.
|
|
1615
1843
|
*
|
|
1616
|
-
* `schemaJson`
|
|
1844
|
+
* `schemaJson` carries the caller-settable fields:
|
|
1617
1845
|
* ```json
|
|
1618
1846
|
* {
|
|
1619
1847
|
* "name": "read_file",
|
|
@@ -2596,10 +2824,21 @@ export function reapStaleAgents(
|
|
|
2596
2824
|
* register an `agent.chat` handler to serve conversational turns, and/or
|
|
2597
2825
|
* a `registerToolHandler` for explicit tool `data` parts.
|
|
2598
2826
|
*
|
|
2827
|
+
* `allow_non_loopback_bind` (boolean, default `false`) is required to bind
|
|
2828
|
+
* anything but loopback. This listener serves NO authentication — there
|
|
2829
|
+
* is no auth parameter, and its router is `NoAuth` — and without
|
|
2830
|
+
* `share_session_runtime` its runtime registers the agent-basics
|
|
2831
|
+
* filesystem tools, so a reachable bind publishes `write_file` /
|
|
2832
|
+
* `edit_file` to anyone who can route to the port. A wildcard bind
|
|
2833
|
+
* (`0.0.0.0:...`) is refused too. To be reachable by other CAR daemons
|
|
2834
|
+
* you want the peer-authenticated, messaging-only listener `car-server`
|
|
2835
|
+
* already runs by default, not this.
|
|
2836
|
+
*
|
|
2599
2837
|
* Returns `'{"bound":"127.0.0.1:8731"}'` on success. Errors if a
|
|
2600
|
-
* server is already running, the bind fails,
|
|
2601
|
-
* is set but no
|
|
2602
|
-
* non-WS path), or
|
|
2838
|
+
* server is already running, the bind fails, the bind is non-loopback
|
|
2839
|
+
* without `allow_non_loopback_bind`, `share_session_runtime` is set but no
|
|
2840
|
+
* session runtime is available (e.g. invoked from a non-WS path), or
|
|
2841
|
+
* `paramsJson` is malformed.
|
|
2603
2842
|
*/
|
|
2604
2843
|
export function startA2AServer(rt: CarRuntime, paramsJson: string): Promise<string>;
|
|
2605
2844
|
|
package/package.json
CHANGED
|
@@ -1,9 +1,23 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "car-runtime",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.53.0",
|
|
4
4
|
"description": "Common Agent Runtime — a deterministic execution layer for AI agents",
|
|
5
5
|
"main": "index.js",
|
|
6
6
|
"types": "index.d.ts",
|
|
7
|
+
"exports": {
|
|
8
|
+
".": {
|
|
9
|
+
"types": "./index.d.ts",
|
|
10
|
+
"require": "./index.js",
|
|
11
|
+
"default": "./index.js"
|
|
12
|
+
},
|
|
13
|
+
"./agent-loop": {
|
|
14
|
+
"types": "./agent-loop.d.ts",
|
|
15
|
+
"import": "./agent-loop.mjs",
|
|
16
|
+
"require": "./agent-loop.js",
|
|
17
|
+
"default": "./agent-loop.js"
|
|
18
|
+
},
|
|
19
|
+
"./package.json": "./package.json"
|
|
20
|
+
},
|
|
7
21
|
"bin": {
|
|
8
22
|
"car-server": "bin/car-server"
|
|
9
23
|
},
|
|
@@ -36,6 +50,9 @@
|
|
|
36
50
|
"files": [
|
|
37
51
|
"index.js",
|
|
38
52
|
"index.d.ts",
|
|
53
|
+
"agent-loop.js",
|
|
54
|
+
"agent-loop.mjs",
|
|
55
|
+
"agent-loop.d.ts",
|
|
39
56
|
"install.js",
|
|
40
57
|
"assets.json",
|
|
41
58
|
"bin/car-server",
|
|
@@ -45,7 +62,8 @@
|
|
|
45
62
|
],
|
|
46
63
|
"scripts": {
|
|
47
64
|
"install": "node install.js",
|
|
48
|
-
"prepack": "node sync-docs.js"
|
|
65
|
+
"prepack": "node sync-docs.js",
|
|
66
|
+
"test": "node --test test/install.test.js"
|
|
49
67
|
},
|
|
50
68
|
"license": "SEE LICENSE IN LICENSE",
|
|
51
69
|
"private": false
|