klyro 1.0.1 → 1.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent/anthropic-adapter.d.ts +13 -5
- package/dist/agent/anthropic-adapter.js +19 -2
- package/dist/agent/capabilities.js +7 -1
- package/dist/agent/orchestrator.d.ts +18 -1
- package/dist/agent/orchestrator.js +51 -4
- package/dist/agent/retry.js +52 -10
- package/dist/agent/runtime.d.ts +34 -0
- package/dist/agent/runtime.js +164 -14
- package/dist/agent/stream-budget.d.ts +36 -0
- package/dist/agent/stream-budget.js +121 -0
- package/dist/checkpoints/store.d.ts +9 -0
- package/dist/checkpoints/store.js +26 -0
- package/dist/cli/commit.d.ts +31 -0
- package/dist/cli/commit.js +142 -0
- package/dist/cli/config.d.ts +45 -0
- package/dist/cli/config.js +82 -0
- package/dist/cli/doctor.d.ts +1 -0
- package/dist/cli/doctor.js +71 -6
- package/dist/cli/hooks.d.ts +47 -0
- package/dist/cli/hooks.js +181 -0
- package/dist/cli/repl.js +41 -1
- package/dist/cli/run.d.ts +6 -0
- package/dist/cli/run.js +76 -3
- package/dist/events/catalog.d.ts +9 -0
- package/dist/events/catalog.js +9 -0
- package/dist/index.js +89 -5
- package/dist/mcp/client.js +1 -1
- package/dist/mcp/registry.d.ts +0 -18
- package/dist/mcp/registry.js +49 -2
- package/dist/policy/engine.d.ts +16 -0
- package/dist/policy/engine.js +74 -1
- package/dist/policy/path-guard.d.ts +24 -0
- package/dist/policy/path-guard.js +46 -0
- package/dist/providers/model-info.d.ts +6 -0
- package/dist/providers/model-info.js +8 -0
- package/dist/tools/fs/apply-patch.js +6 -1
- package/dist/tools/fs/edit-file.js +4 -1
- package/dist/tools/fs/multi-edit.js +4 -1
- package/dist/tools/fs/write-file.js +16 -6
- package/dist/tools/plan/todo-write.js +1 -1
- package/dist/tools/shell/shell-exec.d.ts +28 -0
- package/dist/tools/shell/shell-exec.js +87 -1
- package/dist/trace/writer.d.ts +7 -0
- package/dist/trace/writer.js +7 -0
- package/dist/verification/classify.js +4 -3
- package/dist/verification/engine.d.ts +8 -0
- package/dist/verification/engine.js +25 -0
- package/dist/verification/registry.js +16 -5
- package/dist/verification/scoped.js +36 -5
- package/package.json +1 -1
|
@@ -72,6 +72,9 @@ interface AnthropicRequest {
|
|
|
72
72
|
name: string;
|
|
73
73
|
description: string;
|
|
74
74
|
input_schema: unknown;
|
|
75
|
+
cache_control?: {
|
|
76
|
+
type: 'ephemeral';
|
|
77
|
+
};
|
|
75
78
|
}>;
|
|
76
79
|
max_tokens: number;
|
|
77
80
|
temperature?: number;
|
|
@@ -94,6 +97,14 @@ export declare function anthropicAdapter(opts: AnthropicAdapterOptions): Provide
|
|
|
94
97
|
* Exported via _internal for testing.
|
|
95
98
|
*/
|
|
96
99
|
export declare function buildAnthropicSystem(system: string | undefined, suffix: string | undefined, promptCache: boolean): AnthropicRequest['system'];
|
|
100
|
+
/**
|
|
101
|
+
* Build the Anthropic `tools` array. When prompt caching is enabled, the
|
|
102
|
+
* last tool carries a `cache_control: {type:'ephemeral'}` breakpoint so
|
|
103
|
+
* the (usually stable) tool definitions join the cacheable prefix —
|
|
104
|
+
* mirroring the system-text breakpoint. OpenAI path untouched.
|
|
105
|
+
* Exported via _internal for testing.
|
|
106
|
+
*/
|
|
107
|
+
export declare function buildAnthropicTools(tools: ToolDefinition[], promptCache: boolean): AnthropicRequest['tools'];
|
|
97
108
|
/**
|
|
98
109
|
* Mutable per-stream assembly state. Blocks are keyed by content_block
|
|
99
110
|
* index; the tool id is carried inside the block entry. There is no global
|
|
@@ -118,15 +129,12 @@ interface AnthropicStreamState {
|
|
|
118
129
|
}
|
|
119
130
|
declare function translateSse(event: string, parsed: AnthropicSseEvent, state: AnthropicStreamState): StreamEvent[];
|
|
120
131
|
declare function toAnthropicMessages(messages: Message[]): AnthropicMessage[];
|
|
121
|
-
declare function toAnthropicTool(t: ToolDefinition):
|
|
122
|
-
name: string;
|
|
123
|
-
description: string;
|
|
124
|
-
input_schema: unknown;
|
|
125
|
-
};
|
|
132
|
+
declare function toAnthropicTool(t: ToolDefinition): NonNullable<AnthropicRequest['tools']>[number];
|
|
126
133
|
export declare const _internal: {
|
|
127
134
|
toAnthropicMessages: typeof toAnthropicMessages;
|
|
128
135
|
toAnthropicTool: typeof toAnthropicTool;
|
|
129
136
|
translateSse: typeof translateSse;
|
|
130
137
|
buildAnthropicSystem: typeof buildAnthropicSystem;
|
|
138
|
+
buildAnthropicTools: typeof buildAnthropicTools;
|
|
131
139
|
};
|
|
132
140
|
export {};
|
|
@@ -75,12 +75,29 @@ export function buildAnthropicSystem(system, suffix, promptCache) {
|
|
|
75
75
|
return undefined;
|
|
76
76
|
return promptCache ? [{ type: 'text', text: system, ...breakpoint }] : system;
|
|
77
77
|
}
|
|
78
|
+
/**
|
|
79
|
+
* Build the Anthropic `tools` array. When prompt caching is enabled, the
|
|
80
|
+
* last tool carries a `cache_control: {type:'ephemeral'}` breakpoint so
|
|
81
|
+
* the (usually stable) tool definitions join the cacheable prefix —
|
|
82
|
+
* mirroring the system-text breakpoint. OpenAI path untouched.
|
|
83
|
+
* Exported via _internal for testing.
|
|
84
|
+
*/
|
|
85
|
+
export function buildAnthropicTools(tools, promptCache) {
|
|
86
|
+
if (tools.length === 0)
|
|
87
|
+
return undefined;
|
|
88
|
+
const out = tools.map(toAnthropicTool);
|
|
89
|
+
if (promptCache) {
|
|
90
|
+
const last = out[out.length - 1];
|
|
91
|
+
last.cache_control = { type: 'ephemeral' };
|
|
92
|
+
}
|
|
93
|
+
return out;
|
|
94
|
+
}
|
|
78
95
|
async function* streamAnthropic(req, opts) {
|
|
79
96
|
const body = {
|
|
80
97
|
model: req.model,
|
|
81
98
|
system: buildAnthropicSystem(req.system, req.systemSuffix, opts.promptCache),
|
|
82
99
|
messages: toAnthropicMessages(req.messages),
|
|
83
|
-
tools: req.tools
|
|
100
|
+
tools: buildAnthropicTools(req.tools, opts.promptCache),
|
|
84
101
|
max_tokens: req.maxTokens ?? 4096,
|
|
85
102
|
temperature: req.temperature,
|
|
86
103
|
stream: true,
|
|
@@ -430,4 +447,4 @@ function toAnthropicTool(t) {
|
|
|
430
447
|
};
|
|
431
448
|
}
|
|
432
449
|
// Re-export for testability.
|
|
433
|
-
export const _internal = { toAnthropicMessages, toAnthropicTool, translateSse, buildAnthropicSystem };
|
|
450
|
+
export const _internal = { toAnthropicMessages, toAnthropicTool, translateSse, buildAnthropicSystem, buildAnthropicTools };
|
|
@@ -187,5 +187,11 @@ export const DEFAULT_SPAWN_TOOLS = new Set([
|
|
|
187
187
|
]);
|
|
188
188
|
/** Default deny-list — these are NEVER allowed, even if explicitly requested. */
|
|
189
189
|
export const DEFAULT_DENIED_TOOLS = new Set([
|
|
190
|
-
//
|
|
190
|
+
// Intentionally empty. Deny happens per-pattern (shellDenyRule /
|
|
191
|
+
// DANGEROUS_PATTERNS, .env guards, repair-guard), not per-tool: a
|
|
192
|
+
// tool-granularity deny-all entry (e.g. banning `shell_exec` outright)
|
|
193
|
+
// would break legitimate flows that rely on the allowlist + approval
|
|
194
|
+
// path. Seed candidates considered and rejected: `shell_exec` (needed
|
|
195
|
+
// for tests/builds via approval), `run_verify` (needed by tester/
|
|
196
|
+
// implementer agents), `write_file`/`edit_file` (core agent function).
|
|
191
197
|
]);
|
|
@@ -12,7 +12,7 @@
|
|
|
12
12
|
* The compact result is a `ChildSummary` — a `ToolResult` the parent model
|
|
13
13
|
* can act on — never the full child transcript.
|
|
14
14
|
*/
|
|
15
|
-
import type { RuntimeDeps } from './runtime.js';
|
|
15
|
+
import type { RuntimeDeps, RuntimeEvent } from './runtime.js';
|
|
16
16
|
import type { ToolResult } from '../tools/types.js';
|
|
17
17
|
import { TaskManager, type TaskRecord, type TaskStatus, type TaskSummary } from './task-manager.js';
|
|
18
18
|
import { WorkerSpawner } from './worker-spawner.js';
|
|
@@ -157,6 +157,23 @@ export interface OrchestratorOpts {
|
|
|
157
157
|
*/
|
|
158
158
|
isTui?: boolean;
|
|
159
159
|
}
|
|
160
|
+
/**
|
|
161
|
+
* Build a `subtask.progress` note for one finished tool call.
|
|
162
|
+
* Pure — unit-tested directly (see agent-tools.test.ts).
|
|
163
|
+
*/
|
|
164
|
+
export declare function progressNote(step: number, tool: string, isError: boolean): string;
|
|
165
|
+
/**
|
|
166
|
+
* Build the `RunOptions.onEvent` handler the orchestrator passes into each
|
|
167
|
+
* child's run options. Emits at most one `subtask.progress` per tool call:
|
|
168
|
+
* a `tool_result` is only mirrored when its `tool_call_end` was observed
|
|
169
|
+
* first, so duplicate/late results can never double-emit. (The note needs
|
|
170
|
+
* the ok/ERR outcome, which only `tool_result` carries — `tool_call_end`
|
|
171
|
+
* alone cannot build it — hence the end-gated result throttle.)
|
|
172
|
+
*/
|
|
173
|
+
export declare function createSubtaskProgressEmitter(opts: {
|
|
174
|
+
taskId: string;
|
|
175
|
+
sessionId: string;
|
|
176
|
+
}): (ev: RuntimeEvent) => void;
|
|
160
177
|
export declare class AgentOrchestrator {
|
|
161
178
|
readonly sessionId: string;
|
|
162
179
|
readonly deps: RuntimeDeps;
|
|
@@ -87,6 +87,45 @@ function mapResultStatus(status) {
|
|
|
87
87
|
return 'failed';
|
|
88
88
|
}
|
|
89
89
|
}
|
|
90
|
+
/**
|
|
91
|
+
* Build a `subtask.progress` note for one finished tool call.
|
|
92
|
+
* Pure — unit-tested directly (see agent-tools.test.ts).
|
|
93
|
+
*/
|
|
94
|
+
export function progressNote(step, tool, isError) {
|
|
95
|
+
return `step ${step}: ${tool} ${isError ? 'ERR' : 'ok'}`;
|
|
96
|
+
}
|
|
97
|
+
/**
|
|
98
|
+
* Build the `RunOptions.onEvent` handler the orchestrator passes into each
|
|
99
|
+
* child's run options. Emits at most one `subtask.progress` per tool call:
|
|
100
|
+
* a `tool_result` is only mirrored when its `tool_call_end` was observed
|
|
101
|
+
* first, so duplicate/late results can never double-emit. (The note needs
|
|
102
|
+
* the ok/ERR outcome, which only `tool_result` carries — `tool_call_end`
|
|
103
|
+
* alone cannot build it — hence the end-gated result throttle.)
|
|
104
|
+
*/
|
|
105
|
+
export function createSubtaskProgressEmitter(opts) {
|
|
106
|
+
let step = 0;
|
|
107
|
+
const ended = new Set();
|
|
108
|
+
return (ev) => {
|
|
109
|
+
if (ev.kind === 'step_start') {
|
|
110
|
+
step = ev.step;
|
|
111
|
+
}
|
|
112
|
+
else if (ev.kind === 'tool_call_end') {
|
|
113
|
+
ended.add(ev.id);
|
|
114
|
+
}
|
|
115
|
+
else if (ev.kind === 'tool_result') {
|
|
116
|
+
if (!ended.has(ev.id))
|
|
117
|
+
return;
|
|
118
|
+
ended.delete(ev.id);
|
|
119
|
+
globalBus.emit({
|
|
120
|
+
type: 'subtask.progress',
|
|
121
|
+
ts: Date.now(),
|
|
122
|
+
sessionId: opts.sessionId,
|
|
123
|
+
taskId: opts.taskId,
|
|
124
|
+
note: progressNote(step, ev.name, ev.isError),
|
|
125
|
+
});
|
|
126
|
+
}
|
|
127
|
+
};
|
|
128
|
+
}
|
|
90
129
|
export class AgentOrchestrator {
|
|
91
130
|
sessionId;
|
|
92
131
|
deps;
|
|
@@ -212,14 +251,17 @@ export class AgentOrchestrator {
|
|
|
212
251
|
const registryTools = new Set(this.deps.registry.list().map((t) => t.name));
|
|
213
252
|
const resolved = this.resolveChild(def, parent, registryTools);
|
|
214
253
|
const childModel = input.model ?? resolved.model ?? parent.model;
|
|
215
|
-
// Worktree isolation: write-capable children
|
|
216
|
-
//
|
|
217
|
-
//
|
|
254
|
+
// Worktree isolation: write-capable children get their own worktree —
|
|
255
|
+
// including when the spawn carries an explicit cwd (the worktree is
|
|
256
|
+
// then rooted at the resolved explicit cwd, which containment above
|
|
257
|
+
// already pinned inside the parent). Readonly agents keep the resolved
|
|
258
|
+
// cwd with no worktree. A write-capable spawn outside a git repo is
|
|
259
|
+
// rejected outright.
|
|
218
260
|
const writeCapable = [...resolved.allowed].some((t) => DEFAULT_WRITE_TOOLS.has(t));
|
|
219
261
|
let childCwd = baseCwd;
|
|
220
262
|
let worktree;
|
|
221
263
|
let repoCwd;
|
|
222
|
-
if (!resolved.readonly && writeCapable
|
|
264
|
+
if (!resolved.readonly && writeCapable) {
|
|
223
265
|
const isRepo = await ensureGitRepo(baseCwd).catch(() => false);
|
|
224
266
|
if (!isRepo) {
|
|
225
267
|
return {
|
|
@@ -285,6 +327,11 @@ export class AgentOrchestrator {
|
|
|
285
327
|
maxTimeMs: def.maxTimeMs ?? input.timeoutMs,
|
|
286
328
|
signal: record.abortController.signal,
|
|
287
329
|
nonInteractive: true,
|
|
330
|
+
// Mid-life progress: mirror each finished tool call as one
|
|
331
|
+
// `subtask.progress` bus event (see createSubtaskProgressEmitter).
|
|
332
|
+
// Process-isolated children don't run this closure — only the
|
|
333
|
+
// in-process path reports mid-life progress.
|
|
334
|
+
onEvent: createSubtaskProgressEmitter({ taskId: record.id, sessionId: this.sessionId }),
|
|
288
335
|
// Grandchildren: only children that canSpawn receive the bridge —
|
|
289
336
|
// otherwise tools see NO_ORCHESTRATOR as before.
|
|
290
337
|
...(resolved.canSpawn ? { agentBridge: this.bridgeFor(childRef) } : {}),
|
package/dist/agent/retry.js
CHANGED
|
@@ -17,6 +17,7 @@
|
|
|
17
17
|
*
|
|
18
18
|
* The default policy matches the L6 plan: 5 attempts, 500ms base, 8s cap.
|
|
19
19
|
*/
|
|
20
|
+
import { acquireStreamSlot, noteRateLimited } from './stream-budget.js';
|
|
20
21
|
export const DEFAULT_RETRY = {
|
|
21
22
|
maxAttempts: 5,
|
|
22
23
|
baseMs: 500,
|
|
@@ -88,6 +89,28 @@ export function computeBackoff(attempt, baseMs, maxMs) {
|
|
|
88
89
|
const jitter = exp * 0.25 * (Math.random() * 2 - 1);
|
|
89
90
|
return Math.max(0, Math.floor(exp + jitter));
|
|
90
91
|
}
|
|
92
|
+
/**
|
|
93
|
+
* True when a retryable error event carries a 429 rate-limit signal.
|
|
94
|
+
* Checks the `status` field first, then `code`; accepts numeric values
|
|
95
|
+
* and `'429'` substrings (e.g. `'429'`, `'HTTP_429'`).
|
|
96
|
+
*/
|
|
97
|
+
function isRateLimitedError(ev) {
|
|
98
|
+
if (ev.kind !== 'error')
|
|
99
|
+
return false;
|
|
100
|
+
const rec = ev;
|
|
101
|
+
for (const key of ['status', 'code']) {
|
|
102
|
+
const value = rec[key];
|
|
103
|
+
if (typeof value === 'string' && value.includes('429'))
|
|
104
|
+
return true;
|
|
105
|
+
if (typeof value === 'number' && String(value).includes('429'))
|
|
106
|
+
return true;
|
|
107
|
+
}
|
|
108
|
+
return false;
|
|
109
|
+
}
|
|
110
|
+
function rateLimitDelayMs(ev) {
|
|
111
|
+
const raw = ev['retryAfterMs'];
|
|
112
|
+
return typeof raw === 'number' && Number.isFinite(raw) && raw >= 0 ? raw : undefined;
|
|
113
|
+
}
|
|
91
114
|
export function retryingAdapter(inner, opts = {}) {
|
|
92
115
|
const cfg = { ...DEFAULT_RETRY, ...opts };
|
|
93
116
|
const sleep = opts.sleep ?? defaultSleep;
|
|
@@ -129,20 +152,39 @@ export function retryingAdapter(inner, opts = {}) {
|
|
|
129
152
|
opts.onAttempt?.(attempt);
|
|
130
153
|
if (effectiveSignal?.aborted)
|
|
131
154
|
return;
|
|
155
|
+
// Rate-limit scheduler: hold one stream slot for the duration of
|
|
156
|
+
// this attempt's inner.stream consumption. Abort while queued ends
|
|
157
|
+
// the stream promptly with no inner call.
|
|
158
|
+
let release;
|
|
132
159
|
let sawRetryable = false;
|
|
133
160
|
let lastError = null;
|
|
134
|
-
|
|
135
|
-
|
|
161
|
+
try {
|
|
162
|
+
try {
|
|
163
|
+
release = await acquireStreamSlot(effectiveSignal);
|
|
164
|
+
}
|
|
165
|
+
catch {
|
|
136
166
|
return;
|
|
137
|
-
if (ev.kind === 'error' && ev.retryable) {
|
|
138
|
-
// Buffer the retryable error; don't yield it yet. We'll either
|
|
139
|
-
// re-issue (and the caller will never see the error) or, on
|
|
140
|
-
// final attempt, yield it as the terminal error.
|
|
141
|
-
sawRetryable = true;
|
|
142
|
-
lastError = ev;
|
|
143
|
-
break; // stop consuming; the stream is dead on retryable errors.
|
|
144
167
|
}
|
|
145
|
-
|
|
168
|
+
for await (const ev of streamWithAbort(inner.stream(attemptReq), effectiveSignal)) {
|
|
169
|
+
if (effectiveSignal?.aborted)
|
|
170
|
+
return;
|
|
171
|
+
if (ev.kind === 'error' && ev.retryable) {
|
|
172
|
+
// Adaptive throttling: a 429 collapses the global stream cap
|
|
173
|
+
// to 1 for retryAfterMs (or 60s) — see stream-budget.ts.
|
|
174
|
+
if (isRateLimitedError(ev))
|
|
175
|
+
noteRateLimited(rateLimitDelayMs(ev));
|
|
176
|
+
// Buffer the retryable error; don't yield it yet. We'll either
|
|
177
|
+
// re-issue (and the caller will never see the error) or, on
|
|
178
|
+
// final attempt, yield it as the terminal error.
|
|
179
|
+
sawRetryable = true;
|
|
180
|
+
lastError = ev;
|
|
181
|
+
break; // stop consuming; the stream is dead on retryable errors.
|
|
182
|
+
}
|
|
183
|
+
yield ev;
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
finally {
|
|
187
|
+
release?.();
|
|
146
188
|
}
|
|
147
189
|
if (!sawRetryable)
|
|
148
190
|
return; // success or non-retryable error — done.
|
package/dist/agent/runtime.d.ts
CHANGED
|
@@ -31,6 +31,13 @@ export type VerifyMode = typeof import('../verification/engine.js') extends {
|
|
|
31
31
|
} ? V : 'strict' | 'advisory' | 'off';
|
|
32
32
|
export interface RuntimeDeps {
|
|
33
33
|
adapter: ProviderAdapter;
|
|
34
|
+
/**
|
|
35
|
+
* Ordered failover adapters (L15). When the active adapter ends a step
|
|
36
|
+
* with a terminal provider error, the runtime swaps to the next entry
|
|
37
|
+
* and re-issues the step (bounded by chain length, never loops).
|
|
38
|
+
* Optional — single-adapter callers behave exactly as before.
|
|
39
|
+
*/
|
|
40
|
+
failoverAdapters?: ProviderAdapter[];
|
|
34
41
|
registry: ToolRegistry;
|
|
35
42
|
policy: PolicyEngine;
|
|
36
43
|
approval: ApprovalPrompt;
|
|
@@ -226,6 +233,22 @@ export type RuntimeEvent = {
|
|
|
226
233
|
} | {
|
|
227
234
|
kind: 'checkpoint_saved';
|
|
228
235
|
sessionId: string;
|
|
236
|
+
} | {
|
|
237
|
+
kind: 'status';
|
|
238
|
+
message: string;
|
|
239
|
+
} | {
|
|
240
|
+
kind: 'budget_warning';
|
|
241
|
+
ratio: number;
|
|
242
|
+
threshold: number;
|
|
243
|
+
} | {
|
|
244
|
+
kind: 'provider_failover';
|
|
245
|
+
from: string;
|
|
246
|
+
to: string;
|
|
247
|
+
reason: string;
|
|
248
|
+
} | {
|
|
249
|
+
kind: 'model_override';
|
|
250
|
+
requested: string;
|
|
251
|
+
effective: string;
|
|
229
252
|
};
|
|
230
253
|
export interface RunResult {
|
|
231
254
|
status: 'complete' | 'max_steps' | 'aborted' | 'no_final' | 'verify_failed' | 'limit' | 'blocked' | 'stuck';
|
|
@@ -264,7 +287,18 @@ export declare function toolDefinitions(registry: ToolRegistry): ToolDefinition[
|
|
|
264
287
|
export declare function estimateCost(model: string, usage: {
|
|
265
288
|
input: number;
|
|
266
289
|
output: number;
|
|
290
|
+
cacheRead?: number;
|
|
291
|
+
cacheWrite?: number;
|
|
267
292
|
}): number;
|
|
293
|
+
/**
|
|
294
|
+
* Parallel fan-out cap: approved concurrencySafe tool calls execute in
|
|
295
|
+
* sequential chunks of at most this size. Commit order stays identical
|
|
296
|
+
* (commits run sequentially after execution), so the transcript reads as
|
|
297
|
+
* if the calls ran in order.
|
|
298
|
+
*/
|
|
299
|
+
export declare const MAX_PARALLEL_TOOLS = 8;
|
|
300
|
+
/** Progressive budget-warning thresholds (fraction of maxCost), fired once each per run. */
|
|
301
|
+
export declare const BUDGET_WARNING_THRESHOLDS: readonly [0.4, 0.7, 0.9];
|
|
268
302
|
/** Run the autonomous loop. */
|
|
269
303
|
export declare function run(opts: RunOptions, deps: RuntimeDeps): Promise<RunResult>;
|
|
270
304
|
export declare function defaultSystemPrompt(ctx: {
|
package/dist/agent/runtime.js
CHANGED
|
@@ -23,11 +23,12 @@ import { verify, diagnosticForModel } from '../verification/engine.js';
|
|
|
23
23
|
import { detectVerifyCommand } from '../verification/auto.js';
|
|
24
24
|
import { ensureBaseline, getBaseline } from '../verification/baseline.js';
|
|
25
25
|
import { compressTranscript, totalTokens } from '../context/tokenizer.js';
|
|
26
|
-
import { ratesFor } from '../providers/model-info.js';
|
|
26
|
+
import { ratesFor, isAnthropicModel } from '../providers/model-info.js';
|
|
27
27
|
import { classifyFailure, rerunOnce, gatherRepairContext, guardRepair } from '../verification/classify.js';
|
|
28
28
|
import { findRelatedTests, buildScopedCommand, runScopedVerify, syntaxCheck, checkImports } from '../verification/scoped.js';
|
|
29
29
|
import { globalBus } from '../events/bus.js';
|
|
30
30
|
import { TraceWriter } from '../trace/writer.js';
|
|
31
|
+
import { loadHooks, runHook } from '../cli/hooks.js';
|
|
31
32
|
/** Normalize either systemPrompt shape into {system, suffix}. */
|
|
32
33
|
export function resolveSystemPrompt(fn, ctx) {
|
|
33
34
|
const r = fn(ctx);
|
|
@@ -42,18 +43,30 @@ export function toolDefinitions(registry) {
|
|
|
42
43
|
inputSchema: t.function.parameters,
|
|
43
44
|
}));
|
|
44
45
|
}
|
|
45
|
-
//
|
|
46
|
+
// Model-aware cost estimation, single-sourced from the
|
|
46
47
|
// providers/model-info.ts rate table (local/unknown models are $0).
|
|
47
|
-
//
|
|
48
|
-
//
|
|
49
|
-
//
|
|
50
|
-
// full input rates would overstate spend, silently dropping them
|
|
51
|
-
// understates it, so we keep them visible and out of the math.
|
|
48
|
+
// Cache-aware: for Anthropic-family models (isAnthropicModel), cacheRead
|
|
49
|
+
// bills at 0.1× the input rate and cacheWrite at 1.25×; all other
|
|
50
|
+
// families ignore cache counters (discounted billing, unmodeled).
|
|
52
51
|
/** Estimate USD cost of a usage block given the model name. */
|
|
53
52
|
export function estimateCost(model, usage) {
|
|
54
53
|
const { input: inRate, output: outRate } = ratesFor(model);
|
|
55
|
-
|
|
54
|
+
const base = (usage.input / 1000) * inRate + (usage.output / 1000) * outRate;
|
|
55
|
+
if (!isAnthropicModel(model))
|
|
56
|
+
return base;
|
|
57
|
+
const read = ((usage.cacheRead ?? 0) / 1000) * inRate * 0.1;
|
|
58
|
+
const write = ((usage.cacheWrite ?? 0) / 1000) * inRate * 1.25;
|
|
59
|
+
return base + read + write;
|
|
56
60
|
}
|
|
61
|
+
/**
|
|
62
|
+
* Parallel fan-out cap: approved concurrencySafe tool calls execute in
|
|
63
|
+
* sequential chunks of at most this size. Commit order stays identical
|
|
64
|
+
* (commits run sequentially after execution), so the transcript reads as
|
|
65
|
+
* if the calls ran in order.
|
|
66
|
+
*/
|
|
67
|
+
export const MAX_PARALLEL_TOOLS = 8;
|
|
68
|
+
/** Progressive budget-warning thresholds (fraction of maxCost), fired once each per run. */
|
|
69
|
+
export const BUDGET_WARNING_THRESHOLDS = [0.4, 0.7, 0.9];
|
|
57
70
|
// PERF-002: Memoized token counting cache.
|
|
58
71
|
let tokenCache = {
|
|
59
72
|
lastRef: null,
|
|
@@ -90,6 +103,18 @@ export async function run(opts, deps) {
|
|
|
90
103
|
return [{ role: 'user', content: [text(opts.task)] }];
|
|
91
104
|
})();
|
|
92
105
|
const usage = { input: 0, output: 0 };
|
|
106
|
+
/** Emit one `budget_warning` per threshold the cost ratio has crossed. */
|
|
107
|
+
const checkBudgetWarnings = () => {
|
|
108
|
+
if (maxCost === undefined || maxCost <= 0)
|
|
109
|
+
return;
|
|
110
|
+
const ratio = estimateCost(opts.model, usage) / maxCost;
|
|
111
|
+
for (const threshold of BUDGET_WARNING_THRESHOLDS) {
|
|
112
|
+
if (ratio >= threshold && !firedBudgetWarnings.has(threshold)) {
|
|
113
|
+
firedBudgetWarnings.add(threshold);
|
|
114
|
+
emit?.({ kind: 'budget_warning', ratio, threshold });
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
};
|
|
93
118
|
let steps = 0;
|
|
94
119
|
let toolCallCount = 0;
|
|
95
120
|
let finalText = '';
|
|
@@ -120,6 +145,27 @@ export async function run(opts, deps) {
|
|
|
120
145
|
const emit = opts.onEvent;
|
|
121
146
|
const telemetry = new RuntimeTelemetry();
|
|
122
147
|
telemetry.setMaxSteps(maxSteps);
|
|
148
|
+
// Model-override surfacing (informational): when a parent/orchestrator
|
|
149
|
+
// context carries a model override, emit it once so UIs can show which
|
|
150
|
+
// model actually serves this run.
|
|
151
|
+
if (opts.parentContext?.model) {
|
|
152
|
+
emit?.({ kind: 'model_override', requested: opts.model, effective: opts.parentContext.model });
|
|
153
|
+
}
|
|
154
|
+
// Hooks engine: loaded once per run. Zero-cost fast path — when no hooks
|
|
155
|
+
// file exists, both lists are empty and every hook call site is skipped.
|
|
156
|
+
let runHooks = [];
|
|
157
|
+
try {
|
|
158
|
+
runHooks = loadHooks(opts.cwd);
|
|
159
|
+
}
|
|
160
|
+
catch {
|
|
161
|
+
runHooks = [];
|
|
162
|
+
}
|
|
163
|
+
const preHooks = runHooks.filter((h) => h.event === 'preToolUse');
|
|
164
|
+
const postHooks = runHooks.filter((h) => h.event === 'postToolUse');
|
|
165
|
+
// L15 failover chain: the active adapter starts as deps.adapter; each
|
|
166
|
+
// terminal provider error consumes one fallback. Bounded — never loops.
|
|
167
|
+
let activeAdapter = deps.adapter;
|
|
168
|
+
const failoverQueue = [...(deps.failoverAdapters ?? [])];
|
|
123
169
|
// 3.1 — Event bus + TraceWriter
|
|
124
170
|
const bus = deps.bus ?? globalBus;
|
|
125
171
|
let tracer;
|
|
@@ -140,6 +186,10 @@ export async function run(opts, deps) {
|
|
|
140
186
|
// 5.1 — phases and limits
|
|
141
187
|
const maxCost = opts.maxCost;
|
|
142
188
|
const maxTimeMs = opts.maxTimeMs;
|
|
189
|
+
// Progressive budget warnings: fire once per threshold per run when the
|
|
190
|
+
// cost ratio crosses 0.4 / 0.7 / 0.9 of maxCost (checker defined after
|
|
191
|
+
// `usage` is declared below).
|
|
192
|
+
const firedBudgetWarnings = new Set();
|
|
143
193
|
const startTime = Date.now();
|
|
144
194
|
let phase = 'understanding';
|
|
145
195
|
const setPhase = (p) => {
|
|
@@ -270,7 +320,7 @@ export async function run(opts, deps) {
|
|
|
270
320
|
...(typeof opts.temperature === 'number' ? { temperature: opts.temperature } : {}),
|
|
271
321
|
...(opts.signal ? { signal: opts.signal } : {}),
|
|
272
322
|
};
|
|
273
|
-
const events =
|
|
323
|
+
const events = activeAdapter.stream(req);
|
|
274
324
|
let textBuf = '';
|
|
275
325
|
// Thinking is ephemeral: streamed to the UI live, never stored in the
|
|
276
326
|
// transcript, and cleared when the turn's answer completes.
|
|
@@ -279,6 +329,12 @@ export async function run(opts, deps) {
|
|
|
279
329
|
let lastFinishReason;
|
|
280
330
|
// Set when this step's request must be re-issued after overflow recovery.
|
|
281
331
|
let overflowRetryPending = false;
|
|
332
|
+
// Set when a terminal provider error consumed a failover adapter — the
|
|
333
|
+
// step is re-issued against the next adapter without consuming budget.
|
|
334
|
+
let failoverPending = false;
|
|
335
|
+
let failoverFrom = '';
|
|
336
|
+
let failoverTo = '';
|
|
337
|
+
let failoverReason = '';
|
|
282
338
|
for await (const ev of events) {
|
|
283
339
|
if (opts.signal?.aborted)
|
|
284
340
|
break outer;
|
|
@@ -318,6 +374,7 @@ export async function run(opts, deps) {
|
|
|
318
374
|
...(usage.cacheRead !== undefined ? { cacheRead: usage.cacheRead } : {}),
|
|
319
375
|
...(usage.cacheWrite !== undefined ? { cacheWrite: usage.cacheWrite } : {}),
|
|
320
376
|
});
|
|
377
|
+
checkBudgetWarnings();
|
|
321
378
|
}
|
|
322
379
|
else {
|
|
323
380
|
// Providers that omit usage (Ollama, vLLM, proxies): estimate from
|
|
@@ -329,6 +386,7 @@ export async function run(opts, deps) {
|
|
|
329
386
|
usage.estimated = true;
|
|
330
387
|
telemetry.recordUsage(est.input, est.output);
|
|
331
388
|
emit?.({ kind: 'usage', input: usage.input, output: usage.output, estimated: true });
|
|
389
|
+
checkBudgetWarnings();
|
|
332
390
|
}
|
|
333
391
|
}
|
|
334
392
|
else if (ev.kind === 'error') {
|
|
@@ -357,6 +415,24 @@ export async function run(opts, deps) {
|
|
|
357
415
|
catch { /* ignore — retry with the transcript as-is */ }
|
|
358
416
|
break;
|
|
359
417
|
}
|
|
418
|
+
// L15 failover: a terminal provider error swaps to the next chained
|
|
419
|
+
// adapter and re-issues the step (bounded by chain length). Context
|
|
420
|
+
// overflow is excluded — it owns its own recovery above. By the time
|
|
421
|
+
// an error reaches the runtime, per-adapter retries are exhausted,
|
|
422
|
+
// so any provider error here is terminal for the active adapter.
|
|
423
|
+
if (failoverQueue.length > 0) {
|
|
424
|
+
const next = failoverQueue.shift();
|
|
425
|
+
failoverPending = true;
|
|
426
|
+
failoverFrom = activeAdapter.id;
|
|
427
|
+
failoverTo = next.id;
|
|
428
|
+
failoverReason = `${ev.code}: ${ev.message}`.slice(0, 300);
|
|
429
|
+
activeAdapter = next;
|
|
430
|
+
telemetry.recordError(`failover: ${ev.code}`);
|
|
431
|
+
emit?.({ kind: 'provider_failover', from: failoverFrom, to: failoverTo, reason: failoverReason });
|
|
432
|
+
emit?.({ kind: 'status', message: `provider ${failoverFrom} failed (${ev.code}) — failing over to ${failoverTo}` });
|
|
433
|
+
emitKlyro({ type: 'error', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', code: ev.code, message: `failing over ${failoverFrom} → ${failoverTo}: ${ev.message.slice(0, 200)}` });
|
|
434
|
+
break;
|
|
435
|
+
}
|
|
360
436
|
telemetry.recordError(`stream_error: ${ev.code}`);
|
|
361
437
|
if (store && sessionId) {
|
|
362
438
|
try {
|
|
@@ -386,6 +462,19 @@ export async function run(opts, deps) {
|
|
|
386
462
|
emit?.({ kind: 'step_end', step: steps + 1 });
|
|
387
463
|
continue outer;
|
|
388
464
|
}
|
|
465
|
+
// Failover lands here via `break`: discard the failed attempt's partial
|
|
466
|
+
// output and re-issue the same step against the next adapter, again
|
|
467
|
+
// without consuming the step budget.
|
|
468
|
+
if (failoverPending) {
|
|
469
|
+
failoverPending = false;
|
|
470
|
+
textBuf = '';
|
|
471
|
+
thinkingBuf = '';
|
|
472
|
+
pendingToolCalls.clear();
|
|
473
|
+
lastFinishReason = undefined;
|
|
474
|
+
steps--;
|
|
475
|
+
emit?.({ kind: 'step_end', step: steps + 1 });
|
|
476
|
+
continue outer;
|
|
477
|
+
}
|
|
389
478
|
// Build the assistant message. Tool calls are finalized here: JSON is
|
|
390
479
|
// parsed and schema-validated BEFORE policy/execution. Malformed calls
|
|
391
480
|
// become structured MALFORMED_TOOL_CALL results — garbage arguments must
|
|
@@ -478,7 +567,8 @@ export async function run(opts, deps) {
|
|
|
478
567
|
emit?.({ kind: 'verification_started', command: advisoryCmd });
|
|
479
568
|
let advisoryResult;
|
|
480
569
|
try {
|
|
481
|
-
|
|
570
|
+
// CONTRACT (a): sessionId passthrough to the verify engine.
|
|
571
|
+
advisoryResult = await verify({ cwd: opts.cwd, command: advisoryCmd, timeoutMs: opts.verify?.timeoutMs, ...(sessionId ? { sessionId } : {}) });
|
|
482
572
|
}
|
|
483
573
|
catch (e) {
|
|
484
574
|
const msg = e instanceof Error ? e.message : String(e);
|
|
@@ -557,7 +647,7 @@ export async function run(opts, deps) {
|
|
|
557
647
|
vResult = { ok: sr.ok, exitCode: sr.exitCode, stdout: sr.stdout, stderr: sr.stderr, ...(det ? { failure: det } : {}) };
|
|
558
648
|
}
|
|
559
649
|
else {
|
|
560
|
-
vResult = await verify({ cwd: opts.cwd, command: cmdToRun, timeoutMs: opts.verify?.timeoutMs });
|
|
650
|
+
vResult = await verify({ cwd: opts.cwd, command: cmdToRun, timeoutMs: opts.verify?.timeoutMs, ...(sessionId ? { sessionId } : {}) });
|
|
561
651
|
}
|
|
562
652
|
}
|
|
563
653
|
catch (e) {
|
|
@@ -568,7 +658,7 @@ export async function run(opts, deps) {
|
|
|
568
658
|
// If scoped passed but full may still fail, run full before declaring success
|
|
569
659
|
if (vResult.ok && isScoped) {
|
|
570
660
|
try {
|
|
571
|
-
const full = await verify({ cwd: opts.cwd, command: verifyCmd, timeoutMs: opts.verify?.timeoutMs });
|
|
661
|
+
const full = await verify({ cwd: opts.cwd, command: verifyCmd, timeoutMs: opts.verify?.timeoutMs, ...(sessionId ? { sessionId } : {}) });
|
|
572
662
|
if (!full.ok)
|
|
573
663
|
vResult = full;
|
|
574
664
|
}
|
|
@@ -787,6 +877,32 @@ export async function run(opts, deps) {
|
|
|
787
877
|
const execTool = async (call) => {
|
|
788
878
|
const t0 = Date.now();
|
|
789
879
|
emitKlyro({ type: 'tool.call', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, name: call.name, input: call.input });
|
|
880
|
+
// Hooks: every preToolUse hook runs before execution. A non-zero exit
|
|
881
|
+
// denies the tool with POLICY_DENIED — the real tool never runs.
|
|
882
|
+
if (preHooks.length > 0) {
|
|
883
|
+
for (const hook of preHooks) {
|
|
884
|
+
let exitCode = -1;
|
|
885
|
+
let detail = '';
|
|
886
|
+
try {
|
|
887
|
+
const r = await runHook(hook, { toolName: call.name, input: call.input });
|
|
888
|
+
exitCode = r.exitCode;
|
|
889
|
+
detail = (r.stderr || r.stdout || '').slice(0, 300);
|
|
890
|
+
}
|
|
891
|
+
catch (err) {
|
|
892
|
+
detail = String(err instanceof Error ? err.message : err).slice(0, 300);
|
|
893
|
+
}
|
|
894
|
+
if (exitCode !== 0) {
|
|
895
|
+
const reason = `hook ${hook.name} denied: ${detail || 'hook failed'}`;
|
|
896
|
+
emit?.({ kind: 'policy_decision', id: call.id, name: call.name, action: 'deny', reason });
|
|
897
|
+
emitKlyro({ type: 'permission.decision', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, action: 'deny', reason });
|
|
898
|
+
const latencyMs = Date.now() - t0;
|
|
899
|
+
return {
|
|
900
|
+
obs: { ok: false, error: { code: 'POLICY_DENIED', message: reason } },
|
|
901
|
+
latencyMs,
|
|
902
|
+
};
|
|
903
|
+
}
|
|
904
|
+
}
|
|
905
|
+
}
|
|
790
906
|
let obs;
|
|
791
907
|
try {
|
|
792
908
|
obs = await deps.registry.execute(call.name, call.input, toolCtx);
|
|
@@ -874,6 +990,31 @@ export async function run(opts, deps) {
|
|
|
874
990
|
if (last3.length === 3 && last3[0] === last3[1] && last3[1] === last3[2]) {
|
|
875
991
|
await markStuck(`identical call ×3: ${sig}`);
|
|
876
992
|
}
|
|
993
|
+
// Hooks: postToolUse hooks are best-effort — failures warn on stderr
|
|
994
|
+
// plus a bus event, and never fail the turn.
|
|
995
|
+
if (postHooks.length > 0) {
|
|
996
|
+
for (const hook of postHooks) {
|
|
997
|
+
try {
|
|
998
|
+
const r = await runHook(hook, { toolName: call.name, input: call.input });
|
|
999
|
+
if (!r.ok || r.exitCode !== 0) {
|
|
1000
|
+
const msg = `klyro: hooks: postToolUse ${hook.name} failed (exit ${String(r.exitCode)}): ${(r.stderr || r.stdout || '').slice(0, 200)}\n`;
|
|
1001
|
+
try {
|
|
1002
|
+
process.stderr.write(msg);
|
|
1003
|
+
}
|
|
1004
|
+
catch { /* ignore */ }
|
|
1005
|
+
emitKlyro({ type: 'error', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', code: 'hook_failed', message: msg.slice(0, 300) });
|
|
1006
|
+
}
|
|
1007
|
+
}
|
|
1008
|
+
catch (err) {
|
|
1009
|
+
const msg = `klyro: hooks: postToolUse ${hook.name} error: ${String(err instanceof Error ? err.message : err).slice(0, 200)}\n`;
|
|
1010
|
+
try {
|
|
1011
|
+
process.stderr.write(msg);
|
|
1012
|
+
}
|
|
1013
|
+
catch { /* ignore */ }
|
|
1014
|
+
emitKlyro({ type: 'error', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', code: 'hook_failed', message: msg.slice(0, 300) });
|
|
1015
|
+
}
|
|
1016
|
+
}
|
|
1017
|
+
}
|
|
877
1018
|
};
|
|
878
1019
|
// Sequential path: gate → execute → commit per call, in order.
|
|
879
1020
|
const runOne = async (call) => {
|
|
@@ -897,8 +1038,17 @@ export async function run(opts, deps) {
|
|
|
897
1038
|
break;
|
|
898
1039
|
}
|
|
899
1040
|
if (approved.length > 0 && !opts.signal?.aborted) {
|
|
900
|
-
|
|
901
|
-
|
|
1041
|
+
// Fan-out cap: execute in sequential chunks of MAX_PARALLEL_TOOLS.
|
|
1042
|
+
// Commits below stay in original call order, so the transcript is
|
|
1043
|
+
// unaffected by the chunking.
|
|
1044
|
+
const settled = [];
|
|
1045
|
+
for (let off = 0; off < approved.length; off += MAX_PARALLEL_TOOLS) {
|
|
1046
|
+
if (opts.signal?.aborted)
|
|
1047
|
+
break;
|
|
1048
|
+
const chunk = approved.slice(off, off + MAX_PARALLEL_TOOLS);
|
|
1049
|
+
settled.push(...await Promise.allSettled(chunk.map((c) => execTool(c))));
|
|
1050
|
+
}
|
|
1051
|
+
for (let i = 0; i < settled.length; i++) {
|
|
902
1052
|
const s = settled[i];
|
|
903
1053
|
if (s.status === 'fulfilled') {
|
|
904
1054
|
await commitResult(approved[i], s.value.obs, s.value.latencyMs);
|