klyro 1.0.1 → 1.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/dist/agent/anthropic-adapter.d.ts +13 -5
  2. package/dist/agent/anthropic-adapter.js +19 -2
  3. package/dist/agent/capabilities.js +7 -1
  4. package/dist/agent/orchestrator.d.ts +18 -1
  5. package/dist/agent/orchestrator.js +51 -4
  6. package/dist/agent/retry.js +52 -10
  7. package/dist/agent/runtime.d.ts +34 -0
  8. package/dist/agent/runtime.js +164 -14
  9. package/dist/agent/stream-budget.d.ts +36 -0
  10. package/dist/agent/stream-budget.js +121 -0
  11. package/dist/checkpoints/store.d.ts +9 -0
  12. package/dist/checkpoints/store.js +26 -0
  13. package/dist/cli/commit.d.ts +31 -0
  14. package/dist/cli/commit.js +142 -0
  15. package/dist/cli/config.d.ts +45 -0
  16. package/dist/cli/config.js +82 -0
  17. package/dist/cli/doctor.d.ts +1 -0
  18. package/dist/cli/doctor.js +71 -6
  19. package/dist/cli/hooks.d.ts +47 -0
  20. package/dist/cli/hooks.js +181 -0
  21. package/dist/cli/repl.js +41 -1
  22. package/dist/cli/run.d.ts +6 -0
  23. package/dist/cli/run.js +76 -3
  24. package/dist/events/catalog.d.ts +9 -0
  25. package/dist/events/catalog.js +9 -0
  26. package/dist/index.js +89 -5
  27. package/dist/mcp/client.js +1 -1
  28. package/dist/mcp/registry.d.ts +0 -18
  29. package/dist/mcp/registry.js +49 -2
  30. package/dist/policy/engine.d.ts +16 -0
  31. package/dist/policy/engine.js +74 -1
  32. package/dist/policy/path-guard.d.ts +24 -0
  33. package/dist/policy/path-guard.js +46 -0
  34. package/dist/providers/model-info.d.ts +6 -0
  35. package/dist/providers/model-info.js +8 -0
  36. package/dist/tools/fs/apply-patch.js +6 -1
  37. package/dist/tools/fs/edit-file.js +4 -1
  38. package/dist/tools/fs/multi-edit.js +4 -1
  39. package/dist/tools/fs/write-file.js +16 -6
  40. package/dist/tools/plan/todo-write.js +1 -1
  41. package/dist/tools/shell/shell-exec.d.ts +28 -0
  42. package/dist/tools/shell/shell-exec.js +87 -1
  43. package/dist/trace/writer.d.ts +7 -0
  44. package/dist/trace/writer.js +7 -0
  45. package/dist/verification/classify.js +4 -3
  46. package/dist/verification/engine.d.ts +8 -0
  47. package/dist/verification/engine.js +25 -0
  48. package/dist/verification/registry.js +16 -5
  49. package/dist/verification/scoped.js +36 -5
  50. package/package.json +1 -1
@@ -72,6 +72,9 @@ interface AnthropicRequest {
72
72
  name: string;
73
73
  description: string;
74
74
  input_schema: unknown;
75
+ cache_control?: {
76
+ type: 'ephemeral';
77
+ };
75
78
  }>;
76
79
  max_tokens: number;
77
80
  temperature?: number;
@@ -94,6 +97,14 @@ export declare function anthropicAdapter(opts: AnthropicAdapterOptions): Provide
94
97
  * Exported via _internal for testing.
95
98
  */
96
99
  export declare function buildAnthropicSystem(system: string | undefined, suffix: string | undefined, promptCache: boolean): AnthropicRequest['system'];
100
+ /**
101
+ * Build the Anthropic `tools` array. When prompt caching is enabled, the
102
+ * last tool carries a `cache_control: {type:'ephemeral'}` breakpoint so
103
+ * the (usually stable) tool definitions join the cacheable prefix —
104
+ * mirroring the system-text breakpoint. OpenAI path untouched.
105
+ * Exported via _internal for testing.
106
+ */
107
+ export declare function buildAnthropicTools(tools: ToolDefinition[], promptCache: boolean): AnthropicRequest['tools'];
97
108
  /**
98
109
  * Mutable per-stream assembly state. Blocks are keyed by content_block
99
110
  * index; the tool id is carried inside the block entry. There is no global
@@ -118,15 +129,12 @@ interface AnthropicStreamState {
118
129
  }
119
130
  declare function translateSse(event: string, parsed: AnthropicSseEvent, state: AnthropicStreamState): StreamEvent[];
120
131
  declare function toAnthropicMessages(messages: Message[]): AnthropicMessage[];
121
- declare function toAnthropicTool(t: ToolDefinition): {
122
- name: string;
123
- description: string;
124
- input_schema: unknown;
125
- };
132
+ declare function toAnthropicTool(t: ToolDefinition): NonNullable<AnthropicRequest['tools']>[number];
126
133
  export declare const _internal: {
127
134
  toAnthropicMessages: typeof toAnthropicMessages;
128
135
  toAnthropicTool: typeof toAnthropicTool;
129
136
  translateSse: typeof translateSse;
130
137
  buildAnthropicSystem: typeof buildAnthropicSystem;
138
+ buildAnthropicTools: typeof buildAnthropicTools;
131
139
  };
132
140
  export {};
@@ -75,12 +75,29 @@ export function buildAnthropicSystem(system, suffix, promptCache) {
75
75
  return undefined;
76
76
  return promptCache ? [{ type: 'text', text: system, ...breakpoint }] : system;
77
77
  }
78
+ /**
79
+ * Build the Anthropic `tools` array. When prompt caching is enabled, the
80
+ * last tool carries a `cache_control: {type:'ephemeral'}` breakpoint so
81
+ * the (usually stable) tool definitions join the cacheable prefix —
82
+ * mirroring the system-text breakpoint. OpenAI path untouched.
83
+ * Exported via _internal for testing.
84
+ */
85
+ export function buildAnthropicTools(tools, promptCache) {
86
+ if (tools.length === 0)
87
+ return undefined;
88
+ const out = tools.map(toAnthropicTool);
89
+ if (promptCache) {
90
+ const last = out[out.length - 1];
91
+ last.cache_control = { type: 'ephemeral' };
92
+ }
93
+ return out;
94
+ }
78
95
  async function* streamAnthropic(req, opts) {
79
96
  const body = {
80
97
  model: req.model,
81
98
  system: buildAnthropicSystem(req.system, req.systemSuffix, opts.promptCache),
82
99
  messages: toAnthropicMessages(req.messages),
83
- tools: req.tools.length > 0 ? req.tools.map(toAnthropicTool) : undefined,
100
+ tools: buildAnthropicTools(req.tools, opts.promptCache),
84
101
  max_tokens: req.maxTokens ?? 4096,
85
102
  temperature: req.temperature,
86
103
  stream: true,
@@ -430,4 +447,4 @@ function toAnthropicTool(t) {
430
447
  };
431
448
  }
432
449
  // Re-export for testability.
433
- export const _internal = { toAnthropicMessages, toAnthropicTool, translateSse, buildAnthropicSystem };
450
+ export const _internal = { toAnthropicMessages, toAnthropicTool, translateSse, buildAnthropicSystem, buildAnthropicTools };
@@ -187,5 +187,11 @@ export const DEFAULT_SPAWN_TOOLS = new Set([
187
187
  ]);
188
188
  /** Default deny-list — these are NEVER allowed, even if explicitly requested. */
189
189
  export const DEFAULT_DENIED_TOOLS = new Set([
190
- // Add dangerous tools here. Empty by default — extend as policy matures.
190
+ // Intentionally empty. Deny happens per-pattern (shellDenyRule /
191
+ // DANGEROUS_PATTERNS, .env guards, repair-guard), not per-tool: a
192
+ // tool-granularity deny-all entry (e.g. banning `shell_exec` outright)
193
+ // would break legitimate flows that rely on the allowlist + approval
194
+ // path. Seed candidates considered and rejected: `shell_exec` (needed
195
+ // for tests/builds via approval), `run_verify` (needed by tester/
196
+ // implementer agents), `write_file`/`edit_file` (core agent function).
191
197
  ]);
@@ -12,7 +12,7 @@
12
12
  * The compact result is a `ChildSummary` — a `ToolResult` the parent model
13
13
  * can act on — never the full child transcript.
14
14
  */
15
- import type { RuntimeDeps } from './runtime.js';
15
+ import type { RuntimeDeps, RuntimeEvent } from './runtime.js';
16
16
  import type { ToolResult } from '../tools/types.js';
17
17
  import { TaskManager, type TaskRecord, type TaskStatus, type TaskSummary } from './task-manager.js';
18
18
  import { WorkerSpawner } from './worker-spawner.js';
@@ -157,6 +157,23 @@ export interface OrchestratorOpts {
157
157
  */
158
158
  isTui?: boolean;
159
159
  }
160
+ /**
161
+ * Build a `subtask.progress` note for one finished tool call.
162
+ * Pure — unit-tested directly (see agent-tools.test.ts).
163
+ */
164
+ export declare function progressNote(step: number, tool: string, isError: boolean): string;
165
+ /**
166
+ * Build the `RunOptions.onEvent` handler the orchestrator passes into each
167
+ * child's run options. Emits at most one `subtask.progress` per tool call:
168
+ * a `tool_result` is only mirrored when its `tool_call_end` was observed
169
+ * first, so duplicate/late results can never double-emit. (The note needs
170
+ * the ok/ERR outcome, which only `tool_result` carries — `tool_call_end`
171
+ * alone cannot build it — hence the end-gated result throttle.)
172
+ */
173
+ export declare function createSubtaskProgressEmitter(opts: {
174
+ taskId: string;
175
+ sessionId: string;
176
+ }): (ev: RuntimeEvent) => void;
160
177
  export declare class AgentOrchestrator {
161
178
  readonly sessionId: string;
162
179
  readonly deps: RuntimeDeps;
@@ -87,6 +87,45 @@ function mapResultStatus(status) {
87
87
  return 'failed';
88
88
  }
89
89
  }
90
+ /**
91
+ * Build a `subtask.progress` note for one finished tool call.
92
+ * Pure — unit-tested directly (see agent-tools.test.ts).
93
+ */
94
+ export function progressNote(step, tool, isError) {
95
+ return `step ${step}: ${tool} ${isError ? 'ERR' : 'ok'}`;
96
+ }
97
+ /**
98
+ * Build the `RunOptions.onEvent` handler the orchestrator passes into each
99
+ * child's run options. Emits at most one `subtask.progress` per tool call:
100
+ * a `tool_result` is only mirrored when its `tool_call_end` was observed
101
+ * first, so duplicate/late results can never double-emit. (The note needs
102
+ * the ok/ERR outcome, which only `tool_result` carries — `tool_call_end`
103
+ * alone cannot build it — hence the end-gated result throttle.)
104
+ */
105
+ export function createSubtaskProgressEmitter(opts) {
106
+ let step = 0;
107
+ const ended = new Set();
108
+ return (ev) => {
109
+ if (ev.kind === 'step_start') {
110
+ step = ev.step;
111
+ }
112
+ else if (ev.kind === 'tool_call_end') {
113
+ ended.add(ev.id);
114
+ }
115
+ else if (ev.kind === 'tool_result') {
116
+ if (!ended.has(ev.id))
117
+ return;
118
+ ended.delete(ev.id);
119
+ globalBus.emit({
120
+ type: 'subtask.progress',
121
+ ts: Date.now(),
122
+ sessionId: opts.sessionId,
123
+ taskId: opts.taskId,
124
+ note: progressNote(step, ev.name, ev.isError),
125
+ });
126
+ }
127
+ };
128
+ }
90
129
  export class AgentOrchestrator {
91
130
  sessionId;
92
131
  deps;
@@ -212,14 +251,17 @@ export class AgentOrchestrator {
212
251
  const registryTools = new Set(this.deps.registry.list().map((t) => t.name));
213
252
  const resolved = this.resolveChild(def, parent, registryTools);
214
253
  const childModel = input.model ?? resolved.model ?? parent.model;
215
- // Worktree isolation: write-capable children without an explicit cwd get
216
- // their own worktree; readonly agents keep the parent cwd. A
217
- // write-capable spawn outside a git repo is rejected outright.
254
+ // Worktree isolation: write-capable children get their own worktree —
255
+ // including when the spawn carries an explicit cwd (the worktree is
256
+ // then rooted at the resolved explicit cwd, which containment above
257
+ // already pinned inside the parent). Readonly agents keep the resolved
258
+ // cwd with no worktree. A write-capable spawn outside a git repo is
259
+ // rejected outright.
218
260
  const writeCapable = [...resolved.allowed].some((t) => DEFAULT_WRITE_TOOLS.has(t));
219
261
  let childCwd = baseCwd;
220
262
  let worktree;
221
263
  let repoCwd;
222
- if (!resolved.readonly && writeCapable && !input.cwd) {
264
+ if (!resolved.readonly && writeCapable) {
223
265
  const isRepo = await ensureGitRepo(baseCwd).catch(() => false);
224
266
  if (!isRepo) {
225
267
  return {
@@ -285,6 +327,11 @@ export class AgentOrchestrator {
285
327
  maxTimeMs: def.maxTimeMs ?? input.timeoutMs,
286
328
  signal: record.abortController.signal,
287
329
  nonInteractive: true,
330
+ // Mid-life progress: mirror each finished tool call as one
331
+ // `subtask.progress` bus event (see createSubtaskProgressEmitter).
332
+ // Process-isolated children don't run this closure — only the
333
+ // in-process path reports mid-life progress.
334
+ onEvent: createSubtaskProgressEmitter({ taskId: record.id, sessionId: this.sessionId }),
288
335
  // Grandchildren: only children that canSpawn receive the bridge —
289
336
  // otherwise tools see NO_ORCHESTRATOR as before.
290
337
  ...(resolved.canSpawn ? { agentBridge: this.bridgeFor(childRef) } : {}),
@@ -17,6 +17,7 @@
17
17
  *
18
18
  * The default policy matches the L6 plan: 5 attempts, 500ms base, 8s cap.
19
19
  */
20
+ import { acquireStreamSlot, noteRateLimited } from './stream-budget.js';
20
21
  export const DEFAULT_RETRY = {
21
22
  maxAttempts: 5,
22
23
  baseMs: 500,
@@ -88,6 +89,28 @@ export function computeBackoff(attempt, baseMs, maxMs) {
88
89
  const jitter = exp * 0.25 * (Math.random() * 2 - 1);
89
90
  return Math.max(0, Math.floor(exp + jitter));
90
91
  }
92
+ /**
93
+ * True when a retryable error event carries a 429 rate-limit signal.
94
+ * Checks the `status` field first, then `code`; accepts numeric values
95
+ * and `'429'` substrings (e.g. `'429'`, `'HTTP_429'`).
96
+ */
97
+ function isRateLimitedError(ev) {
98
+ if (ev.kind !== 'error')
99
+ return false;
100
+ const rec = ev;
101
+ for (const key of ['status', 'code']) {
102
+ const value = rec[key];
103
+ if (typeof value === 'string' && value.includes('429'))
104
+ return true;
105
+ if (typeof value === 'number' && String(value).includes('429'))
106
+ return true;
107
+ }
108
+ return false;
109
+ }
110
+ function rateLimitDelayMs(ev) {
111
+ const raw = ev['retryAfterMs'];
112
+ return typeof raw === 'number' && Number.isFinite(raw) && raw >= 0 ? raw : undefined;
113
+ }
91
114
  export function retryingAdapter(inner, opts = {}) {
92
115
  const cfg = { ...DEFAULT_RETRY, ...opts };
93
116
  const sleep = opts.sleep ?? defaultSleep;
@@ -129,20 +152,39 @@ export function retryingAdapter(inner, opts = {}) {
129
152
  opts.onAttempt?.(attempt);
130
153
  if (effectiveSignal?.aborted)
131
154
  return;
155
+ // Rate-limit scheduler: hold one stream slot for the duration of
156
+ // this attempt's inner.stream consumption. Abort while queued ends
157
+ // the stream promptly with no inner call.
158
+ let release;
132
159
  let sawRetryable = false;
133
160
  let lastError = null;
134
- for await (const ev of streamWithAbort(inner.stream(attemptReq), effectiveSignal)) {
135
- if (effectiveSignal?.aborted)
161
+ try {
162
+ try {
163
+ release = await acquireStreamSlot(effectiveSignal);
164
+ }
165
+ catch {
136
166
  return;
137
- if (ev.kind === 'error' && ev.retryable) {
138
- // Buffer the retryable error; don't yield it yet. We'll either
139
- // re-issue (and the caller will never see the error) or, on
140
- // final attempt, yield it as the terminal error.
141
- sawRetryable = true;
142
- lastError = ev;
143
- break; // stop consuming; the stream is dead on retryable errors.
144
167
  }
145
- yield ev;
168
+ for await (const ev of streamWithAbort(inner.stream(attemptReq), effectiveSignal)) {
169
+ if (effectiveSignal?.aborted)
170
+ return;
171
+ if (ev.kind === 'error' && ev.retryable) {
172
+ // Adaptive throttling: a 429 collapses the global stream cap
173
+ // to 1 for retryAfterMs (or 60s) — see stream-budget.ts.
174
+ if (isRateLimitedError(ev))
175
+ noteRateLimited(rateLimitDelayMs(ev));
176
+ // Buffer the retryable error; don't yield it yet. We'll either
177
+ // re-issue (and the caller will never see the error) or, on
178
+ // final attempt, yield it as the terminal error.
179
+ sawRetryable = true;
180
+ lastError = ev;
181
+ break; // stop consuming; the stream is dead on retryable errors.
182
+ }
183
+ yield ev;
184
+ }
185
+ }
186
+ finally {
187
+ release?.();
146
188
  }
147
189
  if (!sawRetryable)
148
190
  return; // success or non-retryable error — done.
@@ -31,6 +31,13 @@ export type VerifyMode = typeof import('../verification/engine.js') extends {
31
31
  } ? V : 'strict' | 'advisory' | 'off';
32
32
  export interface RuntimeDeps {
33
33
  adapter: ProviderAdapter;
34
+ /**
35
+ * Ordered failover adapters (L15). When the active adapter ends a step
36
+ * with a terminal provider error, the runtime swaps to the next entry
37
+ * and re-issues the step (bounded by chain length, never loops).
38
+ * Optional — single-adapter callers behave exactly as before.
39
+ */
40
+ failoverAdapters?: ProviderAdapter[];
34
41
  registry: ToolRegistry;
35
42
  policy: PolicyEngine;
36
43
  approval: ApprovalPrompt;
@@ -226,6 +233,22 @@ export type RuntimeEvent = {
226
233
  } | {
227
234
  kind: 'checkpoint_saved';
228
235
  sessionId: string;
236
+ } | {
237
+ kind: 'status';
238
+ message: string;
239
+ } | {
240
+ kind: 'budget_warning';
241
+ ratio: number;
242
+ threshold: number;
243
+ } | {
244
+ kind: 'provider_failover';
245
+ from: string;
246
+ to: string;
247
+ reason: string;
248
+ } | {
249
+ kind: 'model_override';
250
+ requested: string;
251
+ effective: string;
229
252
  };
230
253
  export interface RunResult {
231
254
  status: 'complete' | 'max_steps' | 'aborted' | 'no_final' | 'verify_failed' | 'limit' | 'blocked' | 'stuck';
@@ -264,7 +287,18 @@ export declare function toolDefinitions(registry: ToolRegistry): ToolDefinition[
264
287
  export declare function estimateCost(model: string, usage: {
265
288
  input: number;
266
289
  output: number;
290
+ cacheRead?: number;
291
+ cacheWrite?: number;
267
292
  }): number;
293
+ /**
294
+ * Parallel fan-out cap: approved concurrencySafe tool calls execute in
295
+ * sequential chunks of at most this size. Commit order stays identical
296
+ * (commits run sequentially after execution), so the transcript reads as
297
+ * if the calls ran in order.
298
+ */
299
+ export declare const MAX_PARALLEL_TOOLS = 8;
300
+ /** Progressive budget-warning thresholds (fraction of maxCost), fired once each per run. */
301
+ export declare const BUDGET_WARNING_THRESHOLDS: readonly [0.4, 0.7, 0.9];
268
302
  /** Run the autonomous loop. */
269
303
  export declare function run(opts: RunOptions, deps: RuntimeDeps): Promise<RunResult>;
270
304
  export declare function defaultSystemPrompt(ctx: {
@@ -23,11 +23,12 @@ import { verify, diagnosticForModel } from '../verification/engine.js';
23
23
  import { detectVerifyCommand } from '../verification/auto.js';
24
24
  import { ensureBaseline, getBaseline } from '../verification/baseline.js';
25
25
  import { compressTranscript, totalTokens } from '../context/tokenizer.js';
26
- import { ratesFor } from '../providers/model-info.js';
26
+ import { ratesFor, isAnthropicModel } from '../providers/model-info.js';
27
27
  import { classifyFailure, rerunOnce, gatherRepairContext, guardRepair } from '../verification/classify.js';
28
28
  import { findRelatedTests, buildScopedCommand, runScopedVerify, syntaxCheck, checkImports } from '../verification/scoped.js';
29
29
  import { globalBus } from '../events/bus.js';
30
30
  import { TraceWriter } from '../trace/writer.js';
31
+ import { loadHooks, runHook } from '../cli/hooks.js';
31
32
  /** Normalize either systemPrompt shape into {system, suffix}. */
32
33
  export function resolveSystemPrompt(fn, ctx) {
33
34
  const r = fn(ctx);
@@ -42,18 +43,30 @@ export function toolDefinitions(registry) {
42
43
  inputSchema: t.function.parameters,
43
44
  }));
44
45
  }
45
- // BUG-005: Model-aware cost estimation, single-sourced from the
46
+ // Model-aware cost estimation, single-sourced from the
46
47
  // providers/model-info.ts rate table (local/unknown models are $0).
47
- // Cost is computed on input/output ONLY: cacheRead/cacheWrite are tracked
48
- // for observability but excluded because cached tokens bill at
49
- // provider-specific discounted rates we don't model — charging them at
50
- // full input rates would overstate spend, silently dropping them
51
- // understates it, so we keep them visible and out of the math.
48
+ // Cache-aware: for Anthropic-family models (isAnthropicModel), cacheRead
49
+ // bills at 0.1× the input rate and cacheWrite at 1.25×; all other
50
+ // families ignore cache counters (discounted billing, unmodeled).
52
51
  /** Estimate USD cost of a usage block given the model name. */
53
52
  export function estimateCost(model, usage) {
54
53
  const { input: inRate, output: outRate } = ratesFor(model);
55
- return (usage.input / 1000) * inRate + (usage.output / 1000) * outRate;
54
+ const base = (usage.input / 1000) * inRate + (usage.output / 1000) * outRate;
55
+ if (!isAnthropicModel(model))
56
+ return base;
57
+ const read = ((usage.cacheRead ?? 0) / 1000) * inRate * 0.1;
58
+ const write = ((usage.cacheWrite ?? 0) / 1000) * inRate * 1.25;
59
+ return base + read + write;
56
60
  }
61
+ /**
62
+ * Parallel fan-out cap: approved concurrencySafe tool calls execute in
63
+ * sequential chunks of at most this size. Commit order stays identical
64
+ * (commits run sequentially after execution), so the transcript reads as
65
+ * if the calls ran in order.
66
+ */
67
+ export const MAX_PARALLEL_TOOLS = 8;
68
+ /** Progressive budget-warning thresholds (fraction of maxCost), fired once each per run. */
69
+ export const BUDGET_WARNING_THRESHOLDS = [0.4, 0.7, 0.9];
57
70
  // PERF-002: Memoized token counting cache.
58
71
  let tokenCache = {
59
72
  lastRef: null,
@@ -90,6 +103,18 @@ export async function run(opts, deps) {
90
103
  return [{ role: 'user', content: [text(opts.task)] }];
91
104
  })();
92
105
  const usage = { input: 0, output: 0 };
106
+ /** Emit one `budget_warning` per threshold the cost ratio has crossed. */
107
+ const checkBudgetWarnings = () => {
108
+ if (maxCost === undefined || maxCost <= 0)
109
+ return;
110
+ const ratio = estimateCost(opts.model, usage) / maxCost;
111
+ for (const threshold of BUDGET_WARNING_THRESHOLDS) {
112
+ if (ratio >= threshold && !firedBudgetWarnings.has(threshold)) {
113
+ firedBudgetWarnings.add(threshold);
114
+ emit?.({ kind: 'budget_warning', ratio, threshold });
115
+ }
116
+ }
117
+ };
93
118
  let steps = 0;
94
119
  let toolCallCount = 0;
95
120
  let finalText = '';
@@ -120,6 +145,27 @@ export async function run(opts, deps) {
120
145
  const emit = opts.onEvent;
121
146
  const telemetry = new RuntimeTelemetry();
122
147
  telemetry.setMaxSteps(maxSteps);
148
+ // Model-override surfacing (informational): when a parent/orchestrator
149
+ // context carries a model override, emit it once so UIs can show which
150
+ // model actually serves this run.
151
+ if (opts.parentContext?.model) {
152
+ emit?.({ kind: 'model_override', requested: opts.model, effective: opts.parentContext.model });
153
+ }
154
+ // Hooks engine: loaded once per run. Zero-cost fast path — when no hooks
155
+ // file exists, both lists are empty and every hook call site is skipped.
156
+ let runHooks = [];
157
+ try {
158
+ runHooks = loadHooks(opts.cwd);
159
+ }
160
+ catch {
161
+ runHooks = [];
162
+ }
163
+ const preHooks = runHooks.filter((h) => h.event === 'preToolUse');
164
+ const postHooks = runHooks.filter((h) => h.event === 'postToolUse');
165
+ // L15 failover chain: the active adapter starts as deps.adapter; each
166
+ // terminal provider error consumes one fallback. Bounded — never loops.
167
+ let activeAdapter = deps.adapter;
168
+ const failoverQueue = [...(deps.failoverAdapters ?? [])];
123
169
  // 3.1 — Event bus + TraceWriter
124
170
  const bus = deps.bus ?? globalBus;
125
171
  let tracer;
@@ -140,6 +186,10 @@ export async function run(opts, deps) {
140
186
  // 5.1 — phases and limits
141
187
  const maxCost = opts.maxCost;
142
188
  const maxTimeMs = opts.maxTimeMs;
189
+ // Progressive budget warnings: fire once per threshold per run when the
190
+ // cost ratio crosses 0.4 / 0.7 / 0.9 of maxCost (checker defined after
191
+ // `usage` is declared below).
192
+ const firedBudgetWarnings = new Set();
143
193
  const startTime = Date.now();
144
194
  let phase = 'understanding';
145
195
  const setPhase = (p) => {
@@ -270,7 +320,7 @@ export async function run(opts, deps) {
270
320
  ...(typeof opts.temperature === 'number' ? { temperature: opts.temperature } : {}),
271
321
  ...(opts.signal ? { signal: opts.signal } : {}),
272
322
  };
273
- const events = deps.adapter.stream(req);
323
+ const events = activeAdapter.stream(req);
274
324
  let textBuf = '';
275
325
  // Thinking is ephemeral: streamed to the UI live, never stored in the
276
326
  // transcript, and cleared when the turn's answer completes.
@@ -279,6 +329,12 @@ export async function run(opts, deps) {
279
329
  let lastFinishReason;
280
330
  // Set when this step's request must be re-issued after overflow recovery.
281
331
  let overflowRetryPending = false;
332
+ // Set when a terminal provider error consumed a failover adapter — the
333
+ // step is re-issued against the next adapter without consuming budget.
334
+ let failoverPending = false;
335
+ let failoverFrom = '';
336
+ let failoverTo = '';
337
+ let failoverReason = '';
282
338
  for await (const ev of events) {
283
339
  if (opts.signal?.aborted)
284
340
  break outer;
@@ -318,6 +374,7 @@ export async function run(opts, deps) {
318
374
  ...(usage.cacheRead !== undefined ? { cacheRead: usage.cacheRead } : {}),
319
375
  ...(usage.cacheWrite !== undefined ? { cacheWrite: usage.cacheWrite } : {}),
320
376
  });
377
+ checkBudgetWarnings();
321
378
  }
322
379
  else {
323
380
  // Providers that omit usage (Ollama, vLLM, proxies): estimate from
@@ -329,6 +386,7 @@ export async function run(opts, deps) {
329
386
  usage.estimated = true;
330
387
  telemetry.recordUsage(est.input, est.output);
331
388
  emit?.({ kind: 'usage', input: usage.input, output: usage.output, estimated: true });
389
+ checkBudgetWarnings();
332
390
  }
333
391
  }
334
392
  else if (ev.kind === 'error') {
@@ -357,6 +415,24 @@ export async function run(opts, deps) {
357
415
  catch { /* ignore — retry with the transcript as-is */ }
358
416
  break;
359
417
  }
418
+ // L15 failover: a terminal provider error swaps to the next chained
419
+ // adapter and re-issues the step (bounded by chain length). Context
420
+ // overflow is excluded — it owns its own recovery above. By the time
421
+ // an error reaches the runtime, per-adapter retries are exhausted,
422
+ // so any provider error here is terminal for the active adapter.
423
+ if (failoverQueue.length > 0) {
424
+ const next = failoverQueue.shift();
425
+ failoverPending = true;
426
+ failoverFrom = activeAdapter.id;
427
+ failoverTo = next.id;
428
+ failoverReason = `${ev.code}: ${ev.message}`.slice(0, 300);
429
+ activeAdapter = next;
430
+ telemetry.recordError(`failover: ${ev.code}`);
431
+ emit?.({ kind: 'provider_failover', from: failoverFrom, to: failoverTo, reason: failoverReason });
432
+ emit?.({ kind: 'status', message: `provider ${failoverFrom} failed (${ev.code}) — failing over to ${failoverTo}` });
433
+ emitKlyro({ type: 'error', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', code: ev.code, message: `failing over ${failoverFrom} → ${failoverTo}: ${ev.message.slice(0, 200)}` });
434
+ break;
435
+ }
360
436
  telemetry.recordError(`stream_error: ${ev.code}`);
361
437
  if (store && sessionId) {
362
438
  try {
@@ -386,6 +462,19 @@ export async function run(opts, deps) {
386
462
  emit?.({ kind: 'step_end', step: steps + 1 });
387
463
  continue outer;
388
464
  }
465
+ // Failover lands here via `break`: discard the failed attempt's partial
466
+ // output and re-issue the same step against the next adapter, again
467
+ // without consuming the step budget.
468
+ if (failoverPending) {
469
+ failoverPending = false;
470
+ textBuf = '';
471
+ thinkingBuf = '';
472
+ pendingToolCalls.clear();
473
+ lastFinishReason = undefined;
474
+ steps--;
475
+ emit?.({ kind: 'step_end', step: steps + 1 });
476
+ continue outer;
477
+ }
389
478
  // Build the assistant message. Tool calls are finalized here: JSON is
390
479
  // parsed and schema-validated BEFORE policy/execution. Malformed calls
391
480
  // become structured MALFORMED_TOOL_CALL results — garbage arguments must
@@ -478,7 +567,8 @@ export async function run(opts, deps) {
478
567
  emit?.({ kind: 'verification_started', command: advisoryCmd });
479
568
  let advisoryResult;
480
569
  try {
481
- advisoryResult = await verify({ cwd: opts.cwd, command: advisoryCmd, timeoutMs: opts.verify?.timeoutMs });
570
+ // CONTRACT (a): sessionId passthrough to the verify engine.
571
+ advisoryResult = await verify({ cwd: opts.cwd, command: advisoryCmd, timeoutMs: opts.verify?.timeoutMs, ...(sessionId ? { sessionId } : {}) });
482
572
  }
483
573
  catch (e) {
484
574
  const msg = e instanceof Error ? e.message : String(e);
@@ -557,7 +647,7 @@ export async function run(opts, deps) {
557
647
  vResult = { ok: sr.ok, exitCode: sr.exitCode, stdout: sr.stdout, stderr: sr.stderr, ...(det ? { failure: det } : {}) };
558
648
  }
559
649
  else {
560
- vResult = await verify({ cwd: opts.cwd, command: cmdToRun, timeoutMs: opts.verify?.timeoutMs });
650
+ vResult = await verify({ cwd: opts.cwd, command: cmdToRun, timeoutMs: opts.verify?.timeoutMs, ...(sessionId ? { sessionId } : {}) });
561
651
  }
562
652
  }
563
653
  catch (e) {
@@ -568,7 +658,7 @@ export async function run(opts, deps) {
568
658
  // If scoped passed but full may still fail, run full before declaring success
569
659
  if (vResult.ok && isScoped) {
570
660
  try {
571
- const full = await verify({ cwd: opts.cwd, command: verifyCmd, timeoutMs: opts.verify?.timeoutMs });
661
+ const full = await verify({ cwd: opts.cwd, command: verifyCmd, timeoutMs: opts.verify?.timeoutMs, ...(sessionId ? { sessionId } : {}) });
572
662
  if (!full.ok)
573
663
  vResult = full;
574
664
  }
@@ -787,6 +877,32 @@ export async function run(opts, deps) {
787
877
  const execTool = async (call) => {
788
878
  const t0 = Date.now();
789
879
  emitKlyro({ type: 'tool.call', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, name: call.name, input: call.input });
880
+ // Hooks: every preToolUse hook runs before execution. A non-zero exit
881
+ // denies the tool with POLICY_DENIED — the real tool never runs.
882
+ if (preHooks.length > 0) {
883
+ for (const hook of preHooks) {
884
+ let exitCode = -1;
885
+ let detail = '';
886
+ try {
887
+ const r = await runHook(hook, { toolName: call.name, input: call.input });
888
+ exitCode = r.exitCode;
889
+ detail = (r.stderr || r.stdout || '').slice(0, 300);
890
+ }
891
+ catch (err) {
892
+ detail = String(err instanceof Error ? err.message : err).slice(0, 300);
893
+ }
894
+ if (exitCode !== 0) {
895
+ const reason = `hook ${hook.name} denied: ${detail || 'hook failed'}`;
896
+ emit?.({ kind: 'policy_decision', id: call.id, name: call.name, action: 'deny', reason });
897
+ emitKlyro({ type: 'permission.decision', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, action: 'deny', reason });
898
+ const latencyMs = Date.now() - t0;
899
+ return {
900
+ obs: { ok: false, error: { code: 'POLICY_DENIED', message: reason } },
901
+ latencyMs,
902
+ };
903
+ }
904
+ }
905
+ }
790
906
  let obs;
791
907
  try {
792
908
  obs = await deps.registry.execute(call.name, call.input, toolCtx);
@@ -874,6 +990,31 @@ export async function run(opts, deps) {
874
990
  if (last3.length === 3 && last3[0] === last3[1] && last3[1] === last3[2]) {
875
991
  await markStuck(`identical call ×3: ${sig}`);
876
992
  }
993
+ // Hooks: postToolUse hooks are best-effort — failures warn on stderr
994
+ // plus a bus event, and never fail the turn.
995
+ if (postHooks.length > 0) {
996
+ for (const hook of postHooks) {
997
+ try {
998
+ const r = await runHook(hook, { toolName: call.name, input: call.input });
999
+ if (!r.ok || r.exitCode !== 0) {
1000
+ const msg = `klyro: hooks: postToolUse ${hook.name} failed (exit ${String(r.exitCode)}): ${(r.stderr || r.stdout || '').slice(0, 200)}\n`;
1001
+ try {
1002
+ process.stderr.write(msg);
1003
+ }
1004
+ catch { /* ignore */ }
1005
+ emitKlyro({ type: 'error', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', code: 'hook_failed', message: msg.slice(0, 300) });
1006
+ }
1007
+ }
1008
+ catch (err) {
1009
+ const msg = `klyro: hooks: postToolUse ${hook.name} error: ${String(err instanceof Error ? err.message : err).slice(0, 200)}\n`;
1010
+ try {
1011
+ process.stderr.write(msg);
1012
+ }
1013
+ catch { /* ignore */ }
1014
+ emitKlyro({ type: 'error', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', code: 'hook_failed', message: msg.slice(0, 300) });
1015
+ }
1016
+ }
1017
+ }
877
1018
  };
878
1019
  // Sequential path: gate → execute → commit per call, in order.
879
1020
  const runOne = async (call) => {
@@ -897,8 +1038,17 @@ export async function run(opts, deps) {
897
1038
  break;
898
1039
  }
899
1040
  if (approved.length > 0 && !opts.signal?.aborted) {
900
- const settled = await Promise.allSettled(approved.map((c) => execTool(c)));
901
- for (let i = 0; i < approved.length; i++) {
1041
+ // Fan-out cap: execute in sequential chunks of MAX_PARALLEL_TOOLS.
1042
+ // Commits below stay in original call order, so the transcript is
1043
+ // unaffected by the chunking.
1044
+ const settled = [];
1045
+ for (let off = 0; off < approved.length; off += MAX_PARALLEL_TOOLS) {
1046
+ if (opts.signal?.aborted)
1047
+ break;
1048
+ const chunk = approved.slice(off, off + MAX_PARALLEL_TOOLS);
1049
+ settled.push(...await Promise.allSettled(chunk.map((c) => execTool(c))));
1050
+ }
1051
+ for (let i = 0; i < settled.length; i++) {
902
1052
  const s = settled[i];
903
1053
  if (s.status === 'fulfilled') {
904
1054
  await commitResult(approved[i], s.value.obs, s.value.latencyMs);