klyro 1.0.0 → 1.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (110) hide show
  1. package/dist/agent/anthropic-adapter.d.ts +49 -5
  2. package/dist/agent/anthropic-adapter.js +86 -17
  3. package/dist/agent/capabilities.d.ts +23 -0
  4. package/dist/agent/capabilities.js +53 -6
  5. package/dist/agent/child-worker.d.ts +104 -0
  6. package/dist/agent/child-worker.js +250 -0
  7. package/dist/agent/orchestrator.d.ts +124 -6
  8. package/dist/agent/orchestrator.js +425 -58
  9. package/dist/agent/provider-adapter.d.ts +8 -0
  10. package/dist/agent/provider-adapter.js +12 -3
  11. package/dist/agent/retry.d.ts +1 -1
  12. package/dist/agent/retry.js +54 -12
  13. package/dist/agent/runtime.d.ts +84 -8
  14. package/dist/agent/runtime.js +352 -39
  15. package/dist/agent/stream-budget.d.ts +36 -0
  16. package/dist/agent/stream-budget.js +121 -0
  17. package/dist/agent/worktree-manager.d.ts +74 -0
  18. package/dist/agent/worktree-manager.js +189 -0
  19. package/dist/checkpoints/store.d.ts +9 -0
  20. package/dist/checkpoints/store.js +56 -5
  21. package/dist/cli/auth.js +16 -1
  22. package/dist/cli/commit.d.ts +31 -0
  23. package/dist/cli/commit.js +142 -0
  24. package/dist/cli/config.d.ts +54 -3
  25. package/dist/cli/config.js +146 -3
  26. package/dist/cli/doctor.d.ts +1 -0
  27. package/dist/cli/doctor.js +71 -6
  28. package/dist/cli/eval.d.ts +6 -1
  29. package/dist/cli/eval.js +9 -0
  30. package/dist/cli/hooks.d.ts +47 -0
  31. package/dist/cli/hooks.js +181 -0
  32. package/dist/cli/repl.js +196 -29
  33. package/dist/cli/run.d.ts +13 -11
  34. package/dist/cli/run.js +144 -20
  35. package/dist/cli/update.d.ts +5 -0
  36. package/dist/cli/update.js +62 -10
  37. package/dist/context/import-graph.d.ts +2 -0
  38. package/dist/context/import-graph.js +31 -3
  39. package/dist/context/klyro-md.js +4 -1
  40. package/dist/context/memory.d.ts +8 -0
  41. package/dist/context/memory.js +50 -2
  42. package/dist/context/project-map.d.ts +6 -0
  43. package/dist/context/project-map.js +50 -2
  44. package/dist/context/repo-map.d.ts +2 -0
  45. package/dist/context/repo-map.js +31 -1
  46. package/dist/events/catalog.d.ts +37 -0
  47. package/dist/events/catalog.js +9 -0
  48. package/dist/index.js +177 -8
  49. package/dist/mcp/client.d.ts +6 -4
  50. package/dist/mcp/client.js +83 -14
  51. package/dist/mcp/config.d.ts +10 -0
  52. package/dist/mcp/config.js +18 -1
  53. package/dist/mcp/registry.d.ts +23 -19
  54. package/dist/mcp/registry.js +127 -8
  55. package/dist/mcp/schema.d.ts +11 -4
  56. package/dist/mcp/schema.js +27 -16
  57. package/dist/mcp/trust.d.ts +20 -0
  58. package/dist/mcp/trust.js +74 -0
  59. package/dist/persistence/audit.d.ts +28 -0
  60. package/dist/persistence/audit.js +101 -1
  61. package/dist/persistence/store.d.ts +26 -2
  62. package/dist/persistence/store.js +140 -13
  63. package/dist/policy/approval.d.ts +14 -0
  64. package/dist/policy/approval.js +44 -2
  65. package/dist/policy/engine.d.ts +17 -0
  66. package/dist/policy/engine.js +162 -9
  67. package/dist/policy/path-guard.d.ts +24 -0
  68. package/dist/policy/path-guard.js +46 -0
  69. package/dist/policy/secret-redactor.js +4 -0
  70. package/dist/providers/model-info.d.ts +23 -0
  71. package/dist/providers/model-info.js +43 -2
  72. package/dist/repl.d.ts +6 -0
  73. package/dist/repl.js +12 -7
  74. package/dist/tools/agent/spawn-agent.js +5 -5
  75. package/dist/tools/agent/task-apply.d.ts +4 -0
  76. package/dist/tools/agent/task-apply.js +44 -0
  77. package/dist/tools/agent/task-stop.d.ts +6 -0
  78. package/dist/tools/agent/task-stop.js +39 -0
  79. package/dist/tools/agent/task-wait.d.ts +17 -0
  80. package/dist/tools/agent/task-wait.js +79 -0
  81. package/dist/tools/fs/apply-patch.js +77 -1
  82. package/dist/tools/fs/edit-file.js +69 -1
  83. package/dist/tools/fs/multi-edit.d.ts +4 -0
  84. package/dist/tools/fs/multi-edit.js +70 -1
  85. package/dist/tools/fs/write-file.js +83 -6
  86. package/dist/tools/plan/todo-write.js +1 -1
  87. package/dist/tools/registry.js +6 -0
  88. package/dist/tools/shell/background.js +6 -3
  89. package/dist/tools/shell/sandbox.d.ts +51 -0
  90. package/dist/tools/shell/sandbox.js +143 -0
  91. package/dist/tools/shell/shell-exec.d.ts +29 -0
  92. package/dist/tools/shell/shell-exec.js +170 -12
  93. package/dist/tools/shell/worker-entry.d.ts +12 -0
  94. package/dist/tools/shell/worker-entry.js +43 -0
  95. package/dist/tools/types.d.ts +6 -0
  96. package/dist/tools/verify/run-verify.js +3 -1
  97. package/dist/trace/writer.d.ts +20 -0
  98. package/dist/trace/writer.js +62 -4
  99. package/dist/tui/app.js +1 -1
  100. package/dist/tui/approval.js +20 -21
  101. package/dist/util.d.ts +1 -0
  102. package/dist/util.js +1 -0
  103. package/dist/verification/baseline.js +17 -3
  104. package/dist/verification/classify.js +27 -15
  105. package/dist/verification/engine.d.ts +8 -0
  106. package/dist/verification/engine.js +28 -1
  107. package/dist/verification/registry.d.ts +2 -0
  108. package/dist/verification/registry.js +44 -0
  109. package/dist/verification/scoped.js +64 -11
  110. package/package.json +1 -1
@@ -23,10 +23,17 @@ import { verify, diagnosticForModel } from '../verification/engine.js';
23
23
  import { detectVerifyCommand } from '../verification/auto.js';
24
24
  import { ensureBaseline, getBaseline } from '../verification/baseline.js';
25
25
  import { compressTranscript, totalTokens } from '../context/tokenizer.js';
26
+ import { ratesFor, isAnthropicModel } from '../providers/model-info.js';
26
27
  import { classifyFailure, rerunOnce, gatherRepairContext, guardRepair } from '../verification/classify.js';
27
28
  import { findRelatedTests, buildScopedCommand, runScopedVerify, syntaxCheck, checkImports } from '../verification/scoped.js';
28
29
  import { globalBus } from '../events/bus.js';
29
30
  import { TraceWriter } from '../trace/writer.js';
31
+ import { loadHooks, runHook } from '../cli/hooks.js';
32
+ /** Normalize either systemPrompt shape into {system, suffix}. */
33
+ export function resolveSystemPrompt(fn, ctx) {
34
+ const r = fn(ctx);
35
+ return typeof r === 'string' ? { system: r } : { system: r.system, suffix: r.suffix };
36
+ }
30
37
  const DEFAULT_MAX_STEPS = 30;
31
38
  /** Convert a registry of tools into ToolDefinitions for the provider. */
32
39
  export function toolDefinitions(registry) {
@@ -36,21 +43,30 @@ export function toolDefinitions(registry) {
36
43
  inputSchema: t.function.parameters,
37
44
  }));
38
45
  }
39
- // BUG-005: Model-aware cost estimation with sensible defaults.
40
- // Rates are per-1K tokens (input / output). Local models are $0.
41
- const MODEL_RATES = [
42
- { test: (m) => /gpt-4/i.test(m), input: 0.003, output: 0.015 },
43
- { test: (m) => /gpt-3\.5/i.test(m), input: 0.0005, output: 0.0015 },
44
- { test: (m) => /claude|anthropic/i.test(m), input: 0.003, output: 0.015 },
45
- { test: (m) => /gemini/i.test(m), input: 0.00075, output: 0.003 },
46
- { test: (m) => /o1/i.test(m), input: 0.015, output: 0.06 },
47
- ];
46
+ // Model-aware cost estimation, single-sourced from the
47
+ // providers/model-info.ts rate table (local/unknown models are $0).
48
+ // Cache-aware: for Anthropic-family models (isAnthropicModel), cacheRead
49
+ // bills at 0.1× the input rate and cacheWrite at 1.25×; all other
50
+ // families ignore cache counters (discounted billing, unmodeled).
48
51
  /** Estimate USD cost of a usage block given the model name. */
49
52
  export function estimateCost(model, usage) {
50
- const match = MODEL_RATES.find((r) => r.test(model));
51
- const { input: inRate, output: outRate } = match ?? { input: 0.003, output: 0.015 };
52
- return (usage.input / 1000) * inRate + (usage.output / 1000) * outRate;
53
+ const { input: inRate, output: outRate } = ratesFor(model);
54
+ const base = (usage.input / 1000) * inRate + (usage.output / 1000) * outRate;
55
+ if (!isAnthropicModel(model))
56
+ return base;
57
+ const read = ((usage.cacheRead ?? 0) / 1000) * inRate * 0.1;
58
+ const write = ((usage.cacheWrite ?? 0) / 1000) * inRate * 1.25;
59
+ return base + read + write;
53
60
  }
61
+ /**
62
+ * Parallel fan-out cap: approved concurrencySafe tool calls execute in
63
+ * sequential chunks of at most this size. Commit order stays identical
64
+ * (commits run sequentially after execution), so the transcript reads as
65
+ * if the calls ran in order.
66
+ */
67
+ export const MAX_PARALLEL_TOOLS = 8;
68
+ /** Progressive budget-warning thresholds (fraction of maxCost), fired once each per run. */
69
+ export const BUDGET_WARNING_THRESHOLDS = [0.4, 0.7, 0.9];
54
70
  // PERF-002: Memoized token counting cache.
55
71
  let tokenCache = {
56
72
  lastRef: null,
@@ -87,15 +103,69 @@ export async function run(opts, deps) {
87
103
  return [{ role: 'user', content: [text(opts.task)] }];
88
104
  })();
89
105
  const usage = { input: 0, output: 0 };
106
+ /** Emit one `budget_warning` per threshold the cost ratio has crossed. */
107
+ const checkBudgetWarnings = () => {
108
+ if (maxCost === undefined || maxCost <= 0)
109
+ return;
110
+ const ratio = estimateCost(opts.model, usage) / maxCost;
111
+ for (const threshold of BUDGET_WARNING_THRESHOLDS) {
112
+ if (ratio >= threshold && !firedBudgetWarnings.has(threshold)) {
113
+ firedBudgetWarnings.add(threshold);
114
+ emit?.({ kind: 'budget_warning', ratio, threshold });
115
+ }
116
+ }
117
+ };
90
118
  let steps = 0;
91
119
  let toolCallCount = 0;
92
120
  let finalText = '';
93
121
  let repairs = 0;
94
122
  let verificationAttempts = 0;
95
123
  let hasEdits = false;
124
+ // Overflow recovery: at most one compress-and-retry per run.
125
+ let overflowRetried = false;
126
+ // Repair ledger + guard: set from the first verification failure until a
127
+ // verification passes. repairUsageStart snapshots usage at failure time so
128
+ // every verification payload can attribute its repairTokens delta.
129
+ let inRepairTurn = false;
130
+ let repairUsageStart;
131
+ const verifyMode = opts.verify?.mode ?? 'strict';
132
+ const markRepairStarted = () => {
133
+ if (!repairUsageStart)
134
+ repairUsageStart = { input: usage.input, output: usage.output };
135
+ inRepairTurn = true;
136
+ };
137
+ const repairTokensNow = () => repairUsageStart
138
+ ? { input: usage.input - repairUsageStart.input, output: usage.output - repairUsageStart.output }
139
+ : undefined;
140
+ /** Attach repairTokens to a verification payload when a repair ledger exists. */
141
+ const withRepairTokens = (v) => {
142
+ const rt = repairTokensNow();
143
+ return rt ? { ...v, repairTokens: rt } : v;
144
+ };
96
145
  const emit = opts.onEvent;
97
146
  const telemetry = new RuntimeTelemetry();
98
147
  telemetry.setMaxSteps(maxSteps);
148
+ // Model-override surfacing (informational): when a parent/orchestrator
149
+ // context carries a model override, emit it once so UIs can show which
150
+ // model actually serves this run.
151
+ if (opts.parentContext?.model) {
152
+ emit?.({ kind: 'model_override', requested: opts.model, effective: opts.parentContext.model });
153
+ }
154
+ // Hooks engine: loaded once per run. Zero-cost fast path — when no hooks
155
+ // file exists, both lists are empty and every hook call site is skipped.
156
+ let runHooks = [];
157
+ try {
158
+ runHooks = loadHooks(opts.cwd);
159
+ }
160
+ catch {
161
+ runHooks = [];
162
+ }
163
+ const preHooks = runHooks.filter((h) => h.event === 'preToolUse');
164
+ const postHooks = runHooks.filter((h) => h.event === 'postToolUse');
165
+ // L15 failover chain: the active adapter starts as deps.adapter; each
166
+ // terminal provider error consumes one fallback. Bounded — never loops.
167
+ let activeAdapter = deps.adapter;
168
+ const failoverQueue = [...(deps.failoverAdapters ?? [])];
99
169
  // 3.1 — Event bus + TraceWriter
100
170
  const bus = deps.bus ?? globalBus;
101
171
  let tracer;
@@ -116,6 +186,10 @@ export async function run(opts, deps) {
116
186
  // 5.1 — phases and limits
117
187
  const maxCost = opts.maxCost;
118
188
  const maxTimeMs = opts.maxTimeMs;
189
+ // Progressive budget warnings: fire once per threshold per run when the
190
+ // cost ratio crosses 0.4 / 0.7 / 0.9 of maxCost (checker defined after
191
+ // `usage` is declared below).
192
+ const firedBudgetWarnings = new Set();
119
193
  const startTime = Date.now();
120
194
  let phase = 'understanding';
121
195
  const setPhase = (p) => {
@@ -200,7 +274,7 @@ export async function run(opts, deps) {
200
274
  catch { /* ignore */ }
201
275
  }
202
276
  await closeTracer();
203
- return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? { ok: false, attempts: verificationAttempts } : undefined, phase: 'blocked' };
277
+ return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined, phase: 'blocked' };
204
278
  }
205
279
  steps++;
206
280
  // 5.1 phase transitions (model-narrated)
@@ -216,13 +290,21 @@ export async function run(opts, deps) {
216
290
  setPhase('verifying');
217
291
  emit?.({ kind: 'step_start', step: steps });
218
292
  telemetry.recordStepStart(steps);
219
- const systemPrompt = deps.systemPrompt({ cwd: opts.cwd, telemetry: steps === 1 ? emptyTelemetryBlock() : telemetry.format() });
293
+ const { system: stableSystem, suffix: telemetrySuffix } = resolveSystemPrompt(deps.systemPrompt, {
294
+ cwd: opts.cwd,
295
+ telemetry: steps === 1 ? emptyTelemetryBlock() : telemetry.format(),
296
+ });
297
+ // Budget accounting sees what the model sees (prefix + suffix); the
298
+ // request itself keeps the halves split for cache-friendly adapters.
299
+ const systemForBudget = telemetrySuffix ? `${stableSystem}\n\n${telemetrySuffix}` : stableSystem;
220
300
  const BUDGET = { total: 120_000, reservedOutput: 4000 };
221
301
  let reqMessages = transcript;
222
- let reqSystem = systemPrompt;
223
- if (cachedTotalTokens(systemPrompt, transcript) > BUDGET.total) {
224
- const c = compressTranscript(systemPrompt, transcript, BUDGET);
302
+ let reqSystem = stableSystem;
303
+ let reqSuffix = telemetrySuffix;
304
+ if (cachedTotalTokens(systemForBudget, transcript) > BUDGET.total) {
305
+ const c = compressTranscript(systemForBudget, transcript, BUDGET);
225
306
  reqSystem = c.system;
307
+ reqSuffix = undefined; // telemetry is regenerable — drop it under pressure
226
308
  reqMessages = c.messages;
227
309
  tokenCache = { lastRef: null, lastSystem: undefined, lastCount: 0 };
228
310
  if (c.dropped > 0)
@@ -231,19 +313,28 @@ export async function run(opts, deps) {
231
313
  const req = {
232
314
  model: opts.parentContext?.model ?? opts.model,
233
315
  system: reqSystem,
316
+ ...(reqSuffix ? { systemSuffix: reqSuffix } : {}),
234
317
  messages: reqMessages,
235
318
  tools: toolDefinitions(deps.registry),
236
319
  ...(opts.maxTokens ? { maxTokens: opts.maxTokens } : {}),
237
320
  ...(typeof opts.temperature === 'number' ? { temperature: opts.temperature } : {}),
238
321
  ...(opts.signal ? { signal: opts.signal } : {}),
239
322
  };
240
- const events = deps.adapter.stream(req);
323
+ const events = activeAdapter.stream(req);
241
324
  let textBuf = '';
242
325
  // Thinking is ephemeral: streamed to the UI live, never stored in the
243
326
  // transcript, and cleared when the turn's answer completes.
244
327
  let thinkingBuf = '';
245
328
  const pendingToolCalls = new Map();
246
329
  let lastFinishReason;
330
+ // Set when this step's request must be re-issued after overflow recovery.
331
+ let overflowRetryPending = false;
332
+ // Set when a terminal provider error consumed a failover adapter — the
333
+ // step is re-issued against the next adapter without consuming budget.
334
+ let failoverPending = false;
335
+ let failoverFrom = '';
336
+ let failoverTo = '';
337
+ let failoverReason = '';
247
338
  for await (const ev of events) {
248
339
  if (opts.signal?.aborted)
249
340
  break outer;
@@ -273,22 +364,75 @@ export async function run(opts, deps) {
273
364
  if (ev.usage) {
274
365
  usage.input += ev.usage.input;
275
366
  usage.output += ev.usage.output;
367
+ if (ev.usage.cacheRead !== undefined)
368
+ usage.cacheRead = (usage.cacheRead ?? 0) + ev.usage.cacheRead;
369
+ if (ev.usage.cacheWrite !== undefined)
370
+ usage.cacheWrite = (usage.cacheWrite ?? 0) + ev.usage.cacheWrite;
276
371
  telemetry.recordUsage(ev.usage.input, ev.usage.output);
277
- emit?.({ kind: 'usage', input: usage.input, output: usage.output });
372
+ emit?.({
373
+ kind: 'usage', input: usage.input, output: usage.output,
374
+ ...(usage.cacheRead !== undefined ? { cacheRead: usage.cacheRead } : {}),
375
+ ...(usage.cacheWrite !== undefined ? { cacheWrite: usage.cacheWrite } : {}),
376
+ });
377
+ checkBudgetWarnings();
278
378
  }
279
379
  else {
280
380
  // Providers that omit usage (Ollama, vLLM, proxies): estimate from
281
381
  // the actual request + generated output so cost accounting never
282
382
  // silently records zero. Marked estimated for the UI/debugging.
283
- const est = estimateTurnUsage(reqSystem, reqMessages, textBuf, pendingToolCalls);
383
+ const est = estimateTurnUsage(systemForBudget, reqMessages, textBuf, pendingToolCalls);
284
384
  usage.input += est.input;
285
385
  usage.output += est.output;
286
386
  usage.estimated = true;
287
387
  telemetry.recordUsage(est.input, est.output);
288
388
  emit?.({ kind: 'usage', input: usage.input, output: usage.output, estimated: true });
389
+ checkBudgetWarnings();
289
390
  }
290
391
  }
291
392
  else if (ev.kind === 'error') {
393
+ // Overflow recovery: on the first REQUEST_TOO_LARGE of a run,
394
+ // aggressively compact the transcript and re-issue the request once.
395
+ // A second occurrence fails normally (returned as no_final below).
396
+ if (ev.code === 'REQUEST_TOO_LARGE' && !overflowRetried) {
397
+ overflowRetried = true;
398
+ overflowRetryPending = true;
399
+ telemetry.recordError('overflow_retry');
400
+ emitKlyro({ type: 'error', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', code: 'REQUEST_TOO_LARGE', message: 'context overflow — aggressively compacting transcript and retrying the request once' });
401
+ try {
402
+ const compacted = compressTranscript(reqSystem, transcript, { total: 30_000, reservedOutput: 4000 });
403
+ if (compacted.messages.length < transcript.length || compacted.dropped > 0) {
404
+ transcript.splice(0, transcript.length, ...compacted.messages);
405
+ }
406
+ else {
407
+ // Transcript already fits the aggressive budget — force it
408
+ // strictly smaller so the retry cannot repeat the overflow.
409
+ const halved = Math.max(4000, Math.floor(totalTokens(reqSystem, transcript) / 2));
410
+ const smaller = compressTranscript(reqSystem, transcript, { total: halved, reservedOutput: 4000 });
411
+ transcript.splice(0, transcript.length, ...smaller.messages);
412
+ }
413
+ tokenCache = { lastRef: null, lastSystem: undefined, lastCount: 0 };
414
+ }
415
+ catch { /* ignore — retry with the transcript as-is */ }
416
+ break;
417
+ }
418
+ // L15 failover: a terminal provider error swaps to the next chained
419
+ // adapter and re-issues the step (bounded by chain length). Context
420
+ // overflow is excluded — it owns its own recovery above. By the time
421
+ // an error reaches the runtime, per-adapter retries are exhausted,
422
+ // so any provider error here is terminal for the active adapter.
423
+ if (failoverQueue.length > 0) {
424
+ const next = failoverQueue.shift();
425
+ failoverPending = true;
426
+ failoverFrom = activeAdapter.id;
427
+ failoverTo = next.id;
428
+ failoverReason = `${ev.code}: ${ev.message}`.slice(0, 300);
429
+ activeAdapter = next;
430
+ telemetry.recordError(`failover: ${ev.code}`);
431
+ emit?.({ kind: 'provider_failover', from: failoverFrom, to: failoverTo, reason: failoverReason });
432
+ emit?.({ kind: 'status', message: `provider ${failoverFrom} failed (${ev.code}) — failing over to ${failoverTo}` });
433
+ emitKlyro({ type: 'error', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', code: ev.code, message: `failing over ${failoverFrom} → ${failoverTo}: ${ev.message.slice(0, 200)}` });
434
+ break;
435
+ }
292
436
  telemetry.recordError(`stream_error: ${ev.code}`);
293
437
  if (store && sessionId) {
294
438
  try {
@@ -306,10 +450,31 @@ export async function run(opts, deps) {
306
450
  hasEdits,
307
451
  usage,
308
452
  repairs,
309
- verification: hasEdits ? { ok: false, attempts: verificationAttempts } : undefined,
453
+ verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined,
310
454
  };
311
455
  }
312
456
  }
457
+ // Overflow recovery lands here via `break`: re-issue the same step
458
+ // without consuming the step budget.
459
+ if (overflowRetryPending) {
460
+ overflowRetryPending = false;
461
+ steps--;
462
+ emit?.({ kind: 'step_end', step: steps + 1 });
463
+ continue outer;
464
+ }
465
+ // Failover lands here via `break`: discard the failed attempt's partial
466
+ // output and re-issue the same step against the next adapter, again
467
+ // without consuming the step budget.
468
+ if (failoverPending) {
469
+ failoverPending = false;
470
+ textBuf = '';
471
+ thinkingBuf = '';
472
+ pendingToolCalls.clear();
473
+ lastFinishReason = undefined;
474
+ steps--;
475
+ emit?.({ kind: 'step_end', step: steps + 1 });
476
+ continue outer;
477
+ }
313
478
  // Build the assistant message. Tool calls are finalized here: JSON is
314
479
  // parsed and schema-validated BEFORE policy/execution. Malformed calls
315
480
  // become structured MALFORMED_TOOL_CALL results — garbage arguments must
@@ -368,7 +533,7 @@ export async function run(opts, deps) {
368
533
  catch { /* ignore */ }
369
534
  }
370
535
  await closeTracer();
371
- return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? { ok: false, attempts: verificationAttempts } : undefined };
536
+ return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined };
372
537
  }
373
538
  if (finalizedCalls.length === 0) {
374
539
  if (hadInvalidTool) {
@@ -380,6 +545,67 @@ export async function run(opts, deps) {
380
545
  continue;
381
546
  }
382
547
  finalText = textBuf;
548
+ // Verify modes: 'off' skips the pipeline entirely (verification stays
549
+ // undefined, no repair turns). 'advisory' runs verify once on
550
+ // completion and reports the outcome without repair turns. 'strict'
551
+ // (default) runs the full pipeline with autonomous repair below.
552
+ if (verifyMode === 'off') {
553
+ emit?.({ kind: 'final_text', text: finalText });
554
+ emit?.({ kind: 'step_end', step: steps });
555
+ if (store && sessionId) {
556
+ try {
557
+ await store.setStatus(sessionId, 'complete', finalText);
558
+ }
559
+ catch { /* ignore */ }
560
+ }
561
+ await closeTracer();
562
+ return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: undefined };
563
+ }
564
+ if (verifyMode === 'advisory') {
565
+ const advisoryCmd = opts.verify?.command ?? detectVerifyCommand(opts.cwd);
566
+ if (opts.verify?.enabled !== false && hasEdits && advisoryCmd) {
567
+ emit?.({ kind: 'verification_started', command: advisoryCmd });
568
+ let advisoryResult;
569
+ try {
570
+ // CONTRACT (a): sessionId passthrough to the verify engine.
571
+ advisoryResult = await verify({ cwd: opts.cwd, command: advisoryCmd, timeoutMs: opts.verify?.timeoutMs, ...(sessionId ? { sessionId } : {}) });
572
+ }
573
+ catch (e) {
574
+ const msg = e instanceof Error ? e.message : String(e);
575
+ advisoryResult = { ok: false, exitCode: -1, stdout: '', stderr: msg, failure: { type: 'unknown', files: [], raw: msg, exitCode: -1 } };
576
+ }
577
+ verificationAttempts++;
578
+ if (advisoryResult.ok) {
579
+ inRepairTurn = false;
580
+ emit?.({ kind: 'verification_succeeded', command: advisoryCmd });
581
+ emit?.({ kind: 'final_text', text: finalText });
582
+ emit?.({ kind: 'step_end', step: steps });
583
+ if (store && sessionId) {
584
+ try {
585
+ await store.setStatus(sessionId, 'complete', finalText);
586
+ }
587
+ catch { /* ignore */ }
588
+ }
589
+ await closeTracer();
590
+ return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: withRepairTokens({ ok: true, command: advisoryCmd, attempts: verificationAttempts }) };
591
+ }
592
+ const advisoryFailure = advisoryResult.failure?.type ?? 'unknown';
593
+ markRepairStarted();
594
+ emit?.({ kind: 'verification_failed', step: String(steps), reason: diagnosticForModel(advisoryResult).slice(0, 800) });
595
+ emit?.({ kind: 'final_text', text: finalText });
596
+ emit?.({ kind: 'step_end', step: steps });
597
+ if (store && sessionId) {
598
+ try {
599
+ await store.setStatus(sessionId, 'complete', finalText);
600
+ }
601
+ catch { /* ignore */ }
602
+ }
603
+ await closeTracer();
604
+ return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: withRepairTokens({ ok: false, command: advisoryCmd, attempts: verificationAttempts, failureType: advisoryFailure }) };
605
+ }
606
+ // No verify applicable (disabled, no edits, or no command) — fall
607
+ // through to the normal completion return below.
608
+ }
383
609
  // Level 8 — Verification + Autonomous Repair (gated on hasEdits below — pure analysis skips verify)
384
610
  const verifyEnabled = opts.verify?.enabled !== false;
385
611
  const verifyCmd = opts.verify?.command ?? detectVerifyCommand(opts.cwd);
@@ -421,7 +647,7 @@ export async function run(opts, deps) {
421
647
  vResult = { ok: sr.ok, exitCode: sr.exitCode, stdout: sr.stdout, stderr: sr.stderr, ...(det ? { failure: det } : {}) };
422
648
  }
423
649
  else {
424
- vResult = await verify({ cwd: opts.cwd, command: cmdToRun, timeoutMs: opts.verify?.timeoutMs });
650
+ vResult = await verify({ cwd: opts.cwd, command: cmdToRun, timeoutMs: opts.verify?.timeoutMs, ...(sessionId ? { sessionId } : {}) });
425
651
  }
426
652
  }
427
653
  catch (e) {
@@ -432,13 +658,14 @@ export async function run(opts, deps) {
432
658
  // If scoped passed but full may still fail, run full before declaring success
433
659
  if (vResult.ok && isScoped) {
434
660
  try {
435
- const full = await verify({ cwd: opts.cwd, command: verifyCmd, timeoutMs: opts.verify?.timeoutMs });
661
+ const full = await verify({ cwd: opts.cwd, command: verifyCmd, timeoutMs: opts.verify?.timeoutMs, ...(sessionId ? { sessionId } : {}) });
436
662
  if (!full.ok)
437
663
  vResult = full;
438
664
  }
439
665
  catch { /* scoped success is enough */ }
440
666
  }
441
667
  if (vResult.ok) {
668
+ inRepairTurn = false;
442
669
  emit?.({ kind: 'verification_succeeded', command: verifyCmd });
443
670
  emit?.({ kind: 'final_text', text: finalText });
444
671
  emit?.({ kind: 'step_end', step: steps });
@@ -449,7 +676,7 @@ export async function run(opts, deps) {
449
676
  catch { /* ignore */ }
450
677
  }
451
678
  await closeTracer();
452
- return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: { ok: true, command: verifyCmd, attempts: verificationAttempts } };
679
+ return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: withRepairTokens({ ok: true, command: verifyCmd, attempts: verificationAttempts }) };
453
680
  }
454
681
  // 6.4 — classify
455
682
  const baseline = await getBaseline(opts.cwd, verifyCmd);
@@ -457,6 +684,7 @@ export async function run(opts, deps) {
457
684
  const cls = classifyFailure({ failure: vResult.failure, stdout: vResult.stdout, stderr: vResult.stderr }, baseline, isFlaky);
458
685
  if (cls === 'flaky') {
459
686
  // rerun succeeded on second try — treat as flaky, don't count as repair
687
+ inRepairTurn = false;
460
688
  emit?.({ kind: 'verification_succeeded', command: verifyCmd });
461
689
  emit?.({ kind: 'final_text', text: finalText });
462
690
  emit?.({ kind: 'step_end', step: steps });
@@ -467,7 +695,7 @@ export async function run(opts, deps) {
467
695
  catch { /* ignore */ }
468
696
  }
469
697
  await closeTracer();
470
- return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: { ok: true, command: verifyCmd, attempts: verificationAttempts } };
698
+ return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: withRepairTokens({ ok: true, command: verifyCmd, attempts: verificationAttempts }) };
471
699
  }
472
700
  if (cls === 'env') {
473
701
  // don't try to repair env failures with code edits
@@ -475,10 +703,11 @@ export async function run(opts, deps) {
475
703
  transcript.push(envMsg);
476
704
  await checkpoint(envMsg);
477
705
  verificationAttempts++;
706
+ markRepairStarted();
478
707
  emit?.({ kind: 'verification_failed', step: String(steps), reason: `env: ${diagnosticForModel(vResult).slice(0, 600)}` });
479
708
  if (verificationAttempts >= maxRepairs) {
480
709
  await closeTracer();
481
- return { status: 'verify_failed', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: { ok: false, command: verifyCmd, attempts: verificationAttempts, failureType: 'env' } };
710
+ return { status: 'verify_failed', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: withRepairTokens({ ok: false, command: verifyCmd, attempts: verificationAttempts, failureType: 'env' }) };
482
711
  }
483
712
  emit?.({ kind: 'step_end', step: steps });
484
713
  continue;
@@ -494,6 +723,7 @@ export async function run(opts, deps) {
494
723
  const failurePaths = new Set(vResult.failure?.files.map((f) => f.path).filter(Boolean) ?? []);
495
724
  const overlaps = [...failurePaths].some((p) => introducedPaths.has(p) || introducedPaths.has(path.basename(p)));
496
725
  if (!overlaps) {
726
+ inRepairTurn = false;
497
727
  emit?.({ kind: 'verification_succeeded', command: verifyCmd });
498
728
  emit?.({ kind: 'final_text', text: finalText });
499
729
  emit?.({ kind: 'step_end', step: steps });
@@ -504,11 +734,12 @@ export async function run(opts, deps) {
504
734
  catch { /* ignore */ }
505
735
  }
506
736
  await closeTracer();
507
- return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: { ok: true, command: verifyCmd, attempts: verificationAttempts } };
737
+ return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: withRepairTokens({ ok: true, command: verifyCmd, attempts: verificationAttempts }) };
508
738
  }
509
739
  }
510
740
  // Failure → repair loop (introduced)
511
741
  verificationAttempts++;
742
+ markRepairStarted();
512
743
  const diagnostic = diagnosticForModel(vResult);
513
744
  const failureType = vResult.failure?.type ?? 'unknown';
514
745
  // 6.4 — gather context
@@ -548,7 +779,7 @@ export async function run(opts, deps) {
548
779
  emit?.({ kind: 'final_text', text: finalText });
549
780
  emit?.({ kind: 'step_end', step: steps });
550
781
  await closeTracer();
551
- return { status: 'verify_failed', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: { ok: false, command: verifyCmd, attempts: verificationAttempts, failureType } };
782
+ return { status: 'verify_failed', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: withRepairTokens({ ok: false, command: verifyCmd, attempts: verificationAttempts, failureType }) };
552
783
  }
553
784
  emit?.({ kind: 'step_end', step: steps });
554
785
  continue; // -> next iteration lets model repair
@@ -562,11 +793,10 @@ export async function run(opts, deps) {
562
793
  catch { /* ignore */ }
563
794
  }
564
795
  await closeTracer();
565
- return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? { ok: true, attempts: verificationAttempts } : undefined };
796
+ return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: true, attempts: verificationAttempts }) : undefined };
566
797
  }
567
798
  // 3.1 — emit turn events
568
799
  emitKlyro({ type: 'turn.start', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', turn: steps, model: opts.model });
569
- // Execute each tool call (after policy) — 3.5 parallel if all concurrencySafe
570
800
  const toolCtx = {
571
801
  cwd: opts.cwd,
572
802
  env: process.env,
@@ -580,7 +810,9 @@ export async function run(opts, deps) {
580
810
  agentDepth: opts.parentContext?.depth ?? 0,
581
811
  agentMaxDepth: opts.parentContext?.maxDepth ?? 1,
582
812
  ...(opts.parentContext?.allowedTools ? { agentAllowedTools: opts.parentContext.allowedTools } : {}),
813
+ ...(opts.parentContext?.allowedPaths ? { agentAllowedPaths: opts.parentContext.allowedPaths } : {}),
583
814
  ...(opts.parentContext?.model ?? opts.model ? { agentModel: opts.parentContext?.model ?? opts.model } : {}),
815
+ repairGuard: { denyTestEdits: inRepairTurn },
584
816
  };
585
817
  const allSafe = finalizedCalls.length > 1 && finalizedCalls.every((c) => deps.registry.get(c.name)?.isConcurrencySafe !== false);
586
818
  // Gate phase: policy decision + approval prompt for one call. Runs
@@ -591,7 +823,12 @@ export async function run(opts, deps) {
591
823
  const decision = await deps.policy.evaluate({ name: call.name, input: call.input }, { cwd: opts.cwd, nonInteractive: opts.nonInteractive });
592
824
  emit?.({ kind: 'policy_decision', id: call.id, name: call.name, action: decision.action, ...(decision.action !== 'allow' ? { reason: decision.reason } : {}) });
593
825
  // Mirror to KlyroEvent bus
594
- emitKlyro({ type: 'permission.decision', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, action: decision.action, ...(decision.action !== 'allow' ? { reason: decision.reason } : {}) });
826
+ if (decision.action === 'allow') {
827
+ emitKlyro({ type: 'permission.decision', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, action: 'allow' });
828
+ }
829
+ else {
830
+ emitKlyro({ type: 'permission.decision', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, action: decision.action, reason: decision.reason });
831
+ }
595
832
  if (decision.action === 'deny') {
596
833
  const denyMsg = {
597
834
  role: 'tool',
@@ -640,6 +877,32 @@ export async function run(opts, deps) {
640
877
  const execTool = async (call) => {
641
878
  const t0 = Date.now();
642
879
  emitKlyro({ type: 'tool.call', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, name: call.name, input: call.input });
880
+ // Hooks: every preToolUse hook runs before execution. A non-zero exit
881
+ // denies the tool with POLICY_DENIED — the real tool never runs.
882
+ if (preHooks.length > 0) {
883
+ for (const hook of preHooks) {
884
+ let exitCode = -1;
885
+ let detail = '';
886
+ try {
887
+ const r = await runHook(hook, { toolName: call.name, input: call.input });
888
+ exitCode = r.exitCode;
889
+ detail = (r.stderr || r.stdout || '').slice(0, 300);
890
+ }
891
+ catch (err) {
892
+ detail = String(err instanceof Error ? err.message : err).slice(0, 300);
893
+ }
894
+ if (exitCode !== 0) {
895
+ const reason = `hook ${hook.name} denied: ${detail || 'hook failed'}`;
896
+ emit?.({ kind: 'policy_decision', id: call.id, name: call.name, action: 'deny', reason });
897
+ emitKlyro({ type: 'permission.decision', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, action: 'deny', reason });
898
+ const latencyMs = Date.now() - t0;
899
+ return {
900
+ obs: { ok: false, error: { code: 'POLICY_DENIED', message: reason } },
901
+ latencyMs,
902
+ };
903
+ }
904
+ }
905
+ }
643
906
  let obs;
644
907
  try {
645
908
  obs = await deps.registry.execute(call.name, call.input, toolCtx);
@@ -727,6 +990,31 @@ export async function run(opts, deps) {
727
990
  if (last3.length === 3 && last3[0] === last3[1] && last3[1] === last3[2]) {
728
991
  await markStuck(`identical call ×3: ${sig}`);
729
992
  }
993
+ // Hooks: postToolUse hooks are best-effort — failures warn on stderr
994
+ // plus a bus event, and never fail the turn.
995
+ if (postHooks.length > 0) {
996
+ for (const hook of postHooks) {
997
+ try {
998
+ const r = await runHook(hook, { toolName: call.name, input: call.input });
999
+ if (!r.ok || r.exitCode !== 0) {
1000
+ const msg = `klyro: hooks: postToolUse ${hook.name} failed (exit ${String(r.exitCode)}): ${(r.stderr || r.stdout || '').slice(0, 200)}\n`;
1001
+ try {
1002
+ process.stderr.write(msg);
1003
+ }
1004
+ catch { /* ignore */ }
1005
+ emitKlyro({ type: 'error', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', code: 'hook_failed', message: msg.slice(0, 300) });
1006
+ }
1007
+ }
1008
+ catch (err) {
1009
+ const msg = `klyro: hooks: postToolUse ${hook.name} error: ${String(err instanceof Error ? err.message : err).slice(0, 200)}\n`;
1010
+ try {
1011
+ process.stderr.write(msg);
1012
+ }
1013
+ catch { /* ignore */ }
1014
+ emitKlyro({ type: 'error', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', code: 'hook_failed', message: msg.slice(0, 300) });
1015
+ }
1016
+ }
1017
+ }
730
1018
  };
731
1019
  // Sequential path: gate → execute → commit per call, in order.
732
1020
  const runOne = async (call) => {
@@ -750,8 +1038,17 @@ export async function run(opts, deps) {
750
1038
  break;
751
1039
  }
752
1040
  if (approved.length > 0 && !opts.signal?.aborted) {
753
- const settled = await Promise.allSettled(approved.map((c) => execTool(c)));
754
- for (let i = 0; i < approved.length; i++) {
1041
+ // Fan-out cap: execute in sequential chunks of MAX_PARALLEL_TOOLS.
1042
+ // Commits below stay in original call order, so the transcript is
1043
+ // unaffected by the chunking.
1044
+ const settled = [];
1045
+ for (let off = 0; off < approved.length; off += MAX_PARALLEL_TOOLS) {
1046
+ if (opts.signal?.aborted)
1047
+ break;
1048
+ const chunk = approved.slice(off, off + MAX_PARALLEL_TOOLS);
1049
+ settled.push(...await Promise.allSettled(chunk.map((c) => execTool(c))));
1050
+ }
1051
+ for (let i = 0; i < settled.length; i++) {
755
1052
  const s = settled[i];
756
1053
  if (s.status === 'fulfilled') {
757
1054
  await commitResult(approved[i], s.value.obs, s.value.latencyMs);
@@ -770,6 +1067,19 @@ export async function run(opts, deps) {
770
1067
  break;
771
1068
  }
772
1069
  }
1070
+ // P0 — drain finished sub-agent completions into parent visibility.
1071
+ // drainCompletions is OPTIONAL on the bridge — guarded with `?.` so
1072
+ // older bridges without it simply yield nothing.
1073
+ try {
1074
+ const completions = opts.agentBridge?.drainCompletions?.() ?? [];
1075
+ for (const c of completions) {
1076
+ const body = c.finalText ?? c.error?.message ?? '';
1077
+ const msg = { role: 'user', content: [text(`Subtask ${c.taskId} (${c.agentName}) ${c.status}: ${body.slice(0, 500)}`)] };
1078
+ transcript.push(msg);
1079
+ await checkpoint(msg);
1080
+ }
1081
+ }
1082
+ catch { /* ignore — completions are best-effort visibility */ }
773
1083
  emit?.({ kind: 'step_end', step: steps });
774
1084
  emitKlyro({ type: 'turn.end', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', turn: steps });
775
1085
  // Level 9 — checkpoint status after each step
@@ -792,7 +1102,7 @@ export async function run(opts, deps) {
792
1102
  catch { /* ignore */ }
793
1103
  }
794
1104
  await closeTracer();
795
- return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? { ok: false, attempts: verificationAttempts } : undefined };
1105
+ return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined };
796
1106
  }
797
1107
  // 5.2 — stuck termination (P0-3): bounded stop instead of unbounded token burn.
798
1108
  if (stuckAbort) {
@@ -804,7 +1114,7 @@ export async function run(opts, deps) {
804
1114
  catch { /* ignore */ }
805
1115
  }
806
1116
  await closeTracer();
807
- return { status: 'stuck', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? { ok: false, attempts: verificationAttempts } : undefined };
1117
+ return { status: 'stuck', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined };
808
1118
  }
809
1119
  emit?.({ kind: 'final_text', text: finalText });
810
1120
  if (store && sessionId) {
@@ -814,7 +1124,7 @@ export async function run(opts, deps) {
814
1124
  catch { /* ignore */ }
815
1125
  }
816
1126
  await closeTracer();
817
- return { status: 'max_steps', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? { ok: false, attempts: verificationAttempts } : undefined };
1127
+ return { status: 'max_steps', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined };
818
1128
  }
819
1129
  /**
820
1130
  * Estimate usage for a turn whose provider omitted it (Ollama, vLLM,
@@ -950,7 +1260,10 @@ export function defaultSystemPrompt(ctx) {
950
1260
  'Do not invent file paths. Do not call tools outside the working directory.',
951
1261
  ].join(' ');
952
1262
  if (ctx.telemetry) {
953
- return base + '\n\n' + ctx.telemetry + '\n\nUse the telemetry above to avoid repeating the same failing call and to keep within the step budget.';
1263
+ return {
1264
+ system: base,
1265
+ suffix: ctx.telemetry + '\n\nUse the telemetry above to avoid repeating the same failing call and to keep within the step budget.',
1266
+ };
954
1267
  }
955
- return base;
1268
+ return { system: base };
956
1269
  }