klyro 0.1.63 → 1.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/dist/agent/anthropic-adapter.d.ts +36 -0
  2. package/dist/agent/anthropic-adapter.js +73 -16
  3. package/dist/agent/capabilities.d.ts +145 -0
  4. package/dist/agent/capabilities.js +191 -0
  5. package/dist/agent/child-worker.d.ts +104 -0
  6. package/dist/agent/child-worker.js +250 -0
  7. package/dist/agent/orchestrator.d.ts +232 -0
  8. package/dist/agent/orchestrator.js +589 -0
  9. package/dist/agent/provider-adapter.d.ts +17 -0
  10. package/dist/agent/provider-adapter.js +35 -3
  11. package/dist/agent/registry.d.ts +1 -0
  12. package/dist/agent/registry.js +1 -0
  13. package/dist/agent/retry.d.ts +13 -2
  14. package/dist/agent/retry.js +21 -3
  15. package/dist/agent/runtime.d.ts +70 -8
  16. package/dist/agent/runtime.js +205 -34
  17. package/dist/agent/scoped-registry.d.ts +22 -0
  18. package/dist/agent/scoped-registry.js +42 -0
  19. package/dist/agent/task-manager.d.ts +115 -0
  20. package/dist/agent/task-manager.js +250 -0
  21. package/dist/agent/worker-spawner.d.ts +17 -12
  22. package/dist/agent/worker-spawner.js +26 -20
  23. package/dist/agent/worktree-manager.d.ts +74 -0
  24. package/dist/agent/worktree-manager.js +189 -0
  25. package/dist/checkpoints/store.js +30 -5
  26. package/dist/cli/auth.js +16 -1
  27. package/dist/cli/config.d.ts +9 -3
  28. package/dist/cli/config.js +64 -3
  29. package/dist/cli/dotenv.d.ts +3 -0
  30. package/dist/cli/dotenv.js +57 -0
  31. package/dist/cli/eval.d.ts +6 -1
  32. package/dist/cli/eval.js +9 -0
  33. package/dist/cli/repl.js +183 -17
  34. package/dist/cli/run.d.ts +10 -11
  35. package/dist/cli/run.js +176 -14
  36. package/dist/cli/update.d.ts +5 -0
  37. package/dist/cli/update.js +62 -10
  38. package/dist/context/import-graph.d.ts +2 -0
  39. package/dist/context/import-graph.js +31 -3
  40. package/dist/context/klyro-md.d.ts +6 -0
  41. package/dist/context/klyro-md.js +25 -16
  42. package/dist/context/memory.d.ts +8 -0
  43. package/dist/context/memory.js +50 -2
  44. package/dist/context/project-map.d.ts +6 -0
  45. package/dist/context/project-map.js +50 -2
  46. package/dist/context/repo-map.d.ts +2 -0
  47. package/dist/context/repo-map.js +31 -1
  48. package/dist/context/trust.d.ts +42 -0
  49. package/dist/context/trust.js +111 -0
  50. package/dist/events/catalog.d.ts +99 -0
  51. package/dist/index.js +93 -4
  52. package/dist/mcp/client.d.ts +55 -0
  53. package/dist/mcp/client.js +294 -0
  54. package/dist/mcp/config.d.ts +40 -0
  55. package/dist/mcp/config.js +99 -0
  56. package/dist/mcp/policy.d.ts +13 -0
  57. package/dist/mcp/policy.js +12 -0
  58. package/dist/mcp/registry.d.ts +72 -0
  59. package/dist/mcp/registry.js +244 -0
  60. package/dist/mcp/schema.d.ts +18 -0
  61. package/dist/mcp/schema.js +57 -0
  62. package/dist/mcp/trust.d.ts +20 -0
  63. package/dist/mcp/trust.js +74 -0
  64. package/dist/persistence/audit.d.ts +28 -0
  65. package/dist/persistence/audit.js +101 -1
  66. package/dist/persistence/store.d.ts +26 -2
  67. package/dist/persistence/store.js +140 -13
  68. package/dist/policy/approval.d.ts +14 -0
  69. package/dist/policy/approval.js +44 -2
  70. package/dist/policy/engine.d.ts +1 -0
  71. package/dist/policy/engine.js +91 -10
  72. package/dist/policy/secret-redactor.js +4 -0
  73. package/dist/providers/model-info.d.ts +17 -0
  74. package/dist/providers/model-info.js +35 -2
  75. package/dist/repl.d.ts +6 -0
  76. package/dist/repl.js +12 -7
  77. package/dist/tools/agent/spawn-agent.d.ts +9 -0
  78. package/dist/tools/agent/spawn-agent.js +50 -0
  79. package/dist/tools/agent/task-apply.d.ts +4 -0
  80. package/dist/tools/agent/task-apply.js +44 -0
  81. package/dist/tools/agent/task-get.d.ts +8 -0
  82. package/dist/tools/agent/task-get.js +40 -0
  83. package/dist/tools/agent/task-list.d.ts +4 -0
  84. package/dist/tools/agent/task-list.js +41 -0
  85. package/dist/tools/agent/task-stop.d.ts +6 -0
  86. package/dist/tools/agent/task-stop.js +39 -0
  87. package/dist/tools/agent/task-wait.d.ts +17 -0
  88. package/dist/tools/agent/task-wait.js +79 -0
  89. package/dist/tools/fs/apply-patch.js +71 -0
  90. package/dist/tools/fs/edit-file.js +65 -0
  91. package/dist/tools/fs/multi-edit.d.ts +4 -0
  92. package/dist/tools/fs/multi-edit.js +66 -0
  93. package/dist/tools/fs/write-file.js +67 -0
  94. package/dist/tools/plan/todo-write.d.ts +1 -1
  95. package/dist/tools/registry.js +12 -0
  96. package/dist/tools/shell/background.js +6 -3
  97. package/dist/tools/shell/sandbox.d.ts +51 -0
  98. package/dist/tools/shell/sandbox.js +143 -0
  99. package/dist/tools/shell/shell-exec.d.ts +1 -0
  100. package/dist/tools/shell/shell-exec.js +83 -11
  101. package/dist/tools/shell/worker-entry.d.ts +12 -0
  102. package/dist/tools/shell/worker-entry.js +43 -0
  103. package/dist/tools/types.d.ts +18 -0
  104. package/dist/tools/verify/run-verify.js +3 -1
  105. package/dist/trace/writer.d.ts +13 -0
  106. package/dist/trace/writer.js +55 -4
  107. package/dist/tui/app.js +1 -1
  108. package/dist/tui/approval.js +20 -21
  109. package/dist/util.d.ts +1 -0
  110. package/dist/util.js +1 -0
  111. package/dist/verification/baseline.js +17 -3
  112. package/dist/verification/classify.js +23 -12
  113. package/dist/verification/engine.js +3 -1
  114. package/dist/verification/registry.d.ts +2 -0
  115. package/dist/verification/registry.js +33 -0
  116. package/dist/verification/scoped.js +28 -6
  117. package/package.json +1 -1
@@ -23,10 +23,16 @@ import { verify, diagnosticForModel } from '../verification/engine.js';
23
23
  import { detectVerifyCommand } from '../verification/auto.js';
24
24
  import { ensureBaseline, getBaseline } from '../verification/baseline.js';
25
25
  import { compressTranscript, totalTokens } from '../context/tokenizer.js';
26
+ import { ratesFor } from '../providers/model-info.js';
26
27
  import { classifyFailure, rerunOnce, gatherRepairContext, guardRepair } from '../verification/classify.js';
27
28
  import { findRelatedTests, buildScopedCommand, runScopedVerify, syntaxCheck, checkImports } from '../verification/scoped.js';
28
29
  import { globalBus } from '../events/bus.js';
29
30
  import { TraceWriter } from '../trace/writer.js';
31
+ /** Normalize either systemPrompt shape into {system, suffix}. */
32
+ export function resolveSystemPrompt(fn, ctx) {
33
+ const r = fn(ctx);
34
+ return typeof r === 'string' ? { system: r } : { system: r.system, suffix: r.suffix };
35
+ }
30
36
  const DEFAULT_MAX_STEPS = 30;
31
37
  /** Convert a registry of tools into ToolDefinitions for the provider. */
32
38
  export function toolDefinitions(registry) {
@@ -36,19 +42,16 @@ export function toolDefinitions(registry) {
36
42
  inputSchema: t.function.parameters,
37
43
  }));
38
44
  }
39
- // BUG-005: Model-aware cost estimation with sensible defaults.
40
- // Rates are per-1K tokens (input / output). Local models are $0.
41
- const MODEL_RATES = [
42
- { test: (m) => /gpt-4/i.test(m), input: 0.003, output: 0.015 },
43
- { test: (m) => /gpt-3\.5/i.test(m), input: 0.0005, output: 0.0015 },
44
- { test: (m) => /claude|anthropic/i.test(m), input: 0.003, output: 0.015 },
45
- { test: (m) => /gemini/i.test(m), input: 0.00075, output: 0.003 },
46
- { test: (m) => /o1/i.test(m), input: 0.015, output: 0.06 },
47
- ];
45
+ // BUG-005: Model-aware cost estimation, single-sourced from the
46
+ // providers/model-info.ts rate table (local/unknown models are $0).
47
+ // Cost is computed on input/output ONLY: cacheRead/cacheWrite are tracked
48
+ // for observability but excluded because cached tokens bill at
49
+ // provider-specific discounted rates we don't model — charging them at
50
+ // full input rates would overstate spend, silently dropping them
51
+ // understates it, so we keep them visible and out of the math.
48
52
  /** Estimate USD cost of a usage block given the model name. */
49
53
  export function estimateCost(model, usage) {
50
- const match = MODEL_RATES.find((r) => r.test(model));
51
- const { input: inRate, output: outRate } = match ?? { input: 0.003, output: 0.015 };
54
+ const { input: inRate, output: outRate } = ratesFor(model);
52
55
  return (usage.input / 1000) * inRate + (usage.output / 1000) * outRate;
53
56
  }
54
57
  // PERF-002: Memoized token counting cache.
@@ -93,6 +96,27 @@ export async function run(opts, deps) {
93
96
  let repairs = 0;
94
97
  let verificationAttempts = 0;
95
98
  let hasEdits = false;
99
+ // Overflow recovery: at most one compress-and-retry per run.
100
+ let overflowRetried = false;
101
+ // Repair ledger + guard: set from the first verification failure until a
102
+ // verification passes. repairUsageStart snapshots usage at failure time so
103
+ // every verification payload can attribute its repairTokens delta.
104
+ let inRepairTurn = false;
105
+ let repairUsageStart;
106
+ const verifyMode = opts.verify?.mode ?? 'strict';
107
+ const markRepairStarted = () => {
108
+ if (!repairUsageStart)
109
+ repairUsageStart = { input: usage.input, output: usage.output };
110
+ inRepairTurn = true;
111
+ };
112
+ const repairTokensNow = () => repairUsageStart
113
+ ? { input: usage.input - repairUsageStart.input, output: usage.output - repairUsageStart.output }
114
+ : undefined;
115
+ /** Attach repairTokens to a verification payload when a repair ledger exists. */
116
+ const withRepairTokens = (v) => {
117
+ const rt = repairTokensNow();
118
+ return rt ? { ...v, repairTokens: rt } : v;
119
+ };
96
120
  const emit = opts.onEvent;
97
121
  const telemetry = new RuntimeTelemetry();
98
122
  telemetry.setMaxSteps(maxSteps);
@@ -200,7 +224,7 @@ export async function run(opts, deps) {
200
224
  catch { /* ignore */ }
201
225
  }
202
226
  await closeTracer();
203
- return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? { ok: false, attempts: verificationAttempts } : undefined, phase: 'blocked' };
227
+ return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined, phase: 'blocked' };
204
228
  }
205
229
  steps++;
206
230
  // 5.1 phase transitions (model-narrated)
@@ -216,21 +240,30 @@ export async function run(opts, deps) {
216
240
  setPhase('verifying');
217
241
  emit?.({ kind: 'step_start', step: steps });
218
242
  telemetry.recordStepStart(steps);
219
- const systemPrompt = deps.systemPrompt({ cwd: opts.cwd, telemetry: steps === 1 ? emptyTelemetryBlock() : telemetry.format() });
243
+ const { system: stableSystem, suffix: telemetrySuffix } = resolveSystemPrompt(deps.systemPrompt, {
244
+ cwd: opts.cwd,
245
+ telemetry: steps === 1 ? emptyTelemetryBlock() : telemetry.format(),
246
+ });
247
+ // Budget accounting sees what the model sees (prefix + suffix); the
248
+ // request itself keeps the halves split for cache-friendly adapters.
249
+ const systemForBudget = telemetrySuffix ? `${stableSystem}\n\n${telemetrySuffix}` : stableSystem;
220
250
  const BUDGET = { total: 120_000, reservedOutput: 4000 };
221
251
  let reqMessages = transcript;
222
- let reqSystem = systemPrompt;
223
- if (cachedTotalTokens(systemPrompt, transcript) > BUDGET.total) {
224
- const c = compressTranscript(systemPrompt, transcript, BUDGET);
252
+ let reqSystem = stableSystem;
253
+ let reqSuffix = telemetrySuffix;
254
+ if (cachedTotalTokens(systemForBudget, transcript) > BUDGET.total) {
255
+ const c = compressTranscript(systemForBudget, transcript, BUDGET);
225
256
  reqSystem = c.system;
257
+ reqSuffix = undefined; // telemetry is regenerable — drop it under pressure
226
258
  reqMessages = c.messages;
227
259
  tokenCache = { lastRef: null, lastSystem: undefined, lastCount: 0 };
228
260
  if (c.dropped > 0)
229
261
  emitKlyro({ type: 'context.compacted', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', dropped: c.dropped });
230
262
  }
231
263
  const req = {
232
- model: opts.model,
264
+ model: opts.parentContext?.model ?? opts.model,
233
265
  system: reqSystem,
266
+ ...(reqSuffix ? { systemSuffix: reqSuffix } : {}),
234
267
  messages: reqMessages,
235
268
  tools: toolDefinitions(deps.registry),
236
269
  ...(opts.maxTokens ? { maxTokens: opts.maxTokens } : {}),
@@ -244,6 +277,8 @@ export async function run(opts, deps) {
244
277
  let thinkingBuf = '';
245
278
  const pendingToolCalls = new Map();
246
279
  let lastFinishReason;
280
+ // Set when this step's request must be re-issued after overflow recovery.
281
+ let overflowRetryPending = false;
247
282
  for await (const ev of events) {
248
283
  if (opts.signal?.aborted)
249
284
  break outer;
@@ -273,14 +308,22 @@ export async function run(opts, deps) {
273
308
  if (ev.usage) {
274
309
  usage.input += ev.usage.input;
275
310
  usage.output += ev.usage.output;
311
+ if (ev.usage.cacheRead !== undefined)
312
+ usage.cacheRead = (usage.cacheRead ?? 0) + ev.usage.cacheRead;
313
+ if (ev.usage.cacheWrite !== undefined)
314
+ usage.cacheWrite = (usage.cacheWrite ?? 0) + ev.usage.cacheWrite;
276
315
  telemetry.recordUsage(ev.usage.input, ev.usage.output);
277
- emit?.({ kind: 'usage', input: usage.input, output: usage.output });
316
+ emit?.({
317
+ kind: 'usage', input: usage.input, output: usage.output,
318
+ ...(usage.cacheRead !== undefined ? { cacheRead: usage.cacheRead } : {}),
319
+ ...(usage.cacheWrite !== undefined ? { cacheWrite: usage.cacheWrite } : {}),
320
+ });
278
321
  }
279
322
  else {
280
323
  // Providers that omit usage (Ollama, vLLM, proxies): estimate from
281
324
  // the actual request + generated output so cost accounting never
282
325
  // silently records zero. Marked estimated for the UI/debugging.
283
- const est = estimateTurnUsage(reqSystem, reqMessages, textBuf, pendingToolCalls);
326
+ const est = estimateTurnUsage(systemForBudget, reqMessages, textBuf, pendingToolCalls);
284
327
  usage.input += est.input;
285
328
  usage.output += est.output;
286
329
  usage.estimated = true;
@@ -289,6 +332,31 @@ export async function run(opts, deps) {
289
332
  }
290
333
  }
291
334
  else if (ev.kind === 'error') {
335
+ // Overflow recovery: on the first REQUEST_TOO_LARGE of a run,
336
+ // aggressively compact the transcript and re-issue the request once.
337
+ // A second occurrence fails normally (returned as no_final below).
338
+ if (ev.code === 'REQUEST_TOO_LARGE' && !overflowRetried) {
339
+ overflowRetried = true;
340
+ overflowRetryPending = true;
341
+ telemetry.recordError('overflow_retry');
342
+ emitKlyro({ type: 'error', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', code: 'REQUEST_TOO_LARGE', message: 'context overflow — aggressively compacting transcript and retrying the request once' });
343
+ try {
344
+ const compacted = compressTranscript(reqSystem, transcript, { total: 30_000, reservedOutput: 4000 });
345
+ if (compacted.messages.length < transcript.length || compacted.dropped > 0) {
346
+ transcript.splice(0, transcript.length, ...compacted.messages);
347
+ }
348
+ else {
349
+ // Transcript already fits the aggressive budget — force it
350
+ // strictly smaller so the retry cannot repeat the overflow.
351
+ const halved = Math.max(4000, Math.floor(totalTokens(reqSystem, transcript) / 2));
352
+ const smaller = compressTranscript(reqSystem, transcript, { total: halved, reservedOutput: 4000 });
353
+ transcript.splice(0, transcript.length, ...smaller.messages);
354
+ }
355
+ tokenCache = { lastRef: null, lastSystem: undefined, lastCount: 0 };
356
+ }
357
+ catch { /* ignore — retry with the transcript as-is */ }
358
+ break;
359
+ }
292
360
  telemetry.recordError(`stream_error: ${ev.code}`);
293
361
  if (store && sessionId) {
294
362
  try {
@@ -306,10 +374,18 @@ export async function run(opts, deps) {
306
374
  hasEdits,
307
375
  usage,
308
376
  repairs,
309
- verification: hasEdits ? { ok: false, attempts: verificationAttempts } : undefined,
377
+ verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined,
310
378
  };
311
379
  }
312
380
  }
381
+ // Overflow recovery lands here via `break`: re-issue the same step
382
+ // without consuming the step budget.
383
+ if (overflowRetryPending) {
384
+ overflowRetryPending = false;
385
+ steps--;
386
+ emit?.({ kind: 'step_end', step: steps + 1 });
387
+ continue outer;
388
+ }
313
389
  // Build the assistant message. Tool calls are finalized here: JSON is
314
390
  // parsed and schema-validated BEFORE policy/execution. Malformed calls
315
391
  // become structured MALFORMED_TOOL_CALL results — garbage arguments must
@@ -368,7 +444,7 @@ export async function run(opts, deps) {
368
444
  catch { /* ignore */ }
369
445
  }
370
446
  await closeTracer();
371
- return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? { ok: false, attempts: verificationAttempts } : undefined };
447
+ return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined };
372
448
  }
373
449
  if (finalizedCalls.length === 0) {
374
450
  if (hadInvalidTool) {
@@ -380,6 +456,66 @@ export async function run(opts, deps) {
380
456
  continue;
381
457
  }
382
458
  finalText = textBuf;
459
+ // Verify modes: 'off' skips the pipeline entirely (verification stays
460
+ // undefined, no repair turns). 'advisory' runs verify once on
461
+ // completion and reports the outcome without repair turns. 'strict'
462
+ // (default) runs the full pipeline with autonomous repair below.
463
+ if (verifyMode === 'off') {
464
+ emit?.({ kind: 'final_text', text: finalText });
465
+ emit?.({ kind: 'step_end', step: steps });
466
+ if (store && sessionId) {
467
+ try {
468
+ await store.setStatus(sessionId, 'complete', finalText);
469
+ }
470
+ catch { /* ignore */ }
471
+ }
472
+ await closeTracer();
473
+ return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: undefined };
474
+ }
475
+ if (verifyMode === 'advisory') {
476
+ const advisoryCmd = opts.verify?.command ?? detectVerifyCommand(opts.cwd);
477
+ if (opts.verify?.enabled !== false && hasEdits && advisoryCmd) {
478
+ emit?.({ kind: 'verification_started', command: advisoryCmd });
479
+ let advisoryResult;
480
+ try {
481
+ advisoryResult = await verify({ cwd: opts.cwd, command: advisoryCmd, timeoutMs: opts.verify?.timeoutMs });
482
+ }
483
+ catch (e) {
484
+ const msg = e instanceof Error ? e.message : String(e);
485
+ advisoryResult = { ok: false, exitCode: -1, stdout: '', stderr: msg, failure: { type: 'unknown', files: [], raw: msg, exitCode: -1 } };
486
+ }
487
+ verificationAttempts++;
488
+ if (advisoryResult.ok) {
489
+ inRepairTurn = false;
490
+ emit?.({ kind: 'verification_succeeded', command: advisoryCmd });
491
+ emit?.({ kind: 'final_text', text: finalText });
492
+ emit?.({ kind: 'step_end', step: steps });
493
+ if (store && sessionId) {
494
+ try {
495
+ await store.setStatus(sessionId, 'complete', finalText);
496
+ }
497
+ catch { /* ignore */ }
498
+ }
499
+ await closeTracer();
500
+ return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: withRepairTokens({ ok: true, command: advisoryCmd, attempts: verificationAttempts }) };
501
+ }
502
+ const advisoryFailure = advisoryResult.failure?.type ?? 'unknown';
503
+ markRepairStarted();
504
+ emit?.({ kind: 'verification_failed', step: String(steps), reason: diagnosticForModel(advisoryResult).slice(0, 800) });
505
+ emit?.({ kind: 'final_text', text: finalText });
506
+ emit?.({ kind: 'step_end', step: steps });
507
+ if (store && sessionId) {
508
+ try {
509
+ await store.setStatus(sessionId, 'complete', finalText);
510
+ }
511
+ catch { /* ignore */ }
512
+ }
513
+ await closeTracer();
514
+ return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: withRepairTokens({ ok: false, command: advisoryCmd, attempts: verificationAttempts, failureType: advisoryFailure }) };
515
+ }
516
+ // No verify applicable (disabled, no edits, or no command) — fall
517
+ // through to the normal completion return below.
518
+ }
383
519
  // Level 8 — Verification + Autonomous Repair (gated on hasEdits below — pure analysis skips verify)
384
520
  const verifyEnabled = opts.verify?.enabled !== false;
385
521
  const verifyCmd = opts.verify?.command ?? detectVerifyCommand(opts.cwd);
@@ -439,6 +575,7 @@ export async function run(opts, deps) {
439
575
  catch { /* scoped success is enough */ }
440
576
  }
441
577
  if (vResult.ok) {
578
+ inRepairTurn = false;
442
579
  emit?.({ kind: 'verification_succeeded', command: verifyCmd });
443
580
  emit?.({ kind: 'final_text', text: finalText });
444
581
  emit?.({ kind: 'step_end', step: steps });
@@ -449,7 +586,7 @@ export async function run(opts, deps) {
449
586
  catch { /* ignore */ }
450
587
  }
451
588
  await closeTracer();
452
- return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: { ok: true, command: verifyCmd, attempts: verificationAttempts } };
589
+ return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: withRepairTokens({ ok: true, command: verifyCmd, attempts: verificationAttempts }) };
453
590
  }
454
591
  // 6.4 — classify
455
592
  const baseline = await getBaseline(opts.cwd, verifyCmd);
@@ -457,6 +594,7 @@ export async function run(opts, deps) {
457
594
  const cls = classifyFailure({ failure: vResult.failure, stdout: vResult.stdout, stderr: vResult.stderr }, baseline, isFlaky);
458
595
  if (cls === 'flaky') {
459
596
  // rerun succeeded on second try — treat as flaky, don't count as repair
597
+ inRepairTurn = false;
460
598
  emit?.({ kind: 'verification_succeeded', command: verifyCmd });
461
599
  emit?.({ kind: 'final_text', text: finalText });
462
600
  emit?.({ kind: 'step_end', step: steps });
@@ -467,7 +605,7 @@ export async function run(opts, deps) {
467
605
  catch { /* ignore */ }
468
606
  }
469
607
  await closeTracer();
470
- return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: { ok: true, command: verifyCmd, attempts: verificationAttempts } };
608
+ return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: withRepairTokens({ ok: true, command: verifyCmd, attempts: verificationAttempts }) };
471
609
  }
472
610
  if (cls === 'env') {
473
611
  // don't try to repair env failures with code edits
@@ -475,10 +613,11 @@ export async function run(opts, deps) {
475
613
  transcript.push(envMsg);
476
614
  await checkpoint(envMsg);
477
615
  verificationAttempts++;
616
+ markRepairStarted();
478
617
  emit?.({ kind: 'verification_failed', step: String(steps), reason: `env: ${diagnosticForModel(vResult).slice(0, 600)}` });
479
618
  if (verificationAttempts >= maxRepairs) {
480
619
  await closeTracer();
481
- return { status: 'verify_failed', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: { ok: false, command: verifyCmd, attempts: verificationAttempts, failureType: 'env' } };
620
+ return { status: 'verify_failed', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: withRepairTokens({ ok: false, command: verifyCmd, attempts: verificationAttempts, failureType: 'env' }) };
482
621
  }
483
622
  emit?.({ kind: 'step_end', step: steps });
484
623
  continue;
@@ -494,6 +633,7 @@ export async function run(opts, deps) {
494
633
  const failurePaths = new Set(vResult.failure?.files.map((f) => f.path).filter(Boolean) ?? []);
495
634
  const overlaps = [...failurePaths].some((p) => introducedPaths.has(p) || introducedPaths.has(path.basename(p)));
496
635
  if (!overlaps) {
636
+ inRepairTurn = false;
497
637
  emit?.({ kind: 'verification_succeeded', command: verifyCmd });
498
638
  emit?.({ kind: 'final_text', text: finalText });
499
639
  emit?.({ kind: 'step_end', step: steps });
@@ -504,11 +644,12 @@ export async function run(opts, deps) {
504
644
  catch { /* ignore */ }
505
645
  }
506
646
  await closeTracer();
507
- return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: { ok: true, command: verifyCmd, attempts: verificationAttempts } };
647
+ return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: withRepairTokens({ ok: true, command: verifyCmd, attempts: verificationAttempts }) };
508
648
  }
509
649
  }
510
650
  // Failure → repair loop (introduced)
511
651
  verificationAttempts++;
652
+ markRepairStarted();
512
653
  const diagnostic = diagnosticForModel(vResult);
513
654
  const failureType = vResult.failure?.type ?? 'unknown';
514
655
  // 6.4 — gather context
@@ -548,7 +689,7 @@ export async function run(opts, deps) {
548
689
  emit?.({ kind: 'final_text', text: finalText });
549
690
  emit?.({ kind: 'step_end', step: steps });
550
691
  await closeTracer();
551
- return { status: 'verify_failed', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: { ok: false, command: verifyCmd, attempts: verificationAttempts, failureType } };
692
+ return { status: 'verify_failed', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: withRepairTokens({ ok: false, command: verifyCmd, attempts: verificationAttempts, failureType }) };
552
693
  }
553
694
  emit?.({ kind: 'step_end', step: steps });
554
695
  continue; // -> next iteration lets model repair
@@ -562,17 +703,26 @@ export async function run(opts, deps) {
562
703
  catch { /* ignore */ }
563
704
  }
564
705
  await closeTracer();
565
- return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? { ok: true, attempts: verificationAttempts } : undefined };
706
+ return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: true, attempts: verificationAttempts }) : undefined };
566
707
  }
567
708
  // 3.1 — emit turn events
568
709
  emitKlyro({ type: 'turn.start', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', turn: steps, model: opts.model });
569
- // Execute each tool call (after policy) — 3.5 parallel if all concurrencySafe
570
710
  const toolCtx = {
571
711
  cwd: opts.cwd,
572
712
  env: process.env,
573
713
  signal: opts.signal,
574
714
  nonInteractive: opts.nonInteractive,
575
715
  sessionId,
716
+ ...(opts.agentBridge ? { agentBridge: opts.agentBridge } : {}),
717
+ ...(opts.parentContext?.taskId
718
+ ? { parentTaskId: opts.parentContext.parentTaskId ?? opts.parentContext.taskId }
719
+ : {}),
720
+ agentDepth: opts.parentContext?.depth ?? 0,
721
+ agentMaxDepth: opts.parentContext?.maxDepth ?? 1,
722
+ ...(opts.parentContext?.allowedTools ? { agentAllowedTools: opts.parentContext.allowedTools } : {}),
723
+ ...(opts.parentContext?.allowedPaths ? { agentAllowedPaths: opts.parentContext.allowedPaths } : {}),
724
+ ...(opts.parentContext?.model ?? opts.model ? { agentModel: opts.parentContext?.model ?? opts.model } : {}),
725
+ repairGuard: { denyTestEdits: inRepairTurn },
576
726
  };
577
727
  const allSafe = finalizedCalls.length > 1 && finalizedCalls.every((c) => deps.registry.get(c.name)?.isConcurrencySafe !== false);
578
728
  // Gate phase: policy decision + approval prompt for one call. Runs
@@ -583,7 +733,12 @@ export async function run(opts, deps) {
583
733
  const decision = await deps.policy.evaluate({ name: call.name, input: call.input }, { cwd: opts.cwd, nonInteractive: opts.nonInteractive });
584
734
  emit?.({ kind: 'policy_decision', id: call.id, name: call.name, action: decision.action, ...(decision.action !== 'allow' ? { reason: decision.reason } : {}) });
585
735
  // Mirror to KlyroEvent bus
586
- emitKlyro({ type: 'permission.decision', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, action: decision.action, ...(decision.action !== 'allow' ? { reason: decision.reason } : {}) });
736
+ if (decision.action === 'allow') {
737
+ emitKlyro({ type: 'permission.decision', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, action: 'allow' });
738
+ }
739
+ else {
740
+ emitKlyro({ type: 'permission.decision', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, action: decision.action, reason: decision.reason });
741
+ }
587
742
  if (decision.action === 'deny') {
588
743
  const denyMsg = {
589
744
  role: 'tool',
@@ -762,6 +917,19 @@ export async function run(opts, deps) {
762
917
  break;
763
918
  }
764
919
  }
920
+ // P0 — drain finished sub-agent completions into parent visibility.
921
+ // drainCompletions is OPTIONAL on the bridge — guarded with `?.` so
922
+ // older bridges without it simply yield nothing.
923
+ try {
924
+ const completions = opts.agentBridge?.drainCompletions?.() ?? [];
925
+ for (const c of completions) {
926
+ const body = c.finalText ?? c.error?.message ?? '';
927
+ const msg = { role: 'user', content: [text(`Subtask ${c.taskId} (${c.agentName}) ${c.status}: ${body.slice(0, 500)}`)] };
928
+ transcript.push(msg);
929
+ await checkpoint(msg);
930
+ }
931
+ }
932
+ catch { /* ignore — completions are best-effort visibility */ }
765
933
  emit?.({ kind: 'step_end', step: steps });
766
934
  emitKlyro({ type: 'turn.end', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', turn: steps });
767
935
  // Level 9 — checkpoint status after each step
@@ -784,7 +952,7 @@ export async function run(opts, deps) {
784
952
  catch { /* ignore */ }
785
953
  }
786
954
  await closeTracer();
787
- return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? { ok: false, attempts: verificationAttempts } : undefined };
955
+ return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined };
788
956
  }
789
957
  // 5.2 — stuck termination (P0-3): bounded stop instead of unbounded token burn.
790
958
  if (stuckAbort) {
@@ -796,7 +964,7 @@ export async function run(opts, deps) {
796
964
  catch { /* ignore */ }
797
965
  }
798
966
  await closeTracer();
799
- return { status: 'stuck', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? { ok: false, attempts: verificationAttempts } : undefined };
967
+ return { status: 'stuck', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined };
800
968
  }
801
969
  emit?.({ kind: 'final_text', text: finalText });
802
970
  if (store && sessionId) {
@@ -806,7 +974,7 @@ export async function run(opts, deps) {
806
974
  catch { /* ignore */ }
807
975
  }
808
976
  await closeTracer();
809
- return { status: 'max_steps', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? { ok: false, attempts: verificationAttempts } : undefined };
977
+ return { status: 'max_steps', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined };
810
978
  }
811
979
  /**
812
980
  * Estimate usage for a turn whose provider omitted it (Ollama, vLLM,
@@ -942,7 +1110,10 @@ export function defaultSystemPrompt(ctx) {
942
1110
  'Do not invent file paths. Do not call tools outside the working directory.',
943
1111
  ].join(' ');
944
1112
  if (ctx.telemetry) {
945
- return base + '\n\n' + ctx.telemetry + '\n\nUse the telemetry above to avoid repeating the same failing call and to keep within the step budget.';
1113
+ return {
1114
+ system: base,
1115
+ suffix: ctx.telemetry + '\n\nUse the telemetry above to avoid repeating the same failing call and to keep within the step budget.',
1116
+ };
946
1117
  }
947
- return base;
1118
+ return { system: base };
948
1119
  }
@@ -0,0 +1,22 @@
1
+ /**
2
+ * ScopedRegistry — a `ToolRegistry` that narrows the surface to a resolved
3
+ * capability set.
4
+ *
5
+ * Both surfaces the runtime relies on (`toolDefinitions(deps.registry)`
6
+ * and `deps.registry.execute(...)`) go through `list()` / `execute()`, so
7
+ * wrapping the child's registry in this class enforces the tool allow-set
8
+ * end-to-end without touching the runtime loop: the model only sees the
9
+ * allowed tools, and any call to a disallowed tool returns `UNKNOWN_TOOL`.
10
+ */
11
+ import type { Tool, ToolContext, ToolResult } from '../tools/types.js';
12
+ import { ToolRegistry } from '../tools/registry.js';
13
+ export declare class ScopedRegistry extends ToolRegistry {
14
+ private readonly parent;
15
+ private readonly allowed;
16
+ constructor(parent: ToolRegistry, allowed: ReadonlySet<string>);
17
+ get(name: string): Tool<unknown, unknown> | undefined;
18
+ list(): Tool<unknown, unknown>[];
19
+ /** Which tools this scope permits, by name (for diagnostics/tests). */
20
+ names(): string[];
21
+ execute(name: string, rawInput: unknown, ctx: ToolContext): Promise<ToolResult<unknown>>;
22
+ }
@@ -0,0 +1,42 @@
1
+ /**
2
+ * ScopedRegistry — a `ToolRegistry` that narrows the surface to a resolved
3
+ * capability set.
4
+ *
5
+ * Both surfaces the runtime relies on (`toolDefinitions(deps.registry)`
6
+ * and `deps.registry.execute(...)`) go through `list()` / `execute()`, so
7
+ * wrapping the child's registry in this class enforces the tool allow-set
8
+ * end-to-end without touching the runtime loop: the model only sees the
9
+ * allowed tools, and any call to a disallowed tool returns `UNKNOWN_TOOL`.
10
+ */
11
+ import { ToolRegistry } from '../tools/registry.js';
12
+ export class ScopedRegistry extends ToolRegistry {
13
+ parent;
14
+ allowed;
15
+ constructor(parent, allowed) {
16
+ super();
17
+ this.parent = parent;
18
+ this.allowed = allowed;
19
+ }
20
+ get(name) {
21
+ return this.allowed.has(name) ? this.parent.get(name) : undefined;
22
+ }
23
+ list() {
24
+ return this.parent.list().filter((t) => this.allowed.has(t.name));
25
+ }
26
+ /** Which tools this scope permits, by name (for diagnostics/tests). */
27
+ names() {
28
+ return [...this.allowed].sort();
29
+ }
30
+ async execute(name, rawInput, ctx) {
31
+ if (!this.allowed.has(name)) {
32
+ return {
33
+ ok: false,
34
+ error: {
35
+ code: 'UNKNOWN_TOOL',
36
+ message: `Tool not available in this agent context: ${name}`,
37
+ },
38
+ };
39
+ }
40
+ return this.parent.execute(name, rawInput, ctx);
41
+ }
42
+ }
@@ -0,0 +1,115 @@
1
+ /**
2
+ * Task manager — tracks the lifecycle of a (possibly nested) agent run.
3
+ *
4
+ * Each `spawn_agent` call creates a `TaskRecord` in a `TaskManager`.
5
+ * The manager owns an `AbortController` per task so that the parent (or
6
+ * the user) can cancel a running child without tearing down the entire
7
+ * process. It enforces a recursion / depth cap and emits `subtask.*`
8
+ * events so the event bus and trace writer can see the task tree.
9
+ *
10
+ * Exactly one task manager exists per session, wired into the orchestrator
11
+ * (src/agent/orchestrator.ts), which the runtime reaches via
12
+ * `parentContext`.
13
+ */
14
+ import { type EventBus } from '../events/bus.js';
15
+ export type TaskStatus = 'queued' | 'running' | 'succeeded' | 'failed' | 'cancelled' | 'timed_out' | 'blocked';
16
+ export interface TaskError {
17
+ code: string;
18
+ message: string;
19
+ details?: unknown;
20
+ }
21
+ /**
22
+ * The full runtime record for one task. `summary` is a free-form progress
23
+ * log; `done` resolves to this very record once the task reaches a terminal
24
+ * status, so `await rec.done` yields the record with its final status.
25
+ */
26
+ export interface TaskRecord {
27
+ id: string;
28
+ parentTaskId?: string;
29
+ sessionId: string;
30
+ agentName: string;
31
+ status: TaskStatus;
32
+ cwd: string;
33
+ depth: number;
34
+ maxDepth: number;
35
+ model?: string;
36
+ startedAt: number;
37
+ finishedAt?: number;
38
+ summary: string[];
39
+ error?: TaskError;
40
+ abortController: AbortController;
41
+ done: Promise<TaskRecord>;
42
+ /** Internal — resolves `done`. Do not call outside TaskManager. */
43
+ _resolveDone: (r: TaskRecord) => void;
44
+ /** Internal cleanup handles, not part of the stable contract. */
45
+ _cleanupParentAbort?: () => void;
46
+ timeoutHandle?: NodeJS.Timeout;
47
+ }
48
+ /** Serialisable, stable subset of a task for logs, tools, and the trace. */
49
+ export interface TaskSummary {
50
+ id: string;
51
+ parentTaskId?: string;
52
+ agentName: string;
53
+ sessionId: string;
54
+ status: TaskStatus;
55
+ cwd: string;
56
+ depth: number;
57
+ model?: string;
58
+ durationMs?: number;
59
+ }
60
+ export interface CreateTaskOpts {
61
+ agentName: string;
62
+ cwd: string;
63
+ depth: number;
64
+ maxDepth: number;
65
+ parentTaskId?: string;
66
+ model?: string;
67
+ timeoutMs?: number;
68
+ /** Abort this task when the given signal fires. */
69
+ abortOnParent?: AbortSignal;
70
+ /** Defaults to true. When false, the task starts in 'queued'. */
71
+ autoStart?: boolean;
72
+ }
73
+ export declare class TaskManager {
74
+ private readonly tasks;
75
+ private readonly watchers;
76
+ private counter;
77
+ private readonly sessionId;
78
+ private readonly bus;
79
+ constructor(opts: {
80
+ sessionId: string;
81
+ bus?: EventBus;
82
+ });
83
+ /** Create a task. Blocks instead of running when the recursion cap is exceeded. */
84
+ create(opts: CreateTaskOpts): TaskRecord;
85
+ /** Look up a live record. */
86
+ get(id: string): TaskRecord | undefined;
87
+ /** List tasks, optionally filtered by parent and/or status. Returns summaries. */
88
+ list(filter?: {
89
+ parentTaskId?: string;
90
+ status?: TaskStatus;
91
+ }): TaskSummary[];
92
+ /** Transition a task to a terminal status and resolve its `done`. */
93
+ finish(id: string, status: 'succeeded' | 'failed' | 'cancelled' | 'timed_out' | 'blocked', patch?: {
94
+ summary?: string[];
95
+ error?: TaskError;
96
+ }): TaskRecord;
97
+ /** Cancel a live task: abort, mark cancelled/timed_out, resolve done. */
98
+ cancel(id: string, reason?: string, finalStatus?: 'cancelled' | 'timed_out'): TaskRecord;
99
+ /** Confirm a task that policy/depth prevented from running. */
100
+ block(id: string, reason: string): TaskRecord;
101
+ /** Cancel `id` and every transitive descendant. */
102
+ cancelTree(id: string, reason?: string): Promise<void>;
103
+ /** Subscribe to status transitions for one task. Delivers the current status immediately. */
104
+ subscribe(id: string, listener: (r: TaskRecord) => void): () => void;
105
+ /** Cancel every live task (used on shutdown / ctrl-c). */
106
+ shutdown(): Promise<void>;
107
+ /** Compact serialisable summary. */
108
+ toSummary(r: TaskRecord): TaskSummary;
109
+ /** Resolve a record's `done` (idempotent). */
110
+ private resolve;
111
+ private _resolveOnce;
112
+ /** Deliver a transition to the task's own watchers, then each ancestor's. */
113
+ private emit;
114
+ private notify;
115
+ }