klyro 1.0.0 → 1.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (96) hide show
  1. package/dist/agent/anthropic-adapter.d.ts +36 -0
  2. package/dist/agent/anthropic-adapter.js +68 -16
  3. package/dist/agent/capabilities.d.ts +23 -0
  4. package/dist/agent/capabilities.js +46 -5
  5. package/dist/agent/child-worker.d.ts +104 -0
  6. package/dist/agent/child-worker.js +250 -0
  7. package/dist/agent/orchestrator.d.ts +106 -5
  8. package/dist/agent/orchestrator.js +378 -58
  9. package/dist/agent/provider-adapter.d.ts +8 -0
  10. package/dist/agent/provider-adapter.js +12 -3
  11. package/dist/agent/retry.d.ts +1 -1
  12. package/dist/agent/retry.js +2 -2
  13. package/dist/agent/runtime.d.ts +50 -8
  14. package/dist/agent/runtime.js +196 -33
  15. package/dist/agent/worktree-manager.d.ts +74 -0
  16. package/dist/agent/worktree-manager.js +189 -0
  17. package/dist/checkpoints/store.js +30 -5
  18. package/dist/cli/auth.js +16 -1
  19. package/dist/cli/config.d.ts +9 -3
  20. package/dist/cli/config.js +64 -3
  21. package/dist/cli/eval.d.ts +6 -1
  22. package/dist/cli/eval.js +9 -0
  23. package/dist/cli/repl.js +155 -28
  24. package/dist/cli/run.d.ts +7 -11
  25. package/dist/cli/run.js +68 -17
  26. package/dist/cli/update.d.ts +5 -0
  27. package/dist/cli/update.js +62 -10
  28. package/dist/context/import-graph.d.ts +2 -0
  29. package/dist/context/import-graph.js +31 -3
  30. package/dist/context/klyro-md.js +4 -1
  31. package/dist/context/memory.d.ts +8 -0
  32. package/dist/context/memory.js +50 -2
  33. package/dist/context/project-map.d.ts +6 -0
  34. package/dist/context/project-map.js +50 -2
  35. package/dist/context/repo-map.d.ts +2 -0
  36. package/dist/context/repo-map.js +31 -1
  37. package/dist/events/catalog.d.ts +28 -0
  38. package/dist/index.js +89 -4
  39. package/dist/mcp/client.d.ts +6 -4
  40. package/dist/mcp/client.js +83 -14
  41. package/dist/mcp/config.d.ts +10 -0
  42. package/dist/mcp/config.js +18 -1
  43. package/dist/mcp/registry.d.ts +23 -1
  44. package/dist/mcp/registry.js +78 -6
  45. package/dist/mcp/schema.d.ts +11 -4
  46. package/dist/mcp/schema.js +27 -16
  47. package/dist/mcp/trust.d.ts +20 -0
  48. package/dist/mcp/trust.js +74 -0
  49. package/dist/persistence/audit.d.ts +28 -0
  50. package/dist/persistence/audit.js +101 -1
  51. package/dist/persistence/store.d.ts +26 -2
  52. package/dist/persistence/store.js +140 -13
  53. package/dist/policy/approval.d.ts +14 -0
  54. package/dist/policy/approval.js +44 -2
  55. package/dist/policy/engine.d.ts +1 -0
  56. package/dist/policy/engine.js +88 -8
  57. package/dist/policy/secret-redactor.js +4 -0
  58. package/dist/providers/model-info.d.ts +17 -0
  59. package/dist/providers/model-info.js +35 -2
  60. package/dist/repl.d.ts +6 -0
  61. package/dist/repl.js +12 -7
  62. package/dist/tools/agent/spawn-agent.js +5 -5
  63. package/dist/tools/agent/task-apply.d.ts +4 -0
  64. package/dist/tools/agent/task-apply.js +44 -0
  65. package/dist/tools/agent/task-stop.d.ts +6 -0
  66. package/dist/tools/agent/task-stop.js +39 -0
  67. package/dist/tools/agent/task-wait.d.ts +17 -0
  68. package/dist/tools/agent/task-wait.js +79 -0
  69. package/dist/tools/fs/apply-patch.js +71 -0
  70. package/dist/tools/fs/edit-file.js +65 -0
  71. package/dist/tools/fs/multi-edit.d.ts +4 -0
  72. package/dist/tools/fs/multi-edit.js +66 -0
  73. package/dist/tools/fs/write-file.js +67 -0
  74. package/dist/tools/registry.js +6 -0
  75. package/dist/tools/shell/background.js +6 -3
  76. package/dist/tools/shell/sandbox.d.ts +51 -0
  77. package/dist/tools/shell/sandbox.js +143 -0
  78. package/dist/tools/shell/shell-exec.d.ts +1 -0
  79. package/dist/tools/shell/shell-exec.js +83 -11
  80. package/dist/tools/shell/worker-entry.d.ts +12 -0
  81. package/dist/tools/shell/worker-entry.js +43 -0
  82. package/dist/tools/types.d.ts +6 -0
  83. package/dist/tools/verify/run-verify.js +3 -1
  84. package/dist/trace/writer.d.ts +13 -0
  85. package/dist/trace/writer.js +55 -4
  86. package/dist/tui/app.js +1 -1
  87. package/dist/tui/approval.js +20 -21
  88. package/dist/util.d.ts +1 -0
  89. package/dist/util.js +1 -0
  90. package/dist/verification/baseline.js +17 -3
  91. package/dist/verification/classify.js +23 -12
  92. package/dist/verification/engine.js +3 -1
  93. package/dist/verification/registry.d.ts +2 -0
  94. package/dist/verification/registry.js +33 -0
  95. package/dist/verification/scoped.js +28 -6
  96. package/package.json +1 -1
@@ -19,6 +19,16 @@ import type { Message } from './message.js';
19
19
  import type { ToolRegistry } from '../tools/registry.js';
20
20
  import type { PolicyEngine } from '../policy/engine.js';
21
21
  import type { ApprovalPrompt } from '../policy/approval.js';
22
+ /**
23
+ * Verification mode. The canonical definition lives in verification/engine.ts
24
+ * (sibling-owned: `export type VerifyMode = 'strict'|'advisory'|'off'`). It is
25
+ * resolved conditionally here so this file typechecks regardless of sibling
26
+ * landing order — and converges to the sibling type automatically once the
27
+ * sibling export exists.
28
+ */
29
+ export type VerifyMode = typeof import('../verification/engine.js') extends {
30
+ VerifyMode: infer V;
31
+ } ? V : 'strict' | 'advisory' | 'off';
22
32
  export interface RuntimeDeps {
23
33
  adapter: ProviderAdapter;
24
34
  registry: ToolRegistry;
@@ -26,15 +36,34 @@ export interface RuntimeDeps {
26
36
  approval: ApprovalPrompt;
27
37
  /**
28
38
  * Build a system prompt given cwd + the current Level-7 runtime telemetry.
29
- * The telemetry block is a compact, in-memory summary of the run so far
30
- * (step count, last tool calls, recent errors). Injected as part of the
31
- * system prompt so the model can see its own state mid-run.
39
+ *
40
+ * TELEMETRY SPLIT: the fn may return either a plain string (legacy —
41
+ * telemetry already concatenated, cache-unfriendly) or
42
+ * `{system, suffix}` where `suffix` is the volatile telemetry block.
43
+ * The runtime forwards both halves via CallRequest (`system` +
44
+ * `systemSuffix`) so adapters can keep the suffix out of the cacheable
45
+ * prefix (Anthropic array form) or concatenate it (OpenAI — unchanged).
46
+ * Both shapes are accepted so custom fns keep compiling.
32
47
  */
33
- systemPrompt: (ctx: {
34
- cwd: string;
35
- telemetry?: string;
36
- }) => string;
48
+ systemPrompt: SystemPromptFn;
37
49
  }
50
+ /** Split system-prompt result: stable prefix + volatile telemetry suffix. */
51
+ export interface SystemPromptResult {
52
+ system: string;
53
+ suffix?: string;
54
+ }
55
+ export type SystemPromptFn = (ctx: {
56
+ cwd: string;
57
+ telemetry?: string;
58
+ }) => string | SystemPromptResult;
59
+ /** Normalize either systemPrompt shape into {system, suffix}. */
60
+ export declare function resolveSystemPrompt(fn: SystemPromptFn, ctx: {
61
+ cwd: string;
62
+ telemetry?: string;
63
+ }): {
64
+ system: string;
65
+ suffix?: string;
66
+ };
38
67
  export type Phase = 'understanding' | 'exploring' | 'planning' | 'implementing' | 'verifying' | 'done' | 'blocked' | 'limit';
39
68
  export interface RunOptions {
40
69
  task: string;
@@ -77,6 +106,9 @@ export interface RunOptions {
77
106
  maxRepairAttempts?: number;
78
107
  timeoutMs?: number;
79
108
  requireVerify?: boolean;
109
+ /** Verification mode (default 'strict'). 'off' skips the pipeline;
110
+ * 'advisory' runs verify once on completion without repair turns. */
111
+ mode?: VerifyMode;
80
112
  };
81
113
  /**
82
114
  * Level 9 — persistence. When a SessionStore is provided, every message
@@ -99,6 +131,8 @@ export interface RunOptions {
99
131
  depth: number;
100
132
  maxDepth: number;
101
133
  allowedTools?: ReadonlySet<string>;
134
+ /** Path allow-set inherited from the parent (sibling C contract). */
135
+ allowedPaths?: readonly string[];
102
136
  model?: string;
103
137
  };
104
138
  /**
@@ -160,6 +194,8 @@ export type RuntimeEvent = {
160
194
  input: number;
161
195
  output: number;
162
196
  estimated?: boolean;
197
+ cacheRead?: number;
198
+ cacheWrite?: number;
163
199
  } | {
164
200
  kind: 'final_text';
165
201
  text: string;
@@ -203,6 +239,8 @@ export interface RunResult {
203
239
  input: number;
204
240
  output: number;
205
241
  estimated?: boolean;
242
+ cacheRead?: number;
243
+ cacheWrite?: number;
206
244
  };
207
245
  /** Number of policy-driven user prompts the user accepted. */
208
246
  repairs?: number;
@@ -212,6 +250,10 @@ export interface RunResult {
212
250
  command?: string;
213
251
  attempts: number;
214
252
  failureType?: string;
253
+ repairTokens?: {
254
+ input: number;
255
+ output: number;
256
+ };
215
257
  };
216
258
  /** 5.1 phase */
217
259
  phase?: Phase;
@@ -228,4 +270,4 @@ export declare function run(opts: RunOptions, deps: RuntimeDeps): Promise<RunRes
228
270
  export declare function defaultSystemPrompt(ctx: {
229
271
  cwd: string;
230
272
  telemetry?: string;
231
- }): string;
273
+ }): SystemPromptResult;
@@ -23,10 +23,16 @@ import { verify, diagnosticForModel } from '../verification/engine.js';
23
23
  import { detectVerifyCommand } from '../verification/auto.js';
24
24
  import { ensureBaseline, getBaseline } from '../verification/baseline.js';
25
25
  import { compressTranscript, totalTokens } from '../context/tokenizer.js';
26
+ import { ratesFor } from '../providers/model-info.js';
26
27
  import { classifyFailure, rerunOnce, gatherRepairContext, guardRepair } from '../verification/classify.js';
27
28
  import { findRelatedTests, buildScopedCommand, runScopedVerify, syntaxCheck, checkImports } from '../verification/scoped.js';
28
29
  import { globalBus } from '../events/bus.js';
29
30
  import { TraceWriter } from '../trace/writer.js';
31
+ /** Normalize either systemPrompt shape into {system, suffix}. */
32
+ export function resolveSystemPrompt(fn, ctx) {
33
+ const r = fn(ctx);
34
+ return typeof r === 'string' ? { system: r } : { system: r.system, suffix: r.suffix };
35
+ }
30
36
  const DEFAULT_MAX_STEPS = 30;
31
37
  /** Convert a registry of tools into ToolDefinitions for the provider. */
32
38
  export function toolDefinitions(registry) {
@@ -36,19 +42,16 @@ export function toolDefinitions(registry) {
36
42
  inputSchema: t.function.parameters,
37
43
  }));
38
44
  }
39
- // BUG-005: Model-aware cost estimation with sensible defaults.
40
- // Rates are per-1K tokens (input / output). Local models are $0.
41
- const MODEL_RATES = [
42
- { test: (m) => /gpt-4/i.test(m), input: 0.003, output: 0.015 },
43
- { test: (m) => /gpt-3\.5/i.test(m), input: 0.0005, output: 0.0015 },
44
- { test: (m) => /claude|anthropic/i.test(m), input: 0.003, output: 0.015 },
45
- { test: (m) => /gemini/i.test(m), input: 0.00075, output: 0.003 },
46
- { test: (m) => /o1/i.test(m), input: 0.015, output: 0.06 },
47
- ];
45
+ // BUG-005: Model-aware cost estimation, single-sourced from the
46
+ // providers/model-info.ts rate table (local/unknown models are $0).
47
+ // Cost is computed on input/output ONLY: cacheRead/cacheWrite are tracked
48
+ // for observability but excluded because cached tokens bill at
49
+ // provider-specific discounted rates we don't model — charging them at
50
+ // full input rates would overstate spend, silently dropping them
51
+ // understates it, so we keep them visible and out of the math.
48
52
  /** Estimate USD cost of a usage block given the model name. */
49
53
  export function estimateCost(model, usage) {
50
- const match = MODEL_RATES.find((r) => r.test(model));
51
- const { input: inRate, output: outRate } = match ?? { input: 0.003, output: 0.015 };
54
+ const { input: inRate, output: outRate } = ratesFor(model);
52
55
  return (usage.input / 1000) * inRate + (usage.output / 1000) * outRate;
53
56
  }
54
57
  // PERF-002: Memoized token counting cache.
@@ -93,6 +96,27 @@ export async function run(opts, deps) {
93
96
  let repairs = 0;
94
97
  let verificationAttempts = 0;
95
98
  let hasEdits = false;
99
+ // Overflow recovery: at most one compress-and-retry per run.
100
+ let overflowRetried = false;
101
+ // Repair ledger + guard: set from the first verification failure until a
102
+ // verification passes. repairUsageStart snapshots usage at failure time so
103
+ // every verification payload can attribute its repairTokens delta.
104
+ let inRepairTurn = false;
105
+ let repairUsageStart;
106
+ const verifyMode = opts.verify?.mode ?? 'strict';
107
+ const markRepairStarted = () => {
108
+ if (!repairUsageStart)
109
+ repairUsageStart = { input: usage.input, output: usage.output };
110
+ inRepairTurn = true;
111
+ };
112
+ const repairTokensNow = () => repairUsageStart
113
+ ? { input: usage.input - repairUsageStart.input, output: usage.output - repairUsageStart.output }
114
+ : undefined;
115
+ /** Attach repairTokens to a verification payload when a repair ledger exists. */
116
+ const withRepairTokens = (v) => {
117
+ const rt = repairTokensNow();
118
+ return rt ? { ...v, repairTokens: rt } : v;
119
+ };
96
120
  const emit = opts.onEvent;
97
121
  const telemetry = new RuntimeTelemetry();
98
122
  telemetry.setMaxSteps(maxSteps);
@@ -200,7 +224,7 @@ export async function run(opts, deps) {
200
224
  catch { /* ignore */ }
201
225
  }
202
226
  await closeTracer();
203
- return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? { ok: false, attempts: verificationAttempts } : undefined, phase: 'blocked' };
227
+ return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined, phase: 'blocked' };
204
228
  }
205
229
  steps++;
206
230
  // 5.1 phase transitions (model-narrated)
@@ -216,13 +240,21 @@ export async function run(opts, deps) {
216
240
  setPhase('verifying');
217
241
  emit?.({ kind: 'step_start', step: steps });
218
242
  telemetry.recordStepStart(steps);
219
- const systemPrompt = deps.systemPrompt({ cwd: opts.cwd, telemetry: steps === 1 ? emptyTelemetryBlock() : telemetry.format() });
243
+ const { system: stableSystem, suffix: telemetrySuffix } = resolveSystemPrompt(deps.systemPrompt, {
244
+ cwd: opts.cwd,
245
+ telemetry: steps === 1 ? emptyTelemetryBlock() : telemetry.format(),
246
+ });
247
+ // Budget accounting sees what the model sees (prefix + suffix); the
248
+ // request itself keeps the halves split for cache-friendly adapters.
249
+ const systemForBudget = telemetrySuffix ? `${stableSystem}\n\n${telemetrySuffix}` : stableSystem;
220
250
  const BUDGET = { total: 120_000, reservedOutput: 4000 };
221
251
  let reqMessages = transcript;
222
- let reqSystem = systemPrompt;
223
- if (cachedTotalTokens(systemPrompt, transcript) > BUDGET.total) {
224
- const c = compressTranscript(systemPrompt, transcript, BUDGET);
252
+ let reqSystem = stableSystem;
253
+ let reqSuffix = telemetrySuffix;
254
+ if (cachedTotalTokens(systemForBudget, transcript) > BUDGET.total) {
255
+ const c = compressTranscript(systemForBudget, transcript, BUDGET);
225
256
  reqSystem = c.system;
257
+ reqSuffix = undefined; // telemetry is regenerable — drop it under pressure
226
258
  reqMessages = c.messages;
227
259
  tokenCache = { lastRef: null, lastSystem: undefined, lastCount: 0 };
228
260
  if (c.dropped > 0)
@@ -231,6 +263,7 @@ export async function run(opts, deps) {
231
263
  const req = {
232
264
  model: opts.parentContext?.model ?? opts.model,
233
265
  system: reqSystem,
266
+ ...(reqSuffix ? { systemSuffix: reqSuffix } : {}),
234
267
  messages: reqMessages,
235
268
  tools: toolDefinitions(deps.registry),
236
269
  ...(opts.maxTokens ? { maxTokens: opts.maxTokens } : {}),
@@ -244,6 +277,8 @@ export async function run(opts, deps) {
244
277
  let thinkingBuf = '';
245
278
  const pendingToolCalls = new Map();
246
279
  let lastFinishReason;
280
+ // Set when this step's request must be re-issued after overflow recovery.
281
+ let overflowRetryPending = false;
247
282
  for await (const ev of events) {
248
283
  if (opts.signal?.aborted)
249
284
  break outer;
@@ -273,14 +308,22 @@ export async function run(opts, deps) {
273
308
  if (ev.usage) {
274
309
  usage.input += ev.usage.input;
275
310
  usage.output += ev.usage.output;
311
+ if (ev.usage.cacheRead !== undefined)
312
+ usage.cacheRead = (usage.cacheRead ?? 0) + ev.usage.cacheRead;
313
+ if (ev.usage.cacheWrite !== undefined)
314
+ usage.cacheWrite = (usage.cacheWrite ?? 0) + ev.usage.cacheWrite;
276
315
  telemetry.recordUsage(ev.usage.input, ev.usage.output);
277
- emit?.({ kind: 'usage', input: usage.input, output: usage.output });
316
+ emit?.({
317
+ kind: 'usage', input: usage.input, output: usage.output,
318
+ ...(usage.cacheRead !== undefined ? { cacheRead: usage.cacheRead } : {}),
319
+ ...(usage.cacheWrite !== undefined ? { cacheWrite: usage.cacheWrite } : {}),
320
+ });
278
321
  }
279
322
  else {
280
323
  // Providers that omit usage (Ollama, vLLM, proxies): estimate from
281
324
  // the actual request + generated output so cost accounting never
282
325
  // silently records zero. Marked estimated for the UI/debugging.
283
- const est = estimateTurnUsage(reqSystem, reqMessages, textBuf, pendingToolCalls);
326
+ const est = estimateTurnUsage(systemForBudget, reqMessages, textBuf, pendingToolCalls);
284
327
  usage.input += est.input;
285
328
  usage.output += est.output;
286
329
  usage.estimated = true;
@@ -289,6 +332,31 @@ export async function run(opts, deps) {
289
332
  }
290
333
  }
291
334
  else if (ev.kind === 'error') {
335
+ // Overflow recovery: on the first REQUEST_TOO_LARGE of a run,
336
+ // aggressively compact the transcript and re-issue the request once.
337
+ // A second occurrence fails normally (returned as no_final below).
338
+ if (ev.code === 'REQUEST_TOO_LARGE' && !overflowRetried) {
339
+ overflowRetried = true;
340
+ overflowRetryPending = true;
341
+ telemetry.recordError('overflow_retry');
342
+ emitKlyro({ type: 'error', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', code: 'REQUEST_TOO_LARGE', message: 'context overflow — aggressively compacting transcript and retrying the request once' });
343
+ try {
344
+ const compacted = compressTranscript(reqSystem, transcript, { total: 30_000, reservedOutput: 4000 });
345
+ if (compacted.messages.length < transcript.length || compacted.dropped > 0) {
346
+ transcript.splice(0, transcript.length, ...compacted.messages);
347
+ }
348
+ else {
349
+ // Transcript already fits the aggressive budget — force it
350
+ // strictly smaller so the retry cannot repeat the overflow.
351
+ const halved = Math.max(4000, Math.floor(totalTokens(reqSystem, transcript) / 2));
352
+ const smaller = compressTranscript(reqSystem, transcript, { total: halved, reservedOutput: 4000 });
353
+ transcript.splice(0, transcript.length, ...smaller.messages);
354
+ }
355
+ tokenCache = { lastRef: null, lastSystem: undefined, lastCount: 0 };
356
+ }
357
+ catch { /* ignore — retry with the transcript as-is */ }
358
+ break;
359
+ }
292
360
  telemetry.recordError(`stream_error: ${ev.code}`);
293
361
  if (store && sessionId) {
294
362
  try {
@@ -306,10 +374,18 @@ export async function run(opts, deps) {
306
374
  hasEdits,
307
375
  usage,
308
376
  repairs,
309
- verification: hasEdits ? { ok: false, attempts: verificationAttempts } : undefined,
377
+ verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined,
310
378
  };
311
379
  }
312
380
  }
381
+ // Overflow recovery lands here via `break`: re-issue the same step
382
+ // without consuming the step budget.
383
+ if (overflowRetryPending) {
384
+ overflowRetryPending = false;
385
+ steps--;
386
+ emit?.({ kind: 'step_end', step: steps + 1 });
387
+ continue outer;
388
+ }
313
389
  // Build the assistant message. Tool calls are finalized here: JSON is
314
390
  // parsed and schema-validated BEFORE policy/execution. Malformed calls
315
391
  // become structured MALFORMED_TOOL_CALL results — garbage arguments must
@@ -368,7 +444,7 @@ export async function run(opts, deps) {
368
444
  catch { /* ignore */ }
369
445
  }
370
446
  await closeTracer();
371
- return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? { ok: false, attempts: verificationAttempts } : undefined };
447
+ return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined };
372
448
  }
373
449
  if (finalizedCalls.length === 0) {
374
450
  if (hadInvalidTool) {
@@ -380,6 +456,66 @@ export async function run(opts, deps) {
380
456
  continue;
381
457
  }
382
458
  finalText = textBuf;
459
+ // Verify modes: 'off' skips the pipeline entirely (verification stays
460
+ // undefined, no repair turns). 'advisory' runs verify once on
461
+ // completion and reports the outcome without repair turns. 'strict'
462
+ // (default) runs the full pipeline with autonomous repair below.
463
+ if (verifyMode === 'off') {
464
+ emit?.({ kind: 'final_text', text: finalText });
465
+ emit?.({ kind: 'step_end', step: steps });
466
+ if (store && sessionId) {
467
+ try {
468
+ await store.setStatus(sessionId, 'complete', finalText);
469
+ }
470
+ catch { /* ignore */ }
471
+ }
472
+ await closeTracer();
473
+ return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: undefined };
474
+ }
475
+ if (verifyMode === 'advisory') {
476
+ const advisoryCmd = opts.verify?.command ?? detectVerifyCommand(opts.cwd);
477
+ if (opts.verify?.enabled !== false && hasEdits && advisoryCmd) {
478
+ emit?.({ kind: 'verification_started', command: advisoryCmd });
479
+ let advisoryResult;
480
+ try {
481
+ advisoryResult = await verify({ cwd: opts.cwd, command: advisoryCmd, timeoutMs: opts.verify?.timeoutMs });
482
+ }
483
+ catch (e) {
484
+ const msg = e instanceof Error ? e.message : String(e);
485
+ advisoryResult = { ok: false, exitCode: -1, stdout: '', stderr: msg, failure: { type: 'unknown', files: [], raw: msg, exitCode: -1 } };
486
+ }
487
+ verificationAttempts++;
488
+ if (advisoryResult.ok) {
489
+ inRepairTurn = false;
490
+ emit?.({ kind: 'verification_succeeded', command: advisoryCmd });
491
+ emit?.({ kind: 'final_text', text: finalText });
492
+ emit?.({ kind: 'step_end', step: steps });
493
+ if (store && sessionId) {
494
+ try {
495
+ await store.setStatus(sessionId, 'complete', finalText);
496
+ }
497
+ catch { /* ignore */ }
498
+ }
499
+ await closeTracer();
500
+ return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: withRepairTokens({ ok: true, command: advisoryCmd, attempts: verificationAttempts }) };
501
+ }
502
+ const advisoryFailure = advisoryResult.failure?.type ?? 'unknown';
503
+ markRepairStarted();
504
+ emit?.({ kind: 'verification_failed', step: String(steps), reason: diagnosticForModel(advisoryResult).slice(0, 800) });
505
+ emit?.({ kind: 'final_text', text: finalText });
506
+ emit?.({ kind: 'step_end', step: steps });
507
+ if (store && sessionId) {
508
+ try {
509
+ await store.setStatus(sessionId, 'complete', finalText);
510
+ }
511
+ catch { /* ignore */ }
512
+ }
513
+ await closeTracer();
514
+ return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: withRepairTokens({ ok: false, command: advisoryCmd, attempts: verificationAttempts, failureType: advisoryFailure }) };
515
+ }
516
+ // No verify applicable (disabled, no edits, or no command) — fall
517
+ // through to the normal completion return below.
518
+ }
383
519
  // Level 8 — Verification + Autonomous Repair (gated on hasEdits below — pure analysis skips verify)
384
520
  const verifyEnabled = opts.verify?.enabled !== false;
385
521
  const verifyCmd = opts.verify?.command ?? detectVerifyCommand(opts.cwd);
@@ -439,6 +575,7 @@ export async function run(opts, deps) {
439
575
  catch { /* scoped success is enough */ }
440
576
  }
441
577
  if (vResult.ok) {
578
+ inRepairTurn = false;
442
579
  emit?.({ kind: 'verification_succeeded', command: verifyCmd });
443
580
  emit?.({ kind: 'final_text', text: finalText });
444
581
  emit?.({ kind: 'step_end', step: steps });
@@ -449,7 +586,7 @@ export async function run(opts, deps) {
449
586
  catch { /* ignore */ }
450
587
  }
451
588
  await closeTracer();
452
- return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: { ok: true, command: verifyCmd, attempts: verificationAttempts } };
589
+ return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: withRepairTokens({ ok: true, command: verifyCmd, attempts: verificationAttempts }) };
453
590
  }
454
591
  // 6.4 — classify
455
592
  const baseline = await getBaseline(opts.cwd, verifyCmd);
@@ -457,6 +594,7 @@ export async function run(opts, deps) {
457
594
  const cls = classifyFailure({ failure: vResult.failure, stdout: vResult.stdout, stderr: vResult.stderr }, baseline, isFlaky);
458
595
  if (cls === 'flaky') {
459
596
  // rerun succeeded on second try — treat as flaky, don't count as repair
597
+ inRepairTurn = false;
460
598
  emit?.({ kind: 'verification_succeeded', command: verifyCmd });
461
599
  emit?.({ kind: 'final_text', text: finalText });
462
600
  emit?.({ kind: 'step_end', step: steps });
@@ -467,7 +605,7 @@ export async function run(opts, deps) {
467
605
  catch { /* ignore */ }
468
606
  }
469
607
  await closeTracer();
470
- return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: { ok: true, command: verifyCmd, attempts: verificationAttempts } };
608
+ return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: withRepairTokens({ ok: true, command: verifyCmd, attempts: verificationAttempts }) };
471
609
  }
472
610
  if (cls === 'env') {
473
611
  // don't try to repair env failures with code edits
@@ -475,10 +613,11 @@ export async function run(opts, deps) {
475
613
  transcript.push(envMsg);
476
614
  await checkpoint(envMsg);
477
615
  verificationAttempts++;
616
+ markRepairStarted();
478
617
  emit?.({ kind: 'verification_failed', step: String(steps), reason: `env: ${diagnosticForModel(vResult).slice(0, 600)}` });
479
618
  if (verificationAttempts >= maxRepairs) {
480
619
  await closeTracer();
481
- return { status: 'verify_failed', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: { ok: false, command: verifyCmd, attempts: verificationAttempts, failureType: 'env' } };
620
+ return { status: 'verify_failed', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: withRepairTokens({ ok: false, command: verifyCmd, attempts: verificationAttempts, failureType: 'env' }) };
482
621
  }
483
622
  emit?.({ kind: 'step_end', step: steps });
484
623
  continue;
@@ -494,6 +633,7 @@ export async function run(opts, deps) {
494
633
  const failurePaths = new Set(vResult.failure?.files.map((f) => f.path).filter(Boolean) ?? []);
495
634
  const overlaps = [...failurePaths].some((p) => introducedPaths.has(p) || introducedPaths.has(path.basename(p)));
496
635
  if (!overlaps) {
636
+ inRepairTurn = false;
497
637
  emit?.({ kind: 'verification_succeeded', command: verifyCmd });
498
638
  emit?.({ kind: 'final_text', text: finalText });
499
639
  emit?.({ kind: 'step_end', step: steps });
@@ -504,11 +644,12 @@ export async function run(opts, deps) {
504
644
  catch { /* ignore */ }
505
645
  }
506
646
  await closeTracer();
507
- return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: { ok: true, command: verifyCmd, attempts: verificationAttempts } };
647
+ return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: withRepairTokens({ ok: true, command: verifyCmd, attempts: verificationAttempts }) };
508
648
  }
509
649
  }
510
650
  // Failure → repair loop (introduced)
511
651
  verificationAttempts++;
652
+ markRepairStarted();
512
653
  const diagnostic = diagnosticForModel(vResult);
513
654
  const failureType = vResult.failure?.type ?? 'unknown';
514
655
  // 6.4 — gather context
@@ -548,7 +689,7 @@ export async function run(opts, deps) {
548
689
  emit?.({ kind: 'final_text', text: finalText });
549
690
  emit?.({ kind: 'step_end', step: steps });
550
691
  await closeTracer();
551
- return { status: 'verify_failed', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: { ok: false, command: verifyCmd, attempts: verificationAttempts, failureType } };
692
+ return { status: 'verify_failed', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: withRepairTokens({ ok: false, command: verifyCmd, attempts: verificationAttempts, failureType }) };
552
693
  }
553
694
  emit?.({ kind: 'step_end', step: steps });
554
695
  continue; // -> next iteration lets model repair
@@ -562,11 +703,10 @@ export async function run(opts, deps) {
562
703
  catch { /* ignore */ }
563
704
  }
564
705
  await closeTracer();
565
- return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? { ok: true, attempts: verificationAttempts } : undefined };
706
+ return { status: 'complete', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: true, attempts: verificationAttempts }) : undefined };
566
707
  }
567
708
  // 3.1 — emit turn events
568
709
  emitKlyro({ type: 'turn.start', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', turn: steps, model: opts.model });
569
- // Execute each tool call (after policy) — 3.5 parallel if all concurrencySafe
570
710
  const toolCtx = {
571
711
  cwd: opts.cwd,
572
712
  env: process.env,
@@ -580,7 +720,9 @@ export async function run(opts, deps) {
580
720
  agentDepth: opts.parentContext?.depth ?? 0,
581
721
  agentMaxDepth: opts.parentContext?.maxDepth ?? 1,
582
722
  ...(opts.parentContext?.allowedTools ? { agentAllowedTools: opts.parentContext.allowedTools } : {}),
723
+ ...(opts.parentContext?.allowedPaths ? { agentAllowedPaths: opts.parentContext.allowedPaths } : {}),
583
724
  ...(opts.parentContext?.model ?? opts.model ? { agentModel: opts.parentContext?.model ?? opts.model } : {}),
725
+ repairGuard: { denyTestEdits: inRepairTurn },
584
726
  };
585
727
  const allSafe = finalizedCalls.length > 1 && finalizedCalls.every((c) => deps.registry.get(c.name)?.isConcurrencySafe !== false);
586
728
  // Gate phase: policy decision + approval prompt for one call. Runs
@@ -591,7 +733,12 @@ export async function run(opts, deps) {
591
733
  const decision = await deps.policy.evaluate({ name: call.name, input: call.input }, { cwd: opts.cwd, nonInteractive: opts.nonInteractive });
592
734
  emit?.({ kind: 'policy_decision', id: call.id, name: call.name, action: decision.action, ...(decision.action !== 'allow' ? { reason: decision.reason } : {}) });
593
735
  // Mirror to KlyroEvent bus
594
- emitKlyro({ type: 'permission.decision', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, action: decision.action, ...(decision.action !== 'allow' ? { reason: decision.reason } : {}) });
736
+ if (decision.action === 'allow') {
737
+ emitKlyro({ type: 'permission.decision', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, action: 'allow' });
738
+ }
739
+ else {
740
+ emitKlyro({ type: 'permission.decision', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, action: decision.action, reason: decision.reason });
741
+ }
595
742
  if (decision.action === 'deny') {
596
743
  const denyMsg = {
597
744
  role: 'tool',
@@ -770,6 +917,19 @@ export async function run(opts, deps) {
770
917
  break;
771
918
  }
772
919
  }
920
+ // P0 — drain finished sub-agent completions into parent visibility.
921
+ // drainCompletions is OPTIONAL on the bridge — guarded with `?.` so
922
+ // older bridges without it simply yield nothing.
923
+ try {
924
+ const completions = opts.agentBridge?.drainCompletions?.() ?? [];
925
+ for (const c of completions) {
926
+ const body = c.finalText ?? c.error?.message ?? '';
927
+ const msg = { role: 'user', content: [text(`Subtask ${c.taskId} (${c.agentName}) ${c.status}: ${body.slice(0, 500)}`)] };
928
+ transcript.push(msg);
929
+ await checkpoint(msg);
930
+ }
931
+ }
932
+ catch { /* ignore — completions are best-effort visibility */ }
773
933
  emit?.({ kind: 'step_end', step: steps });
774
934
  emitKlyro({ type: 'turn.end', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', turn: steps });
775
935
  // Level 9 — checkpoint status after each step
@@ -792,7 +952,7 @@ export async function run(opts, deps) {
792
952
  catch { /* ignore */ }
793
953
  }
794
954
  await closeTracer();
795
- return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? { ok: false, attempts: verificationAttempts } : undefined };
955
+ return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined };
796
956
  }
797
957
  // 5.2 — stuck termination (P0-3): bounded stop instead of unbounded token burn.
798
958
  if (stuckAbort) {
@@ -804,7 +964,7 @@ export async function run(opts, deps) {
804
964
  catch { /* ignore */ }
805
965
  }
806
966
  await closeTracer();
807
- return { status: 'stuck', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? { ok: false, attempts: verificationAttempts } : undefined };
967
+ return { status: 'stuck', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined };
808
968
  }
809
969
  emit?.({ kind: 'final_text', text: finalText });
810
970
  if (store && sessionId) {
@@ -814,7 +974,7 @@ export async function run(opts, deps) {
814
974
  catch { /* ignore */ }
815
975
  }
816
976
  await closeTracer();
817
- return { status: 'max_steps', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? { ok: false, attempts: verificationAttempts } : undefined };
977
+ return { status: 'max_steps', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined };
818
978
  }
819
979
  /**
820
980
  * Estimate usage for a turn whose provider omitted it (Ollama, vLLM,
@@ -950,7 +1110,10 @@ export function defaultSystemPrompt(ctx) {
950
1110
  'Do not invent file paths. Do not call tools outside the working directory.',
951
1111
  ].join(' ');
952
1112
  if (ctx.telemetry) {
953
- return base + '\n\n' + ctx.telemetry + '\n\nUse the telemetry above to avoid repeating the same failing call and to keep within the step budget.';
1113
+ return {
1114
+ system: base,
1115
+ suffix: ctx.telemetry + '\n\nUse the telemetry above to avoid repeating the same failing call and to keep within the step budget.',
1116
+ };
954
1117
  }
955
- return base;
1118
+ return { system: base };
956
1119
  }
@@ -0,0 +1,74 @@
1
+ /**
2
+ * Git worktree isolation for parallel write-capable child agents.
3
+ *
4
+ * Each write-capable child gets its own worktree on a branch
5
+ * `klyro/<taskId>` so concurrent children never clobber each other's files.
6
+ * `task_apply` merges the branch back into the parent tree; failures,
7
+ * cancellations, and timeouts remove the worktree best-effort.
8
+ *
9
+ * Semantics worth knowing:
10
+ * - Dirty parent: the merge runs in the parent checkout, so uncommitted
11
+ * parent edits can conflict. On conflict the merge is ABORTED (no
12
+ * commit), the branch is kept, and the caller gets MERGE_CONFLICT with
13
+ * the file list — resolve and re-apply.
14
+ * - Stale branch: a leftover `klyro/<taskId>` branch (e.g. from a crashed
15
+ * session) makes `createWorktree` fail. Delete the stale branch
16
+ * (`deleteBranch`) or sweep orphaned worktrees (`pruneStaleWorktrees`)
17
+ * before retrying.
18
+ * - No-apply: tasks that never reach `succeeded` are NEVER merged — their
19
+ * worktrees are removed best-effort and `task_apply` reports NOT_READY.
20
+ *
21
+ * All git invocations go through `execFile` argv (no shell) with a ~30s
22
+ * timeout, so this is safe on Windows.
23
+ */
24
+ /** Per-command timeout for every git invocation. */
25
+ export declare const GIT_TIMEOUT_MS = 30000;
26
+ export interface WorktreeInfo {
27
+ worktreePath: string;
28
+ branch: string;
29
+ }
30
+ export interface MergeResult {
31
+ merged: boolean;
32
+ conflictFiles: string[];
33
+ }
34
+ /** True when `cwd` is inside a git working tree. Never throws — false on any failure. */
35
+ export declare function ensureGitRepo(cwd: string): Promise<boolean>;
36
+ /** Resolve the repo top-level for any path inside it. Throws when not in a repo. */
37
+ export declare function repoTopLevel(cwd: string): Promise<string>;
38
+ /**
39
+ * Create an isolated worktree for a task on branch `klyro/<taskId>`,
40
+ * rooted at `<repo>/.klyro/worktrees/<taskId>` from HEAD.
41
+ */
42
+ export declare function createWorktree(opts: {
43
+ repoCwd: string;
44
+ taskId: string;
45
+ }): Promise<WorktreeInfo>;
46
+ /**
47
+ * Merge a task branch into the current checkout (`repoCwd`) with
48
+ * `git merge --no-ff --no-edit`. On conflict, parses `git status
49
+ * --porcelain` for unmerged paths (UU/AA/DD), aborts the merge WITHOUT
50
+ * committing, and returns `{ merged: false, conflictFiles }`.
51
+ */
52
+ export declare function mergeWorktree(opts: {
53
+ repoCwd: string;
54
+ branch: string;
55
+ }): Promise<MergeResult>;
56
+ /** Remove a worktree and prune its metadata. Throws with context on failure. */
57
+ export declare function removeWorktree(opts: {
58
+ repoCwd: string;
59
+ worktreePath: string;
60
+ force?: boolean;
61
+ }): Promise<void>;
62
+ /** Delete a task branch (best-effort cleanup after a successful merge). Never throws. */
63
+ export declare function deleteBranch(opts: {
64
+ repoCwd: string;
65
+ branch: string;
66
+ }): Promise<void>;
67
+ /**
68
+ * Remove worktrees under `.klyro/worktrees` whose task id is NOT in
69
+ * `activeTaskIds` (orphans from crashed/timed-out sessions). Lists
70
+ * `git worktree list --porcelain`, force-removes each stale entry, and
71
+ * returns the removed paths. Never throws — per-worktree failures are
72
+ * skipped and a failing `git` itself yields `[]`.
73
+ */
74
+ export declare function pruneStaleWorktrees(repoCwd: string, activeTaskIds: readonly string[]): Promise<string[]>;