klyro 1.0.1 → 1.0.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. package/dist/agent/anthropic-adapter.d.ts +13 -5
  2. package/dist/agent/anthropic-adapter.js +19 -2
  3. package/dist/agent/capabilities.js +7 -1
  4. package/dist/agent/orchestrator.d.ts +44 -4
  5. package/dist/agent/orchestrator.js +102 -8
  6. package/dist/agent/retry.js +52 -10
  7. package/dist/agent/runtime.d.ts +34 -0
  8. package/dist/agent/runtime.js +169 -15
  9. package/dist/agent/stream-budget.d.ts +36 -0
  10. package/dist/agent/stream-budget.js +121 -0
  11. package/dist/checkpoints/store.d.ts +9 -0
  12. package/dist/checkpoints/store.js +26 -0
  13. package/dist/cli/commit.d.ts +31 -0
  14. package/dist/cli/commit.js +142 -0
  15. package/dist/cli/config.d.ts +45 -0
  16. package/dist/cli/config.js +82 -0
  17. package/dist/cli/doctor.d.ts +1 -0
  18. package/dist/cli/doctor.js +71 -6
  19. package/dist/cli/hooks.d.ts +47 -0
  20. package/dist/cli/hooks.js +181 -0
  21. package/dist/cli/markdown.js +29 -1
  22. package/dist/cli/repl.js +41 -1
  23. package/dist/cli/run.d.ts +6 -0
  24. package/dist/cli/run.js +76 -3
  25. package/dist/context/tokenizer.d.ts +18 -0
  26. package/dist/context/tokenizer.js +50 -1
  27. package/dist/events/bus.d.ts +5 -0
  28. package/dist/events/bus.js +10 -0
  29. package/dist/events/catalog.d.ts +9 -0
  30. package/dist/events/catalog.js +9 -0
  31. package/dist/index.js +89 -5
  32. package/dist/mcp/client.js +1 -1
  33. package/dist/mcp/registry.d.ts +0 -18
  34. package/dist/mcp/registry.js +49 -2
  35. package/dist/policy/engine.d.ts +16 -0
  36. package/dist/policy/engine.js +74 -1
  37. package/dist/policy/path-guard.d.ts +24 -0
  38. package/dist/policy/path-guard.js +46 -0
  39. package/dist/providers/model-info.d.ts +6 -0
  40. package/dist/providers/model-info.js +8 -0
  41. package/dist/tools/fs/apply-patch.js +6 -1
  42. package/dist/tools/fs/edit-file.js +4 -1
  43. package/dist/tools/fs/multi-edit.js +4 -1
  44. package/dist/tools/fs/write-file.js +16 -6
  45. package/dist/tools/plan/todo-write.js +1 -1
  46. package/dist/tools/shell/sandbox.d.ts +4 -3
  47. package/dist/tools/shell/sandbox.js +23 -3
  48. package/dist/tools/shell/shell-exec.d.ts +28 -0
  49. package/dist/tools/shell/shell-exec.js +87 -1
  50. package/dist/trace/writer.d.ts +7 -0
  51. package/dist/trace/writer.js +7 -0
  52. package/dist/tui/app.js +20 -1
  53. package/dist/tui/app.test.js +18 -12
  54. package/dist/tui/approval.test.js +20 -3
  55. package/dist/tui/markdown.d.ts +13 -0
  56. package/dist/tui/markdown.js +169 -2
  57. package/dist/tui/scroll-flow.test.js +3 -1
  58. package/dist/verification/classify.js +4 -3
  59. package/dist/verification/engine.d.ts +8 -0
  60. package/dist/verification/engine.js +25 -0
  61. package/dist/verification/registry.js +16 -5
  62. package/dist/verification/scoped.js +36 -5
  63. package/package.json +1 -1
package/dist/cli/repl.js CHANGED
@@ -185,6 +185,27 @@ export async function startRepl(opts = {}) {
185
185
  },
186
186
  });
187
187
  let adapter = buildAdapter(currentProvider, currentBaseUrl, currentApiKey);
188
+ // L15 failover — initial adapter only: extra `providers.failover` entries
189
+ // become fallback adapters for agent runs in this session. Live
190
+ // /provider switches rebuild a single adapter (buildAdapter) and bypass
191
+ // the chain — documented, not a bug.
192
+ let replFailoverAdapters = [];
193
+ try {
194
+ const { resolveProviderChain } = await import('./config.js');
195
+ const chain = await resolveProviderChain(cwd);
196
+ for (const entry of chain.slice(1)) {
197
+ if (!entry.apiKey)
198
+ continue;
199
+ try {
200
+ replFailoverAdapters.push(buildAdapter(entry.provider, entry.baseURL ?? currentBaseUrl, entry.apiKey));
201
+ }
202
+ catch { /* skip unbuildable entries */ }
203
+ }
204
+ if (replFailoverAdapters.length > 0) {
205
+ process.stderr.write(`klyro: failover chain: ${replFailoverAdapters.map((b) => b.id).join(' → ')}\n`);
206
+ }
207
+ }
208
+ catch { /* best-effort — single-provider session proceeds */ }
188
209
  const ctxBlock = await buildLevel6Context({ cwd });
189
210
  let ctxPrefix = ctxBlock.formatted ? `\n\n<context>\n${ctxBlock.formatted}\n</context>` : '';
190
211
  // 4.4 KLYRO.md hierarchy (mutable — /reload refreshes). Content is gated
@@ -646,6 +667,19 @@ export async function startRepl(opts = {}) {
646
667
  if (isMounted && directHooks)
647
668
  directHooks.clearTranscript();
648
669
  }
670
+ // Update nudge (best-effort, cached 24h, silent fail): a single stderr
671
+ // line when a newer version exists. No prompt, no blocking — the check
672
+ // itself is fire-and-forget.
673
+ void (async () => {
674
+ try {
675
+ const { checkForUpdate } = await import('./update.js');
676
+ const cur = readVersion();
677
+ const latest = await checkForUpdate(cur);
678
+ if (latest)
679
+ process.stderr.write(`Update available: ${cur} → ${latest} (klyro update)\n`);
680
+ }
681
+ catch { /* silent */ }
682
+ })();
649
683
  app = render(React.createElement(App, {
650
684
  initialModel: model,
651
685
  maxSteps: currentMaxSteps,
@@ -897,8 +931,14 @@ export async function startRepl(opts = {}) {
897
931
  else if (ev.kind === 'aborted') {
898
932
  queuedStatus({ status: 'aborted' });
899
933
  }
934
+ else if (ev.kind === 'model_override') {
935
+ queuedAppend({ id: `movr-${Date.now()}`, kind: 'text', text: `[model] override: requested ${ev.requested} → effective ${ev.effective}`, role: 'assistant' });
936
+ }
900
937
  },
901
- }, { adapter, registry, policy, approval, systemPrompt: systemPromptFn });
938
+ }, {
939
+ adapter, registry, policy, approval, systemPrompt: systemPromptFn,
940
+ ...(replFailoverAdapters.length > 0 ? { failoverAdapters: replFailoverAdapters } : {}),
941
+ });
902
942
  if (result.finalText)
903
943
  lastAssistantText = result.finalText;
904
944
  if (result.verification) {
package/dist/cli/run.d.ts CHANGED
@@ -70,6 +70,12 @@ export interface RunCliOptions {
70
70
  agent?: string;
71
71
  maxDepth?: number;
72
72
  }
73
+ /**
74
+ * Double-Ctrl+C detector (pure, exported for tests): the second SIGINT
75
+ * within 1500ms of the first forces `process.exit(130)`. The live handler
76
+ * below owns the timestamp closure; tests exercise only this predicate.
77
+ */
78
+ export declare function shouldForceExit(lastSigintAt: number | undefined, now: number): boolean;
73
79
  export declare function runOnce(opts: RunCliOptions): Promise<number>;
74
80
  /** Wrap a system-prompt fn to inject Level-6 context (project map etc.) + KLYRO.md (4.4). */
75
81
  export declare function makeRunSystemPrompt(cwd: string, base: SystemPromptFn): Promise<SystemPromptFn>;
package/dist/cli/run.js CHANGED
@@ -16,6 +16,7 @@ import { run } from '../agent/runtime.js';
16
16
  import { builtinRegistry } from '../tools/registry.js';
17
17
  import { builtinRules, clonePolicyConfig, PolicyEngine } from '../policy/engine.js';
18
18
  import { DenyAllApprovalPrompt } from '../policy/approval.js';
19
+ import { redact } from '../policy/secret-redactor.js';
19
20
  import { buildLevel6Context } from '../context/level6.js';
20
21
  import { memoryBlock } from '../context/memory.js';
21
22
  import { resolveSessionId } from '../persistence/session.js';
@@ -24,6 +25,14 @@ function readEnv(name, fallback) {
24
25
  const v = process.env[name];
25
26
  return v && v.length > 0 ? v : fallback;
26
27
  }
28
+ /**
29
+ * Double-Ctrl+C detector (pure, exported for tests): the second SIGINT
30
+ * within 1500ms of the first forces `process.exit(130)`. The live handler
31
+ * below owns the timestamp closure; tests exercise only this predicate.
32
+ */
33
+ export function shouldForceExit(lastSigintAt, now) {
34
+ return lastSigintAt !== undefined && now - lastSigintAt < 1500;
35
+ }
27
36
  export async function runOnce(opts) {
28
37
  // P0.5 — load <cwd>/.env first so KLYRO_* vars resolve without `export`.
29
38
  // Never throws (missing file is a no-op); explicit env wins (no-clobber).
@@ -88,6 +97,36 @@ export async function runOnce(opts) {
88
97
  adapter = retryingAdapter(httpChatAdapter({ baseURL: baseUrl, apiKey, timeoutMs: opts.timeoutMs ?? 60_000 }), { onRetry: onRetryEmit });
89
98
  }
90
99
  }
100
+ // L15 provider failover: extra chain entries (after the primary) become
101
+ // fallback adapters for the runtime. Custom injected adapters (tests)
102
+ // skip chain wiring. Failures resolving the chain never block the run.
103
+ let failoverAdapters;
104
+ if (!opts.adapter) {
105
+ try {
106
+ const { resolveProviderChain } = await import('./config.js');
107
+ const chain = await resolveProviderChain(opts.cwd);
108
+ const fallbacks = chain.slice(1);
109
+ if (fallbacks.length > 0) {
110
+ const built = [];
111
+ for (const entry of fallbacks) {
112
+ if (!entry.apiKey)
113
+ continue;
114
+ const base = entry.provider === 'anthropic'
115
+ ? anthropicAdapter({ baseURL: entry.baseURL, apiKey: entry.apiKey, timeoutMs: opts.timeoutMs ?? 60_000 })
116
+ : entry.baseURL
117
+ ? httpChatAdapter({ baseURL: entry.baseURL, apiKey: entry.apiKey, timeoutMs: opts.timeoutMs ?? 60_000 })
118
+ : null;
119
+ if (base)
120
+ built.push(retryingAdapter(base, { onRetry: onRetryEmit }));
121
+ }
122
+ if (built.length > 0) {
123
+ failoverAdapters = built;
124
+ stderr.write(`klyro: failover chain: ${built.map((b) => b.id).join(' → ')}\n`);
125
+ }
126
+ }
127
+ }
128
+ catch { /* best-effort — single-provider run proceeds */ }
129
+ }
91
130
  const registry = builtinRegistry();
92
131
  const policy = new PolicyEngine(builtinRules(), clonePolicyConfig());
93
132
  // Persisted "always allow" patterns apply to one-shot runs too.
@@ -149,6 +188,16 @@ export async function runOnce(opts) {
149
188
  return 2;
150
189
  }
151
190
  sessionId = full;
191
+ // M22 resume-lock warning (read-only probe — takeover proceeds anyway).
192
+ try {
193
+ const { readSessionLock } = await import('../persistence/store.js');
194
+ const { getDefaultSessionsDir } = await import('../persistence/session.js');
195
+ const lock = readSessionLock(opts.sessionsDir ?? getDefaultSessionsDir(), sessionId);
196
+ if (lock.held && lock.alive) {
197
+ stderr.write(`klyro: session ${sessionId.slice(0, 8)} is locked by live pid ${lock.pid ?? '?'} — taking over (proceeding anyway)\n`);
198
+ }
199
+ }
200
+ catch { /* probe is best-effort; never block resume */ }
152
201
  const msgs = await store.loadMessages(sessionId);
153
202
  // Convert StoredMessage to Message
154
203
  initialTranscript = msgs.map((m) => ({ role: m.role, content: m.content }));
@@ -176,7 +225,17 @@ export async function runOnce(opts) {
176
225
  // persistence is disabled or the session was never created).
177
226
  sessionIdForRetry.id = sessionId ?? 'ephemeral';
178
227
  const ac = new AbortController();
228
+ // Double-Ctrl+C: first press aborts the run; a second press within
229
+ // 1500ms forces process.exit(130) (shouldForceExit owns the predicate).
230
+ let lastSigintAt;
179
231
  const onSigint = () => {
232
+ const now = Date.now();
233
+ if (shouldForceExit(lastSigintAt, now)) {
234
+ stderr.write('\nklyro: SIGINT twice — forcing exit\n');
235
+ process.exit(130);
236
+ return;
237
+ }
238
+ lastSigintAt = now;
180
239
  stderr.write('\nklyro: SIGINT — aborting\n');
181
240
  ac.abort();
182
241
  };
@@ -298,8 +357,20 @@ export async function runOnce(opts) {
298
357
  if (output === 'human')
299
358
  stderr.write(`[session ${ev.sessionId.slice(0, 8)} checkpoint]\n`);
300
359
  }
360
+ else if (ev.kind === 'provider_failover') {
361
+ stderr.write(`[failover] ${ev.from} → ${ev.to}: ${ev.reason.slice(0, 200)}\n`);
362
+ }
363
+ else if (ev.kind === 'budget_warning') {
364
+ stderr.write(`[budget] ${(ev.ratio * 100).toFixed(0)}% of max cost used (threshold ${(ev.threshold * 100).toFixed(0)}%)\n`);
365
+ }
366
+ else if (ev.kind === 'model_override') {
367
+ stderr.write(`[model] override: requested ${ev.requested} → effective ${ev.effective}\n`);
368
+ }
301
369
  },
302
- }, { adapter, registry, policy, approval: new DenyAllApprovalPrompt(), systemPrompt });
370
+ }, {
371
+ adapter, registry, policy, approval: new DenyAllApprovalPrompt(), systemPrompt,
372
+ ...(failoverAdapters ? { failoverAdapters } : {}),
373
+ });
303
374
  }
304
375
  finally {
305
376
  doneSigint();
@@ -404,8 +475,10 @@ async function dryRunReport(opts) {
404
475
  maxSteps: opts.maxSteps,
405
476
  maxTokens: opts.maxTokens,
406
477
  temperature: opts.temperature,
407
- systemPrompt,
408
- task: opts.task,
478
+ // Secrets must never leak into a printable report: redact both the
479
+ // assembled system prompt and the task before printing.
480
+ systemPrompt: redact(systemPrompt),
481
+ task: redact(opts.task),
409
482
  toolCount: registry.list().length,
410
483
  toolNames: registry.list().map((t) => t.name),
411
484
  policyRules: rules.map((r) => r.name),
@@ -7,6 +7,14 @@
7
7
  * never overflows the model's true window because the heuristic
8
8
  * overestimates mixed text.
9
9
  *
10
+ * R3 — accuracy improvement: the heuristic is *calibrated* against the
11
+ * provider's reported usage when available. The runtime calls
12
+ * `calibrateEstimate` after each message_end that carries usage, which
13
+ * adjusts the per-character ratio toward the real value observed for this
14
+ * model. The calibration is bounded (0.1–0.6 chars/token) so a bad sample
15
+ * can't make the budget non-conservative. Until the first sample arrives,
16
+ * the safe chars/4 default is used.
17
+ *
10
18
  * Strategy:
11
19
  * 1. Always preserve the system prompt, the latest user task, and the
12
20
  * latest assistant message.
@@ -28,10 +36,20 @@ export interface BudgetCheck {
28
36
  used: number;
29
37
  cap: number;
30
38
  }
39
+ /**
40
+ * Calibrate the heuristic against provider-reported usage. Pass the actual
41
+ * input token count and the char length of the transcript that was sent.
42
+ * The ratio self-corrects toward the model's true tokenizer behavior.
43
+ */
44
+ export declare function calibrateEstimate(usedChars: number, reportedInputTokens: number): number;
45
+ /** Current chars/token ratio (after calibration, if any). */
46
+ export declare function charsPerTokenRatio(): number;
31
47
  export declare function estimateTokens(s: string): number;
32
48
  export declare function estimateMessage(m: Message): number;
33
49
  /** Total input tokens for a transcript + optional system prompt. */
34
50
  export declare function totalTokens(system: string | undefined, messages: Message[]): number;
51
+ /** Count the raw character length of a transcript for calibration. */
52
+ export declare function transcriptCharLength(system: string | undefined, messages: Message[]): number;
35
53
  /** True if the input fits under the budget cap. */
36
54
  export declare function withinBudget(system: string | undefined, messages: Message[], budget: TokenBudget): BudgetCheck;
37
55
  /**
@@ -7,6 +7,14 @@
7
7
  * never overflows the model's true window because the heuristic
8
8
  * overestimates mixed text.
9
9
  *
10
+ * R3 — accuracy improvement: the heuristic is *calibrated* against the
11
+ * provider's reported usage when available. The runtime calls
12
+ * `calibrateEstimate` after each message_end that carries usage, which
13
+ * adjusts the per-character ratio toward the real value observed for this
14
+ * model. The calibration is bounded (0.1–0.6 chars/token) so a bad sample
15
+ * can't make the budget non-conservative. Until the first sample arrives,
16
+ * the safe chars/4 default is used.
17
+ *
10
18
  * Strategy:
11
19
  * 1. Always preserve the system prompt, the latest user task, and the
12
20
  * latest assistant message.
@@ -16,8 +24,32 @@
16
24
  * 3. If still over budget, summarize the surviving tail into a single
17
25
  * user message ("Earlier in this session: …").
18
26
  */
27
+ /** Calibration ratio: chars per token. Starts at 4.0 (the classic heuristic)
28
+ * and self-corrects toward the provider's reported usage. Bounded to
29
+ * [MIN_RATIO, MAX_RATIO] = [2.0, 6.0] so a pathological sample (a tool dump
30
+ * that tokenizes densely, or a sparse prompt) can't drive the budget into a
31
+ * non-conservative regime. Real tokenizers sit around 3.5–4.5 chars/token. */
32
+ const MIN_RATIO = 2.0;
33
+ const MAX_RATIO = 6.0;
34
+ let charsPerToken = 4.0;
35
+ /**
36
+ * Calibrate the heuristic against provider-reported usage. Pass the actual
37
+ * input token count and the char length of the transcript that was sent.
38
+ * The ratio self-corrects toward the model's true tokenizer behavior.
39
+ */
40
+ export function calibrateEstimate(usedChars, reportedInputTokens) {
41
+ if (reportedInputTokens <= 0 || usedChars <= 0)
42
+ return charsPerToken;
43
+ const newRatio = usedChars / reportedInputTokens;
44
+ charsPerToken = Math.max(MIN_RATIO, Math.min(MAX_RATIO, newRatio));
45
+ return charsPerToken;
46
+ }
47
+ /** Current chars/token ratio (after calibration, if any). */
48
+ export function charsPerTokenRatio() {
49
+ return charsPerToken;
50
+ }
19
51
  export function estimateTokens(s) {
20
- return Math.ceil(s.length / 4);
52
+ return Math.ceil(s.length / charsPerToken);
21
53
  }
22
54
  export function estimateMessage(m) {
23
55
  let n = 4; // role + structural overhead
@@ -43,6 +75,23 @@ export function totalTokens(system, messages) {
43
75
  n += estimateMessage(m);
44
76
  return n;
45
77
  }
78
+ /** Count the raw character length of a transcript for calibration. */
79
+ export function transcriptCharLength(system, messages) {
80
+ let n = system ? system.length : 0;
81
+ for (const m of messages) {
82
+ for (const b of m.content) {
83
+ if (b.kind === 'text')
84
+ n += b.text.length;
85
+ else if (b.kind === 'tool_use')
86
+ n += b.name.length + JSON.stringify(b.input).length;
87
+ else if (b.kind === 'tool_result') {
88
+ const out = typeof b.output === 'string' ? b.output : JSON.stringify(b.output ?? '');
89
+ n += out.length + (b.name?.length ?? 0);
90
+ }
91
+ }
92
+ }
93
+ return n;
94
+ }
46
95
  /** True if the input fits under the budget cap. */
47
96
  export function withinBudget(system, messages, budget) {
48
97
  const used = totalTokens(system, messages);
@@ -4,6 +4,11 @@
4
4
  */
5
5
  import type { KlyroEvent } from './catalog.js';
6
6
  type Listener = (ev: KlyroEvent) => void;
7
+ /** Cap for the retained event history. Long sessions emit a lot of
8
+ * stream.delta / tool.result events; keeping them all grows memory
9
+ * without bound. We retain a fixed recent window plus the structural
10
+ * events (phase, verify, error) so replays stay coherent. */
11
+ export declare const HISTORY_CAP = 10000;
7
12
  export declare class EventBus {
8
13
  private listeners;
9
14
  private history;
@@ -2,11 +2,21 @@
2
2
  * 3.1 — core/events emitter
3
3
  * In-memory pub/sub for KlyroEvents. Sync delivery, no buffering.
4
4
  */
5
+ /** Cap for the retained event history. Long sessions emit a lot of
6
+ * stream.delta / tool.result events; keeping them all grows memory
7
+ * without bound. We retain a fixed recent window plus the structural
8
+ * events (phase, verify, error) so replays stay coherent. */
9
+ export const HISTORY_CAP = 10_000;
5
10
  export class EventBus {
6
11
  listeners = new Set();
7
12
  history = [];
8
13
  emit(ev) {
9
14
  this.history.push(ev);
15
+ if (this.history.length > HISTORY_CAP) {
16
+ // Cheap uniform prune: drop every other oldest event so a flood of
17
+ // deltas can't force a reallocation on each emit.
18
+ this.history = this.history.filter((_, i) => i % 2 === 1);
19
+ }
10
20
  for (const l of [...this.listeners]) {
11
21
  try {
12
22
  l(ev);
@@ -1,6 +1,15 @@
1
1
  /**
2
2
  * 3.1 — KlyroEvent catalog (Appendix C)
3
3
  * Every observable action in the harness is a typed event.
4
+ *
5
+ * Reserved-future members (declared but intentionally unproduced — no
6
+ * emitter exists yet, so do not treat their absence as a bug):
7
+ * `subtask.tool_call`, `subtask.tool_result`.
8
+ * `subtask.progress` IS emitted (throttled: max 1 per child tool call,
9
+ * in-process children only; process-isolated children don't run the emitter).
10
+ * The orchestrator otherwise emits started/completed/failed/cancelled/
11
+ * timed_out/merged only; grep for emitters before assuming one of the
12
+ * reserved members fires.
4
13
  */
5
14
  export type KlyroEvent = {
6
15
  type: 'session.start';
@@ -1,5 +1,14 @@
1
1
  /**
2
2
  * 3.1 — KlyroEvent catalog (Appendix C)
3
3
  * Every observable action in the harness is a typed event.
4
+ *
5
+ * Reserved-future members (declared but intentionally unproduced — no
6
+ * emitter exists yet, so do not treat their absence as a bug):
7
+ * `subtask.tool_call`, `subtask.tool_result`.
8
+ * `subtask.progress` IS emitted (throttled: max 1 per child tool call,
9
+ * in-process children only; process-isolated children don't run the emitter).
10
+ * The orchestrator otherwise emits started/completed/failed/cancelled/
11
+ * timed_out/merged only; grep for emitters before assuming one of the
12
+ * reserved members fires.
4
13
  */
5
14
  export {};
package/dist/index.js CHANGED
@@ -349,7 +349,7 @@ async function main() {
349
349
  });
350
350
  program
351
351
  .command('chat [prompt]')
352
- .description('Legacy streamed chat. Without a prompt, start an interactive REPL.')
352
+ .description('Legacy streamed chat. Without a prompt, start an interactive REPL. (deprecated: history truncation is approximate; prefer `klyro tui`)')
353
353
  .option('-s, --system <text>', 'System message', 'You are a helpful assistant.')
354
354
  .option('-m, --model <id>', 'Override the model (default: env KLYRO_MODEL)')
355
355
  .option('-t, --timeout <ms>', 'Request timeout in ms (default: env KLYRO_TIMEOUT_MS or 60000)', (v) => parsePositiveInt('-t/--timeout', v))
@@ -613,11 +613,78 @@ async function main() {
613
613
  }
614
614
  });
615
615
  mcp.command('add <name> <url>').description('Add MCP server').action(async (name) => { process.stdout.write(`added mcp ${name} (stub)\n`); });
616
+ mcp.command('probe <name>').description('Connect to an MCP server (15s timeout), list its tools, print count+names').action(async (name) => {
617
+ const { loadMcpServers } = await import('./mcp/config.js');
618
+ const cfg = loadMcpServers(process.cwd());
619
+ const spec = cfg.servers[name];
620
+ if (!spec) {
621
+ process.stderr.write(`klyro: mcp server not found: ${name}\n`);
622
+ process.exit(2);
623
+ }
624
+ const { McpClient } = await import('./mcp/client.js');
625
+ const client = new McpClient(name, spec);
626
+ // 15s overall probe budget (connect has its own internal timeout too).
627
+ const timer = setTimeout(() => {
628
+ process.stderr.write(`klyro: mcp probe ${name} timed out after 15s\n`);
629
+ process.exit(2);
630
+ }, 15_000);
631
+ try {
632
+ await client.connect();
633
+ const tools = await client.listTools();
634
+ const names = tools.map((t) => t.name);
635
+ process.stdout.write(`${name}: ${tools.length} tool(s)${names.length > 0 ? `: ${names.join(', ')}` : ''}\n`);
636
+ process.exit(0);
637
+ }
638
+ catch (err) {
639
+ process.stderr.write(`klyro: mcp probe ${name} failed: ${err instanceof Error ? err.message : String(err)}\n`);
640
+ process.exit(2);
641
+ }
642
+ finally {
643
+ clearTimeout(timer);
644
+ try {
645
+ await client.close();
646
+ }
647
+ catch { /* ignore */ }
648
+ }
649
+ });
616
650
  mcp.command('serve').description('Serve as MCP server').action(async () => { process.stdout.write('klyro mcp serve — exposing tools (stub)\n'); });
617
- // 10.2 — Hooks / agents
618
- program.command('hooks').description('List hooks (10.2)').action(async () => { process.stdout.write('hooks: SessionStart UserPromptSubmit PreToolUse PostToolUse (stub)\n'); });
619
- program.command('agents [name]').description('List agents (10.2) or show one agent definition').action(async (name) => {
651
+ // 10.2 — Hooks: list configured preToolUse/postToolUse hooks.
652
+ program.command('hooks [cmd]').description('Hooks (10.2): `klyro hooks` or `klyro hooks list` prints configured hooks').action(async (cmd) => {
653
+ if (cmd && cmd !== 'list') {
654
+ process.stderr.write(`klyro: unknown hooks command: ${cmd} (usage: klyro hooks [list])\n`);
655
+ process.exit(2);
656
+ }
657
+ const { loadHooks } = await import('./cli/hooks.js');
658
+ const hooks = loadHooks(process.cwd());
659
+ if (hooks.length === 0) {
660
+ process.stdout.write('hooks: none configured (.klyro/hooks.json, ~/.klyro/hooks.json)\n');
661
+ return;
662
+ }
663
+ for (const h of hooks)
664
+ process.stdout.write(`${h.name} ${h.event} ${h.command}\n`);
665
+ });
666
+ program.command('agents [name] [extra...]').description('List agents (10.2), show one, or run: agents run <name> <task...>').action(async (name, extra) => {
620
667
  const { BUILTIN_AGENTS } = await import('./agent/orchestrator.js');
668
+ // `klyro agents run <name> <task...>`: one-shot run under a named agent.
669
+ if (name === 'run') {
670
+ const [agentName, ...taskParts] = extra ?? [];
671
+ if (!agentName || !BUILTIN_AGENTS.some((a) => a.id === agentName)) {
672
+ process.stderr.write(`klyro: unknown agent: ${agentName ?? '(missing)'} (known: ${BUILTIN_AGENTS.map((a) => a.id).join(', ')})\n`);
673
+ process.exit(2);
674
+ }
675
+ const task = (taskParts ?? []).join(' ').trim();
676
+ if (!task) {
677
+ process.stderr.write('klyro: agents run requires a task (usage: klyro agents run <name> <task...>)\n');
678
+ process.exit(2);
679
+ }
680
+ const model = process.env.KLYRO_MODEL;
681
+ if (!model) {
682
+ process.stderr.write('klyro: KLYRO_MODEL is not set (or pass --model via klyro run)\n');
683
+ process.exit(2);
684
+ }
685
+ const code = await runOnce({ task, cwd: process.cwd(), model, agent: agentName });
686
+ process.exit(code);
687
+ }
621
688
  if (!name) {
622
689
  for (const a of BUILTIN_AGENTS)
623
690
  process.stdout.write(`${a.id} — ${a.description}\n`);
@@ -642,7 +709,24 @@ async function main() {
642
709
  process.stdout.write(lines.join('\n') + '\n');
643
710
  });
644
711
  // 10.3 — Web / git workflows / SDK
645
- program.command('commit').description('Create commit (10.3)').action(async () => { process.stdout.write('commit — conventional message (stub, use /commit)\n'); });
712
+ program
713
+ .command('commit')
714
+ .description('Commit staged changes with a conventional message (verification hooks always run)')
715
+ .option('--dry-run', 'Print the message + files without committing')
716
+ .option('--message <msg>', 'Summary for the conventional message (default: update <n> files)')
717
+ .option('--force-secret', 'Commit even if the staged diff looks like it contains a secret')
718
+ .action(async (opts) => {
719
+ const { runCommit } = await import('./cli/commit.js');
720
+ const globalYes = program.opts().yes ?? process.env.KLYRO_YES === '1';
721
+ const code = await runCommit({
722
+ cwd: process.cwd(),
723
+ yes: !!globalYes,
724
+ dryRun: !!opts.dryRun,
725
+ ...(opts.message !== undefined ? { message: opts.message } : {}),
726
+ forceSecret: !!opts.forceSecret,
727
+ });
728
+ process.exit(code);
729
+ });
646
730
  program.command('audit [session]').description('Verify audit chain (13.4)').action(async (session) => {
647
731
  if (!session) {
648
732
  process.stderr.write('klyro: audit requires a session id (usage: klyro audit <session>)\n');
@@ -90,7 +90,7 @@ export class McpClient {
90
90
  await this.request('initialize', {
91
91
  protocolVersion: PROTOCOL_VERSION,
92
92
  capabilities: {},
93
- clientInfo: { name: 'klyro', version: '1.0.1' },
93
+ clientInfo: { name: 'klyro', version: '1.0.2' },
94
94
  }, timeoutMs);
95
95
  this.notify('notifications/initialized', {});
96
96
  this.connected = true;
@@ -1,21 +1,3 @@
1
- /**
2
- * P1 — MCP tool registration (r-11-17.md §3.1).
3
- *
4
- * Exposes MCP server tools through the {@link ToolRegistry} under
5
- * `mcp__<server>__<tool>` names. Security properties (non-negotiable):
6
- *
7
- * 1. Deny-by-default: `evaluateMcpPolicy` runs BEFORE any client I/O; a
8
- * denied tool returns POLICY_DENIED without touching the subprocess.
9
- * 2. Redact-before-transcript: every success value AND error message passes
10
- * through `redact()` before it can reach the model or trace.
11
- * 3. `requireApproval` servers add an ask-rule to the PolicyEngine so the
12
- * runtime loop prompts before executing.
13
- *
14
- * Permission class: no code in src consumes `Tool.permission` (it is
15
- * write-only metadata today), so we pick the most restrictive class,
16
- * 'admin' — the same class as `spawn_agent`, since MCP tools execute
17
- * arbitrary external side effects (read/write/network) outside our control.
18
- */
19
1
  import { type McpClientLike } from './client.js';
20
2
  import { type McpServerSpec } from './config.js';
21
3
  import type { PolicyEngine } from '../policy/engine.js';
@@ -15,7 +15,18 @@
15
15
  * write-only metadata today), so we pick the most restrictive class,
16
16
  * 'admin' — the same class as `spawn_agent`, since MCP tools execute
17
17
  * arbitrary external side effects (read/write/network) outside our control.
18
+ *
19
+ * Debug capture: when `KLYRO_MCP_DEBUG=1` is set, every MCP tool success
20
+ * AND error ALSO writes the UNREDACTED raw JSON payload (pre-redaction,
21
+ * may contain secrets — handle accordingly) to
22
+ * `<configDir>/tool-output/mcp-<server>-<ts>.json` (mode 0600) and notes
23
+ * the path on stderr. `<configDir>` is `$KLYRO_CONFIG_DIR` when set,
24
+ * otherwise `~/.klyro`. Default off: with the flag unset (or any value
25
+ * other than `1`) no file is written and behaviour is unchanged.
18
26
  */
27
+ import * as fs from 'node:fs';
28
+ import * as os from 'node:os';
29
+ import * as path from 'node:path';
19
30
  import { McpClient, McpError } from './client.js';
20
31
  import { loadMcpServers } from './config.js';
21
32
  import { evaluateMcpPolicy } from './policy.js';
@@ -48,7 +59,39 @@ export function sanitizeMcpName(server, tool) {
48
59
  function errMessage(err) {
49
60
  return err instanceof Error ? err.message : String(err);
50
61
  }
51
- async function executeMcpTool(spec, toolDef, client, input, ctx) {
62
+ /** Debug-capture filename disambiguator when Date.now() collides. */
63
+ let mcpDebugCounter = 0;
64
+ /**
65
+ * `KLYRO_MCP_DEBUG=1` capture: write the UNREDACTED raw payload to
66
+ * `<configDir>/tool-output/mcp-<server>-<ts>.json` (mode 0600) + a stderr
67
+ * note. Best-effort and synchronous — never throws into the tool path.
68
+ */
69
+ function captureMcpDebug(server, tool, payload) {
70
+ if (process.env.KLYRO_MCP_DEBUG !== '1')
71
+ return;
72
+ try {
73
+ const base = process.env.KLYRO_CONFIG_DIR ?? path.join(os.homedir() || process.cwd(), '.klyro');
74
+ const dir = path.join(base, 'tool-output');
75
+ fs.mkdirSync(dir, { recursive: true });
76
+ const safe = server.replace(/[^A-Za-z0-9_.-]/g, '_').slice(0, 32) || 'server';
77
+ let file = path.join(dir, `mcp-${safe}-${Date.now()}.json`);
78
+ if (fs.existsSync(file)) {
79
+ mcpDebugCounter += 1;
80
+ file = path.join(dir, `mcp-${safe}-${Date.now()}-${mcpDebugCounter}.json`);
81
+ }
82
+ fs.writeFileSync(file, JSON.stringify({ server, tool, payload }, null, 2), { mode: 0o600 });
83
+ try {
84
+ process.stderr.write(`klyro: mcp debug captured ${server}/${tool} -> ${file}\n`);
85
+ }
86
+ catch {
87
+ /* ignore */
88
+ }
89
+ }
90
+ catch {
91
+ /* best-effort only */
92
+ }
93
+ }
94
+ async function executeMcpTool(server, spec, toolDef, client, input, ctx) {
52
95
  try {
53
96
  // (1) Deny-by-default — runs BEFORE any client I/O.
54
97
  const decision = evaluateMcpPolicy(spec.policy, toolDef.name);
@@ -67,16 +110,20 @@ async function executeMcpTool(spec, toolDef, client, input, ctx) {
67
110
  // (3) Typed server failures keep their code; everything else is TOOL_ERROR.
68
111
  // Redact BEFORE the message can reach the model/trace.
69
112
  if (err instanceof McpError) {
113
+ captureMcpDebug(server, toolDef.name, { code: err.code, message: err.message, details: err.details });
70
114
  return { ok: false, error: { code: err.code, message: redact(err.message) } };
71
115
  }
116
+ captureMcpDebug(server, toolDef.name, { message: errMessage(err) });
72
117
  return { ok: false, error: { code: 'TOOL_ERROR', message: redact(errMessage(err)) } };
73
118
  }
74
119
  // (4) Server-reported error → TOOL_ERROR, redacted, bounded.
75
120
  if (res.isError) {
121
+ captureMcpDebug(server, toolDef.name, res.raw);
76
122
  return { ok: false, error: { code: 'TOOL_ERROR', message: redact(res.text).slice(0, 2000) } };
77
123
  }
78
124
  // (5) Success — redact BEFORE the value reaches the model/trace, then
79
125
  // truncate to a bounded size with a marker.
126
+ captureMcpDebug(server, toolDef.name, res.raw);
80
127
  const redacted = redact(res.text);
81
128
  if (redacted.length > MCP_SUCCESS_MAX_CHARS) {
82
129
  return {
@@ -177,7 +224,7 @@ export async function registerMcpServers(cfg, opts) {
177
224
  // 'admin': most restrictive class — MCP tools run arbitrary
178
225
  // external side effects; nothing in src reads this field yet.
179
226
  permission: 'admin',
180
- execute: (input, ctx) => executeMcpTool(boundSpec, boundDef, boundClient, input, ctx),
227
+ execute: (input, ctx) => executeMcpTool(server, boundSpec, boundDef, boundClient, input, ctx),
181
228
  });
182
229
  registry.register(tool);
183
230
  registered.push(name);
@@ -84,6 +84,22 @@ export declare class PolicyEngine {
84
84
  /** Builtin set of rules. Order matters: first match wins. */
85
85
  export declare function builtinRules(): PolicyRule[];
86
86
  export declare function matchesGlobRule(call: ToolCallLike, rule: string): boolean;
87
+ /** Branches that `git push` must never target without an explicit opt-out. */
88
+ export declare const PROTECTED_BRANCHES: string[];
89
+ /** Matches `push` with a protected branch name later on the same line. */
90
+ export declare const PROTECTED_PUSH_RE: RegExp;
91
+ /**
92
+ * True when the command is a `git push` with NO ref/positional args
93
+ * (e.g. `git push`, `git push -f`) — i.e. it pushes whatever is checked
94
+ * out. `git push origin feature` is explicit, not bare.
95
+ */
96
+ export declare function isBareGitPush(cmd: string): boolean;
97
+ /**
98
+ * Resolve the currently checked-out branch via `git branch --show-current`.
99
+ * Returns null on any failure (not a repo, git missing) — callers treat
100
+ * null as "unknown" and allow the normal policy flow to continue.
101
+ */
102
+ export declare function currentGitBranch(cwd: string): string | null;
87
103
  /** Hard-deny for obviously destructive shell patterns. */
88
104
  export declare const shellDenyRule: PolicyRule;
89
105
  /** Allowlist for shell — exact or prefix. Anything not in the list asks. */