klyro 1.0.5 → 1.0.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. package/README.md +13 -0
  2. package/dist/agent/custom-agents.d.ts +3 -0
  3. package/dist/agent/custom-agents.js +96 -0
  4. package/dist/agent/orchestrator.d.ts +26 -0
  5. package/dist/agent/orchestrator.js +41 -4
  6. package/dist/agent/runtime.d.ts +15 -0
  7. package/dist/agent/runtime.js +232 -61
  8. package/dist/chat.d.ts +10 -0
  9. package/dist/chat.js +39 -7
  10. package/dist/checkpoints/store.d.ts +11 -0
  11. package/dist/checkpoints/store.js +32 -0
  12. package/dist/cli/auth.d.ts +10 -3
  13. package/dist/cli/auth.js +43 -5
  14. package/dist/cli/completion.js +2 -2
  15. package/dist/cli/config.d.ts +4 -4
  16. package/dist/cli/doctor.js +0 -1
  17. package/dist/cli/eval.d.ts +15 -1
  18. package/dist/cli/eval.js +43 -5
  19. package/dist/cli/hooks.d.ts +74 -5
  20. package/dist/cli/hooks.js +118 -7
  21. package/dist/cli/init.d.ts +6 -0
  22. package/dist/cli/init.js +60 -0
  23. package/dist/cli/keychain.d.ts +10 -0
  24. package/dist/cli/keychain.js +86 -0
  25. package/dist/cli/repl.js +188 -30
  26. package/dist/cli/run.d.ts +7 -1
  27. package/dist/cli/run.js +92 -50
  28. package/dist/cli/setup.js +3 -2
  29. package/dist/cli/slash/custom.d.ts +25 -0
  30. package/dist/cli/slash/custom.js +166 -0
  31. package/dist/cli/slash/parser.d.ts +9 -1
  32. package/dist/cli/slash/parser.js +34 -9
  33. package/dist/cli/update.d.ts +3 -1
  34. package/dist/cli/update.js +16 -1
  35. package/dist/context/accounting.d.ts +6 -0
  36. package/dist/context/accounting.js +8 -2
  37. package/dist/context/compaction.d.ts +2 -1
  38. package/dist/context/compaction.js +39 -12
  39. package/dist/context/memory.d.ts +11 -0
  40. package/dist/context/memory.js +59 -4
  41. package/dist/eval/harness.d.ts +40 -5
  42. package/dist/eval/harness.js +103 -10
  43. package/dist/eval/judge.d.ts +32 -0
  44. package/dist/eval/judge.js +63 -0
  45. package/dist/eval/tasks.js +134 -0
  46. package/dist/index.js +239 -130
  47. package/dist/mcp/auth.d.ts +85 -0
  48. package/dist/mcp/auth.js +249 -0
  49. package/dist/mcp/client.d.ts +15 -0
  50. package/dist/mcp/client.js +42 -2
  51. package/dist/mcp/config.d.ts +31 -1
  52. package/dist/mcp/config.js +84 -1
  53. package/dist/mcp/registry.d.ts +19 -0
  54. package/dist/mcp/registry.js +118 -2
  55. package/dist/mcp/remote.d.ts +36 -0
  56. package/dist/mcp/remote.js +207 -0
  57. package/dist/mcp/sse.d.ts +42 -0
  58. package/dist/mcp/sse.js +310 -0
  59. package/dist/persistence/audit.d.ts +15 -3
  60. package/dist/persistence/audit.js +84 -13
  61. package/dist/persistence/store.d.ts +9 -0
  62. package/dist/persistence/store.js +17 -0
  63. package/dist/policy/approval.d.ts +15 -1
  64. package/dist/policy/approval.js +8 -0
  65. package/dist/policy/engine.js +9 -0
  66. package/dist/providers/endpoints.d.ts +43 -0
  67. package/dist/providers/endpoints.js +104 -0
  68. package/dist/providers.js +17 -14
  69. package/dist/tools/shell/shell-exec.d.ts +13 -0
  70. package/dist/tools/shell/shell-exec.js +64 -2
  71. package/dist/tui/app.js +172 -15
  72. package/dist/tui/app.test.js +27 -2
  73. package/dist/tui/approval.js +55 -1
  74. package/dist/tui/scroll-model.d.ts +2 -2
  75. package/dist/tui/scroll-model.js +9 -3
  76. package/dist/tui/tokens.d.ts +8 -11
  77. package/dist/tui/tokens.js +18 -11
  78. package/package.json +1 -1
@@ -23,14 +23,15 @@ import { verify, diagnosticForModel } from '../verification/engine.js';
23
23
  import { detectVerifyCommand } from '../verification/auto.js';
24
24
  import { ensureBaseline, getBaseline } from '../verification/baseline.js';
25
25
  import { compressTranscript, totalTokens, calibrateEstimate, transcriptCharLength } from '../context/tokenizer.js';
26
- import { capForModel } from '../context/accounting.js';
26
+ import { capForModel, RESERVE_OUTPUT_TOKENS } from '../context/accounting.js';
27
+ import { shouldRemind, reminderForTodos } from '../context/memory.js';
27
28
  import { ratesFor, isAnthropicModel } from '../providers/model-info.js';
28
29
  import { classifyFailure, rerunOnce, gatherRepairContext, guardRepair } from '../verification/classify.js';
29
30
  import { findRelatedTests, buildScopedCommand, runScopedVerify, syntaxCheck, checkImports } from '../verification/scoped.js';
30
31
  import { globalBus } from '../events/bus.js';
31
32
  import { TraceWriter } from '../trace/writer.js';
32
33
  import { killAllJobs } from '../tools/shell/background.js';
33
- import { loadHooks, runHook } from '../cli/hooks.js';
34
+ import { loadHooks, runHook, hooksForEvent } from '../cli/hooks.js';
34
35
  /** Normalize either systemPrompt shape into {system, suffix}. */
35
36
  export function resolveSystemPrompt(fn, ctx) {
36
37
  const r = fn(ctx);
@@ -118,6 +119,7 @@ export async function run(opts, deps) {
118
119
  }
119
120
  };
120
121
  let steps = 0;
122
+ let lastRemindTurn = 0;
121
123
  let toolCallCount = 0;
122
124
  let finalText = '';
123
125
  let repairs = 0;
@@ -154,16 +156,21 @@ export async function run(opts, deps) {
154
156
  emit?.({ kind: 'model_override', requested: opts.model, effective: opts.parentContext.model });
155
157
  }
156
158
  // Hooks engine: loaded once per run. Zero-cost fast path — when no hooks
157
- // file exists, both lists are empty and every hook call site is skipped.
159
+ // file exists, the list is empty and every hook call site is skipped.
160
+ // --bare skips hooks entirely (deterministic runs).
161
+ // Tool-event hooks are matched per tool at the call sites below
162
+ // (hooksForEvent over runHooks); lifecycle events run at their own points.
158
163
  let runHooks = [];
159
- try {
160
- runHooks = loadHooks(opts.cwd);
161
- }
162
- catch {
163
- runHooks = [];
164
+ if (!opts.bare) {
165
+ try {
166
+ runHooks = loadHooks(opts.cwd);
167
+ }
168
+ catch {
169
+ runHooks = [];
170
+ }
164
171
  }
165
- const preHooks = runHooks.filter((h) => h.event === 'preToolUse');
166
- const postHooks = runHooks.filter((h) => h.event === 'postToolUse');
172
+ // Tool-event hooks are matched per tool at the call sites below
173
+ // (hooksForEvent over runHooks); lifecycle events run at their own points.
167
174
  // L15 failover chain: the active adapter starts as deps.adapter; each
168
175
  // terminal provider error consumes one fallback. Bounded — never loops.
169
176
  let activeAdapter = deps.adapter;
@@ -179,6 +186,16 @@ export async function run(opts, deps) {
179
186
  bus.emit(ev);
180
187
  tracer?.write(ev).catch(() => undefined);
181
188
  };
189
+ // Light audit writer: mirrors policy decisions + tool results into the
190
+ // chained audit log. Best-effort — audit errors must not break the run.
191
+ const auditLog = opts.audit?.log;
192
+ const auditSessionId = opts.audit?.sessionId ?? opts.persist?.sessionId;
193
+ const writeAudit = (ev) => {
194
+ if (!auditLog || !auditSessionId)
195
+ return;
196
+ // fire-and-forget — the log serializes its own chain internally
197
+ void auditLog.write(ev).catch(() => undefined);
198
+ };
182
199
  const closeTracer = async () => {
183
200
  try {
184
201
  await tracer?.close();
@@ -247,11 +264,35 @@ export async function run(opts, deps) {
247
264
  // Fire-and-forget initial checkpoint (don't await to block loop start)
248
265
  void checkpoint(transcript[transcript.length - 1]);
249
266
  }
267
+ // sessionStart: prerequisite gate. A non-zero exit aborts the run before
268
+ // step 1 with status 'blocked' (e.g. missing toolchain, dirty tree).
269
+ {
270
+ const starters = hooksForEvent(runHooks, 'sessionStart');
271
+ for (const hook of starters) {
272
+ let r = null;
273
+ try {
274
+ r = await runHook(hook, { toolName: '', input: {} }, { event: 'sessionStart', sessionId, cwd: opts.cwd, task: opts.task });
275
+ }
276
+ catch {
277
+ r = null;
278
+ }
279
+ if (r && !r.ok) {
280
+ const reason = (r.stderr || r.stdout || 'sessionStart hook failed').slice(0, 500);
281
+ emitKlyro({ type: 'error', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', code: 'session_blocked', message: reason });
282
+ await closeTracer();
283
+ return { status: 'blocked', steps, toolCalls: toolCallCount, finalText: `Blocked by sessionStart hook ${hook.name}: ${reason}`, transcript, hasEdits, usage, repairs, phase: 'blocked' };
284
+ }
285
+ }
286
+ }
250
287
  // 5.2 — stuck detection state
251
288
  const callHistory = [];
252
289
  const fileEditCounts = new Map();
253
290
  let stuckTriggers = 0;
254
291
  let stuckAbort = false;
292
+ // Steerable stop: a stop hook's `{"continue":true}` verdict carries one
293
+ // more turn. Consumed once at the completion point, max 3 per run.
294
+ let stopCont = null;
295
+ let stopContUsed = 0;
255
296
  outer: while (steps < maxSteps) {
256
297
  // 5.1 limits: max-cost, max-time
257
298
  if (maxCost !== undefined) {
@@ -282,6 +323,33 @@ export async function run(opts, deps) {
282
323
  return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined, phase: 'blocked' };
283
324
  }
284
325
  steps++;
326
+ // 8.4 — stale-todo reminder: every 20 turns, re-inject pending plan
327
+ // items from `.klyro/plans/todos.json` (written by todo_write) so a
328
+ // long run cannot silently drop its checklist. Best-effort + tiny.
329
+ if (shouldRemind(steps, lastRemindTurn)) {
330
+ lastRemindTurn = steps;
331
+ try {
332
+ const { readFileSync } = await import('node:fs');
333
+ const { join } = await import('node:path');
334
+ const rawTodos = JSON.parse(readFileSync(join(opts.cwd, '.klyro', 'plans', 'todos.json'), 'utf-8'));
335
+ if (Array.isArray(rawTodos)) {
336
+ const planSteps = rawTodos
337
+ .filter((t) => typeof t.title === 'string')
338
+ .map((t, i) => ({
339
+ id: `todo-${i}`,
340
+ title: t.title,
341
+ status: ['pending', 'in_progress', 'done', 'failed', 'skipped'].includes(t.status) ? t.status : 'pending',
342
+ }));
343
+ const reminder = reminderForTodos(planSteps);
344
+ if (reminder) {
345
+ const note = { role: 'user', content: [text(`[system note] ${reminder}`)] };
346
+ transcript.push(note);
347
+ await checkpoint(note);
348
+ }
349
+ }
350
+ }
351
+ catch { /* no todos file — nothing to remind */ }
352
+ }
285
353
  // 5.1 phase transitions (model-narrated)
286
354
  if (steps === 1)
287
355
  setPhase('understanding');
@@ -304,7 +372,7 @@ export async function run(opts, deps) {
304
372
  const systemForBudget = telemetrySuffix ? `${stableSystem}\n\n${telemetrySuffix}` : stableSystem;
305
373
  // Window-aware ceiling (was a hardcoded 120k that overflowed 8k local
306
374
  // models): size the input budget to the model's context window.
307
- const BUDGET = { total: capForModel(opts.model, 4000), reservedOutput: 4000 };
375
+ const BUDGET = { total: capForModel(opts.model, RESERVE_OUTPUT_TOKENS), reservedOutput: RESERVE_OUTPUT_TOKENS };
308
376
  let reqMessages = transcript;
309
377
  let reqSystem = stableSystem;
310
378
  let reqSuffix = telemetrySuffix;
@@ -410,7 +478,7 @@ export async function run(opts, deps) {
410
478
  telemetry.recordError('overflow_retry');
411
479
  emitKlyro({ type: 'error', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', code: 'REQUEST_TOO_LARGE', message: 'context overflow — aggressively compacting transcript and retrying the request once' });
412
480
  try {
413
- const compacted = compressTranscript(reqSystem, transcript, { total: 30_000, reservedOutput: 4000 });
481
+ const compacted = compressTranscript(reqSystem, transcript, { total: 30_000, reservedOutput: RESERVE_OUTPUT_TOKENS });
414
482
  if (compacted.messages.length < transcript.length || compacted.dropped > 0) {
415
483
  transcript.splice(0, transcript.length, ...compacted.messages);
416
484
  }
@@ -418,7 +486,7 @@ export async function run(opts, deps) {
418
486
  // Transcript already fits the aggressive budget — force it
419
487
  // strictly smaller so the retry cannot repeat the overflow.
420
488
  const halved = Math.max(4000, Math.floor(totalTokens(reqSystem, transcript) / 2));
421
- const smaller = compressTranscript(reqSystem, transcript, { total: halved, reservedOutput: 4000 });
489
+ const smaller = compressTranscript(reqSystem, transcript, { total: halved, reservedOutput: RESERVE_OUTPUT_TOKENS });
422
490
  transcript.splice(0, transcript.length, ...smaller.messages);
423
491
  }
424
492
  tokenCache = { lastRef: null, lastSystem: undefined, lastCount: 0 };
@@ -549,6 +617,17 @@ export async function run(opts, deps) {
549
617
  return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined };
550
618
  }
551
619
  if (finalizedCalls.length === 0) {
620
+ // Steerable stop: a stop hook asked for one more turn instead of
621
+ // completing. Consumed once per verdict, max 3 per run.
622
+ if (stopCont !== null && stopContUsed < 3) {
623
+ stopContUsed++;
624
+ const contMsg = { role: 'user', content: [text(`[system note] ${stopCont}`)] };
625
+ stopCont = null;
626
+ transcript.push(contMsg);
627
+ await checkpoint(contMsg);
628
+ emit?.({ kind: 'step_end', step: steps });
629
+ continue;
630
+ }
552
631
  if (hadInvalidTool) {
553
632
  // The model attempted a tool call that failed validation; the error is
554
633
  // already a tool_result in the transcript — loop so the model can
@@ -833,39 +912,49 @@ export async function run(opts, deps) {
833
912
  // results immediately — gate runs in call order so these stay ordered.
834
913
  // Returns true when the call is approved for execution.
835
914
  const gateCall = async (call) => {
836
- const decision = await deps.policy.evaluate({ name: call.name, input: call.input, permission: deps.registry.get(call.name)?.permission }, { cwd: opts.cwd, nonInteractive: opts.nonInteractive });
837
- emit?.({ kind: 'policy_decision', id: call.id, name: call.name, action: decision.action, ...(decision.action !== 'allow' ? { reason: decision.reason } : {}) });
838
- // Mirror to KlyroEvent bus
839
- if (decision.action === 'allow') {
840
- emitKlyro({ type: 'permission.decision', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, action: 'allow' });
841
- }
842
- else {
843
- emitKlyro({ type: 'permission.decision', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, action: decision.action, reason: decision.reason });
844
- }
845
- if (decision.action === 'deny') {
846
- const denyMsg = {
847
- role: 'tool',
848
- content: [
849
- mkToolResult(call.id, call.name, { error: 'POLICY_DENIED', reason: decision.reason }, true),
850
- ],
851
- };
852
- transcript.push(denyMsg);
853
- await checkpoint(denyMsg, { toolCallId: call.id, toolName: call.name, input: call.input, output: { error: 'POLICY_DENIED', reason: decision.reason }, isError: true });
854
- emitKlyro({ type: 'tool.result', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, name: call.name, output: { error: 'POLICY_DENIED' }, isError: true, latencyMs: 0 });
855
- telemetry.recordToolError(call, 'policy_denied');
856
- emit?.({ kind: 'tool_result', id: call.id, name: call.name, output: { error: 'POLICY_DENIED', reason: decision.reason }, isError: true, latencyMs: 0 });
857
- return false;
858
- }
859
- if (decision.action === 'ask') {
915
+ // Edit-and-retry loop: an `e`dit choice re-validates + re-evaluates
916
+ // policy on the edited input (bounded to 3 rounds so a user can't be
917
+ // re-prompted forever). `call.input` is updated in place so the
918
+ // executed + checkpointed call reflects what was approved.
919
+ let effectiveInput = call.input;
920
+ for (let round = 0; round < 3; round++) {
921
+ const decision = await deps.policy.evaluate({ name: call.name, input: effectiveInput, permission: deps.registry.get(call.name)?.permission }, { cwd: opts.cwd, nonInteractive: opts.nonInteractive });
922
+ emit?.({ kind: 'policy_decision', id: call.id, name: call.name, action: decision.action, ...(decision.action !== 'allow' ? { reason: decision.reason } : {}) });
923
+ // Mirror to KlyroEvent bus
924
+ if (decision.action === 'allow') {
925
+ emitKlyro({ type: 'permission.decision', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, action: 'allow' });
926
+ }
927
+ else {
928
+ emitKlyro({ type: 'permission.decision', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, action: decision.action, reason: decision.reason });
929
+ }
930
+ writeAudit({ kind: 'policy_decision', sessionId: auditSessionId ?? 'ephemeral', callId: call.id, action: decision.action, ts: Date.now() });
931
+ if (decision.action === 'deny') {
932
+ const denyMsg = {
933
+ role: 'tool',
934
+ content: [
935
+ mkToolResult(call.id, call.name, { error: 'POLICY_DENIED', reason: decision.reason }, true),
936
+ ],
937
+ };
938
+ transcript.push(denyMsg);
939
+ await checkpoint(denyMsg, { toolCallId: call.id, toolName: call.name, input: effectiveInput, output: { error: 'POLICY_DENIED', reason: decision.reason }, isError: true });
940
+ emitKlyro({ type: 'tool.result', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, name: call.name, output: { error: 'POLICY_DENIED' }, isError: true, latencyMs: 0 });
941
+ telemetry.recordToolError(call, 'policy_denied');
942
+ emit?.({ kind: 'tool_result', id: call.id, name: call.name, output: { error: 'POLICY_DENIED', reason: decision.reason }, isError: true, latencyMs: 0 });
943
+ return false;
944
+ }
945
+ if (decision.action === 'allow') {
946
+ call.input = effectiveInput;
947
+ return true;
948
+ }
949
+ // decision.action === 'ask'
860
950
  emitKlyro({ type: 'permission.ask', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, name: call.name, reason: decision.reason });
861
951
  const choice = await deps.approval.ask({
862
952
  toolName: call.name,
863
953
  reason: decision.reason,
864
- summary: summarizeToolCall(call),
865
- input: call.input,
866
- pattern: patternForCall(call.name, call.input),
954
+ summary: summarizeToolCall({ ...call, input: effectiveInput }),
955
+ input: effectiveInput,
956
+ pattern: patternForCall(call.name, effectiveInput),
867
957
  });
868
- // Approval UI in TUI handles y/a/A/n/e/? — e edits input, ? explains
869
958
  if (choice === 'deny') {
870
959
  const denyMsg2 = {
871
960
  role: 'tool',
@@ -874,15 +963,49 @@ export async function run(opts, deps) {
874
963
  ],
875
964
  };
876
965
  transcript.push(denyMsg2);
877
- await checkpoint(denyMsg2, { toolCallId: call.id, toolName: call.name, input: call.input, output: { error: 'POLICY_DENIED', reason: 'user denied' }, isError: true });
966
+ await checkpoint(denyMsg2, { toolCallId: call.id, toolName: call.name, input: effectiveInput, output: { error: 'POLICY_DENIED', reason: 'user denied' }, isError: true });
878
967
  telemetry.recordToolError(call, 'user_denied');
879
968
  emit?.({ kind: 'tool_result', id: call.id, name: call.name, output: { error: 'POLICY_DENIED', reason: 'user denied' }, isError: true, latencyMs: 0 });
880
969
  return false;
881
970
  }
882
- // Handle 'edit' choice: for now treat as allow with edited input (future: re-prompt)
971
+ if (typeof choice === 'object' && choice.kind === 'edit') {
972
+ // Re-validate the edited input against the tool schema before it
973
+ // goes anywhere — a malformed edit denies instead of executing.
974
+ const tool = deps.registry.get(call.name);
975
+ const parsed = tool?.inputSchema.safeParse(choice.editedInput);
976
+ if (!parsed || !parsed.success) {
977
+ const denyMsg3 = {
978
+ role: 'tool',
979
+ content: [
980
+ mkToolResult(call.id, call.name, { error: 'POLICY_DENIED', reason: 'edited input failed tool schema validation' }, true),
981
+ ],
982
+ };
983
+ transcript.push(denyMsg3);
984
+ await checkpoint(denyMsg3, { toolCallId: call.id, toolName: call.name, input: effectiveInput, output: { error: 'POLICY_DENIED', reason: 'edited input invalid' }, isError: true });
985
+ telemetry.recordToolError(call, 'edit_invalid');
986
+ emit?.({ kind: 'tool_result', id: call.id, name: call.name, output: { error: 'POLICY_DENIED', reason: 'edited input invalid' }, isError: true, latencyMs: 0 });
987
+ return false;
988
+ }
989
+ effectiveInput = parsed.data;
990
+ continue; // re-evaluate policy on the edited input
991
+ }
992
+ // allow / always / always-persist — approved with (possibly edited) input.
883
993
  repairs++;
994
+ call.input = effectiveInput;
995
+ return true;
884
996
  }
885
- return true;
997
+ // Edit rounds exhausted without approval — deny rather than loop forever.
998
+ const denyMsg4 = {
999
+ role: 'tool',
1000
+ content: [
1001
+ mkToolResult(call.id, call.name, { error: 'POLICY_DENIED', reason: 'approval rounds exhausted' }, true),
1002
+ ],
1003
+ };
1004
+ transcript.push(denyMsg4);
1005
+ await checkpoint(denyMsg4, { toolCallId: call.id, toolName: call.name, input: effectiveInput, output: { error: 'POLICY_DENIED', reason: 'approval rounds exhausted' }, isError: true });
1006
+ telemetry.recordToolError(call, 'approval_exhausted');
1007
+ emit?.({ kind: 'tool_result', id: call.id, name: call.name, output: { error: 'POLICY_DENIED', reason: 'approval rounds exhausted' }, isError: true, latencyMs: 0 });
1008
+ return false;
886
1009
  };
887
1010
  // Execute phase: run the tool with no transcript writes, so concurrent
888
1011
  // executions can't interleave. A throw here becomes a tool error (an
@@ -890,30 +1013,43 @@ export async function run(opts, deps) {
890
1013
  const execTool = async (call) => {
891
1014
  const t0 = Date.now();
892
1015
  emitKlyro({ type: 'tool.call', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, name: call.name, input: call.input });
893
- // Hooks: every preToolUse hook runs before execution. A non-zero exit
894
- // denies the tool with POLICY_DENIED — the real tool never runs.
895
- if (preHooks.length > 0) {
896
- for (const hook of preHooks) {
1016
+ // Hooks: matching preToolUse hooks run before execution. A non-zero
1017
+ // exit denies the tool with POLICY_DENIED — the real tool never runs.
1018
+ // Matchers scope hooks per tool; stdin carries the structured payload.
1019
+ // A structured JSON verdict wins over the exit code: deny blocks with
1020
+ // its message, allow+context attaches model-visible context.
1021
+ const hookContext = [];
1022
+ const matchingPre = hooksForEvent(runHooks, 'preToolUse', call.name);
1023
+ if (matchingPre.length > 0) {
1024
+ for (const hook of matchingPre) {
897
1025
  let exitCode = -1;
898
1026
  let detail = '';
1027
+ let verdict;
899
1028
  try {
900
- const r = await runHook(hook, { toolName: call.name, input: call.input });
1029
+ const r = await runHook(hook, { toolName: call.name, input: call.input }, { event: 'preToolUse', tool: call.name, input: call.input, sessionId, cwd: opts.cwd });
901
1030
  exitCode = r.exitCode;
902
1031
  detail = (r.stderr || r.stdout || '').slice(0, 300);
1032
+ if (r.verdict)
1033
+ verdict = r.verdict;
903
1034
  }
904
1035
  catch (err) {
905
1036
  detail = String(err instanceof Error ? err.message : err).slice(0, 300);
906
1037
  }
907
- if (exitCode !== 0) {
908
- const reason = `hook ${hook.name} denied: ${detail || 'hook failed'}`;
1038
+ if (verdict?.decision === 'deny' || exitCode !== 0) {
1039
+ const reason = `hook ${hook.name} denied: ${verdict?.message || detail || 'hook failed'}`;
909
1040
  emit?.({ kind: 'policy_decision', id: call.id, name: call.name, action: 'deny', reason });
910
1041
  emitKlyro({ type: 'permission.decision', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, action: 'deny', reason });
911
1042
  const latencyMs = Date.now() - t0;
912
1043
  return {
913
1044
  obs: { ok: false, error: { code: 'POLICY_DENIED', message: reason } },
914
1045
  latencyMs,
1046
+ hookContext,
915
1047
  };
916
1048
  }
1049
+ // Allow verdicts may carry model-visible context (sliced at parse).
1050
+ if (verdict?.decision !== 'deny' && typeof verdict?.context === 'string' && verdict.context) {
1051
+ hookContext.push({ hook: hook.name, context: verdict.context });
1052
+ }
917
1053
  }
918
1054
  }
919
1055
  let obs;
@@ -924,7 +1060,7 @@ export async function run(opts, deps) {
924
1060
  obs = { ok: false, error: { code: 'EXEC_CRASH', message: err instanceof Error ? err.message : String(err) } };
925
1061
  }
926
1062
  const latencyMs = Date.now() - t0;
927
- return { obs, latencyMs };
1063
+ return { obs, latencyMs, hookContext };
928
1064
  };
929
1065
  // 5.2 stuck termination (P0-3): the FIRST detection injects one
930
1066
  // "change approach" synthetic message for the next model turn; the SECOND
@@ -945,7 +1081,7 @@ export async function run(opts, deps) {
945
1081
  };
946
1082
  // Commit phase: fold one execution result into the transcript, in original
947
1083
  // call order. The only writer — call sequentially, never concurrently.
948
- const commitResult = async (call, obs, latencyMs) => {
1084
+ const commitResult = async (call, obs, latencyMs, hookContext = []) => {
949
1085
  const output = obs.ok ? redactOutput(obs.value) : redactOutput({ error: obs.error });
950
1086
  const toolMsg = {
951
1087
  role: 'tool',
@@ -953,6 +1089,16 @@ export async function run(opts, deps) {
953
1089
  };
954
1090
  transcript.push(toolMsg);
955
1091
  await checkpoint(toolMsg, { toolCallId: call.id, toolName: call.name, input: call.input, output, isError: !obs.ok });
1092
+ // Hook-injected context rides as its own user message right after the
1093
+ // tool result (uniform across output shapes — no result surgery).
1094
+ if (hookContext.length > 0) {
1095
+ const note = {
1096
+ role: 'user',
1097
+ content: [text(hookContext.map((h) => `[hook ${h.hook} context]\n${h.context}`).join('\n\n'))],
1098
+ };
1099
+ transcript.push(note);
1100
+ await checkpoint(note);
1101
+ }
956
1102
  if (obs.ok) {
957
1103
  telemetry.recordToolCall(call, latencyMs, false);
958
1104
  if (call.name === 'write_file' || call.name === 'edit_file' || call.name === 'multi_edit' || call.name === 'apply_patch') {
@@ -975,6 +1121,8 @@ export async function run(opts, deps) {
975
1121
  }
976
1122
  emitKlyro({ type: 'tool.result', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, name: call.name, output, isError: !obs.ok, latencyMs });
977
1123
  emit?.({ kind: 'tool_result', id: call.id, name: call.name, output, isError: !obs.ok, latencyMs });
1124
+ // Light audit: every executed call completes into the chained log.
1125
+ writeAudit({ kind: 'tool_call_completed', sessionId: auditSessionId ?? 'ephemeral', callId: call.id, isError: !obs.ok, latencyMs, ts: Date.now() });
978
1126
  if (obs.ok) {
979
1127
  const fileChanged = inferFileChanged(call.name, call.input, obs.value);
980
1128
  if (fileChanged) {
@@ -1003,12 +1151,13 @@ export async function run(opts, deps) {
1003
1151
  if (last3.length === 3 && last3[0] === last3[1] && last3[1] === last3[2]) {
1004
1152
  await markStuck(`identical call ×3: ${sig}`);
1005
1153
  }
1006
- // Hooks: postToolUse hooks are best-effort — failures warn on stderr
1007
- // plus a bus event, and never fail the turn.
1008
- if (postHooks.length > 0) {
1009
- for (const hook of postHooks) {
1154
+ // Hooks: matching postToolUse hooks are best-effort — failures warn
1155
+ // on stderr plus a bus event, and never fail the turn.
1156
+ const matchingPost = hooksForEvent(runHooks, 'postToolUse', call.name);
1157
+ if (matchingPost.length > 0) {
1158
+ for (const hook of matchingPost) {
1010
1159
  try {
1011
- const r = await runHook(hook, { toolName: call.name, input: call.input });
1160
+ const r = await runHook(hook, { toolName: call.name, input: call.input }, { event: 'postToolUse', tool: call.name, input: call.input, sessionId, cwd: opts.cwd });
1012
1161
  if (!r.ok || r.exitCode !== 0) {
1013
1162
  const msg = `klyro: hooks: postToolUse ${hook.name} failed (exit ${String(r.exitCode)}): ${(r.stderr || r.stdout || '').slice(0, 200)}\n`;
1014
1163
  try {
@@ -1033,8 +1182,8 @@ export async function run(opts, deps) {
1033
1182
  const runOne = async (call) => {
1034
1183
  if (!(await gateCall(call)))
1035
1184
  return;
1036
- const { obs, latencyMs } = await execTool(call);
1037
- await commitResult(call, obs, latencyMs);
1185
+ const { obs, latencyMs, hookContext } = await execTool(call);
1186
+ await commitResult(call, obs, latencyMs, hookContext);
1038
1187
  };
1039
1188
  // 3.5 — parallel when every call is concurrencySafe, sequential otherwise.
1040
1189
  // Gate runs sequentially in both paths (approval UI is one-at-a-time).
@@ -1064,7 +1213,7 @@ export async function run(opts, deps) {
1064
1213
  for (let i = 0; i < settled.length; i++) {
1065
1214
  const s = settled[i];
1066
1215
  if (s.status === 'fulfilled') {
1067
- await commitResult(approved[i], s.value.obs, s.value.latencyMs);
1216
+ await commitResult(approved[i], s.value.obs, s.value.latencyMs, s.value.hookContext);
1068
1217
  }
1069
1218
  else {
1070
1219
  await commitResult(approved[i], { ok: false, error: { code: 'EXEC_CRASH', message: String(s.reason) } }, 0);
@@ -1093,6 +1242,28 @@ export async function run(opts, deps) {
1093
1242
  }
1094
1243
  }
1095
1244
  catch { /* ignore — completions are best-effort visibility */ }
1245
+ // stop hooks: run once per completed step (blocking). A structured
1246
+ // `{"continue": true, "message": ...}` verdict asks for one more turn
1247
+ // instead of completing — consumed at the completion point below,
1248
+ // bounded to 3 continuations per run so a hook can't loop forever.
1249
+ stopCont = null; // fresh verdict per step; stale ones never carry over
1250
+ for (const hook of hooksForEvent(runHooks, 'stop')) {
1251
+ try {
1252
+ const r = await runHook(hook, { toolName: '', input: {} }, { event: 'stop', sessionId, cwd: opts.cwd, step: steps, status: 'open' });
1253
+ if (!r.ok) {
1254
+ try {
1255
+ process.stderr.write(`klyro: hooks: stop ${hook.name} failed (exit ${String(r.exitCode)})\n`);
1256
+ }
1257
+ catch { /* ignore */ }
1258
+ }
1259
+ else if (r.verdict?.cont === true && stopContUsed < 3) {
1260
+ stopCont = typeof r.verdict.message === 'string' && r.verdict.message
1261
+ ? r.verdict.message
1262
+ : `stop hook ${hook.name} requested continuation`;
1263
+ }
1264
+ }
1265
+ catch { /* ignore — stop hooks never fail the turn */ }
1266
+ }
1096
1267
  emit?.({ kind: 'step_end', step: steps });
1097
1268
  emitKlyro({ type: 'turn.end', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', turn: steps });
1098
1269
  // Level 9 — checkpoint status after each step
package/dist/chat.d.ts CHANGED
@@ -20,6 +20,8 @@ export interface ChatOptions {
20
20
  }
21
21
  /** Strip a trailing slash so we can append /chat/completions cleanly. */
22
22
  export declare function normalizeBaseURL(url: string): string;
23
+ /** True for loopback hostnames (localhost / 127.0.0.0/8 / ::1). */
24
+ export declare function isLoopbackHost(host: string): boolean;
23
25
  /**
24
26
  * Validate that the base URL is HTTPS (or localhost over HTTP for local LLMs).
25
27
  * Refuses to send the bearer token over a plaintext remote connection.
@@ -31,6 +33,14 @@ export declare function normalizeBaseURL(url: string): string;
31
33
  export declare function assertSafeBaseURL(url: string, opts?: {
32
34
  allowInsecure?: boolean;
33
35
  }): void;
36
+ /**
37
+ * Remote MCP URL guard. Mirrors `assertSafeBaseURL`'s fail-closed posture but
38
+ * is named for MCP config/transports so callers do not have to import provider
39
+ * URL text.
40
+ */
41
+ export declare function assertSafeRemoteURL(url: string, opts?: {
42
+ allowInsecure?: boolean;
43
+ }): void;
34
44
  export declare function chat(prompt: string, system: string, modelOverride?: string, opts?: ChatOptions): Promise<void>;
35
45
  /**
36
46
  * Parse SSE frames and write text deltas to stdout. Handles
package/dist/chat.js CHANGED
@@ -18,6 +18,17 @@ const MAX_ERROR_BODY_BYTES = 4_000;
18
18
  export function normalizeBaseURL(url) {
19
19
  return url.replace(/\/+$/, '');
20
20
  }
21
+ /** True for loopback hostnames (localhost / 127.0.0.0/8 / ::1). */
22
+ export function isLoopbackHost(host) {
23
+ const h = host.toLowerCase().replace(/^\[|\]$/g, '');
24
+ if (h === 'localhost' || h === '127.0.0.1' || h === '::1' || h === '0.0.0.0' || h === '::')
25
+ return true;
26
+ if (h.startsWith('127.')) {
27
+ const parts = h.split('.');
28
+ return parts.length === 4 && parts.every((p) => /^\d+$/.test(p) && Number(p) >= 0 && Number(p) <= 255);
29
+ }
30
+ return false;
31
+ }
21
32
  /**
22
33
  * Validate that the base URL is HTTPS (or localhost over HTTP for local LLMs).
23
34
  * Refuses to send the bearer token over a plaintext remote connection.
@@ -41,14 +52,10 @@ export function assertSafeBaseURL(url, opts) {
41
52
  if (opts?.allowInsecure === true || process.env.KLYRO_ALLOW_INSECURE === '1')
42
53
  return;
43
54
  const host = parsed.hostname.toLowerCase();
44
- // Allow loopback and private networks without extra flag
45
- if (host === 'localhost' || host === '127.0.0.1' || host === '::1' || host === '0.0.0.0' || host === '::' || host === '[::]')
55
+ // Allow loopback without flag (shared helper); private LAN ranges keep
56
+ // the provider-path behavior below.
57
+ if (isLoopbackHost(host))
46
58
  return;
47
- if (host.startsWith('127.')) {
48
- const parts = host.split('.');
49
- if (parts.length === 4 && parts.every((p) => /^\d+$/.test(p) && Number(p) >= 0 && Number(p) <= 255))
50
- return;
51
- }
52
59
  // Private ranges 10/8, 192.168/16, 172.16-31/12
53
60
  if (/^10\.\d+\.\d+\.\d+$/.test(host))
54
61
  return;
@@ -62,6 +69,31 @@ export function assertSafeBaseURL(url, opts) {
62
69
  }
63
70
  throw new Error(`Unsupported KLYRO_BASE_URL protocol: ${parsed.protocol}`);
64
71
  }
72
+ /**
73
+ * Remote MCP URL guard. Mirrors `assertSafeBaseURL`'s fail-closed posture but
74
+ * is named for MCP config/transports so callers do not have to import provider
75
+ * URL text.
76
+ */
77
+ export function assertSafeRemoteURL(url, opts) {
78
+ let parsed;
79
+ try {
80
+ parsed = new URL(url);
81
+ }
82
+ catch {
83
+ throw new Error(`invalid remote MCP URL: ${url}`);
84
+ }
85
+ if (parsed.protocol === 'https:')
86
+ return;
87
+ if (parsed.protocol === 'http:') {
88
+ if (isLoopbackHost(parsed.hostname))
89
+ return;
90
+ if (opts?.allowInsecure === true || process.env.KLYRO_ALLOW_INSECURE === '1')
91
+ return;
92
+ throw new Error(`Refusing to connect to remote MCP server over plaintext HTTP to ${parsed.hostname}. ` +
93
+ `Use https:// or a loopback URL, or set KLYRO_ALLOW_INSECURE=1 for this terminal only (not recommended).`);
94
+ }
95
+ throw new Error(`unsupported remote MCP URL protocol: ${parsed.protocol}`);
96
+ }
65
97
  export async function chat(prompt, system, modelOverride, opts = {}) {
66
98
  const baseURL = opts.baseURL ?? process.env.KLYRO_BASE_URL ?? 'https://api.openai.com/v1';
67
99
  const apiKey = opts.apiKey ?? process.env.KLYRO_API_KEY;
@@ -12,6 +12,17 @@
12
12
  */
13
13
  export declare function snapshot(cwd: string, files: string[]): Promise<string>;
14
14
  export declare function listCheckpoints(cwd: string): Promise<string[]>;
15
+ export interface CheckpointInfo {
16
+ /** 1-based index from the latest (1 = newest, like `undo(n)`). */
17
+ index: number;
18
+ id: string;
19
+ ts: number;
20
+ files: number;
21
+ }
22
+ /** Numbered snapshot list for `/checkpoints` and the `/rewind` menu. */
23
+ export declare function listCheckpointInfo(cwd: string): Promise<CheckpointInfo[]>;
24
+ /** File paths a snapshot would restore (for `/rewind <n> preview`). */
25
+ export declare function snapshotFiles(cwd: string, id: string): Promise<string[]>;
15
26
  export declare function diff(cwd: string, id?: string): Promise<string>;
16
27
  export declare function undo(cwd: string, n?: number): Promise<void>;
17
28
  export declare function rewind(cwd: string): Promise<void>;
@@ -154,6 +154,38 @@ export async function listCheckpoints(cwd) {
154
154
  return [];
155
155
  }
156
156
  }
157
+ /** Numbered snapshot list for `/checkpoints` and the `/rewind` menu. */
158
+ export async function listCheckpointInfo(cwd) {
159
+ const ids = await listCheckpoints(cwd);
160
+ const out = [];
161
+ for (let i = ids.length - 1; i >= 0; i--) {
162
+ const id = ids[i];
163
+ let ts = 0;
164
+ let files = 0;
165
+ try {
166
+ const meta = JSON.parse(await fs.readFile(path.join(ckptDir(cwd), id, '.meta.json'), 'utf-8'));
167
+ if (typeof meta.ts === 'number')
168
+ ts = meta.ts;
169
+ if (Array.isArray(meta.files))
170
+ files = meta.files.length;
171
+ }
172
+ catch { /* best-effort */ }
173
+ out.push({ index: ids.length - i, id, ts, files });
174
+ }
175
+ return out;
176
+ }
177
+ /** File paths a snapshot would restore (for `/rewind <n> preview`). */
178
+ export async function snapshotFiles(cwd, id) {
179
+ try {
180
+ const meta = JSON.parse(await fs.readFile(path.join(ckptDir(cwd), id, '.meta.json'), 'utf-8'));
181
+ const files = Array.isArray(meta.files) ? meta.files : [];
182
+ const missing = Array.isArray(meta.missing) ? meta.missing.map((f) => `${f} (deleted)`) : [];
183
+ return [...files, ...missing];
184
+ }
185
+ catch {
186
+ return [];
187
+ }
188
+ }
157
189
  export async function diff(cwd, id) {
158
190
  const ckpts = await listCheckpoints(cwd);
159
191
  const target = id ?? ckpts[ckpts.length - 1];