klyro 1.0.5 → 1.0.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +13 -0
- package/dist/agent/custom-agents.d.ts +3 -0
- package/dist/agent/custom-agents.js +96 -0
- package/dist/agent/orchestrator.d.ts +26 -0
- package/dist/agent/orchestrator.js +41 -4
- package/dist/agent/runtime.d.ts +15 -0
- package/dist/agent/runtime.js +232 -61
- package/dist/chat.d.ts +10 -0
- package/dist/chat.js +39 -7
- package/dist/checkpoints/store.d.ts +11 -0
- package/dist/checkpoints/store.js +32 -0
- package/dist/cli/auth.d.ts +10 -3
- package/dist/cli/auth.js +43 -5
- package/dist/cli/completion.js +2 -2
- package/dist/cli/config.d.ts +4 -4
- package/dist/cli/doctor.js +0 -1
- package/dist/cli/eval.d.ts +15 -1
- package/dist/cli/eval.js +43 -5
- package/dist/cli/hooks.d.ts +74 -5
- package/dist/cli/hooks.js +118 -7
- package/dist/cli/init.d.ts +6 -0
- package/dist/cli/init.js +60 -0
- package/dist/cli/keychain.d.ts +10 -0
- package/dist/cli/keychain.js +86 -0
- package/dist/cli/repl.js +188 -30
- package/dist/cli/run.d.ts +7 -1
- package/dist/cli/run.js +92 -50
- package/dist/cli/setup.js +3 -2
- package/dist/cli/slash/custom.d.ts +25 -0
- package/dist/cli/slash/custom.js +166 -0
- package/dist/cli/slash/parser.d.ts +9 -1
- package/dist/cli/slash/parser.js +34 -9
- package/dist/cli/update.d.ts +3 -1
- package/dist/cli/update.js +16 -1
- package/dist/context/accounting.d.ts +6 -0
- package/dist/context/accounting.js +8 -2
- package/dist/context/compaction.d.ts +2 -1
- package/dist/context/compaction.js +39 -12
- package/dist/context/memory.d.ts +11 -0
- package/dist/context/memory.js +59 -4
- package/dist/eval/harness.d.ts +40 -5
- package/dist/eval/harness.js +103 -10
- package/dist/eval/judge.d.ts +32 -0
- package/dist/eval/judge.js +63 -0
- package/dist/eval/tasks.js +134 -0
- package/dist/index.js +239 -130
- package/dist/mcp/auth.d.ts +85 -0
- package/dist/mcp/auth.js +249 -0
- package/dist/mcp/client.d.ts +15 -0
- package/dist/mcp/client.js +42 -2
- package/dist/mcp/config.d.ts +31 -1
- package/dist/mcp/config.js +84 -1
- package/dist/mcp/registry.d.ts +19 -0
- package/dist/mcp/registry.js +118 -2
- package/dist/mcp/remote.d.ts +36 -0
- package/dist/mcp/remote.js +207 -0
- package/dist/mcp/sse.d.ts +42 -0
- package/dist/mcp/sse.js +310 -0
- package/dist/persistence/audit.d.ts +15 -3
- package/dist/persistence/audit.js +84 -13
- package/dist/persistence/store.d.ts +9 -0
- package/dist/persistence/store.js +17 -0
- package/dist/policy/approval.d.ts +15 -1
- package/dist/policy/approval.js +8 -0
- package/dist/policy/engine.js +9 -0
- package/dist/providers/endpoints.d.ts +43 -0
- package/dist/providers/endpoints.js +104 -0
- package/dist/providers.js +17 -14
- package/dist/tools/shell/shell-exec.d.ts +13 -0
- package/dist/tools/shell/shell-exec.js +64 -2
- package/dist/tui/app.js +172 -15
- package/dist/tui/app.test.js +27 -2
- package/dist/tui/approval.js +55 -1
- package/dist/tui/scroll-model.d.ts +2 -2
- package/dist/tui/scroll-model.js +9 -3
- package/dist/tui/tokens.d.ts +8 -11
- package/dist/tui/tokens.js +18 -11
- package/package.json +1 -1
package/dist/agent/runtime.js
CHANGED
|
@@ -23,14 +23,15 @@ import { verify, diagnosticForModel } from '../verification/engine.js';
|
|
|
23
23
|
import { detectVerifyCommand } from '../verification/auto.js';
|
|
24
24
|
import { ensureBaseline, getBaseline } from '../verification/baseline.js';
|
|
25
25
|
import { compressTranscript, totalTokens, calibrateEstimate, transcriptCharLength } from '../context/tokenizer.js';
|
|
26
|
-
import { capForModel } from '../context/accounting.js';
|
|
26
|
+
import { capForModel, RESERVE_OUTPUT_TOKENS } from '../context/accounting.js';
|
|
27
|
+
import { shouldRemind, reminderForTodos } from '../context/memory.js';
|
|
27
28
|
import { ratesFor, isAnthropicModel } from '../providers/model-info.js';
|
|
28
29
|
import { classifyFailure, rerunOnce, gatherRepairContext, guardRepair } from '../verification/classify.js';
|
|
29
30
|
import { findRelatedTests, buildScopedCommand, runScopedVerify, syntaxCheck, checkImports } from '../verification/scoped.js';
|
|
30
31
|
import { globalBus } from '../events/bus.js';
|
|
31
32
|
import { TraceWriter } from '../trace/writer.js';
|
|
32
33
|
import { killAllJobs } from '../tools/shell/background.js';
|
|
33
|
-
import { loadHooks, runHook } from '../cli/hooks.js';
|
|
34
|
+
import { loadHooks, runHook, hooksForEvent } from '../cli/hooks.js';
|
|
34
35
|
/** Normalize either systemPrompt shape into {system, suffix}. */
|
|
35
36
|
export function resolveSystemPrompt(fn, ctx) {
|
|
36
37
|
const r = fn(ctx);
|
|
@@ -118,6 +119,7 @@ export async function run(opts, deps) {
|
|
|
118
119
|
}
|
|
119
120
|
};
|
|
120
121
|
let steps = 0;
|
|
122
|
+
let lastRemindTurn = 0;
|
|
121
123
|
let toolCallCount = 0;
|
|
122
124
|
let finalText = '';
|
|
123
125
|
let repairs = 0;
|
|
@@ -154,16 +156,21 @@ export async function run(opts, deps) {
|
|
|
154
156
|
emit?.({ kind: 'model_override', requested: opts.model, effective: opts.parentContext.model });
|
|
155
157
|
}
|
|
156
158
|
// Hooks engine: loaded once per run. Zero-cost fast path — when no hooks
|
|
157
|
-
// file exists,
|
|
159
|
+
// file exists, the list is empty and every hook call site is skipped.
|
|
160
|
+
// --bare skips hooks entirely (deterministic runs).
|
|
161
|
+
// Tool-event hooks are matched per tool at the call sites below
|
|
162
|
+
// (hooksForEvent over runHooks); lifecycle events run at their own points.
|
|
158
163
|
let runHooks = [];
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
+
if (!opts.bare) {
|
|
165
|
+
try {
|
|
166
|
+
runHooks = loadHooks(opts.cwd);
|
|
167
|
+
}
|
|
168
|
+
catch {
|
|
169
|
+
runHooks = [];
|
|
170
|
+
}
|
|
164
171
|
}
|
|
165
|
-
|
|
166
|
-
|
|
172
|
+
// Tool-event hooks are matched per tool at the call sites below
|
|
173
|
+
// (hooksForEvent over runHooks); lifecycle events run at their own points.
|
|
167
174
|
// L15 failover chain: the active adapter starts as deps.adapter; each
|
|
168
175
|
// terminal provider error consumes one fallback. Bounded — never loops.
|
|
169
176
|
let activeAdapter = deps.adapter;
|
|
@@ -179,6 +186,16 @@ export async function run(opts, deps) {
|
|
|
179
186
|
bus.emit(ev);
|
|
180
187
|
tracer?.write(ev).catch(() => undefined);
|
|
181
188
|
};
|
|
189
|
+
// Light audit writer: mirrors policy decisions + tool results into the
|
|
190
|
+
// chained audit log. Best-effort — audit errors must not break the run.
|
|
191
|
+
const auditLog = opts.audit?.log;
|
|
192
|
+
const auditSessionId = opts.audit?.sessionId ?? opts.persist?.sessionId;
|
|
193
|
+
const writeAudit = (ev) => {
|
|
194
|
+
if (!auditLog || !auditSessionId)
|
|
195
|
+
return;
|
|
196
|
+
// fire-and-forget — the log serializes its own chain internally
|
|
197
|
+
void auditLog.write(ev).catch(() => undefined);
|
|
198
|
+
};
|
|
182
199
|
const closeTracer = async () => {
|
|
183
200
|
try {
|
|
184
201
|
await tracer?.close();
|
|
@@ -247,11 +264,35 @@ export async function run(opts, deps) {
|
|
|
247
264
|
// Fire-and-forget initial checkpoint (don't await to block loop start)
|
|
248
265
|
void checkpoint(transcript[transcript.length - 1]);
|
|
249
266
|
}
|
|
267
|
+
// sessionStart: prerequisite gate. A non-zero exit aborts the run before
|
|
268
|
+
// step 1 with status 'blocked' (e.g. missing toolchain, dirty tree).
|
|
269
|
+
{
|
|
270
|
+
const starters = hooksForEvent(runHooks, 'sessionStart');
|
|
271
|
+
for (const hook of starters) {
|
|
272
|
+
let r = null;
|
|
273
|
+
try {
|
|
274
|
+
r = await runHook(hook, { toolName: '', input: {} }, { event: 'sessionStart', sessionId, cwd: opts.cwd, task: opts.task });
|
|
275
|
+
}
|
|
276
|
+
catch {
|
|
277
|
+
r = null;
|
|
278
|
+
}
|
|
279
|
+
if (r && !r.ok) {
|
|
280
|
+
const reason = (r.stderr || r.stdout || 'sessionStart hook failed').slice(0, 500);
|
|
281
|
+
emitKlyro({ type: 'error', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', code: 'session_blocked', message: reason });
|
|
282
|
+
await closeTracer();
|
|
283
|
+
return { status: 'blocked', steps, toolCalls: toolCallCount, finalText: `Blocked by sessionStart hook ${hook.name}: ${reason}`, transcript, hasEdits, usage, repairs, phase: 'blocked' };
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
}
|
|
250
287
|
// 5.2 — stuck detection state
|
|
251
288
|
const callHistory = [];
|
|
252
289
|
const fileEditCounts = new Map();
|
|
253
290
|
let stuckTriggers = 0;
|
|
254
291
|
let stuckAbort = false;
|
|
292
|
+
// Steerable stop: a stop hook's `{"continue":true}` verdict carries one
|
|
293
|
+
// more turn. Consumed once at the completion point, max 3 per run.
|
|
294
|
+
let stopCont = null;
|
|
295
|
+
let stopContUsed = 0;
|
|
255
296
|
outer: while (steps < maxSteps) {
|
|
256
297
|
// 5.1 limits: max-cost, max-time
|
|
257
298
|
if (maxCost !== undefined) {
|
|
@@ -282,6 +323,33 @@ export async function run(opts, deps) {
|
|
|
282
323
|
return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined, phase: 'blocked' };
|
|
283
324
|
}
|
|
284
325
|
steps++;
|
|
326
|
+
// 8.4 — stale-todo reminder: every 20 turns, re-inject pending plan
|
|
327
|
+
// items from `.klyro/plans/todos.json` (written by todo_write) so a
|
|
328
|
+
// long run cannot silently drop its checklist. Best-effort + tiny.
|
|
329
|
+
if (shouldRemind(steps, lastRemindTurn)) {
|
|
330
|
+
lastRemindTurn = steps;
|
|
331
|
+
try {
|
|
332
|
+
const { readFileSync } = await import('node:fs');
|
|
333
|
+
const { join } = await import('node:path');
|
|
334
|
+
const rawTodos = JSON.parse(readFileSync(join(opts.cwd, '.klyro', 'plans', 'todos.json'), 'utf-8'));
|
|
335
|
+
if (Array.isArray(rawTodos)) {
|
|
336
|
+
const planSteps = rawTodos
|
|
337
|
+
.filter((t) => typeof t.title === 'string')
|
|
338
|
+
.map((t, i) => ({
|
|
339
|
+
id: `todo-${i}`,
|
|
340
|
+
title: t.title,
|
|
341
|
+
status: ['pending', 'in_progress', 'done', 'failed', 'skipped'].includes(t.status) ? t.status : 'pending',
|
|
342
|
+
}));
|
|
343
|
+
const reminder = reminderForTodos(planSteps);
|
|
344
|
+
if (reminder) {
|
|
345
|
+
const note = { role: 'user', content: [text(`[system note] ${reminder}`)] };
|
|
346
|
+
transcript.push(note);
|
|
347
|
+
await checkpoint(note);
|
|
348
|
+
}
|
|
349
|
+
}
|
|
350
|
+
}
|
|
351
|
+
catch { /* no todos file — nothing to remind */ }
|
|
352
|
+
}
|
|
285
353
|
// 5.1 phase transitions (model-narrated)
|
|
286
354
|
if (steps === 1)
|
|
287
355
|
setPhase('understanding');
|
|
@@ -304,7 +372,7 @@ export async function run(opts, deps) {
|
|
|
304
372
|
const systemForBudget = telemetrySuffix ? `${stableSystem}\n\n${telemetrySuffix}` : stableSystem;
|
|
305
373
|
// Window-aware ceiling (was a hardcoded 120k that overflowed 8k local
|
|
306
374
|
// models): size the input budget to the model's context window.
|
|
307
|
-
const BUDGET = { total: capForModel(opts.model,
|
|
375
|
+
const BUDGET = { total: capForModel(opts.model, RESERVE_OUTPUT_TOKENS), reservedOutput: RESERVE_OUTPUT_TOKENS };
|
|
308
376
|
let reqMessages = transcript;
|
|
309
377
|
let reqSystem = stableSystem;
|
|
310
378
|
let reqSuffix = telemetrySuffix;
|
|
@@ -410,7 +478,7 @@ export async function run(opts, deps) {
|
|
|
410
478
|
telemetry.recordError('overflow_retry');
|
|
411
479
|
emitKlyro({ type: 'error', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', code: 'REQUEST_TOO_LARGE', message: 'context overflow — aggressively compacting transcript and retrying the request once' });
|
|
412
480
|
try {
|
|
413
|
-
const compacted = compressTranscript(reqSystem, transcript, { total: 30_000, reservedOutput:
|
|
481
|
+
const compacted = compressTranscript(reqSystem, transcript, { total: 30_000, reservedOutput: RESERVE_OUTPUT_TOKENS });
|
|
414
482
|
if (compacted.messages.length < transcript.length || compacted.dropped > 0) {
|
|
415
483
|
transcript.splice(0, transcript.length, ...compacted.messages);
|
|
416
484
|
}
|
|
@@ -418,7 +486,7 @@ export async function run(opts, deps) {
|
|
|
418
486
|
// Transcript already fits the aggressive budget — force it
|
|
419
487
|
// strictly smaller so the retry cannot repeat the overflow.
|
|
420
488
|
const halved = Math.max(4000, Math.floor(totalTokens(reqSystem, transcript) / 2));
|
|
421
|
-
const smaller = compressTranscript(reqSystem, transcript, { total: halved, reservedOutput:
|
|
489
|
+
const smaller = compressTranscript(reqSystem, transcript, { total: halved, reservedOutput: RESERVE_OUTPUT_TOKENS });
|
|
422
490
|
transcript.splice(0, transcript.length, ...smaller.messages);
|
|
423
491
|
}
|
|
424
492
|
tokenCache = { lastRef: null, lastSystem: undefined, lastCount: 0 };
|
|
@@ -549,6 +617,17 @@ export async function run(opts, deps) {
|
|
|
549
617
|
return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined };
|
|
550
618
|
}
|
|
551
619
|
if (finalizedCalls.length === 0) {
|
|
620
|
+
// Steerable stop: a stop hook asked for one more turn instead of
|
|
621
|
+
// completing. Consumed once per verdict, max 3 per run.
|
|
622
|
+
if (stopCont !== null && stopContUsed < 3) {
|
|
623
|
+
stopContUsed++;
|
|
624
|
+
const contMsg = { role: 'user', content: [text(`[system note] ${stopCont}`)] };
|
|
625
|
+
stopCont = null;
|
|
626
|
+
transcript.push(contMsg);
|
|
627
|
+
await checkpoint(contMsg);
|
|
628
|
+
emit?.({ kind: 'step_end', step: steps });
|
|
629
|
+
continue;
|
|
630
|
+
}
|
|
552
631
|
if (hadInvalidTool) {
|
|
553
632
|
// The model attempted a tool call that failed validation; the error is
|
|
554
633
|
// already a tool_result in the transcript — loop so the model can
|
|
@@ -833,39 +912,49 @@ export async function run(opts, deps) {
|
|
|
833
912
|
// results immediately — gate runs in call order so these stay ordered.
|
|
834
913
|
// Returns true when the call is approved for execution.
|
|
835
914
|
const gateCall = async (call) => {
|
|
836
|
-
|
|
837
|
-
|
|
838
|
-
//
|
|
839
|
-
|
|
840
|
-
|
|
841
|
-
|
|
842
|
-
|
|
843
|
-
|
|
844
|
-
|
|
845
|
-
|
|
846
|
-
|
|
847
|
-
|
|
848
|
-
|
|
849
|
-
|
|
850
|
-
|
|
851
|
-
};
|
|
852
|
-
|
|
853
|
-
|
|
854
|
-
|
|
855
|
-
|
|
856
|
-
|
|
857
|
-
|
|
858
|
-
|
|
859
|
-
|
|
915
|
+
// Edit-and-retry loop: an `e`dit choice re-validates + re-evaluates
|
|
916
|
+
// policy on the edited input (bounded to 3 rounds so a user can't be
|
|
917
|
+
// re-prompted forever). `call.input` is updated in place so the
|
|
918
|
+
// executed + checkpointed call reflects what was approved.
|
|
919
|
+
let effectiveInput = call.input;
|
|
920
|
+
for (let round = 0; round < 3; round++) {
|
|
921
|
+
const decision = await deps.policy.evaluate({ name: call.name, input: effectiveInput, permission: deps.registry.get(call.name)?.permission }, { cwd: opts.cwd, nonInteractive: opts.nonInteractive });
|
|
922
|
+
emit?.({ kind: 'policy_decision', id: call.id, name: call.name, action: decision.action, ...(decision.action !== 'allow' ? { reason: decision.reason } : {}) });
|
|
923
|
+
// Mirror to KlyroEvent bus
|
|
924
|
+
if (decision.action === 'allow') {
|
|
925
|
+
emitKlyro({ type: 'permission.decision', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, action: 'allow' });
|
|
926
|
+
}
|
|
927
|
+
else {
|
|
928
|
+
emitKlyro({ type: 'permission.decision', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, action: decision.action, reason: decision.reason });
|
|
929
|
+
}
|
|
930
|
+
writeAudit({ kind: 'policy_decision', sessionId: auditSessionId ?? 'ephemeral', callId: call.id, action: decision.action, ts: Date.now() });
|
|
931
|
+
if (decision.action === 'deny') {
|
|
932
|
+
const denyMsg = {
|
|
933
|
+
role: 'tool',
|
|
934
|
+
content: [
|
|
935
|
+
mkToolResult(call.id, call.name, { error: 'POLICY_DENIED', reason: decision.reason }, true),
|
|
936
|
+
],
|
|
937
|
+
};
|
|
938
|
+
transcript.push(denyMsg);
|
|
939
|
+
await checkpoint(denyMsg, { toolCallId: call.id, toolName: call.name, input: effectiveInput, output: { error: 'POLICY_DENIED', reason: decision.reason }, isError: true });
|
|
940
|
+
emitKlyro({ type: 'tool.result', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, name: call.name, output: { error: 'POLICY_DENIED' }, isError: true, latencyMs: 0 });
|
|
941
|
+
telemetry.recordToolError(call, 'policy_denied');
|
|
942
|
+
emit?.({ kind: 'tool_result', id: call.id, name: call.name, output: { error: 'POLICY_DENIED', reason: decision.reason }, isError: true, latencyMs: 0 });
|
|
943
|
+
return false;
|
|
944
|
+
}
|
|
945
|
+
if (decision.action === 'allow') {
|
|
946
|
+
call.input = effectiveInput;
|
|
947
|
+
return true;
|
|
948
|
+
}
|
|
949
|
+
// decision.action === 'ask'
|
|
860
950
|
emitKlyro({ type: 'permission.ask', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, name: call.name, reason: decision.reason });
|
|
861
951
|
const choice = await deps.approval.ask({
|
|
862
952
|
toolName: call.name,
|
|
863
953
|
reason: decision.reason,
|
|
864
|
-
summary: summarizeToolCall(call),
|
|
865
|
-
input:
|
|
866
|
-
pattern: patternForCall(call.name,
|
|
954
|
+
summary: summarizeToolCall({ ...call, input: effectiveInput }),
|
|
955
|
+
input: effectiveInput,
|
|
956
|
+
pattern: patternForCall(call.name, effectiveInput),
|
|
867
957
|
});
|
|
868
|
-
// Approval UI in TUI handles y/a/A/n/e/? — e edits input, ? explains
|
|
869
958
|
if (choice === 'deny') {
|
|
870
959
|
const denyMsg2 = {
|
|
871
960
|
role: 'tool',
|
|
@@ -874,15 +963,49 @@ export async function run(opts, deps) {
|
|
|
874
963
|
],
|
|
875
964
|
};
|
|
876
965
|
transcript.push(denyMsg2);
|
|
877
|
-
await checkpoint(denyMsg2, { toolCallId: call.id, toolName: call.name, input:
|
|
966
|
+
await checkpoint(denyMsg2, { toolCallId: call.id, toolName: call.name, input: effectiveInput, output: { error: 'POLICY_DENIED', reason: 'user denied' }, isError: true });
|
|
878
967
|
telemetry.recordToolError(call, 'user_denied');
|
|
879
968
|
emit?.({ kind: 'tool_result', id: call.id, name: call.name, output: { error: 'POLICY_DENIED', reason: 'user denied' }, isError: true, latencyMs: 0 });
|
|
880
969
|
return false;
|
|
881
970
|
}
|
|
882
|
-
|
|
971
|
+
if (typeof choice === 'object' && choice.kind === 'edit') {
|
|
972
|
+
// Re-validate the edited input against the tool schema before it
|
|
973
|
+
// goes anywhere — a malformed edit denies instead of executing.
|
|
974
|
+
const tool = deps.registry.get(call.name);
|
|
975
|
+
const parsed = tool?.inputSchema.safeParse(choice.editedInput);
|
|
976
|
+
if (!parsed || !parsed.success) {
|
|
977
|
+
const denyMsg3 = {
|
|
978
|
+
role: 'tool',
|
|
979
|
+
content: [
|
|
980
|
+
mkToolResult(call.id, call.name, { error: 'POLICY_DENIED', reason: 'edited input failed tool schema validation' }, true),
|
|
981
|
+
],
|
|
982
|
+
};
|
|
983
|
+
transcript.push(denyMsg3);
|
|
984
|
+
await checkpoint(denyMsg3, { toolCallId: call.id, toolName: call.name, input: effectiveInput, output: { error: 'POLICY_DENIED', reason: 'edited input invalid' }, isError: true });
|
|
985
|
+
telemetry.recordToolError(call, 'edit_invalid');
|
|
986
|
+
emit?.({ kind: 'tool_result', id: call.id, name: call.name, output: { error: 'POLICY_DENIED', reason: 'edited input invalid' }, isError: true, latencyMs: 0 });
|
|
987
|
+
return false;
|
|
988
|
+
}
|
|
989
|
+
effectiveInput = parsed.data;
|
|
990
|
+
continue; // re-evaluate policy on the edited input
|
|
991
|
+
}
|
|
992
|
+
// allow / always / always-persist — approved with (possibly edited) input.
|
|
883
993
|
repairs++;
|
|
994
|
+
call.input = effectiveInput;
|
|
995
|
+
return true;
|
|
884
996
|
}
|
|
885
|
-
|
|
997
|
+
// Edit rounds exhausted without approval — deny rather than loop forever.
|
|
998
|
+
const denyMsg4 = {
|
|
999
|
+
role: 'tool',
|
|
1000
|
+
content: [
|
|
1001
|
+
mkToolResult(call.id, call.name, { error: 'POLICY_DENIED', reason: 'approval rounds exhausted' }, true),
|
|
1002
|
+
],
|
|
1003
|
+
};
|
|
1004
|
+
transcript.push(denyMsg4);
|
|
1005
|
+
await checkpoint(denyMsg4, { toolCallId: call.id, toolName: call.name, input: effectiveInput, output: { error: 'POLICY_DENIED', reason: 'approval rounds exhausted' }, isError: true });
|
|
1006
|
+
telemetry.recordToolError(call, 'approval_exhausted');
|
|
1007
|
+
emit?.({ kind: 'tool_result', id: call.id, name: call.name, output: { error: 'POLICY_DENIED', reason: 'approval rounds exhausted' }, isError: true, latencyMs: 0 });
|
|
1008
|
+
return false;
|
|
886
1009
|
};
|
|
887
1010
|
// Execute phase: run the tool with no transcript writes, so concurrent
|
|
888
1011
|
// executions can't interleave. A throw here becomes a tool error (an
|
|
@@ -890,30 +1013,43 @@ export async function run(opts, deps) {
|
|
|
890
1013
|
const execTool = async (call) => {
|
|
891
1014
|
const t0 = Date.now();
|
|
892
1015
|
emitKlyro({ type: 'tool.call', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, name: call.name, input: call.input });
|
|
893
|
-
// Hooks:
|
|
894
|
-
// denies the tool with POLICY_DENIED — the real tool never runs.
|
|
895
|
-
|
|
896
|
-
|
|
1016
|
+
// Hooks: matching preToolUse hooks run before execution. A non-zero
|
|
1017
|
+
// exit denies the tool with POLICY_DENIED — the real tool never runs.
|
|
1018
|
+
// Matchers scope hooks per tool; stdin carries the structured payload.
|
|
1019
|
+
// A structured JSON verdict wins over the exit code: deny blocks with
|
|
1020
|
+
// its message, allow+context attaches model-visible context.
|
|
1021
|
+
const hookContext = [];
|
|
1022
|
+
const matchingPre = hooksForEvent(runHooks, 'preToolUse', call.name);
|
|
1023
|
+
if (matchingPre.length > 0) {
|
|
1024
|
+
for (const hook of matchingPre) {
|
|
897
1025
|
let exitCode = -1;
|
|
898
1026
|
let detail = '';
|
|
1027
|
+
let verdict;
|
|
899
1028
|
try {
|
|
900
|
-
const r = await runHook(hook, { toolName: call.name, input: call.input });
|
|
1029
|
+
const r = await runHook(hook, { toolName: call.name, input: call.input }, { event: 'preToolUse', tool: call.name, input: call.input, sessionId, cwd: opts.cwd });
|
|
901
1030
|
exitCode = r.exitCode;
|
|
902
1031
|
detail = (r.stderr || r.stdout || '').slice(0, 300);
|
|
1032
|
+
if (r.verdict)
|
|
1033
|
+
verdict = r.verdict;
|
|
903
1034
|
}
|
|
904
1035
|
catch (err) {
|
|
905
1036
|
detail = String(err instanceof Error ? err.message : err).slice(0, 300);
|
|
906
1037
|
}
|
|
907
|
-
if (exitCode !== 0) {
|
|
908
|
-
const reason = `hook ${hook.name} denied: ${detail || 'hook failed'}`;
|
|
1038
|
+
if (verdict?.decision === 'deny' || exitCode !== 0) {
|
|
1039
|
+
const reason = `hook ${hook.name} denied: ${verdict?.message || detail || 'hook failed'}`;
|
|
909
1040
|
emit?.({ kind: 'policy_decision', id: call.id, name: call.name, action: 'deny', reason });
|
|
910
1041
|
emitKlyro({ type: 'permission.decision', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, action: 'deny', reason });
|
|
911
1042
|
const latencyMs = Date.now() - t0;
|
|
912
1043
|
return {
|
|
913
1044
|
obs: { ok: false, error: { code: 'POLICY_DENIED', message: reason } },
|
|
914
1045
|
latencyMs,
|
|
1046
|
+
hookContext,
|
|
915
1047
|
};
|
|
916
1048
|
}
|
|
1049
|
+
// Allow verdicts may carry model-visible context (sliced at parse).
|
|
1050
|
+
if (verdict?.decision !== 'deny' && typeof verdict?.context === 'string' && verdict.context) {
|
|
1051
|
+
hookContext.push({ hook: hook.name, context: verdict.context });
|
|
1052
|
+
}
|
|
917
1053
|
}
|
|
918
1054
|
}
|
|
919
1055
|
let obs;
|
|
@@ -924,7 +1060,7 @@ export async function run(opts, deps) {
|
|
|
924
1060
|
obs = { ok: false, error: { code: 'EXEC_CRASH', message: err instanceof Error ? err.message : String(err) } };
|
|
925
1061
|
}
|
|
926
1062
|
const latencyMs = Date.now() - t0;
|
|
927
|
-
return { obs, latencyMs };
|
|
1063
|
+
return { obs, latencyMs, hookContext };
|
|
928
1064
|
};
|
|
929
1065
|
// 5.2 stuck termination (P0-3): the FIRST detection injects one
|
|
930
1066
|
// "change approach" synthetic message for the next model turn; the SECOND
|
|
@@ -945,7 +1081,7 @@ export async function run(opts, deps) {
|
|
|
945
1081
|
};
|
|
946
1082
|
// Commit phase: fold one execution result into the transcript, in original
|
|
947
1083
|
// call order. The only writer — call sequentially, never concurrently.
|
|
948
|
-
const commitResult = async (call, obs, latencyMs) => {
|
|
1084
|
+
const commitResult = async (call, obs, latencyMs, hookContext = []) => {
|
|
949
1085
|
const output = obs.ok ? redactOutput(obs.value) : redactOutput({ error: obs.error });
|
|
950
1086
|
const toolMsg = {
|
|
951
1087
|
role: 'tool',
|
|
@@ -953,6 +1089,16 @@ export async function run(opts, deps) {
|
|
|
953
1089
|
};
|
|
954
1090
|
transcript.push(toolMsg);
|
|
955
1091
|
await checkpoint(toolMsg, { toolCallId: call.id, toolName: call.name, input: call.input, output, isError: !obs.ok });
|
|
1092
|
+
// Hook-injected context rides as its own user message right after the
|
|
1093
|
+
// tool result (uniform across output shapes — no result surgery).
|
|
1094
|
+
if (hookContext.length > 0) {
|
|
1095
|
+
const note = {
|
|
1096
|
+
role: 'user',
|
|
1097
|
+
content: [text(hookContext.map((h) => `[hook ${h.hook} context]\n${h.context}`).join('\n\n'))],
|
|
1098
|
+
};
|
|
1099
|
+
transcript.push(note);
|
|
1100
|
+
await checkpoint(note);
|
|
1101
|
+
}
|
|
956
1102
|
if (obs.ok) {
|
|
957
1103
|
telemetry.recordToolCall(call, latencyMs, false);
|
|
958
1104
|
if (call.name === 'write_file' || call.name === 'edit_file' || call.name === 'multi_edit' || call.name === 'apply_patch') {
|
|
@@ -975,6 +1121,8 @@ export async function run(opts, deps) {
|
|
|
975
1121
|
}
|
|
976
1122
|
emitKlyro({ type: 'tool.result', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', callId: call.id, name: call.name, output, isError: !obs.ok, latencyMs });
|
|
977
1123
|
emit?.({ kind: 'tool_result', id: call.id, name: call.name, output, isError: !obs.ok, latencyMs });
|
|
1124
|
+
// Light audit: every executed call completes into the chained log.
|
|
1125
|
+
writeAudit({ kind: 'tool_call_completed', sessionId: auditSessionId ?? 'ephemeral', callId: call.id, isError: !obs.ok, latencyMs, ts: Date.now() });
|
|
978
1126
|
if (obs.ok) {
|
|
979
1127
|
const fileChanged = inferFileChanged(call.name, call.input, obs.value);
|
|
980
1128
|
if (fileChanged) {
|
|
@@ -1003,12 +1151,13 @@ export async function run(opts, deps) {
|
|
|
1003
1151
|
if (last3.length === 3 && last3[0] === last3[1] && last3[1] === last3[2]) {
|
|
1004
1152
|
await markStuck(`identical call ×3: ${sig}`);
|
|
1005
1153
|
}
|
|
1006
|
-
// Hooks: postToolUse hooks are best-effort — failures warn
|
|
1007
|
-
// plus a bus event, and never fail the turn.
|
|
1008
|
-
|
|
1009
|
-
|
|
1154
|
+
// Hooks: matching postToolUse hooks are best-effort — failures warn
|
|
1155
|
+
// on stderr plus a bus event, and never fail the turn.
|
|
1156
|
+
const matchingPost = hooksForEvent(runHooks, 'postToolUse', call.name);
|
|
1157
|
+
if (matchingPost.length > 0) {
|
|
1158
|
+
for (const hook of matchingPost) {
|
|
1010
1159
|
try {
|
|
1011
|
-
const r = await runHook(hook, { toolName: call.name, input: call.input });
|
|
1160
|
+
const r = await runHook(hook, { toolName: call.name, input: call.input }, { event: 'postToolUse', tool: call.name, input: call.input, sessionId, cwd: opts.cwd });
|
|
1012
1161
|
if (!r.ok || r.exitCode !== 0) {
|
|
1013
1162
|
const msg = `klyro: hooks: postToolUse ${hook.name} failed (exit ${String(r.exitCode)}): ${(r.stderr || r.stdout || '').slice(0, 200)}\n`;
|
|
1014
1163
|
try {
|
|
@@ -1033,8 +1182,8 @@ export async function run(opts, deps) {
|
|
|
1033
1182
|
const runOne = async (call) => {
|
|
1034
1183
|
if (!(await gateCall(call)))
|
|
1035
1184
|
return;
|
|
1036
|
-
const { obs, latencyMs } = await execTool(call);
|
|
1037
|
-
await commitResult(call, obs, latencyMs);
|
|
1185
|
+
const { obs, latencyMs, hookContext } = await execTool(call);
|
|
1186
|
+
await commitResult(call, obs, latencyMs, hookContext);
|
|
1038
1187
|
};
|
|
1039
1188
|
// 3.5 — parallel when every call is concurrencySafe, sequential otherwise.
|
|
1040
1189
|
// Gate runs sequentially in both paths (approval UI is one-at-a-time).
|
|
@@ -1064,7 +1213,7 @@ export async function run(opts, deps) {
|
|
|
1064
1213
|
for (let i = 0; i < settled.length; i++) {
|
|
1065
1214
|
const s = settled[i];
|
|
1066
1215
|
if (s.status === 'fulfilled') {
|
|
1067
|
-
await commitResult(approved[i], s.value.obs, s.value.latencyMs);
|
|
1216
|
+
await commitResult(approved[i], s.value.obs, s.value.latencyMs, s.value.hookContext);
|
|
1068
1217
|
}
|
|
1069
1218
|
else {
|
|
1070
1219
|
await commitResult(approved[i], { ok: false, error: { code: 'EXEC_CRASH', message: String(s.reason) } }, 0);
|
|
@@ -1093,6 +1242,28 @@ export async function run(opts, deps) {
|
|
|
1093
1242
|
}
|
|
1094
1243
|
}
|
|
1095
1244
|
catch { /* ignore — completions are best-effort visibility */ }
|
|
1245
|
+
// stop hooks: run once per completed step (blocking). A structured
|
|
1246
|
+
// `{"continue": true, "message": ...}` verdict asks for one more turn
|
|
1247
|
+
// instead of completing — consumed at the completion point below,
|
|
1248
|
+
// bounded to 3 continuations per run so a hook can't loop forever.
|
|
1249
|
+
stopCont = null; // fresh verdict per step; stale ones never carry over
|
|
1250
|
+
for (const hook of hooksForEvent(runHooks, 'stop')) {
|
|
1251
|
+
try {
|
|
1252
|
+
const r = await runHook(hook, { toolName: '', input: {} }, { event: 'stop', sessionId, cwd: opts.cwd, step: steps, status: 'open' });
|
|
1253
|
+
if (!r.ok) {
|
|
1254
|
+
try {
|
|
1255
|
+
process.stderr.write(`klyro: hooks: stop ${hook.name} failed (exit ${String(r.exitCode)})\n`);
|
|
1256
|
+
}
|
|
1257
|
+
catch { /* ignore */ }
|
|
1258
|
+
}
|
|
1259
|
+
else if (r.verdict?.cont === true && stopContUsed < 3) {
|
|
1260
|
+
stopCont = typeof r.verdict.message === 'string' && r.verdict.message
|
|
1261
|
+
? r.verdict.message
|
|
1262
|
+
: `stop hook ${hook.name} requested continuation`;
|
|
1263
|
+
}
|
|
1264
|
+
}
|
|
1265
|
+
catch { /* ignore — stop hooks never fail the turn */ }
|
|
1266
|
+
}
|
|
1096
1267
|
emit?.({ kind: 'step_end', step: steps });
|
|
1097
1268
|
emitKlyro({ type: 'turn.end', ts: Date.now(), sessionId: sessionId ?? 'ephemeral', turn: steps });
|
|
1098
1269
|
// Level 9 — checkpoint status after each step
|
package/dist/chat.d.ts
CHANGED
|
@@ -20,6 +20,8 @@ export interface ChatOptions {
|
|
|
20
20
|
}
|
|
21
21
|
/** Strip a trailing slash so we can append /chat/completions cleanly. */
|
|
22
22
|
export declare function normalizeBaseURL(url: string): string;
|
|
23
|
+
/** True for loopback hostnames (localhost / 127.0.0.0/8 / ::1). */
|
|
24
|
+
export declare function isLoopbackHost(host: string): boolean;
|
|
23
25
|
/**
|
|
24
26
|
* Validate that the base URL is HTTPS (or localhost over HTTP for local LLMs).
|
|
25
27
|
* Refuses to send the bearer token over a plaintext remote connection.
|
|
@@ -31,6 +33,14 @@ export declare function normalizeBaseURL(url: string): string;
|
|
|
31
33
|
export declare function assertSafeBaseURL(url: string, opts?: {
|
|
32
34
|
allowInsecure?: boolean;
|
|
33
35
|
}): void;
|
|
36
|
+
/**
|
|
37
|
+
* Remote MCP URL guard. Mirrors `assertSafeBaseURL`'s fail-closed posture but
|
|
38
|
+
* is named for MCP config/transports so callers do not have to import provider
|
|
39
|
+
* URL text.
|
|
40
|
+
*/
|
|
41
|
+
export declare function assertSafeRemoteURL(url: string, opts?: {
|
|
42
|
+
allowInsecure?: boolean;
|
|
43
|
+
}): void;
|
|
34
44
|
export declare function chat(prompt: string, system: string, modelOverride?: string, opts?: ChatOptions): Promise<void>;
|
|
35
45
|
/**
|
|
36
46
|
* Parse SSE frames and write text deltas to stdout. Handles
|
package/dist/chat.js
CHANGED
|
@@ -18,6 +18,17 @@ const MAX_ERROR_BODY_BYTES = 4_000;
|
|
|
18
18
|
export function normalizeBaseURL(url) {
|
|
19
19
|
return url.replace(/\/+$/, '');
|
|
20
20
|
}
|
|
21
|
+
/** True for loopback hostnames (localhost / 127.0.0.0/8 / ::1). */
|
|
22
|
+
export function isLoopbackHost(host) {
|
|
23
|
+
const h = host.toLowerCase().replace(/^\[|\]$/g, '');
|
|
24
|
+
if (h === 'localhost' || h === '127.0.0.1' || h === '::1' || h === '0.0.0.0' || h === '::')
|
|
25
|
+
return true;
|
|
26
|
+
if (h.startsWith('127.')) {
|
|
27
|
+
const parts = h.split('.');
|
|
28
|
+
return parts.length === 4 && parts.every((p) => /^\d+$/.test(p) && Number(p) >= 0 && Number(p) <= 255);
|
|
29
|
+
}
|
|
30
|
+
return false;
|
|
31
|
+
}
|
|
21
32
|
/**
|
|
22
33
|
* Validate that the base URL is HTTPS (or localhost over HTTP for local LLMs).
|
|
23
34
|
* Refuses to send the bearer token over a plaintext remote connection.
|
|
@@ -41,14 +52,10 @@ export function assertSafeBaseURL(url, opts) {
|
|
|
41
52
|
if (opts?.allowInsecure === true || process.env.KLYRO_ALLOW_INSECURE === '1')
|
|
42
53
|
return;
|
|
43
54
|
const host = parsed.hostname.toLowerCase();
|
|
44
|
-
// Allow loopback
|
|
45
|
-
|
|
55
|
+
// Allow loopback without flag (shared helper); private LAN ranges keep
|
|
56
|
+
// the provider-path behavior below.
|
|
57
|
+
if (isLoopbackHost(host))
|
|
46
58
|
return;
|
|
47
|
-
if (host.startsWith('127.')) {
|
|
48
|
-
const parts = host.split('.');
|
|
49
|
-
if (parts.length === 4 && parts.every((p) => /^\d+$/.test(p) && Number(p) >= 0 && Number(p) <= 255))
|
|
50
|
-
return;
|
|
51
|
-
}
|
|
52
59
|
// Private ranges 10/8, 192.168/16, 172.16-31/12
|
|
53
60
|
if (/^10\.\d+\.\d+\.\d+$/.test(host))
|
|
54
61
|
return;
|
|
@@ -62,6 +69,31 @@ export function assertSafeBaseURL(url, opts) {
|
|
|
62
69
|
}
|
|
63
70
|
throw new Error(`Unsupported KLYRO_BASE_URL protocol: ${parsed.protocol}`);
|
|
64
71
|
}
|
|
72
|
+
/**
|
|
73
|
+
* Remote MCP URL guard. Mirrors `assertSafeBaseURL`'s fail-closed posture but
|
|
74
|
+
* is named for MCP config/transports so callers do not have to import provider
|
|
75
|
+
* URL text.
|
|
76
|
+
*/
|
|
77
|
+
export function assertSafeRemoteURL(url, opts) {
|
|
78
|
+
let parsed;
|
|
79
|
+
try {
|
|
80
|
+
parsed = new URL(url);
|
|
81
|
+
}
|
|
82
|
+
catch {
|
|
83
|
+
throw new Error(`invalid remote MCP URL: ${url}`);
|
|
84
|
+
}
|
|
85
|
+
if (parsed.protocol === 'https:')
|
|
86
|
+
return;
|
|
87
|
+
if (parsed.protocol === 'http:') {
|
|
88
|
+
if (isLoopbackHost(parsed.hostname))
|
|
89
|
+
return;
|
|
90
|
+
if (opts?.allowInsecure === true || process.env.KLYRO_ALLOW_INSECURE === '1')
|
|
91
|
+
return;
|
|
92
|
+
throw new Error(`Refusing to connect to remote MCP server over plaintext HTTP to ${parsed.hostname}. ` +
|
|
93
|
+
`Use https:// or a loopback URL, or set KLYRO_ALLOW_INSECURE=1 for this terminal only (not recommended).`);
|
|
94
|
+
}
|
|
95
|
+
throw new Error(`unsupported remote MCP URL protocol: ${parsed.protocol}`);
|
|
96
|
+
}
|
|
65
97
|
export async function chat(prompt, system, modelOverride, opts = {}) {
|
|
66
98
|
const baseURL = opts.baseURL ?? process.env.KLYRO_BASE_URL ?? 'https://api.openai.com/v1';
|
|
67
99
|
const apiKey = opts.apiKey ?? process.env.KLYRO_API_KEY;
|
|
@@ -12,6 +12,17 @@
|
|
|
12
12
|
*/
|
|
13
13
|
export declare function snapshot(cwd: string, files: string[]): Promise<string>;
|
|
14
14
|
export declare function listCheckpoints(cwd: string): Promise<string[]>;
|
|
15
|
+
export interface CheckpointInfo {
|
|
16
|
+
/** 1-based index from the latest (1 = newest, like `undo(n)`). */
|
|
17
|
+
index: number;
|
|
18
|
+
id: string;
|
|
19
|
+
ts: number;
|
|
20
|
+
files: number;
|
|
21
|
+
}
|
|
22
|
+
/** Numbered snapshot list for `/checkpoints` and the `/rewind` menu. */
|
|
23
|
+
export declare function listCheckpointInfo(cwd: string): Promise<CheckpointInfo[]>;
|
|
24
|
+
/** File paths a snapshot would restore (for `/rewind <n> preview`). */
|
|
25
|
+
export declare function snapshotFiles(cwd: string, id: string): Promise<string[]>;
|
|
15
26
|
export declare function diff(cwd: string, id?: string): Promise<string>;
|
|
16
27
|
export declare function undo(cwd: string, n?: number): Promise<void>;
|
|
17
28
|
export declare function rewind(cwd: string): Promise<void>;
|
|
@@ -154,6 +154,38 @@ export async function listCheckpoints(cwd) {
|
|
|
154
154
|
return [];
|
|
155
155
|
}
|
|
156
156
|
}
|
|
157
|
+
/** Numbered snapshot list for `/checkpoints` and the `/rewind` menu. */
|
|
158
|
+
export async function listCheckpointInfo(cwd) {
|
|
159
|
+
const ids = await listCheckpoints(cwd);
|
|
160
|
+
const out = [];
|
|
161
|
+
for (let i = ids.length - 1; i >= 0; i--) {
|
|
162
|
+
const id = ids[i];
|
|
163
|
+
let ts = 0;
|
|
164
|
+
let files = 0;
|
|
165
|
+
try {
|
|
166
|
+
const meta = JSON.parse(await fs.readFile(path.join(ckptDir(cwd), id, '.meta.json'), 'utf-8'));
|
|
167
|
+
if (typeof meta.ts === 'number')
|
|
168
|
+
ts = meta.ts;
|
|
169
|
+
if (Array.isArray(meta.files))
|
|
170
|
+
files = meta.files.length;
|
|
171
|
+
}
|
|
172
|
+
catch { /* best-effort */ }
|
|
173
|
+
out.push({ index: ids.length - i, id, ts, files });
|
|
174
|
+
}
|
|
175
|
+
return out;
|
|
176
|
+
}
|
|
177
|
+
/** File paths a snapshot would restore (for `/rewind <n> preview`). */
|
|
178
|
+
export async function snapshotFiles(cwd, id) {
|
|
179
|
+
try {
|
|
180
|
+
const meta = JSON.parse(await fs.readFile(path.join(ckptDir(cwd), id, '.meta.json'), 'utf-8'));
|
|
181
|
+
const files = Array.isArray(meta.files) ? meta.files : [];
|
|
182
|
+
const missing = Array.isArray(meta.missing) ? meta.missing.map((f) => `${f} (deleted)`) : [];
|
|
183
|
+
return [...files, ...missing];
|
|
184
|
+
}
|
|
185
|
+
catch {
|
|
186
|
+
return [];
|
|
187
|
+
}
|
|
188
|
+
}
|
|
157
189
|
export async function diff(cwd, id) {
|
|
158
190
|
const ckpts = await listCheckpoints(cwd);
|
|
159
191
|
const target = id ?? ckpts[ckpts.length - 1];
|