klyro 1.0.17 → 1.0.18
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent/runtime.d.ts +7 -0
- package/dist/agent/runtime.js +41 -1
- package/dist/checkpoints/store.d.ts +15 -1
- package/dist/checkpoints/store.js +7 -2
- package/dist/cli/args.d.ts +7 -0
- package/dist/cli/args.js +10 -0
- package/dist/cli/eval.d.ts +32 -0
- package/dist/cli/eval.js +82 -3
- package/dist/cli/run.d.ts +5 -0
- package/dist/cli/run.js +4 -0
- package/dist/context/klyro-md.d.ts +11 -0
- package/dist/context/klyro-md.js +59 -4
- package/dist/eval/baseline.d.ts +47 -0
- package/dist/eval/baseline.js +55 -0
- package/dist/eval/harness.d.ts +14 -2
- package/dist/eval/harness.js +40 -2
- package/dist/index.js +5 -2
- package/dist/tools/git/git-blame.d.ts +5 -0
- package/dist/tools/git/git-blame.js +22 -0
- package/dist/tools/git/git-diff.js +3 -20
- package/dist/tools/git/git-log.js +5 -19
- package/dist/tools/git/git-show.d.ts +5 -0
- package/dist/tools/git/git-show.js +22 -0
- package/dist/tools/git/git-status.js +4 -21
- package/dist/tools/git/run-git.d.ts +5 -0
- package/dist/tools/git/run-git.js +18 -0
- package/dist/tools/registry.js +4 -0
- package/dist/tools/search/glob.js +8 -3
- package/dist/tools/search/grep.js +10 -3
- package/dist/tools/search/ignore.d.ts +8 -0
- package/dist/tools/search/ignore.js +49 -0
- package/dist/tools/search/search-files.js +10 -3
- package/dist/tui/app.js +37 -1
- package/dist/tui/app.test.js +17 -3
- package/dist/tui/transcript-commands.d.ts +14 -0
- package/dist/tui/transcript-commands.js +31 -0
- package/package.json +1 -1
package/dist/agent/runtime.d.ts
CHANGED
|
@@ -271,6 +271,13 @@ export type RuntimeEvent = {
|
|
|
271
271
|
kind: 'model_override';
|
|
272
272
|
requested: string;
|
|
273
273
|
effective: string;
|
|
274
|
+
} | {
|
|
275
|
+
kind: 'turn.summary';
|
|
276
|
+
turn: number;
|
|
277
|
+
toolCalls: number;
|
|
278
|
+
inputTokens: number;
|
|
279
|
+
outputTokens: number;
|
|
280
|
+
durationMs: number;
|
|
274
281
|
};
|
|
275
282
|
export interface RunResult {
|
|
276
283
|
status: 'complete' | 'max_steps' | 'aborted' | 'no_final' | 'verify_failed' | 'limit' | 'blocked' | 'stuck';
|
package/dist/agent/runtime.js
CHANGED
|
@@ -119,6 +119,8 @@ export async function run(opts, deps) {
|
|
|
119
119
|
}
|
|
120
120
|
};
|
|
121
121
|
let steps = 0;
|
|
122
|
+
// 4.5c — per-turn start time for turn.summary durationMs.
|
|
123
|
+
let turnStartMs = Date.now();
|
|
122
124
|
let lastRemindTurn = 0;
|
|
123
125
|
let toolCallCount = 0;
|
|
124
126
|
let finalText = '';
|
|
@@ -247,6 +249,23 @@ export async function run(opts, deps) {
|
|
|
247
249
|
// best-effort — don't crash runtime on persistence failure
|
|
248
250
|
}
|
|
249
251
|
}
|
|
252
|
+
/**
|
|
253
|
+
* 4.5c — end-of-turn summary: emit a turn.summary event (cumulative
|
|
254
|
+
* toolCalls + token counts, per-turn durationMs) and record a compact
|
|
255
|
+
* one-line summary in the transcript.
|
|
256
|
+
*/
|
|
257
|
+
async function emitTurnSummary() {
|
|
258
|
+
const durationMs = Date.now() - turnStartMs;
|
|
259
|
+
emit?.({ kind: 'turn.summary', turn: steps, toolCalls: toolCallCount, inputTokens: usage.input, outputTokens: usage.output, durationMs });
|
|
260
|
+
const msg = {
|
|
261
|
+
role: 'user',
|
|
262
|
+
content: [text(`[turn ${steps} summary] ${toolCallCount} tool call(s), ${usage.input} in / ${usage.output} out tokens, ${durationMs}ms`)],
|
|
263
|
+
};
|
|
264
|
+
transcript.push(msg);
|
|
265
|
+
await checkpoint(msg);
|
|
266
|
+
// Invalidate token cache since transcript changed
|
|
267
|
+
tokenCache = { lastRef: null, lastSystem: undefined, lastCount: 0 };
|
|
268
|
+
}
|
|
250
269
|
// Persist initial user message
|
|
251
270
|
if (store && sessionId && transcript.length > 0) {
|
|
252
271
|
// Fire-and-forget initial checkpoint (don't await to block loop start)
|
|
@@ -322,6 +341,7 @@ export async function run(opts, deps) {
|
|
|
322
341
|
return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined, phase: 'blocked' };
|
|
323
342
|
}
|
|
324
343
|
steps++;
|
|
344
|
+
turnStartMs = Date.now();
|
|
325
345
|
// 8.4 — stale-todo reminder: every 20 turns, re-inject pending plan
|
|
326
346
|
// items from `.klyro/plans/todos.json` (written by todo_write) so a
|
|
327
347
|
// long run cannot silently drop its checklist. Best-effort + tiny.
|
|
@@ -615,6 +635,9 @@ export async function run(opts, deps) {
|
|
|
615
635
|
return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined };
|
|
616
636
|
}
|
|
617
637
|
if (finalizedCalls.length === 0) {
|
|
638
|
+
// 4.5c — the turn completed (final text, malformed-only, or
|
|
639
|
+
// stop-hook continuation): summarize before the verify branching.
|
|
640
|
+
await emitTurnSummary();
|
|
618
641
|
// Steerable stop: a stop hook asked for one more turn instead of
|
|
619
642
|
// completing. Consumed once per verdict, max 3 per run.
|
|
620
643
|
if (stopCont !== null && stopContUsed < 3) {
|
|
@@ -1051,6 +1074,18 @@ export async function run(opts, deps) {
|
|
|
1051
1074
|
}
|
|
1052
1075
|
}
|
|
1053
1076
|
}
|
|
1077
|
+
// 4.5a — shell/git mutations bypass file-change inference: take a
|
|
1078
|
+
// pre-execution snapshot so the turn stays restorable. Best-effort.
|
|
1079
|
+
if (call.name === 'shell_exec' || call.name.startsWith('git_')) {
|
|
1080
|
+
try {
|
|
1081
|
+
const { snapshot } = await import('../checkpoints/store.js');
|
|
1082
|
+
await snapshot(opts.cwd, [...fileEditCounts.keys()].slice(-20), {
|
|
1083
|
+
...(sessionId !== undefined ? { sessionId } : {}),
|
|
1084
|
+
eventId: call.id,
|
|
1085
|
+
});
|
|
1086
|
+
}
|
|
1087
|
+
catch { /* ignore */ }
|
|
1088
|
+
}
|
|
1054
1089
|
let obs;
|
|
1055
1090
|
try {
|
|
1056
1091
|
obs = await deps.registry.execute(call.name, call.input, toolCtx);
|
|
@@ -1138,7 +1173,10 @@ export async function run(opts, deps) {
|
|
|
1138
1173
|
// 4.5 — checkpoint snapshot after each mutation
|
|
1139
1174
|
try {
|
|
1140
1175
|
const { snapshot } = await import('../checkpoints/store.js');
|
|
1141
|
-
await snapshot(opts.cwd, [fileChanged.path]
|
|
1176
|
+
await snapshot(opts.cwd, [fileChanged.path], {
|
|
1177
|
+
...(sessionId !== undefined ? { sessionId } : {}),
|
|
1178
|
+
eventId: call.id,
|
|
1179
|
+
});
|
|
1142
1180
|
}
|
|
1143
1181
|
catch { /* ignore */ }
|
|
1144
1182
|
// 5.2 file edit count
|
|
@@ -1251,6 +1289,8 @@ export async function run(opts, deps) {
|
|
|
1251
1289
|
break;
|
|
1252
1290
|
}
|
|
1253
1291
|
}
|
|
1292
|
+
// 4.5c — the tool turn completed: summarize (cumulative counts).
|
|
1293
|
+
await emitTurnSummary();
|
|
1254
1294
|
// P0 — drain finished sub-agent completions into parent visibility.
|
|
1255
1295
|
// drainCompletions is OPTIONAL on the bridge — guarded with `?.` so
|
|
1256
1296
|
// older bridges without it simply yield nothing.
|
|
@@ -10,7 +10,21 @@
|
|
|
10
10
|
* enforced at the trace/persist boundaries (TraceWriter, SessionStore)
|
|
11
11
|
* instead of here.
|
|
12
12
|
*/
|
|
13
|
-
|
|
13
|
+
/** Optional provenance recorded into a checkpoint's .meta.json. */
|
|
14
|
+
export interface SnapshotOptions {
|
|
15
|
+
sessionId?: string;
|
|
16
|
+
eventId?: string;
|
|
17
|
+
}
|
|
18
|
+
/** Checkpoint meta on disk — sessionId/eventId are absent on old metas. */
|
|
19
|
+
export interface CheckpointMeta {
|
|
20
|
+
id: string;
|
|
21
|
+
files: string[];
|
|
22
|
+
missing: string[];
|
|
23
|
+
ts: number;
|
|
24
|
+
sessionId?: string;
|
|
25
|
+
eventId?: string;
|
|
26
|
+
}
|
|
27
|
+
export declare function snapshot(cwd: string, files: string[], opts?: SnapshotOptions): Promise<string>;
|
|
14
28
|
export declare function listCheckpoints(cwd: string): Promise<string[]>;
|
|
15
29
|
export interface CheckpointInfo {
|
|
16
30
|
/** 1-based index from the latest (1 = newest, like `undo(n)`). */
|
|
@@ -55,7 +55,7 @@ function containedPath(cwd, base, rel) {
|
|
|
55
55
|
return null;
|
|
56
56
|
return out;
|
|
57
57
|
}
|
|
58
|
-
export async function snapshot(cwd, files) {
|
|
58
|
+
export async function snapshot(cwd, files, opts) {
|
|
59
59
|
const dir = ckptDir(cwd);
|
|
60
60
|
await fs.mkdir(dir, { recursive: true });
|
|
61
61
|
lockDown(dir, 0o700);
|
|
@@ -92,7 +92,12 @@ export async function snapshot(cwd, files) {
|
|
|
92
92
|
// leaves a truncated .meta.json that undo() then trusts).
|
|
93
93
|
const metaPath = path.join(dest, '.meta.json');
|
|
94
94
|
const metaTmp = `${metaPath}.tmp-${Date.now()}-${Math.random().toString(36).slice(2, 6)}`;
|
|
95
|
-
|
|
95
|
+
const meta = { id, files: kept, missing, ts: Date.now() };
|
|
96
|
+
if (opts?.sessionId !== undefined)
|
|
97
|
+
meta.sessionId = opts.sessionId;
|
|
98
|
+
if (opts?.eventId !== undefined)
|
|
99
|
+
meta.eventId = opts.eventId;
|
|
100
|
+
await fs.writeFile(metaTmp, JSON.stringify(meta, null, 2));
|
|
96
101
|
lockDown(metaTmp, 0o600);
|
|
97
102
|
await fsyncFile(metaTmp);
|
|
98
103
|
try {
|
package/dist/cli/args.d.ts
CHANGED
|
@@ -1 +1,8 @@
|
|
|
1
1
|
export declare function parsePositiveInt(name: string, v: string): number;
|
|
2
|
+
/**
|
|
3
|
+
* 5.3 — `--auto-answer <text>` wiring. `ask_user` honors
|
|
4
|
+
* `KLYRO_AUTO_ANSWER` in headless runs; `klyro run` / `klyro eval` set it
|
|
5
|
+
* from the flag via this helper before the run starts (explicit env wins
|
|
6
|
+
* when the flag is omitted). Exported for tests.
|
|
7
|
+
*/
|
|
8
|
+
export declare function applyAutoAnswer(value: string | undefined): void;
|
package/dist/cli/args.js
CHANGED
|
@@ -12,3 +12,13 @@ export function parsePositiveInt(name, v) {
|
|
|
12
12
|
}
|
|
13
13
|
return n;
|
|
14
14
|
}
|
|
15
|
+
/**
|
|
16
|
+
* 5.3 — `--auto-answer <text>` wiring. `ask_user` honors
|
|
17
|
+
* `KLYRO_AUTO_ANSWER` in headless runs; `klyro run` / `klyro eval` set it
|
|
18
|
+
* from the flag via this helper before the run starts (explicit env wins
|
|
19
|
+
* when the flag is omitted). Exported for tests.
|
|
20
|
+
*/
|
|
21
|
+
export function applyAutoAnswer(value) {
|
|
22
|
+
if (value !== undefined)
|
|
23
|
+
process.env.KLYRO_AUTO_ANSWER = value;
|
|
24
|
+
}
|
package/dist/cli/eval.d.ts
CHANGED
|
@@ -81,6 +81,22 @@ export interface EvalResult {
|
|
|
81
81
|
};
|
|
82
82
|
/** Isolated workdir the scenario ran in (tmp unless --cwd). Debugging aid. */
|
|
83
83
|
workDir?: string;
|
|
84
|
+
/** 5.4a — provider token usage for the scenario run. */
|
|
85
|
+
tokens?: {
|
|
86
|
+
input: number;
|
|
87
|
+
output: number;
|
|
88
|
+
};
|
|
89
|
+
/** 5.4a — estimated USD cost (runtime pricing helper, same rates as budgets). */
|
|
90
|
+
costUsd?: number;
|
|
91
|
+
/** 5.4a — distinct file paths referenced by tool-call inputs (best-effort). */
|
|
92
|
+
filesTouched?: string[];
|
|
93
|
+
/** 5.4a — verification outcome passthrough. */
|
|
94
|
+
verification?: {
|
|
95
|
+
ok: boolean;
|
|
96
|
+
attempts: number;
|
|
97
|
+
};
|
|
98
|
+
/** 5.4a — repair count passthrough. */
|
|
99
|
+
repairs?: number;
|
|
84
100
|
}
|
|
85
101
|
export interface RunEvalOptions {
|
|
86
102
|
inputPath: string;
|
|
@@ -98,10 +114,26 @@ export interface RunEvalOptions {
|
|
|
98
114
|
* touch the caller's directory. Pass explicitly to inspect artifacts.
|
|
99
115
|
*/
|
|
100
116
|
cwd?: string;
|
|
117
|
+
/**
|
|
118
|
+
* 5.3 — `--auto-answer <text>`: sets `KLYRO_AUTO_ANSWER` before the run
|
|
119
|
+
* starts so headless `ask_user` calls resolve without prompting.
|
|
120
|
+
*/
|
|
121
|
+
autoAnswer?: string;
|
|
101
122
|
}
|
|
123
|
+
/**
|
|
124
|
+
* 5.5a — curated stable subset of `evals/fixtures` for `--suite core`.
|
|
125
|
+
* Every entry has task.md + check.sh (verified 2026-09-24).
|
|
126
|
+
*/
|
|
127
|
+
export declare const CORE_SUITE_FIXTURES: string[];
|
|
128
|
+
/**
|
|
129
|
+
* Filter directory names down to the core suite (in curated order).
|
|
130
|
+
* Exported for tests.
|
|
131
|
+
*/
|
|
132
|
+
export declare function selectCoreFixtures(entries: string[]): string[];
|
|
102
133
|
export declare function runEval(opts: RunEvalOptions): Promise<number>;
|
|
103
134
|
export declare function scriptedAdapterFromSpec(spec: Array<Array<unknown[]>> | undefined): ProviderAdapter;
|
|
104
135
|
export declare function runScenario(sc: EvalScenario, judgeOpts?: {
|
|
105
136
|
adapter: ProviderAdapter;
|
|
106
137
|
model: string;
|
|
107
138
|
}, workDir?: string): Promise<EvalResult>;
|
|
139
|
+
export declare function collectFilesTouched(transcript: RunResult['transcript']): string[];
|
package/dist/cli/eval.js
CHANGED
|
@@ -45,11 +45,26 @@ import * as os from 'node:os';
|
|
|
45
45
|
import * as path from 'node:path';
|
|
46
46
|
import * as readline from 'node:readline/promises';
|
|
47
47
|
import { stdin as input, stdout, stderr } from 'node:process';
|
|
48
|
-
import { run } from '../agent/runtime.js';
|
|
48
|
+
import { run, estimateCost } from '../agent/runtime.js';
|
|
49
|
+
import { applyAutoAnswer } from './args.js';
|
|
49
50
|
import { builtinRegistry } from '../tools/registry.js';
|
|
50
51
|
import { builtinRules, DEFAULT_POLICY_CONFIG, PolicyEngine } from '../policy/engine.js';
|
|
51
52
|
import { DenyAllApprovalPrompt } from '../policy/approval.js';
|
|
53
|
+
/**
|
|
54
|
+
* 5.5a — curated stable subset of `evals/fixtures` for `--suite core`.
|
|
55
|
+
* Every entry has task.md + check.sh (verified 2026-09-24).
|
|
56
|
+
*/
|
|
57
|
+
export const CORE_SUITE_FIXTURES = ['read-answer', 'add-fn-test', 'fix-failing-test', 'rename', 'cli-flag'];
|
|
58
|
+
/**
|
|
59
|
+
* Filter directory names down to the core suite (in curated order).
|
|
60
|
+
* Exported for tests.
|
|
61
|
+
*/
|
|
62
|
+
export function selectCoreFixtures(entries) {
|
|
63
|
+
return CORE_SUITE_FIXTURES.filter((id) => entries.includes(id));
|
|
64
|
+
}
|
|
52
65
|
export async function runEval(opts) {
|
|
66
|
+
// 5.3 — --auto-answer sets KLYRO_AUTO_ANSWER before anything runs.
|
|
67
|
+
applyAutoAnswer(opts.autoAnswer);
|
|
53
68
|
// Live judge adapter (shared by suite + JSONL paths) for `judge.rubric`.
|
|
54
69
|
let judgeAdapter;
|
|
55
70
|
if (opts.judgeModel) {
|
|
@@ -65,6 +80,7 @@ export async function runEval(opts) {
|
|
|
65
80
|
}
|
|
66
81
|
// 5.4 — suite mode: load from evals/fixtures
|
|
67
82
|
if (opts.suite) {
|
|
83
|
+
const suiteStart = Date.now();
|
|
68
84
|
const fs = await import('node:fs/promises');
|
|
69
85
|
const path = await import('node:path');
|
|
70
86
|
// For smoke, use the 10 fixtures directly
|
|
@@ -75,8 +91,13 @@ export async function runEval(opts) {
|
|
|
75
91
|
for (const e of entries) {
|
|
76
92
|
if (opts.filter && !e.includes(opts.filter))
|
|
77
93
|
continue;
|
|
78
|
-
//
|
|
79
|
-
if (opts.suite
|
|
94
|
+
// 5.5a — core = curated stable subset (see CORE_SUITE_FIXTURES).
|
|
95
|
+
if (opts.suite === 'core') {
|
|
96
|
+
if (!CORE_SUITE_FIXTURES.includes(e))
|
|
97
|
+
continue;
|
|
98
|
+
// 6.5 — suite filter: smoke = type smoke, l6 = prefix l6-introduce
|
|
99
|
+
}
|
|
100
|
+
else if (opts.suite && opts.suite !== 'smoke') {
|
|
80
101
|
if (!e.startsWith(opts.suite) && !e.includes(opts.suite))
|
|
81
102
|
continue;
|
|
82
103
|
}
|
|
@@ -141,6 +162,16 @@ export async function runEval(opts) {
|
|
|
141
162
|
await fs.mkdir(outDir, { recursive: true });
|
|
142
163
|
const ts = new Date().toISOString().replace(/[:.]/g, '-');
|
|
143
164
|
await fs.writeFile(path.join(outDir, `${ts}.json`), JSON.stringify({ suite: opts.suite, results }, null, 2));
|
|
165
|
+
// 5.5b — refresh the tracked baseline with environment metadata.
|
|
166
|
+
const { collectBaselineMetadata, buildEvalBaseline, writeEvalBaseline } = await import('../eval/baseline.js');
|
|
167
|
+
const meta = collectBaselineMetadata({ model: opts.model });
|
|
168
|
+
await writeEvalBaseline(outDir, buildEvalBaseline(meta, {
|
|
169
|
+
suite: opts.suite ?? '',
|
|
170
|
+
total: results.length,
|
|
171
|
+
passed,
|
|
172
|
+
failed: results.length - passed,
|
|
173
|
+
durationMs: Date.now() - suiteStart,
|
|
174
|
+
}));
|
|
144
175
|
}
|
|
145
176
|
catch { /* ignore */ }
|
|
146
177
|
return passed === results.length ? 0 : 1;
|
|
@@ -165,6 +196,18 @@ export async function runEval(opts) {
|
|
|
165
196
|
stdout.write(`[${tag}] ${r.name} (${r.durationMs}ms, ${r.steps} steps, ${r.toolCalls} tools)\n`);
|
|
166
197
|
for (const f of r.failures)
|
|
167
198
|
stdout.write(` - ${f}\n`);
|
|
199
|
+
// 5.4a — rich record extras (tokens/cost/files/verification).
|
|
200
|
+
const extras = [];
|
|
201
|
+
if (r.tokens)
|
|
202
|
+
extras.push(`${r.tokens.input} in/${r.tokens.output} out`);
|
|
203
|
+
if (r.costUsd !== undefined)
|
|
204
|
+
extras.push(`$${r.costUsd.toFixed(4)}`);
|
|
205
|
+
if (r.filesTouched && r.filesTouched.length > 0)
|
|
206
|
+
extras.push(`${r.filesTouched.length} files`);
|
|
207
|
+
if (r.verification)
|
|
208
|
+
extras.push(`verify:${r.verification.ok ? 'ok' : 'fail'}`);
|
|
209
|
+
if (extras.length > 0)
|
|
210
|
+
stdout.write(` · ${extras.join(' · ')}\n`);
|
|
168
211
|
}
|
|
169
212
|
}
|
|
170
213
|
const passed = results.filter((r) => r.passed).length;
|
|
@@ -332,5 +375,41 @@ export async function runScenario(sc, judgeOpts, workDir) {
|
|
|
332
375
|
durationMs: 0,
|
|
333
376
|
workDir: cwd,
|
|
334
377
|
...(judge ? { judge } : {}),
|
|
378
|
+
// 5.4a — rich records mapped from RunResult (cost reuses the runtime
|
|
379
|
+
// pricing helper, the same rates the budget code uses).
|
|
380
|
+
tokens: { input: result.usage.input, output: result.usage.output },
|
|
381
|
+
costUsd: estimateCost(model, result.usage),
|
|
382
|
+
filesTouched: collectFilesTouched(result.transcript),
|
|
383
|
+
...(result.verification
|
|
384
|
+
? { verification: { ok: result.verification.ok, attempts: result.verification.attempts } }
|
|
385
|
+
: {}),
|
|
386
|
+
...(result.repairs !== undefined ? { repairs: result.repairs } : {}),
|
|
335
387
|
};
|
|
336
388
|
}
|
|
389
|
+
/**
|
|
390
|
+
* 5.4a — best-effort distinct file paths from tool-call inputs in a run
|
|
391
|
+
* transcript. Looks for common path-ish keys on `tool_use` blocks.
|
|
392
|
+
* Exported for tests.
|
|
393
|
+
*/
|
|
394
|
+
const FILE_PATH_KEYS = ['path', 'file', 'filePath', 'filename', 'target'];
|
|
395
|
+
export function collectFilesTouched(transcript) {
|
|
396
|
+
const seen = new Set();
|
|
397
|
+
for (const m of transcript) {
|
|
398
|
+
for (const b of m.content) {
|
|
399
|
+
if (b.kind !== 'tool_use')
|
|
400
|
+
continue;
|
|
401
|
+
const input = b.input;
|
|
402
|
+
for (const k of FILE_PATH_KEYS) {
|
|
403
|
+
const v = input[k];
|
|
404
|
+
if (typeof v === 'string' && v.length > 0)
|
|
405
|
+
seen.add(v);
|
|
406
|
+
else if (Array.isArray(v)) {
|
|
407
|
+
for (const e of v)
|
|
408
|
+
if (typeof e === 'string' && e.length > 0)
|
|
409
|
+
seen.add(e);
|
|
410
|
+
}
|
|
411
|
+
}
|
|
412
|
+
}
|
|
413
|
+
}
|
|
414
|
+
return [...seen].sort();
|
|
415
|
+
}
|
package/dist/cli/run.d.ts
CHANGED
|
@@ -91,6 +91,11 @@ export interface RunCliOptions {
|
|
|
91
91
|
/** P1.4 — run the task under a named child-capable orchestrator context. */
|
|
92
92
|
agent?: string;
|
|
93
93
|
maxDepth?: number;
|
|
94
|
+
/**
|
|
95
|
+
* 5.3 — `--auto-answer <text>`: answer `ask_user` prompts headlessly.
|
|
96
|
+
* Applied to `process.env.KLYRO_AUTO_ANSWER` at the top of `runOnce`.
|
|
97
|
+
*/
|
|
98
|
+
autoAnswer?: string;
|
|
94
99
|
}
|
|
95
100
|
/**
|
|
96
101
|
* Split a comma-separated tool/dir list (string or string[]) into names.
|
package/dist/cli/run.js
CHANGED
|
@@ -23,6 +23,7 @@ import { memoryBlock } from '../context/memory.js';
|
|
|
23
23
|
import { estimateCost } from '../providers/model-info.js';
|
|
24
24
|
import { resolveModelAlias } from '../providers/model-info.js';
|
|
25
25
|
import { logDebug, logInfo } from '../util/log.js';
|
|
26
|
+
import { applyAutoAnswer } from './args.js';
|
|
26
27
|
import { resolveSessionId } from '../persistence/session.js';
|
|
27
28
|
import * as fs from 'node:fs';
|
|
28
29
|
/**
|
|
@@ -71,6 +72,9 @@ export function shouldForceExit(lastSigintAt, now) {
|
|
|
71
72
|
return lastSigintAt !== undefined && now - lastSigintAt < 1500;
|
|
72
73
|
}
|
|
73
74
|
export async function runOnce(opts) {
|
|
75
|
+
// 5.3 — --auto-answer sets KLYRO_AUTO_ANSWER before anything runs so
|
|
76
|
+
// headless ask_user calls resolve without prompting.
|
|
77
|
+
applyAutoAnswer(opts.autoAnswer);
|
|
74
78
|
// P0.5 — load <cwd>/.env first so KLYRO_* vars resolve without `export`.
|
|
75
79
|
// Never throws (missing file is a no-op); explicit env wins (no-clobber).
|
|
76
80
|
try {
|
|
@@ -12,5 +12,16 @@ export interface KlyroMdFile {
|
|
|
12
12
|
}
|
|
13
13
|
/** Per-file loader (used by the trust gate to approve/hash individual files). */
|
|
14
14
|
export declare function loadKlyroMdFiles(cwd: string): Promise<KlyroMdFile[]>;
|
|
15
|
+
/**
|
|
16
|
+
* 4.4a — Subdirectory instructions: root files PLUS instruction files found
|
|
17
|
+
* in ancestor directories of targetPath from cwd down to targetPath's dir
|
|
18
|
+
* (nearest-wins order root→leaf, each capped like cap()).
|
|
19
|
+
*
|
|
20
|
+
* Laziness preserved: plain loadKlyroMdFiles(cwd) never reads subdirs —
|
|
21
|
+
* callers pass the relevant active path explicitly.
|
|
22
|
+
*
|
|
23
|
+
* Containment: a targetPath escaping cwd is rejected (root files only).
|
|
24
|
+
*/
|
|
25
|
+
export declare function loadKlyroMdForPath(cwd: string, targetPath: string): Promise<KlyroMdFile[]>;
|
|
15
26
|
export declare function loadKlyroMd(cwd: string): Promise<string>;
|
|
16
27
|
export declare function handleInit(cwd: string): Promise<string>;
|
package/dist/context/klyro-md.js
CHANGED
|
@@ -30,16 +30,65 @@ export async function loadKlyroMdFiles(cwd) {
|
|
|
30
30
|
// Klyro's own instruction file is KLYRO.md. The rest are read-only
|
|
31
31
|
// fallbacks for imported repos that follow other conventions — Klyro
|
|
32
32
|
// never writes them. KLYRO.md wins by load order.
|
|
33
|
-
for (const name of
|
|
33
|
+
for (const name of INSTRUCTION_NAMES) {
|
|
34
34
|
const p = path.join(cwd, name);
|
|
35
35
|
try {
|
|
36
36
|
const t = await fs.readFile(p, 'utf-8');
|
|
37
|
-
files.push({ path: p, content: cap(await resolveImports(t, path.dirname(p), cwd), MAX_FILE_CHARS) });
|
|
37
|
+
files.push({ path: p, content: cap(await resolveImports(t, path.dirname(p), cwd, 0, [p]), MAX_FILE_CHARS) });
|
|
38
38
|
}
|
|
39
39
|
catch { /* ignore */ }
|
|
40
40
|
}
|
|
41
41
|
return files;
|
|
42
42
|
}
|
|
43
|
+
/** Instruction file names read at every level (root + subdirectories). */
|
|
44
|
+
const INSTRUCTION_NAMES = ['KLYRO.md', 'KLYRO.local.md', 'CLAUDE.md', 'CLAUDE.local.md', 'AGENTS.md', '.cursorrules'];
|
|
45
|
+
/**
|
|
46
|
+
* 4.4a — Subdirectory instructions: root files PLUS instruction files found
|
|
47
|
+
* in ancestor directories of targetPath from cwd down to targetPath's dir
|
|
48
|
+
* (nearest-wins order root→leaf, each capped like cap()).
|
|
49
|
+
*
|
|
50
|
+
* Laziness preserved: plain loadKlyroMdFiles(cwd) never reads subdirs —
|
|
51
|
+
* callers pass the relevant active path explicitly.
|
|
52
|
+
*
|
|
53
|
+
* Containment: a targetPath escaping cwd is rejected (root files only).
|
|
54
|
+
*/
|
|
55
|
+
export async function loadKlyroMdForPath(cwd, targetPath) {
|
|
56
|
+
const files = await loadKlyroMdFiles(cwd);
|
|
57
|
+
const root = path.resolve(cwd);
|
|
58
|
+
const absTarget = path.resolve(root, targetPath);
|
|
59
|
+
const relTarget = path.relative(root, absTarget);
|
|
60
|
+
if (relTarget.startsWith('..') || path.isAbsolute(relTarget))
|
|
61
|
+
return files;
|
|
62
|
+
// Target dir: the target itself when it is an existing directory,
|
|
63
|
+
// otherwise its parent dir (i.e. the target is a file being edited).
|
|
64
|
+
let targetDir = absTarget;
|
|
65
|
+
try {
|
|
66
|
+
const st = await fs.stat(absTarget);
|
|
67
|
+
if (!st.isDirectory())
|
|
68
|
+
targetDir = path.dirname(absTarget);
|
|
69
|
+
}
|
|
70
|
+
catch {
|
|
71
|
+
targetDir = path.dirname(absTarget);
|
|
72
|
+
}
|
|
73
|
+
const relDir = path.relative(root, targetDir);
|
|
74
|
+
if (relDir.startsWith('..') || path.isAbsolute(relDir))
|
|
75
|
+
return files;
|
|
76
|
+
const segs = relDir.split(path.sep).filter((s) => s && s !== '.');
|
|
77
|
+
let prefix = root;
|
|
78
|
+
for (const seg of segs) {
|
|
79
|
+
prefix = path.join(prefix, seg);
|
|
80
|
+
const dir = prefix;
|
|
81
|
+
for (const name of INSTRUCTION_NAMES) {
|
|
82
|
+
const p = path.join(dir, name);
|
|
83
|
+
try {
|
|
84
|
+
const t = await fs.readFile(p, 'utf-8');
|
|
85
|
+
files.push({ path: p, content: cap(await resolveImports(t, path.dirname(p), root, 0, [p]), MAX_FILE_CHARS) });
|
|
86
|
+
}
|
|
87
|
+
catch { /* ignore */ }
|
|
88
|
+
}
|
|
89
|
+
}
|
|
90
|
+
return files;
|
|
91
|
+
}
|
|
43
92
|
export async function loadKlyroMd(cwd) {
|
|
44
93
|
const parts = [];
|
|
45
94
|
let total = 0;
|
|
@@ -56,7 +105,7 @@ export async function loadKlyroMd(cwd) {
|
|
|
56
105
|
}
|
|
57
106
|
return parts.join('\n\n---\n\n');
|
|
58
107
|
}
|
|
59
|
-
async function resolveImports(text, base, root, depth = 0) {
|
|
108
|
+
async function resolveImports(text, base, root, depth = 0, chain = []) {
|
|
60
109
|
if (depth > 5)
|
|
61
110
|
return text;
|
|
62
111
|
const importRe = /^@import\s+(.+)$/gm;
|
|
@@ -69,9 +118,15 @@ async function resolveImports(text, base, root, depth = 0) {
|
|
|
69
118
|
const relToRoot = path.relative(root, p);
|
|
70
119
|
if (relToRoot.startsWith('..') || path.isAbsolute(relToRoot))
|
|
71
120
|
continue;
|
|
121
|
+
// Cycle guard: an @import target already on the current chain resolves
|
|
122
|
+
// to a marker instead of recursing (depth cap stays as backstop).
|
|
123
|
+
if (chain.includes(p)) {
|
|
124
|
+
out = out.replace(m[0], `<!-- klyro: import cycle skipped: ${rel} -->`);
|
|
125
|
+
continue;
|
|
126
|
+
}
|
|
72
127
|
try {
|
|
73
128
|
const t = await fs.readFile(p, 'utf-8');
|
|
74
|
-
const resolved = await resolveImports(cap(t, MAX_FILE_CHARS), path.dirname(p), root, depth + 1);
|
|
129
|
+
const resolved = await resolveImports(cap(t, MAX_FILE_CHARS), path.dirname(p), root, depth + 1, [...chain, p]);
|
|
75
130
|
out = out.replace(m[0], resolved);
|
|
76
131
|
}
|
|
77
132
|
catch { /* ignore missing */ }
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* 5.5b — eval baseline writer. Records the environment alongside suite
|
|
3
|
+
* results so regressions can be attributed (model/node/platform/commit).
|
|
4
|
+
*
|
|
5
|
+
* The tracked `evals/results/baseline.json` is refreshed by `klyro eval
|
|
6
|
+
* --suite <name>` (best-effort; suite failures never fail the write path).
|
|
7
|
+
*/
|
|
8
|
+
export interface EvalBaselineMeta {
|
|
9
|
+
/** Model id the suite ran under (KLYRO_MODEL fallback, then 'mock'). */
|
|
10
|
+
model: string;
|
|
11
|
+
/** Node version (process.version). */
|
|
12
|
+
node: string;
|
|
13
|
+
/** OS platform (process.platform). */
|
|
14
|
+
platform: string;
|
|
15
|
+
/** Git HEAD, best-effort ('unknown' outside a repo). */
|
|
16
|
+
commit: string;
|
|
17
|
+
/** ISO timestamp of the run. */
|
|
18
|
+
timestamp: string;
|
|
19
|
+
}
|
|
20
|
+
export interface EvalBaseline extends EvalBaselineMeta {
|
|
21
|
+
suite: string;
|
|
22
|
+
total: number;
|
|
23
|
+
passed: number;
|
|
24
|
+
failed: number;
|
|
25
|
+
passRate: number;
|
|
26
|
+
durationMs: number;
|
|
27
|
+
}
|
|
28
|
+
/** Best-effort `git rev-parse HEAD` (never throws). Exported for tests. */
|
|
29
|
+
export declare function gitCommit(cwd?: string): string;
|
|
30
|
+
/** Collect the environment half of a baseline record. Exported for tests. */
|
|
31
|
+
export declare function collectBaselineMetadata(opts?: {
|
|
32
|
+
model?: string;
|
|
33
|
+
cwd?: string;
|
|
34
|
+
}): EvalBaselineMeta;
|
|
35
|
+
/** Combine metadata + suite counts into a full baseline record. */
|
|
36
|
+
export declare function buildEvalBaseline(meta: EvalBaselineMeta, suite: {
|
|
37
|
+
suite: string;
|
|
38
|
+
total: number;
|
|
39
|
+
passed: number;
|
|
40
|
+
failed: number;
|
|
41
|
+
durationMs: number;
|
|
42
|
+
}): EvalBaseline;
|
|
43
|
+
/**
|
|
44
|
+
* Write `baseline.json` into `dir` (created if needed). Returns the path.
|
|
45
|
+
* Never throws for missing dirs — the caller wraps in try/catch anyway.
|
|
46
|
+
*/
|
|
47
|
+
export declare function writeEvalBaseline(dir: string, baseline: EvalBaseline): Promise<string>;
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* 5.5b — eval baseline writer. Records the environment alongside suite
|
|
3
|
+
* results so regressions can be attributed (model/node/platform/commit).
|
|
4
|
+
*
|
|
5
|
+
* The tracked `evals/results/baseline.json` is refreshed by `klyro eval
|
|
6
|
+
* --suite <name>` (best-effort; suite failures never fail the write path).
|
|
7
|
+
*/
|
|
8
|
+
import * as fsp from 'node:fs/promises';
|
|
9
|
+
import * as path from 'node:path';
|
|
10
|
+
import { execFileSync } from 'node:child_process';
|
|
11
|
+
/** Best-effort `git rev-parse HEAD` (never throws). Exported for tests. */
|
|
12
|
+
export function gitCommit(cwd = process.cwd()) {
|
|
13
|
+
try {
|
|
14
|
+
const out = execFileSync('git', ['rev-parse', 'HEAD'], {
|
|
15
|
+
cwd,
|
|
16
|
+
stdio: ['ignore', 'pipe', 'ignore'],
|
|
17
|
+
}).toString().trim();
|
|
18
|
+
return out || 'unknown';
|
|
19
|
+
}
|
|
20
|
+
catch {
|
|
21
|
+
return 'unknown';
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
/** Collect the environment half of a baseline record. Exported for tests. */
|
|
25
|
+
export function collectBaselineMetadata(opts = {}) {
|
|
26
|
+
return {
|
|
27
|
+
model: opts.model ?? process.env.KLYRO_MODEL ?? 'mock',
|
|
28
|
+
node: process.version,
|
|
29
|
+
platform: process.platform,
|
|
30
|
+
commit: gitCommit(opts.cwd),
|
|
31
|
+
timestamp: new Date().toISOString(),
|
|
32
|
+
};
|
|
33
|
+
}
|
|
34
|
+
/** Combine metadata + suite counts into a full baseline record. */
|
|
35
|
+
export function buildEvalBaseline(meta, suite) {
|
|
36
|
+
return {
|
|
37
|
+
...meta,
|
|
38
|
+
suite: suite.suite,
|
|
39
|
+
total: suite.total,
|
|
40
|
+
passed: suite.passed,
|
|
41
|
+
failed: suite.failed,
|
|
42
|
+
passRate: suite.total === 0 ? 0 : suite.passed / suite.total,
|
|
43
|
+
durationMs: suite.durationMs,
|
|
44
|
+
};
|
|
45
|
+
}
|
|
46
|
+
/**
|
|
47
|
+
* Write `baseline.json` into `dir` (created if needed). Returns the path.
|
|
48
|
+
* Never throws for missing dirs — the caller wraps in try/catch anyway.
|
|
49
|
+
*/
|
|
50
|
+
export async function writeEvalBaseline(dir, baseline) {
|
|
51
|
+
await fsp.mkdir(dir, { recursive: true });
|
|
52
|
+
const out = path.join(dir, 'baseline.json');
|
|
53
|
+
await fsp.writeFile(out, JSON.stringify(baseline, null, 2) + '\n');
|
|
54
|
+
return out;
|
|
55
|
+
}
|
package/dist/eval/harness.d.ts
CHANGED
|
@@ -77,10 +77,21 @@ export interface FileFixture {
|
|
|
77
77
|
script?: StreamEvent[][];
|
|
78
78
|
}
|
|
79
79
|
export declare function loadFileFixture(dir: string): Promise<FileFixture>;
|
|
80
|
-
export declare function
|
|
80
|
+
export declare function sanitizedFixtureEnv(base?: NodeJS.ProcessEnv): NodeJS.ProcessEnv;
|
|
81
|
+
/**
|
|
82
|
+
* 5.4b — guard: fixture workdirs must stay inside `os.tmpdir()` (the
|
|
83
|
+
* harness relies on that isolation). Throws otherwise unless the caller
|
|
84
|
+
* explicitly passes `allowWorkdirOutsideTmp`. Exported for tests.
|
|
85
|
+
*/
|
|
86
|
+
export declare function assertTmpWorkdir(dir: string, opts?: {
|
|
87
|
+
allowWorkdirOutsideTmp?: boolean;
|
|
88
|
+
}): void;
|
|
89
|
+
export interface FileFixtureOptions {
|
|
81
90
|
runs?: number;
|
|
82
91
|
parallel?: number;
|
|
83
|
-
|
|
92
|
+
allowWorkdirOutsideTmp?: boolean;
|
|
93
|
+
}
|
|
94
|
+
export declare function runFileFixture(fixture: FileFixture, _opts?: FileFixtureOptions): Promise<TaskResult>;
|
|
84
95
|
/**
|
|
85
96
|
* Agent-driven fixture: seed a tmp repo, run the fixture's canned script
|
|
86
97
|
* through the REAL runtime + tools, then assert outcomes with check.sh
|
|
@@ -90,6 +101,7 @@ export declare function runFileFixture(fixture: FileFixture, _opts?: {
|
|
|
90
101
|
export declare function runAgentFixture(fixture: FileFixture, opts?: {
|
|
91
102
|
judgeAdapter?: ProviderAdapter;
|
|
92
103
|
judgeModel?: string;
|
|
104
|
+
allowWorkdirOutsideTmp?: boolean;
|
|
93
105
|
}): Promise<TaskResult>;
|
|
94
106
|
/** Hook the harness up to a session + audit log so task runs are durable. */ export declare function runHarnessWithPersistence(tasks: ScriptedTask[], opts: {
|
|
95
107
|
storeDir: string;
|