klyro 1.0.17 → 1.0.18

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -271,6 +271,13 @@ export type RuntimeEvent = {
271
271
  kind: 'model_override';
272
272
  requested: string;
273
273
  effective: string;
274
+ } | {
275
+ kind: 'turn.summary';
276
+ turn: number;
277
+ toolCalls: number;
278
+ inputTokens: number;
279
+ outputTokens: number;
280
+ durationMs: number;
274
281
  };
275
282
  export interface RunResult {
276
283
  status: 'complete' | 'max_steps' | 'aborted' | 'no_final' | 'verify_failed' | 'limit' | 'blocked' | 'stuck';
@@ -119,6 +119,8 @@ export async function run(opts, deps) {
119
119
  }
120
120
  };
121
121
  let steps = 0;
122
+ // 4.5c — per-turn start time for turn.summary durationMs.
123
+ let turnStartMs = Date.now();
122
124
  let lastRemindTurn = 0;
123
125
  let toolCallCount = 0;
124
126
  let finalText = '';
@@ -247,6 +249,23 @@ export async function run(opts, deps) {
247
249
  // best-effort — don't crash runtime on persistence failure
248
250
  }
249
251
  }
252
+ /**
253
+ * 4.5c — end-of-turn summary: emit a turn.summary event (cumulative
254
+ * toolCalls + token counts, per-turn durationMs) and record a compact
255
+ * one-line summary in the transcript.
256
+ */
257
+ async function emitTurnSummary() {
258
+ const durationMs = Date.now() - turnStartMs;
259
+ emit?.({ kind: 'turn.summary', turn: steps, toolCalls: toolCallCount, inputTokens: usage.input, outputTokens: usage.output, durationMs });
260
+ const msg = {
261
+ role: 'user',
262
+ content: [text(`[turn ${steps} summary] ${toolCallCount} tool call(s), ${usage.input} in / ${usage.output} out tokens, ${durationMs}ms`)],
263
+ };
264
+ transcript.push(msg);
265
+ await checkpoint(msg);
266
+ // Invalidate token cache since transcript changed
267
+ tokenCache = { lastRef: null, lastSystem: undefined, lastCount: 0 };
268
+ }
250
269
  // Persist initial user message
251
270
  if (store && sessionId && transcript.length > 0) {
252
271
  // Fire-and-forget initial checkpoint (don't await to block loop start)
@@ -322,6 +341,7 @@ export async function run(opts, deps) {
322
341
  return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined, phase: 'blocked' };
323
342
  }
324
343
  steps++;
344
+ turnStartMs = Date.now();
325
345
  // 8.4 — stale-todo reminder: every 20 turns, re-inject pending plan
326
346
  // items from `.klyro/plans/todos.json` (written by todo_write) so a
327
347
  // long run cannot silently drop its checklist. Best-effort + tiny.
@@ -615,6 +635,9 @@ export async function run(opts, deps) {
615
635
  return { status: 'aborted', steps, toolCalls: toolCallCount, finalText, transcript, hasEdits, usage, repairs, verification: hasEdits ? withRepairTokens({ ok: false, attempts: verificationAttempts }) : undefined };
616
636
  }
617
637
  if (finalizedCalls.length === 0) {
638
+ // 4.5c — the turn completed (final text, malformed-only, or
639
+ // stop-hook continuation): summarize before the verify branching.
640
+ await emitTurnSummary();
618
641
  // Steerable stop: a stop hook asked for one more turn instead of
619
642
  // completing. Consumed once per verdict, max 3 per run.
620
643
  if (stopCont !== null && stopContUsed < 3) {
@@ -1051,6 +1074,18 @@ export async function run(opts, deps) {
1051
1074
  }
1052
1075
  }
1053
1076
  }
1077
+ // 4.5a — shell/git mutations bypass file-change inference: take a
1078
+ // pre-execution snapshot so the turn stays restorable. Best-effort.
1079
+ if (call.name === 'shell_exec' || call.name.startsWith('git_')) {
1080
+ try {
1081
+ const { snapshot } = await import('../checkpoints/store.js');
1082
+ await snapshot(opts.cwd, [...fileEditCounts.keys()].slice(-20), {
1083
+ ...(sessionId !== undefined ? { sessionId } : {}),
1084
+ eventId: call.id,
1085
+ });
1086
+ }
1087
+ catch { /* ignore */ }
1088
+ }
1054
1089
  let obs;
1055
1090
  try {
1056
1091
  obs = await deps.registry.execute(call.name, call.input, toolCtx);
@@ -1138,7 +1173,10 @@ export async function run(opts, deps) {
1138
1173
  // 4.5 — checkpoint snapshot after each mutation
1139
1174
  try {
1140
1175
  const { snapshot } = await import('../checkpoints/store.js');
1141
- await snapshot(opts.cwd, [fileChanged.path]);
1176
+ await snapshot(opts.cwd, [fileChanged.path], {
1177
+ ...(sessionId !== undefined ? { sessionId } : {}),
1178
+ eventId: call.id,
1179
+ });
1142
1180
  }
1143
1181
  catch { /* ignore */ }
1144
1182
  // 5.2 file edit count
@@ -1251,6 +1289,8 @@ export async function run(opts, deps) {
1251
1289
  break;
1252
1290
  }
1253
1291
  }
1292
+ // 4.5c — the tool turn completed: summarize (cumulative counts).
1293
+ await emitTurnSummary();
1254
1294
  // P0 — drain finished sub-agent completions into parent visibility.
1255
1295
  // drainCompletions is OPTIONAL on the bridge — guarded with `?.` so
1256
1296
  // older bridges without it simply yield nothing.
@@ -10,7 +10,21 @@
10
10
  * enforced at the trace/persist boundaries (TraceWriter, SessionStore)
11
11
  * instead of here.
12
12
  */
13
- export declare function snapshot(cwd: string, files: string[]): Promise<string>;
13
+ /** Optional provenance recorded into a checkpoint's .meta.json. */
14
+ export interface SnapshotOptions {
15
+ sessionId?: string;
16
+ eventId?: string;
17
+ }
18
+ /** Checkpoint meta on disk — sessionId/eventId are absent on old metas. */
19
+ export interface CheckpointMeta {
20
+ id: string;
21
+ files: string[];
22
+ missing: string[];
23
+ ts: number;
24
+ sessionId?: string;
25
+ eventId?: string;
26
+ }
27
+ export declare function snapshot(cwd: string, files: string[], opts?: SnapshotOptions): Promise<string>;
14
28
  export declare function listCheckpoints(cwd: string): Promise<string[]>;
15
29
  export interface CheckpointInfo {
16
30
  /** 1-based index from the latest (1 = newest, like `undo(n)`). */
@@ -55,7 +55,7 @@ function containedPath(cwd, base, rel) {
55
55
  return null;
56
56
  return out;
57
57
  }
58
- export async function snapshot(cwd, files) {
58
+ export async function snapshot(cwd, files, opts) {
59
59
  const dir = ckptDir(cwd);
60
60
  await fs.mkdir(dir, { recursive: true });
61
61
  lockDown(dir, 0o700);
@@ -92,7 +92,12 @@ export async function snapshot(cwd, files) {
92
92
  // leaves a truncated .meta.json that undo() then trusts).
93
93
  const metaPath = path.join(dest, '.meta.json');
94
94
  const metaTmp = `${metaPath}.tmp-${Date.now()}-${Math.random().toString(36).slice(2, 6)}`;
95
- await fs.writeFile(metaTmp, JSON.stringify({ id, files: kept, missing, ts: Date.now() }, null, 2));
95
+ const meta = { id, files: kept, missing, ts: Date.now() };
96
+ if (opts?.sessionId !== undefined)
97
+ meta.sessionId = opts.sessionId;
98
+ if (opts?.eventId !== undefined)
99
+ meta.eventId = opts.eventId;
100
+ await fs.writeFile(metaTmp, JSON.stringify(meta, null, 2));
96
101
  lockDown(metaTmp, 0o600);
97
102
  await fsyncFile(metaTmp);
98
103
  try {
@@ -1 +1,8 @@
1
1
  export declare function parsePositiveInt(name: string, v: string): number;
2
+ /**
3
+ * 5.3 — `--auto-answer <text>` wiring. `ask_user` honors
4
+ * `KLYRO_AUTO_ANSWER` in headless runs; `klyro run` / `klyro eval` set it
5
+ * from the flag via this helper before the run starts (explicit env wins
6
+ * when the flag is omitted). Exported for tests.
7
+ */
8
+ export declare function applyAutoAnswer(value: string | undefined): void;
package/dist/cli/args.js CHANGED
@@ -12,3 +12,13 @@ export function parsePositiveInt(name, v) {
12
12
  }
13
13
  return n;
14
14
  }
15
+ /**
16
+ * 5.3 — `--auto-answer <text>` wiring. `ask_user` honors
17
+ * `KLYRO_AUTO_ANSWER` in headless runs; `klyro run` / `klyro eval` set it
18
+ * from the flag via this helper before the run starts (explicit env wins
19
+ * when the flag is omitted). Exported for tests.
20
+ */
21
+ export function applyAutoAnswer(value) {
22
+ if (value !== undefined)
23
+ process.env.KLYRO_AUTO_ANSWER = value;
24
+ }
@@ -81,6 +81,22 @@ export interface EvalResult {
81
81
  };
82
82
  /** Isolated workdir the scenario ran in (tmp unless --cwd). Debugging aid. */
83
83
  workDir?: string;
84
+ /** 5.4a — provider token usage for the scenario run. */
85
+ tokens?: {
86
+ input: number;
87
+ output: number;
88
+ };
89
+ /** 5.4a — estimated USD cost (runtime pricing helper, same rates as budgets). */
90
+ costUsd?: number;
91
+ /** 5.4a — distinct file paths referenced by tool-call inputs (best-effort). */
92
+ filesTouched?: string[];
93
+ /** 5.4a — verification outcome passthrough. */
94
+ verification?: {
95
+ ok: boolean;
96
+ attempts: number;
97
+ };
98
+ /** 5.4a — repair count passthrough. */
99
+ repairs?: number;
84
100
  }
85
101
  export interface RunEvalOptions {
86
102
  inputPath: string;
@@ -98,10 +114,26 @@ export interface RunEvalOptions {
98
114
  * touch the caller's directory. Pass explicitly to inspect artifacts.
99
115
  */
100
116
  cwd?: string;
117
+ /**
118
+ * 5.3 — `--auto-answer <text>`: sets `KLYRO_AUTO_ANSWER` before the run
119
+ * starts so headless `ask_user` calls resolve without prompting.
120
+ */
121
+ autoAnswer?: string;
101
122
  }
123
+ /**
124
+ * 5.5a — curated stable subset of `evals/fixtures` for `--suite core`.
125
+ * Every entry has task.md + check.sh (verified 2026-09-24).
126
+ */
127
+ export declare const CORE_SUITE_FIXTURES: string[];
128
+ /**
129
+ * Filter directory names down to the core suite (in curated order).
130
+ * Exported for tests.
131
+ */
132
+ export declare function selectCoreFixtures(entries: string[]): string[];
102
133
  export declare function runEval(opts: RunEvalOptions): Promise<number>;
103
134
  export declare function scriptedAdapterFromSpec(spec: Array<Array<unknown[]>> | undefined): ProviderAdapter;
104
135
  export declare function runScenario(sc: EvalScenario, judgeOpts?: {
105
136
  adapter: ProviderAdapter;
106
137
  model: string;
107
138
  }, workDir?: string): Promise<EvalResult>;
139
+ export declare function collectFilesTouched(transcript: RunResult['transcript']): string[];
package/dist/cli/eval.js CHANGED
@@ -45,11 +45,26 @@ import * as os from 'node:os';
45
45
  import * as path from 'node:path';
46
46
  import * as readline from 'node:readline/promises';
47
47
  import { stdin as input, stdout, stderr } from 'node:process';
48
- import { run } from '../agent/runtime.js';
48
+ import { run, estimateCost } from '../agent/runtime.js';
49
+ import { applyAutoAnswer } from './args.js';
49
50
  import { builtinRegistry } from '../tools/registry.js';
50
51
  import { builtinRules, DEFAULT_POLICY_CONFIG, PolicyEngine } from '../policy/engine.js';
51
52
  import { DenyAllApprovalPrompt } from '../policy/approval.js';
53
+ /**
54
+ * 5.5a — curated stable subset of `evals/fixtures` for `--suite core`.
55
+ * Every entry has task.md + check.sh (verified 2026-09-24).
56
+ */
57
+ export const CORE_SUITE_FIXTURES = ['read-answer', 'add-fn-test', 'fix-failing-test', 'rename', 'cli-flag'];
58
+ /**
59
+ * Filter directory names down to the core suite (in curated order).
60
+ * Exported for tests.
61
+ */
62
+ export function selectCoreFixtures(entries) {
63
+ return CORE_SUITE_FIXTURES.filter((id) => entries.includes(id));
64
+ }
52
65
  export async function runEval(opts) {
66
+ // 5.3 — --auto-answer sets KLYRO_AUTO_ANSWER before anything runs.
67
+ applyAutoAnswer(opts.autoAnswer);
53
68
  // Live judge adapter (shared by suite + JSONL paths) for `judge.rubric`.
54
69
  let judgeAdapter;
55
70
  if (opts.judgeModel) {
@@ -65,6 +80,7 @@ export async function runEval(opts) {
65
80
  }
66
81
  // 5.4 — suite mode: load from evals/fixtures
67
82
  if (opts.suite) {
83
+ const suiteStart = Date.now();
68
84
  const fs = await import('node:fs/promises');
69
85
  const path = await import('node:path');
70
86
  // For smoke, use the 10 fixtures directly
@@ -75,8 +91,13 @@ export async function runEval(opts) {
75
91
  for (const e of entries) {
76
92
  if (opts.filter && !e.includes(opts.filter))
77
93
  continue;
78
- // 6.5 — suite filter: smoke = type smoke, l6 = prefix l6-introduce
79
- if (opts.suite && opts.suite !== 'smoke') {
94
+ // 5.5a — core = curated stable subset (see CORE_SUITE_FIXTURES).
95
+ if (opts.suite === 'core') {
96
+ if (!CORE_SUITE_FIXTURES.includes(e))
97
+ continue;
98
+ // 6.5 — suite filter: smoke = type smoke, l6 = prefix l6-introduce
99
+ }
100
+ else if (opts.suite && opts.suite !== 'smoke') {
80
101
  if (!e.startsWith(opts.suite) && !e.includes(opts.suite))
81
102
  continue;
82
103
  }
@@ -141,6 +162,16 @@ export async function runEval(opts) {
141
162
  await fs.mkdir(outDir, { recursive: true });
142
163
  const ts = new Date().toISOString().replace(/[:.]/g, '-');
143
164
  await fs.writeFile(path.join(outDir, `${ts}.json`), JSON.stringify({ suite: opts.suite, results }, null, 2));
165
+ // 5.5b — refresh the tracked baseline with environment metadata.
166
+ const { collectBaselineMetadata, buildEvalBaseline, writeEvalBaseline } = await import('../eval/baseline.js');
167
+ const meta = collectBaselineMetadata({ model: opts.model });
168
+ await writeEvalBaseline(outDir, buildEvalBaseline(meta, {
169
+ suite: opts.suite ?? '',
170
+ total: results.length,
171
+ passed,
172
+ failed: results.length - passed,
173
+ durationMs: Date.now() - suiteStart,
174
+ }));
144
175
  }
145
176
  catch { /* ignore */ }
146
177
  return passed === results.length ? 0 : 1;
@@ -165,6 +196,18 @@ export async function runEval(opts) {
165
196
  stdout.write(`[${tag}] ${r.name} (${r.durationMs}ms, ${r.steps} steps, ${r.toolCalls} tools)\n`);
166
197
  for (const f of r.failures)
167
198
  stdout.write(` - ${f}\n`);
199
+ // 5.4a — rich record extras (tokens/cost/files/verification).
200
+ const extras = [];
201
+ if (r.tokens)
202
+ extras.push(`${r.tokens.input} in/${r.tokens.output} out`);
203
+ if (r.costUsd !== undefined)
204
+ extras.push(`$${r.costUsd.toFixed(4)}`);
205
+ if (r.filesTouched && r.filesTouched.length > 0)
206
+ extras.push(`${r.filesTouched.length} files`);
207
+ if (r.verification)
208
+ extras.push(`verify:${r.verification.ok ? 'ok' : 'fail'}`);
209
+ if (extras.length > 0)
210
+ stdout.write(` · ${extras.join(' · ')}\n`);
168
211
  }
169
212
  }
170
213
  const passed = results.filter((r) => r.passed).length;
@@ -332,5 +375,41 @@ export async function runScenario(sc, judgeOpts, workDir) {
332
375
  durationMs: 0,
333
376
  workDir: cwd,
334
377
  ...(judge ? { judge } : {}),
378
+ // 5.4a — rich records mapped from RunResult (cost reuses the runtime
379
+ // pricing helper, the same rates the budget code uses).
380
+ tokens: { input: result.usage.input, output: result.usage.output },
381
+ costUsd: estimateCost(model, result.usage),
382
+ filesTouched: collectFilesTouched(result.transcript),
383
+ ...(result.verification
384
+ ? { verification: { ok: result.verification.ok, attempts: result.verification.attempts } }
385
+ : {}),
386
+ ...(result.repairs !== undefined ? { repairs: result.repairs } : {}),
335
387
  };
336
388
  }
389
+ /**
390
+ * 5.4a — best-effort distinct file paths from tool-call inputs in a run
391
+ * transcript. Looks for common path-ish keys on `tool_use` blocks.
392
+ * Exported for tests.
393
+ */
394
+ const FILE_PATH_KEYS = ['path', 'file', 'filePath', 'filename', 'target'];
395
+ export function collectFilesTouched(transcript) {
396
+ const seen = new Set();
397
+ for (const m of transcript) {
398
+ for (const b of m.content) {
399
+ if (b.kind !== 'tool_use')
400
+ continue;
401
+ const input = b.input;
402
+ for (const k of FILE_PATH_KEYS) {
403
+ const v = input[k];
404
+ if (typeof v === 'string' && v.length > 0)
405
+ seen.add(v);
406
+ else if (Array.isArray(v)) {
407
+ for (const e of v)
408
+ if (typeof e === 'string' && e.length > 0)
409
+ seen.add(e);
410
+ }
411
+ }
412
+ }
413
+ }
414
+ return [...seen].sort();
415
+ }
package/dist/cli/run.d.ts CHANGED
@@ -91,6 +91,11 @@ export interface RunCliOptions {
91
91
  /** P1.4 — run the task under a named child-capable orchestrator context. */
92
92
  agent?: string;
93
93
  maxDepth?: number;
94
+ /**
95
+ * 5.3 — `--auto-answer <text>`: answer `ask_user` prompts headlessly.
96
+ * Applied to `process.env.KLYRO_AUTO_ANSWER` at the top of `runOnce`.
97
+ */
98
+ autoAnswer?: string;
94
99
  }
95
100
  /**
96
101
  * Split a comma-separated tool/dir list (string or string[]) into names.
package/dist/cli/run.js CHANGED
@@ -23,6 +23,7 @@ import { memoryBlock } from '../context/memory.js';
23
23
  import { estimateCost } from '../providers/model-info.js';
24
24
  import { resolveModelAlias } from '../providers/model-info.js';
25
25
  import { logDebug, logInfo } from '../util/log.js';
26
+ import { applyAutoAnswer } from './args.js';
26
27
  import { resolveSessionId } from '../persistence/session.js';
27
28
  import * as fs from 'node:fs';
28
29
  /**
@@ -71,6 +72,9 @@ export function shouldForceExit(lastSigintAt, now) {
71
72
  return lastSigintAt !== undefined && now - lastSigintAt < 1500;
72
73
  }
73
74
  export async function runOnce(opts) {
75
+ // 5.3 — --auto-answer sets KLYRO_AUTO_ANSWER before anything runs so
76
+ // headless ask_user calls resolve without prompting.
77
+ applyAutoAnswer(opts.autoAnswer);
74
78
  // P0.5 — load <cwd>/.env first so KLYRO_* vars resolve without `export`.
75
79
  // Never throws (missing file is a no-op); explicit env wins (no-clobber).
76
80
  try {
@@ -12,5 +12,16 @@ export interface KlyroMdFile {
12
12
  }
13
13
  /** Per-file loader (used by the trust gate to approve/hash individual files). */
14
14
  export declare function loadKlyroMdFiles(cwd: string): Promise<KlyroMdFile[]>;
15
+ /**
16
+ * 4.4a — Subdirectory instructions: root files PLUS instruction files found
17
+ * in ancestor directories of targetPath from cwd down to targetPath's dir
18
+ * (nearest-wins order root→leaf, each capped like cap()).
19
+ *
20
+ * Laziness preserved: plain loadKlyroMdFiles(cwd) never reads subdirs —
21
+ * callers pass the relevant active path explicitly.
22
+ *
23
+ * Containment: a targetPath escaping cwd is rejected (root files only).
24
+ */
25
+ export declare function loadKlyroMdForPath(cwd: string, targetPath: string): Promise<KlyroMdFile[]>;
15
26
  export declare function loadKlyroMd(cwd: string): Promise<string>;
16
27
  export declare function handleInit(cwd: string): Promise<string>;
@@ -30,16 +30,65 @@ export async function loadKlyroMdFiles(cwd) {
30
30
  // Klyro's own instruction file is KLYRO.md. The rest are read-only
31
31
  // fallbacks for imported repos that follow other conventions — Klyro
32
32
  // never writes them. KLYRO.md wins by load order.
33
- for (const name of ['KLYRO.md', 'KLYRO.local.md', 'CLAUDE.md', 'CLAUDE.local.md', 'AGENTS.md', '.cursorrules']) {
33
+ for (const name of INSTRUCTION_NAMES) {
34
34
  const p = path.join(cwd, name);
35
35
  try {
36
36
  const t = await fs.readFile(p, 'utf-8');
37
- files.push({ path: p, content: cap(await resolveImports(t, path.dirname(p), cwd), MAX_FILE_CHARS) });
37
+ files.push({ path: p, content: cap(await resolveImports(t, path.dirname(p), cwd, 0, [p]), MAX_FILE_CHARS) });
38
38
  }
39
39
  catch { /* ignore */ }
40
40
  }
41
41
  return files;
42
42
  }
43
+ /** Instruction file names read at every level (root + subdirectories). */
44
+ const INSTRUCTION_NAMES = ['KLYRO.md', 'KLYRO.local.md', 'CLAUDE.md', 'CLAUDE.local.md', 'AGENTS.md', '.cursorrules'];
45
+ /**
46
+ * 4.4a — Subdirectory instructions: root files PLUS instruction files found
47
+ * in ancestor directories of targetPath from cwd down to targetPath's dir
48
+ * (nearest-wins order root→leaf, each capped like cap()).
49
+ *
50
+ * Laziness preserved: plain loadKlyroMdFiles(cwd) never reads subdirs —
51
+ * callers pass the relevant active path explicitly.
52
+ *
53
+ * Containment: a targetPath escaping cwd is rejected (root files only).
54
+ */
55
+ export async function loadKlyroMdForPath(cwd, targetPath) {
56
+ const files = await loadKlyroMdFiles(cwd);
57
+ const root = path.resolve(cwd);
58
+ const absTarget = path.resolve(root, targetPath);
59
+ const relTarget = path.relative(root, absTarget);
60
+ if (relTarget.startsWith('..') || path.isAbsolute(relTarget))
61
+ return files;
62
+ // Target dir: the target itself when it is an existing directory,
63
+ // otherwise its parent dir (i.e. the target is a file being edited).
64
+ let targetDir = absTarget;
65
+ try {
66
+ const st = await fs.stat(absTarget);
67
+ if (!st.isDirectory())
68
+ targetDir = path.dirname(absTarget);
69
+ }
70
+ catch {
71
+ targetDir = path.dirname(absTarget);
72
+ }
73
+ const relDir = path.relative(root, targetDir);
74
+ if (relDir.startsWith('..') || path.isAbsolute(relDir))
75
+ return files;
76
+ const segs = relDir.split(path.sep).filter((s) => s && s !== '.');
77
+ let prefix = root;
78
+ for (const seg of segs) {
79
+ prefix = path.join(prefix, seg);
80
+ const dir = prefix;
81
+ for (const name of INSTRUCTION_NAMES) {
82
+ const p = path.join(dir, name);
83
+ try {
84
+ const t = await fs.readFile(p, 'utf-8');
85
+ files.push({ path: p, content: cap(await resolveImports(t, path.dirname(p), root, 0, [p]), MAX_FILE_CHARS) });
86
+ }
87
+ catch { /* ignore */ }
88
+ }
89
+ }
90
+ return files;
91
+ }
43
92
  export async function loadKlyroMd(cwd) {
44
93
  const parts = [];
45
94
  let total = 0;
@@ -56,7 +105,7 @@ export async function loadKlyroMd(cwd) {
56
105
  }
57
106
  return parts.join('\n\n---\n\n');
58
107
  }
59
- async function resolveImports(text, base, root, depth = 0) {
108
+ async function resolveImports(text, base, root, depth = 0, chain = []) {
60
109
  if (depth > 5)
61
110
  return text;
62
111
  const importRe = /^@import\s+(.+)$/gm;
@@ -69,9 +118,15 @@ async function resolveImports(text, base, root, depth = 0) {
69
118
  const relToRoot = path.relative(root, p);
70
119
  if (relToRoot.startsWith('..') || path.isAbsolute(relToRoot))
71
120
  continue;
121
+ // Cycle guard: an @import target already on the current chain resolves
122
+ // to a marker instead of recursing (depth cap stays as backstop).
123
+ if (chain.includes(p)) {
124
+ out = out.replace(m[0], `<!-- klyro: import cycle skipped: ${rel} -->`);
125
+ continue;
126
+ }
72
127
  try {
73
128
  const t = await fs.readFile(p, 'utf-8');
74
- const resolved = await resolveImports(cap(t, MAX_FILE_CHARS), path.dirname(p), root, depth + 1);
129
+ const resolved = await resolveImports(cap(t, MAX_FILE_CHARS), path.dirname(p), root, depth + 1, [...chain, p]);
75
130
  out = out.replace(m[0], resolved);
76
131
  }
77
132
  catch { /* ignore missing */ }
@@ -0,0 +1,47 @@
1
+ /**
2
+ * 5.5b — eval baseline writer. Records the environment alongside suite
3
+ * results so regressions can be attributed (model/node/platform/commit).
4
+ *
5
+ * The tracked `evals/results/baseline.json` is refreshed by `klyro eval
6
+ * --suite <name>` (best-effort; suite failures never fail the write path).
7
+ */
8
+ export interface EvalBaselineMeta {
9
+ /** Model id the suite ran under (KLYRO_MODEL fallback, then 'mock'). */
10
+ model: string;
11
+ /** Node version (process.version). */
12
+ node: string;
13
+ /** OS platform (process.platform). */
14
+ platform: string;
15
+ /** Git HEAD, best-effort ('unknown' outside a repo). */
16
+ commit: string;
17
+ /** ISO timestamp of the run. */
18
+ timestamp: string;
19
+ }
20
+ export interface EvalBaseline extends EvalBaselineMeta {
21
+ suite: string;
22
+ total: number;
23
+ passed: number;
24
+ failed: number;
25
+ passRate: number;
26
+ durationMs: number;
27
+ }
28
+ /** Best-effort `git rev-parse HEAD` (never throws). Exported for tests. */
29
+ export declare function gitCommit(cwd?: string): string;
30
+ /** Collect the environment half of a baseline record. Exported for tests. */
31
+ export declare function collectBaselineMetadata(opts?: {
32
+ model?: string;
33
+ cwd?: string;
34
+ }): EvalBaselineMeta;
35
+ /** Combine metadata + suite counts into a full baseline record. */
36
+ export declare function buildEvalBaseline(meta: EvalBaselineMeta, suite: {
37
+ suite: string;
38
+ total: number;
39
+ passed: number;
40
+ failed: number;
41
+ durationMs: number;
42
+ }): EvalBaseline;
43
+ /**
44
+ * Write `baseline.json` into `dir` (created if needed). Returns the path.
45
+ * Never throws for missing dirs — the caller wraps in try/catch anyway.
46
+ */
47
+ export declare function writeEvalBaseline(dir: string, baseline: EvalBaseline): Promise<string>;
@@ -0,0 +1,55 @@
1
+ /**
2
+ * 5.5b — eval baseline writer. Records the environment alongside suite
3
+ * results so regressions can be attributed (model/node/platform/commit).
4
+ *
5
+ * The tracked `evals/results/baseline.json` is refreshed by `klyro eval
6
+ * --suite <name>` (best-effort; suite failures never fail the write path).
7
+ */
8
+ import * as fsp from 'node:fs/promises';
9
+ import * as path from 'node:path';
10
+ import { execFileSync } from 'node:child_process';
11
+ /** Best-effort `git rev-parse HEAD` (never throws). Exported for tests. */
12
+ export function gitCommit(cwd = process.cwd()) {
13
+ try {
14
+ const out = execFileSync('git', ['rev-parse', 'HEAD'], {
15
+ cwd,
16
+ stdio: ['ignore', 'pipe', 'ignore'],
17
+ }).toString().trim();
18
+ return out || 'unknown';
19
+ }
20
+ catch {
21
+ return 'unknown';
22
+ }
23
+ }
24
+ /** Collect the environment half of a baseline record. Exported for tests. */
25
+ export function collectBaselineMetadata(opts = {}) {
26
+ return {
27
+ model: opts.model ?? process.env.KLYRO_MODEL ?? 'mock',
28
+ node: process.version,
29
+ platform: process.platform,
30
+ commit: gitCommit(opts.cwd),
31
+ timestamp: new Date().toISOString(),
32
+ };
33
+ }
34
+ /** Combine metadata + suite counts into a full baseline record. */
35
+ export function buildEvalBaseline(meta, suite) {
36
+ return {
37
+ ...meta,
38
+ suite: suite.suite,
39
+ total: suite.total,
40
+ passed: suite.passed,
41
+ failed: suite.failed,
42
+ passRate: suite.total === 0 ? 0 : suite.passed / suite.total,
43
+ durationMs: suite.durationMs,
44
+ };
45
+ }
46
+ /**
47
+ * Write `baseline.json` into `dir` (created if needed). Returns the path.
48
+ * Never throws for missing dirs — the caller wraps in try/catch anyway.
49
+ */
50
+ export async function writeEvalBaseline(dir, baseline) {
51
+ await fsp.mkdir(dir, { recursive: true });
52
+ const out = path.join(dir, 'baseline.json');
53
+ await fsp.writeFile(out, JSON.stringify(baseline, null, 2) + '\n');
54
+ return out;
55
+ }
@@ -77,10 +77,21 @@ export interface FileFixture {
77
77
  script?: StreamEvent[][];
78
78
  }
79
79
  export declare function loadFileFixture(dir: string): Promise<FileFixture>;
80
- export declare function runFileFixture(fixture: FileFixture, _opts?: {
80
+ export declare function sanitizedFixtureEnv(base?: NodeJS.ProcessEnv): NodeJS.ProcessEnv;
81
+ /**
82
+ * 5.4b — guard: fixture workdirs must stay inside `os.tmpdir()` (the
83
+ * harness relies on that isolation). Throws otherwise unless the caller
84
+ * explicitly passes `allowWorkdirOutsideTmp`. Exported for tests.
85
+ */
86
+ export declare function assertTmpWorkdir(dir: string, opts?: {
87
+ allowWorkdirOutsideTmp?: boolean;
88
+ }): void;
89
+ export interface FileFixtureOptions {
81
90
  runs?: number;
82
91
  parallel?: number;
83
- }): Promise<TaskResult>;
92
+ allowWorkdirOutsideTmp?: boolean;
93
+ }
94
+ export declare function runFileFixture(fixture: FileFixture, _opts?: FileFixtureOptions): Promise<TaskResult>;
84
95
  /**
85
96
  * Agent-driven fixture: seed a tmp repo, run the fixture's canned script
86
97
  * through the REAL runtime + tools, then assert outcomes with check.sh
@@ -90,6 +101,7 @@ export declare function runFileFixture(fixture: FileFixture, _opts?: {
90
101
  export declare function runAgentFixture(fixture: FileFixture, opts?: {
91
102
  judgeAdapter?: ProviderAdapter;
92
103
  judgeModel?: string;
104
+ allowWorkdirOutsideTmp?: boolean;
93
105
  }): Promise<TaskResult>;
94
106
  /** Hook the harness up to a session + audit log so task runs are durable. */ export declare function runHarnessWithPersistence(tasks: ScriptedTask[], opts: {
95
107
  storeDir: string;