klyro 1.0.5 → 1.0.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. package/README.md +13 -0
  2. package/dist/agent/custom-agents.d.ts +3 -0
  3. package/dist/agent/custom-agents.js +96 -0
  4. package/dist/agent/orchestrator.d.ts +26 -0
  5. package/dist/agent/orchestrator.js +41 -4
  6. package/dist/agent/runtime.d.ts +15 -0
  7. package/dist/agent/runtime.js +232 -61
  8. package/dist/chat.d.ts +10 -0
  9. package/dist/chat.js +39 -7
  10. package/dist/checkpoints/store.d.ts +11 -0
  11. package/dist/checkpoints/store.js +32 -0
  12. package/dist/cli/auth.d.ts +10 -3
  13. package/dist/cli/auth.js +43 -5
  14. package/dist/cli/completion.js +2 -2
  15. package/dist/cli/config.d.ts +4 -4
  16. package/dist/cli/doctor.js +0 -1
  17. package/dist/cli/eval.d.ts +15 -1
  18. package/dist/cli/eval.js +43 -5
  19. package/dist/cli/hooks.d.ts +74 -5
  20. package/dist/cli/hooks.js +118 -7
  21. package/dist/cli/init.d.ts +6 -0
  22. package/dist/cli/init.js +60 -0
  23. package/dist/cli/keychain.d.ts +10 -0
  24. package/dist/cli/keychain.js +86 -0
  25. package/dist/cli/repl.js +188 -30
  26. package/dist/cli/run.d.ts +7 -1
  27. package/dist/cli/run.js +92 -50
  28. package/dist/cli/setup.js +3 -2
  29. package/dist/cli/slash/custom.d.ts +25 -0
  30. package/dist/cli/slash/custom.js +166 -0
  31. package/dist/cli/slash/parser.d.ts +9 -1
  32. package/dist/cli/slash/parser.js +34 -9
  33. package/dist/cli/update.d.ts +3 -1
  34. package/dist/cli/update.js +16 -1
  35. package/dist/context/accounting.d.ts +6 -0
  36. package/dist/context/accounting.js +8 -2
  37. package/dist/context/compaction.d.ts +2 -1
  38. package/dist/context/compaction.js +39 -12
  39. package/dist/context/memory.d.ts +11 -0
  40. package/dist/context/memory.js +59 -4
  41. package/dist/eval/harness.d.ts +40 -5
  42. package/dist/eval/harness.js +103 -10
  43. package/dist/eval/judge.d.ts +32 -0
  44. package/dist/eval/judge.js +63 -0
  45. package/dist/eval/tasks.js +134 -0
  46. package/dist/index.js +239 -130
  47. package/dist/mcp/auth.d.ts +85 -0
  48. package/dist/mcp/auth.js +249 -0
  49. package/dist/mcp/client.d.ts +15 -0
  50. package/dist/mcp/client.js +42 -2
  51. package/dist/mcp/config.d.ts +31 -1
  52. package/dist/mcp/config.js +84 -1
  53. package/dist/mcp/registry.d.ts +19 -0
  54. package/dist/mcp/registry.js +118 -2
  55. package/dist/mcp/remote.d.ts +36 -0
  56. package/dist/mcp/remote.js +207 -0
  57. package/dist/mcp/sse.d.ts +42 -0
  58. package/dist/mcp/sse.js +310 -0
  59. package/dist/persistence/audit.d.ts +15 -3
  60. package/dist/persistence/audit.js +84 -13
  61. package/dist/persistence/store.d.ts +9 -0
  62. package/dist/persistence/store.js +17 -0
  63. package/dist/policy/approval.d.ts +15 -1
  64. package/dist/policy/approval.js +8 -0
  65. package/dist/policy/engine.js +9 -0
  66. package/dist/providers/endpoints.d.ts +43 -0
  67. package/dist/providers/endpoints.js +104 -0
  68. package/dist/providers.js +17 -14
  69. package/dist/tools/shell/shell-exec.d.ts +13 -0
  70. package/dist/tools/shell/shell-exec.js +64 -2
  71. package/dist/tui/app.js +172 -15
  72. package/dist/tui/app.test.js +27 -2
  73. package/dist/tui/approval.js +55 -1
  74. package/dist/tui/scroll-model.d.ts +2 -2
  75. package/dist/tui/scroll-model.js +9 -3
  76. package/dist/tui/tokens.d.ts +8 -11
  77. package/dist/tui/tokens.js +18 -11
  78. package/package.json +1 -1
@@ -4,8 +4,14 @@
4
4
  */
5
5
  import { totalTokens } from './tokenizer.js';
6
6
  import { getModelInfo } from '../providers/model-info.js';
7
+ /**
8
+ * Single output-reserve used by every budget in the harness (displayed
9
+ * accounting, runtime enforcement, compaction elide). One constant so the
10
+ * meter and the enforcer can never disagree.
11
+ */
12
+ export const RESERVE_OUTPUT_TOKENS = 8000;
7
13
  export function accounting(system, messages, opts = {}) {
8
- const reserveOutput = opts.reserveOutput ?? 16_000;
14
+ const reserveOutput = opts.reserveOutput ?? RESERVE_OUTPUT_TOKENS;
9
15
  // Window-aware default: the legacy 120k ceiling overflows small-window
10
16
  // models (e.g. 8k local models) and wastes large ones — size to the model.
11
17
  const cap = opts.cap ?? capForModel(opts.model, reserveOutput);
@@ -21,7 +27,7 @@ export function accounting(system, messages, opts = {}) {
21
27
  * tiny windows still function. Unknown models use the registry fallback
22
28
  * window (100k); a missing model name keeps the legacy 120k.
23
29
  */
24
- export function capForModel(model, reserveOutput = 16_000) {
30
+ export function capForModel(model, reserveOutput = RESERVE_OUTPUT_TOKENS) {
25
31
  if (!model)
26
32
  return 120_000;
27
33
  const window = getModelInfo(model).contextWindow;
@@ -2,13 +2,14 @@
2
2
  * 8.3 — Auto-compaction: (a) elide → (b) summarize 60% with model.small → (c) keep last N verbatim
3
3
  * Trigger at compactAt (80%) or /compact [focus]. Validates summary mentions every checkpointed file else fallback.
4
4
  */
5
- import type { Message } from '../agent/message.js';
5
+ import type { Message, ContentBlock } from '../agent/message.js';
6
6
  export interface CompactionResult {
7
7
  messages: Message[];
8
8
  summary: string;
9
9
  dropped: number;
10
10
  method: 'elide' | 'summarize' | 'fallback';
11
11
  }
12
+ export declare function textBlock(s: string): ContentBlock;
12
13
  export declare function compact(messages: Message[], opts: {
13
14
  system?: string;
14
15
  cap?: number;
@@ -1,14 +1,37 @@
1
1
  import { compressTranscript } from './tokenizer.js';
2
- import { capForModel } from './accounting.js';
2
+ import { capForModel, RESERVE_OUTPUT_TOKENS } from './accounting.js';
3
+ export function textBlock(s) {
4
+ return { kind: 'text', text: s };
5
+ }
6
+ /**
7
+ * Tightened checkpoint validation: every checkpointed file must be mentioned
8
+ * by a path segment, not a loose basename substring. `src/util.ts` is matched
9
+ * by "util.ts" only when it names a segment; `foo` never matches `foo.png`
10
+ * nor a path nested inside it.
11
+ */
12
+ function mentionsFile(summary, filePath) {
13
+ const segs = filePath.split('/').filter(Boolean);
14
+ const tail = segs[segs.length - 1] ?? filePath;
15
+ // Match the full basename OR any path segment borne by the file.
16
+ // Prefer exact-path segments (src/util.ts or util.ts) to avoid the
17
+ // substring false-positive the old `includes(basename)` produced.
18
+ const quoted = filePath.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
19
+ const quotedTail = tail.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
20
+ return new RegExp(`(^|[\\W_])${quotedTail}([\\W_]|$)`).test(summary) || summary.includes(quoted);
21
+ }
3
22
  export async function compact(messages, opts) {
4
23
  const cap = opts.cap ?? capForModel(opts.model);
5
24
  // (a) elide old tool results
6
- const elided = compressTranscript(opts.system, messages, { total: cap, reservedOutput: 16_000 });
25
+ const elided = compressTranscript(opts.system, messages, { total: cap, reservedOutput: RESERVE_OUTPUT_TOKENS });
7
26
  if (opts.checkpointedFiles && elided.dropped > 0) {
8
- // (b) try summarize oldest 60% using strict template (mock small model)
9
- const n = Math.floor(messages.length * 0.6);
10
- const oldest = messages.slice(0, n);
11
- const newest = messages.slice(n);
27
+ // (b) try summarize oldest 60% using strict template (mock small model).
28
+ // Slice from the ELIDED sequence so the summarize pass operates on what
29
+ // actually survives elision (previously it sliced the raw `messages`,
30
+ // silently dropping the elide benefit on the oldest portion).
31
+ const elidedBest = elided.messages;
32
+ const n = Math.floor(elidedBest.length * 0.6);
33
+ const oldest = elidedBest.slice(0, n);
34
+ const newest = elidedBest.slice(n);
12
35
  const template = `Summarize the following ${oldest.length} messages, mentioning every file: ${opts.checkpointedFiles.join(', ')}.\nFocus: ${opts.focus ?? 'general'}\n\n` + oldest.map((m) => JSON.stringify(m.content).slice(0, 500)).join('\n');
13
36
  let summary = `Earlier in session: ${oldest.length} turns covering ${opts.checkpointedFiles.join(', ')}`;
14
37
  if (opts.summarizeFn) {
@@ -17,24 +40,28 @@ export async function compact(messages, opts) {
17
40
  }
18
41
  catch { /* fallback */ }
19
42
  }
20
- // validate: every checkpointed file mentioned
21
- const missing = opts.checkpointedFiles.filter((f) => !summary.includes(f.split('/').pop() ?? f));
43
+ // validate: every checkpointed file mentioned (path-segment match).
44
+ const missing = opts.checkpointedFiles.filter((f) => !mentionsFile(summary, f));
22
45
  if (missing.length === 0) {
23
- // (c) keep last N verbatim + summary as first message
24
- const summaryMsg = { role: 'user', content: [{ kind: 'text', text: summary }] };
46
+ // (c) keep last N verbatim + summary as first message (typed block).
47
+ const summaryMsg = { role: 'user', content: [textBlock(summary)] };
25
48
  return { messages: [summaryMsg, ...newest], summary, dropped: elided.dropped, method: 'summarize' };
26
49
  }
27
50
  // retry once then fallback to elide
28
51
  if (opts.summarizeFn) {
29
52
  try {
30
53
  const retry = await opts.summarizeFn(template + '\nEnsure to mention: ' + missing.join(', '));
31
- if (missing.every((f) => retry.includes(f.split('/').pop() ?? f))) {
32
- const summaryMsg2 = { role: 'user', content: [{ kind: 'text', text: retry }] };
54
+ if (missing.every((f) => mentionsFile(retry, f))) {
55
+ const summaryMsg2 = { role: 'user', content: [textBlock(retry)] };
33
56
  return { messages: [summaryMsg2, ...newest], summary: retry, dropped: elided.dropped, method: 'summarize' };
34
57
  }
35
58
  }
36
59
  catch { /* fallback */ }
37
60
  }
38
61
  }
62
+ // Post-check the elide path too: if it still exceeds cap, keep eliding.
63
+ if (elided.dropped === 0) {
64
+ return { messages, summary: '', dropped: 0, method: 'fallback' };
65
+ }
39
66
  return { messages: elided.messages, summary: elided.dropped > 0 ? `Elided ${elided.dropped} observations` : '', dropped: elided.dropped, method: elided.dropped > 0 ? 'elide' : 'fallback' };
40
67
  }
@@ -3,6 +3,17 @@ import type { PlanStep } from '../agent/runtime.js';
3
3
  export declare const MEMORY_WRITE_LIMIT_CHARS = 8000;
4
4
  /** Steady-state cap (~1k tokens ≈ 4k chars) kept via tail slice. */
5
5
  export declare const MEMORY_STEADY_STATE_CHARS = 4000;
6
+ /** Archive index filename (recall without an LLM pass). */
7
+ export declare const MEMORY_ARCHIVE_INDEX = "archive-index.json";
8
+ /** Max retained archives. */
9
+ export declare const MEMORY_ARCHIVE_KEEP = 20;
10
+ export interface MemoryArchiveEntry {
11
+ file: string;
12
+ ts: number;
13
+ chars: number;
14
+ /** First line of the archived head — lets the model decide what to re-read. */
15
+ preview: string;
16
+ }
6
17
  export declare function memoryWrite(cwd: string, content: string): Promise<string>;
7
18
  export declare function loadMemory(cwd: string): Promise<string>;
8
19
  /** Synchronous read for the system-prompt build path (run on every turn). */
@@ -10,22 +10,63 @@ import * as fs from 'node:fs/promises';
10
10
  import { readFileSync } from 'node:fs';
11
11
  import * as path from 'node:path';
12
12
  import { redact } from '../policy/secret-redactor.js';
13
+ import { resolveAndFollowSymlinks, assertNotSymlink } from '../policy/path-guard.js';
13
14
  /** Single-write ceiling — anything larger is rejected, not sliced. */
14
15
  export const MEMORY_WRITE_LIMIT_CHARS = 8000;
15
16
  /** Steady-state cap (~1k tokens ≈ 4k chars) kept via tail slice. */
16
17
  export const MEMORY_STEADY_STATE_CHARS = 4000;
18
+ /** Archive index filename (recall without an LLM pass). */
19
+ export const MEMORY_ARCHIVE_INDEX = 'archive-index.json';
20
+ /** Max retained archives. */
21
+ export const MEMORY_ARCHIVE_KEEP = 20;
17
22
  export async function memoryWrite(cwd, content) {
18
23
  if (content.length > MEMORY_WRITE_LIMIT_CHARS) {
19
24
  throw new Error(`memory budget exceeded: single write is ${content.length} chars (max ${MEMORY_WRITE_LIMIT_CHARS}) — split it into smaller notes`);
20
25
  }
26
+ // B4 — symlink/TOCTOU guard: the memory dir must stay inside cwd even if
27
+ // `.klyro` or `.klyro/memory` is (or becomes) a symlink. Resolve+follow up
28
+ // the existing parent chain first (defeats an escaping parent symlink), then
29
+ // mkdir, then a final realpath + lstat: a swapped-in final symlink is
30
+ // refused before any write lands outside cwd.
21
31
  const dir = path.join(cwd, '.klyro', 'memory');
32
+ await resolveAndFollowSymlinks(cwd, '.klyro'); // throws if `.klyro` escapes cwd
22
33
  await fs.mkdir(dir, { recursive: true });
23
- const p = path.join(dir, 'session-notes.md');
34
+ const { resolved } = await resolveAndFollowSymlinks(cwd, '.klyro/memory');
35
+ await assertNotSymlink(resolved);
36
+ const p = path.join(resolved, 'session-notes.md');
24
37
  const prev = await fs.readFile(p, 'utf-8').catch(() => '');
25
38
  // S4-at-rest: redact before appending — redact() only fires on secret
26
39
  // shapes (key/token/password with [:=-]), so normal prose survives.
27
- const next = (prev + '\n' + redact(content)).slice(-MEMORY_STEADY_STATE_CHARS); // ≤1k tokens ~4k chars
28
- const tmp = path.join(dir, `.session-notes.md.tmp-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2, 8)}`);
40
+ const full = prev + '\n' + redact(content);
41
+ let next = full;
42
+ if (full.length > MEMORY_STEADY_STATE_CHARS) {
43
+ // Rotation, not silent loss: the dropped head is archived to a dated
44
+ // file (pruned to the latest 20) instead of vanishing, and the archive
45
+ // index is refreshed so the model knows what exists to re-read.
46
+ const head = full.slice(0, full.length - MEMORY_STEADY_STATE_CHARS);
47
+ next = full.slice(-MEMORY_STEADY_STATE_CHARS);
48
+ try {
49
+ const stamp = new Date().toISOString().replace(/[:.]/g, '-');
50
+ const file = `archive-${stamp}.md`;
51
+ await fs.writeFile(path.join(resolved, file), head, 'utf-8');
52
+ const entries = (await fs.readdir(resolved)).filter((e) => e.startsWith('archive-') && e.endsWith('.md')).sort();
53
+ for (const old of entries.slice(0, Math.max(0, entries.length - MEMORY_ARCHIVE_KEEP))) {
54
+ await fs.unlink(path.join(resolved, old)).catch(() => undefined);
55
+ }
56
+ const index = [];
57
+ for (const e of (await fs.readdir(resolved)).filter((e) => e.startsWith('archive-') && e.endsWith('.md')).sort()) {
58
+ try {
59
+ const content = await fs.readFile(path.join(resolved, e), 'utf-8');
60
+ const stat = await fs.stat(path.join(resolved, e));
61
+ index.push({ file: e, ts: Math.round(stat.mtimeMs), chars: content.length, preview: content.split('\n')[0]?.slice(0, 160) ?? '' });
62
+ }
63
+ catch { /* skip unreadable entries */ }
64
+ }
65
+ await fs.writeFile(path.join(resolved, MEMORY_ARCHIVE_INDEX), JSON.stringify(index, null, 2), 'utf-8');
66
+ }
67
+ catch { /* archive is best-effort; the live notes still persist */ }
68
+ }
69
+ const tmp = path.join(resolved, `.session-notes.md.tmp-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2, 8)}`);
29
70
  await fs.writeFile(tmp, next, 'utf-8');
30
71
  try {
31
72
  const fh = await fs.open(tmp, 'r+');
@@ -66,7 +107,21 @@ export function loadMemorySync(cwd) {
66
107
  /** Wrap persisted notes into the injected prompt block; '' when empty. */
67
108
  export function memoryBlock(cwd) {
68
109
  const notes = loadMemorySync(cwd).trim();
69
- return notes ? `\n\n<memory>\n${notes.slice(0, 4000)}\n</memory>` : '';
110
+ if (!notes)
111
+ return '';
112
+ // Recall aid: tell the model which archives exist (with previews) so it
113
+ // can re-read them via read_file instead of losing rotated knowledge.
114
+ let archived = '';
115
+ try {
116
+ const raw = readFileSync(path.join(cwd, '.klyro', 'memory', MEMORY_ARCHIVE_INDEX), 'utf-8');
117
+ const entries = JSON.parse(raw);
118
+ if (Array.isArray(entries) && entries.length > 0) {
119
+ const lines = entries.slice(-5).map((e) => ` - .klyro/memory/${e.file} (${e.chars} chars): ${e.preview}`);
120
+ archived = `\nEarlier notes archived (read one with read_file for detail):\n${lines.join('\n')}`;
121
+ }
122
+ }
123
+ catch { /* no index — nothing archived yet */ }
124
+ return `\n\n<memory>\n${notes.slice(0, 4000)}${archived}\n</memory>`;
70
125
  }
71
126
  export function shouldRemind(turn, lastRemindTurn) {
72
127
  return turn - lastRemindTurn >= 20;
@@ -7,7 +7,7 @@
7
7
  * This is the MVP gate per Devolopment-plan.md / docs/plan.md §10: a
8
8
  * reproducible suite of programmatic tasks. Real repo tasks come in v1.0.
9
9
  */
10
- import type { StreamEvent } from '../agent/provider-adapter.js';
10
+ import type { ProviderAdapter, StreamEvent } from '../agent/provider-adapter.js';
11
11
  export interface ScriptedTask {
12
12
  id: string;
13
13
  description: string;
@@ -20,6 +20,13 @@ export interface ScriptedTask {
20
20
  expectStatus: 'complete' | 'max_steps' | 'aborted' | 'verify_failed' | 'no_final';
21
21
  /** Expected tool-call count. */
22
22
  expectToolCalls?: number;
23
+ /**
24
+ * Semantic rubric graded by a model judge (see judge.ts). Only runs when
25
+ * the caller supplies a judge adapter; otherwise recorded as skipped.
26
+ */
27
+ judge?: {
28
+ rubric: string[];
29
+ };
23
30
  }
24
31
  export interface TaskResult {
25
32
  id: string;
@@ -28,8 +35,17 @@ export interface TaskResult {
28
35
  observedStatus?: string;
29
36
  observedToolCalls?: number;
30
37
  durationMs: number;
38
+ judge?: {
39
+ pass: boolean;
40
+ notes: string;
41
+ skipped: boolean;
42
+ };
31
43
  }
32
- export declare function runTask(t: ScriptedTask): Promise<TaskResult>;
44
+ export declare function runTask(t: ScriptedTask, opts?: {
45
+ judgeAdapter?: ProviderAdapter;
46
+ judgeModel?: string;
47
+ workDir?: string;
48
+ }): Promise<TaskResult>;
33
49
  export interface HarnessSummary {
34
50
  total: number;
35
51
  passed: number;
@@ -38,7 +54,10 @@ export interface HarnessSummary {
38
54
  results: TaskResult[];
39
55
  durationMs: number;
40
56
  }
41
- export declare function runHarness(tasks: ScriptedTask[]): Promise<HarnessSummary>;
57
+ export declare function runHarness(tasks: ScriptedTask[], opts?: {
58
+ judgeAdapter?: ProviderAdapter;
59
+ judgeModel?: string;
60
+ }): Promise<HarnessSummary>;
42
61
  /** Format a harness summary as a markdown report. */
43
62
  export declare function formatReport(summary: HarnessSummary): string;
44
63
  /** 5.4 — File-based fixture support: repo|repo.json, task.md, check.sh, meta.json */
@@ -48,14 +67,30 @@ export interface FileFixture {
48
67
  checkSh: string;
49
68
  meta: Record<string, unknown>;
50
69
  repo?: string;
70
+ /**
71
+ * Optional canned agent turns (StreamEvent[][], same shape as
72
+ * ScriptedTask.script). When present the fixture is agent-driven: the
73
+ * runtime executes the script with real tools in a seeded tmp repo and
74
+ * check.sh then asserts the filesystem outcomes.
75
+ */
76
+ script?: StreamEvent[][];
51
77
  }
52
78
  export declare function loadFileFixture(dir: string): Promise<FileFixture>;
53
79
  export declare function runFileFixture(fixture: FileFixture, opts?: {
54
80
  runs?: number;
55
81
  parallel?: number;
56
82
  }): Promise<TaskResult>;
57
- /** Hook the harness up to a session + audit log so task runs are durable. */
58
- export declare function runHarnessWithPersistence(tasks: ScriptedTask[], opts: {
83
+ /**
84
+ * Agent-driven fixture: seed a tmp repo, run the fixture's canned script
85
+ * through the REAL runtime + tools, then assert outcomes with check.sh
86
+ * (plus optional meta judge rubric). This is what makes fixtures more than
87
+ * static `exit 0` stubs: the agent loop genuinely executes.
88
+ */
89
+ export declare function runAgentFixture(fixture: FileFixture, opts?: {
90
+ judgeAdapter?: ProviderAdapter;
91
+ judgeModel?: string;
92
+ }): Promise<TaskResult>;
93
+ /** Hook the harness up to a session + audit log so task runs are durable. */ export declare function runHarnessWithPersistence(tasks: ScriptedTask[], opts: {
59
94
  storeDir: string;
60
95
  auditPath: string;
61
96
  }): Promise<HarnessSummary>;
@@ -42,9 +42,13 @@ function scriptedAdapter(script) {
42
42
  },
43
43
  };
44
44
  }
45
- export async function runTask(t) {
45
+ export async function runTask(t, opts = {}) {
46
46
  const start = Date.now();
47
- const cwd = path.join(os.tmpdir(), 'klyro-eval-' + t.id + '-' + Math.random().toString(36).slice(2));
47
+ // workDir lets callers seed a repo and inspect outcomes afterwards
48
+ // (agent-driven fixtures); otherwise an isolated tmp dir is used and
49
+ // removed.
50
+ const owned = !opts.workDir;
51
+ const cwd = opts.workDir ?? path.join(os.tmpdir(), 'klyro-eval-' + t.id + '-' + Math.random().toString(36).slice(2));
48
52
  await fs.mkdir(cwd, { recursive: true });
49
53
  const reg = new ToolRegistry()
50
54
  .register(readFileTool).register(writeFileTool).register(editFileTool)
@@ -72,6 +76,33 @@ export async function runTask(t) {
72
76
  if (verifyFailure) {
73
77
  details += ` verifyFailure=${verifyFailure};`;
74
78
  }
79
+ // Model-graded semantic check (opt-in: needs a live judge adapter).
80
+ let judge;
81
+ if (t.judge && t.judge.rubric.length > 0) {
82
+ if (opts.judgeAdapter) {
83
+ const { runJudge } = await import('./judge.js');
84
+ const texts = result.transcript
85
+ .filter((m) => m.role === 'assistant')
86
+ .flatMap((m) => m.content)
87
+ .filter((b) => b.kind === 'text')
88
+ .map((b) => b.text)
89
+ .join('\n');
90
+ const v = await runJudge(opts.judgeAdapter, opts.judgeModel ?? 'mock-judge', {
91
+ task: t.task,
92
+ finalText: result.finalText,
93
+ toolCalls: result.toolCalls,
94
+ extra: `assistant transcript:\n${texts.slice(0, 3000)}`,
95
+ }, t.judge.rubric);
96
+ judge = { pass: v.pass, notes: v.notes, skipped: v.skipped };
97
+ if (!v.pass) {
98
+ details += ` judge=fail (${v.notes || 'rubric unmet'});`;
99
+ pass = false;
100
+ }
101
+ }
102
+ else {
103
+ judge = { pass: true, notes: 'no judge adapter — skipped', skipped: true };
104
+ }
105
+ }
75
106
  return {
76
107
  id: t.id,
77
108
  status: pass ? 'pass' : 'fail',
@@ -79,20 +110,23 @@ export async function runTask(t) {
79
110
  observedStatus,
80
111
  observedToolCalls: result.toolCalls,
81
112
  durationMs: Date.now() - start,
113
+ ...(judge ? { judge } : {}),
82
114
  };
83
115
  }
84
116
  finally {
85
- try {
86
- await fs.rm(cwd, { recursive: true, force: true });
117
+ if (owned) {
118
+ try {
119
+ await fs.rm(cwd, { recursive: true, force: true });
120
+ }
121
+ catch { }
87
122
  }
88
- catch { }
89
123
  }
90
124
  }
91
- export async function runHarness(tasks) {
125
+ export async function runHarness(tasks, opts = {}) {
92
126
  const start = Date.now();
93
127
  const results = [];
94
128
  for (const t of tasks)
95
- results.push(await runTask(t));
129
+ results.push(await runTask(t, opts));
96
130
  const passed = results.filter((r) => r.status === 'pass').length;
97
131
  return {
98
132
  total: results.length,
@@ -130,7 +164,14 @@ export async function loadFileFixture(dir) {
130
164
  repo = await fs.readFile(path.join(dir, 'repo'), 'utf-8');
131
165
  }
132
166
  catch { /* ignore */ }
133
- return { dir, task: task.trim(), checkSh, meta, repo };
167
+ let script;
168
+ try {
169
+ const raw = JSON.parse(await fs.readFile(path.join(dir, 'script.json'), 'utf-8'));
170
+ if (Array.isArray(raw))
171
+ script = raw;
172
+ }
173
+ catch { /* no script → static fixture */ }
174
+ return { dir, task: task.trim(), checkSh, meta, ...(repo ? { repo } : {}), ...(script ? { script } : {}) };
134
175
  }
135
176
  export async function runFileFixture(fixture, opts = {}) {
136
177
  const start = Date.now();
@@ -168,8 +209,60 @@ export async function runFileFixture(fixture, opts = {}) {
168
209
  await fs.rm(tmp, { recursive: true, force: true }).catch(() => undefined);
169
210
  return result;
170
211
  }
171
- /** Hook the harness up to a session + audit log so task runs are durable. */
172
- export async function runHarnessWithPersistence(tasks, opts) {
212
+ /**
213
+ * Agent-driven fixture: seed a tmp repo, run the fixture's canned script
214
+ * through the REAL runtime + tools, then assert outcomes with check.sh
215
+ * (plus optional meta judge rubric). This is what makes fixtures more than
216
+ * static `exit 0` stubs: the agent loop genuinely executes.
217
+ */
218
+ export async function runAgentFixture(fixture, opts = {}) {
219
+ const start = Date.now();
220
+ const id = path.basename(fixture.dir);
221
+ if (!fixture.script) {
222
+ return { id, status: 'fail', details: 'agent fixture needs script.json', durationMs: Date.now() - start };
223
+ }
224
+ const tmp = path.join(os.tmpdir(), 'klyro-eval-agent-' + Math.random().toString(36).slice(2));
225
+ await fs.mkdir(tmp, { recursive: true });
226
+ try {
227
+ if (fixture.repo) {
228
+ const src = path.isAbsolute(fixture.repo) ? fixture.repo : path.join(fixture.dir, fixture.repo);
229
+ await fs.cp(src, tmp, { recursive: true }).catch(() => undefined);
230
+ }
231
+ const meta = fixture.meta;
232
+ const judgeRubric = meta['judge'] && typeof meta['judge'] === 'object'
233
+ ? meta['judge'].rubric
234
+ : undefined;
235
+ const r = await runTask({
236
+ id,
237
+ description: fixture.task.slice(0, 120),
238
+ task: fixture.task,
239
+ script: fixture.script,
240
+ expectStatus: (typeof meta['expectStatus'] === 'string' ? meta['expectStatus'] : 'complete'),
241
+ ...(typeof meta['expectToolCalls'] === 'number' ? { expectToolCalls: meta['expectToolCalls'] } : {}),
242
+ ...(Array.isArray(judgeRubric) && judgeRubric.every((s) => typeof s === 'string') ? { judge: { rubric: judgeRubric } } : {}),
243
+ }, { workDir: tmp, ...(opts.judgeAdapter ? { judgeAdapter: opts.judgeAdapter, judgeModel: opts.judgeModel } : {}) });
244
+ if (r.status === 'fail')
245
+ return { ...r, durationMs: Date.now() - start };
246
+ // Structural pass — now assert real filesystem outcomes.
247
+ const { spawn } = await import('node:child_process');
248
+ const out = await new Promise((resolve) => {
249
+ const child = spawn('bash', ['-c', fixture.checkSh], { cwd: tmp, shell: false });
250
+ let buf = '';
251
+ child.stdout?.on('data', (b) => { buf += b.toString(); });
252
+ child.stderr?.on('data', (b) => { buf += b.toString(); });
253
+ child.on('close', (code) => resolve(`exit=${code ?? -1} ${buf.slice(0, 500)}`));
254
+ child.on('error', (err) => resolve(`spawn-error: ${String(err)}`));
255
+ });
256
+ if (!out.startsWith('exit=0')) {
257
+ return { ...r, status: 'fail', details: `${r.details} check.sh: ${out}`.trim(), durationMs: Date.now() - start };
258
+ }
259
+ return { ...r, durationMs: Date.now() - start };
260
+ }
261
+ finally {
262
+ await fs.rm(tmp, { recursive: true, force: true }).catch(() => undefined);
263
+ }
264
+ }
265
+ /** Hook the harness up to a session + audit log so task runs are durable. */ export async function runHarnessWithPersistence(tasks, opts) {
173
266
  const store = new SessionStore(opts.storeDir);
174
267
  const audit = new AuditLog(opts.auditPath);
175
268
  const start = Date.now();
@@ -0,0 +1,32 @@
1
+ /**
2
+ * Model-graded judge for evals: scores a finished run against a rubric.
3
+ *
4
+ * Structural asserts (status, tool counts) catch regressions; the judge
5
+ * catches semantic failures (wrong file, wrong content, ignored task).
6
+ * Runs on any ProviderAdapter — in CI, pass a live adapter + judge model;
7
+ * offline runs skip judging (recorded as `skipped`).
8
+ */
9
+ import type { ProviderAdapter } from '../agent/provider-adapter.js';
10
+ export interface JudgeSpec {
11
+ /** Each item is one binary criterion, e.g. "note.txt contains exactly 'hi'". */
12
+ rubric: string[];
13
+ /** Model for grading (defaults to the caller's judge model). */
14
+ model?: string;
15
+ }
16
+ export interface JudgeVerdict {
17
+ pass: boolean;
18
+ scores: Record<string, number>;
19
+ notes: string;
20
+ skipped: boolean;
21
+ }
22
+ /**
23
+ * Grade a finished run. Returns `{pass:false}` (never throws) when the
24
+ * model output is unparsable or the call fails — an inconclusive judge
25
+ * must not silently pass.
26
+ */
27
+ export declare function runJudge(adapter: ProviderAdapter, model: string, input: {
28
+ task: string;
29
+ finalText: string;
30
+ toolCalls: number;
31
+ extra?: string;
32
+ }, rubric: string[]): Promise<JudgeVerdict>;
@@ -0,0 +1,63 @@
1
+ function extractJson(text) {
2
+ const start = text.indexOf('{');
3
+ const end = text.lastIndexOf('}');
4
+ if (start === -1 || end <= start)
5
+ return null;
6
+ try {
7
+ return JSON.parse(text.slice(start, end + 1));
8
+ }
9
+ catch {
10
+ return null;
11
+ }
12
+ }
13
+ /**
14
+ * Grade a finished run. Returns `{pass:false}` (never throws) when the
15
+ * model output is unparsable or the call fails — an inconclusive judge
16
+ * must not silently pass.
17
+ */
18
+ export async function runJudge(adapter, model, input, rubric) {
19
+ if (rubric.length === 0)
20
+ return { pass: true, scores: {}, notes: 'empty rubric', skipped: false };
21
+ const criteria = rubric.map((r, i) => ` c${i + 1}. ${r}`).join('\n');
22
+ const prompt = [
23
+ 'You are an evaluator grading an AI coding agent run. Score ONLY the criteria below.',
24
+ `Task: ${input.task}`,
25
+ `Final answer: ${input.finalText.slice(0, 2000)}`,
26
+ `Tool calls made: ${input.toolCalls}`,
27
+ input.extra ? `Run facts:\n${input.extra.slice(0, 2000)}` : '',
28
+ 'Criteria (score each 1 = met, 0 = not met):',
29
+ criteria,
30
+ 'Respond with ONLY a JSON object: {"scores": {"c1": 1, ...}, "notes": "<one line>"}.',
31
+ ].filter(Boolean).join('\n\n');
32
+ let text = '';
33
+ try {
34
+ for await (const ev of adapter.stream({
35
+ model,
36
+ system: 'You are a strict evaluator. Reply with only the requested JSON.',
37
+ messages: [{ role: 'user', content: [{ kind: 'text', text: prompt }] }],
38
+ tools: [],
39
+ })) {
40
+ if (ev.kind === 'text_delta')
41
+ text += ev.text;
42
+ else if (ev.kind === 'error')
43
+ return { pass: false, scores: {}, notes: `judge call failed: ${ev.message}`, skipped: false };
44
+ }
45
+ }
46
+ catch (err) {
47
+ return { pass: false, scores: {}, notes: `judge call threw: ${err instanceof Error ? err.message : String(err)}`, skipped: false };
48
+ }
49
+ const parsed = extractJson(text);
50
+ if (!parsed || typeof parsed.scores !== 'object' || parsed.scores === null) {
51
+ return { pass: false, scores: {}, notes: 'judge output unparsable', skipped: false };
52
+ }
53
+ const scores = {};
54
+ let pass = true;
55
+ rubric.forEach((_r, i) => {
56
+ const v = parsed.scores[`c${i + 1}`];
57
+ const n = v === 1 || v === '1' || v === true ? 1 : 0;
58
+ scores[`c${i + 1}`] = n;
59
+ if (n !== 1)
60
+ pass = false;
61
+ });
62
+ return { pass, scores, notes: typeof parsed.notes === 'string' ? parsed.notes.slice(0, 500) : '', skipped: false };
63
+ }