klyro 1.0.5 → 1.0.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +13 -0
- package/dist/agent/custom-agents.d.ts +3 -0
- package/dist/agent/custom-agents.js +96 -0
- package/dist/agent/orchestrator.d.ts +26 -0
- package/dist/agent/orchestrator.js +41 -4
- package/dist/agent/runtime.d.ts +15 -0
- package/dist/agent/runtime.js +232 -61
- package/dist/chat.d.ts +10 -0
- package/dist/chat.js +39 -7
- package/dist/checkpoints/store.d.ts +11 -0
- package/dist/checkpoints/store.js +32 -0
- package/dist/cli/auth.d.ts +10 -3
- package/dist/cli/auth.js +43 -5
- package/dist/cli/completion.js +2 -2
- package/dist/cli/config.d.ts +4 -4
- package/dist/cli/doctor.js +0 -1
- package/dist/cli/eval.d.ts +15 -1
- package/dist/cli/eval.js +43 -5
- package/dist/cli/hooks.d.ts +74 -5
- package/dist/cli/hooks.js +118 -7
- package/dist/cli/init.d.ts +6 -0
- package/dist/cli/init.js +60 -0
- package/dist/cli/keychain.d.ts +10 -0
- package/dist/cli/keychain.js +86 -0
- package/dist/cli/repl.js +188 -30
- package/dist/cli/run.d.ts +7 -1
- package/dist/cli/run.js +92 -50
- package/dist/cli/setup.js +3 -2
- package/dist/cli/slash/custom.d.ts +25 -0
- package/dist/cli/slash/custom.js +166 -0
- package/dist/cli/slash/parser.d.ts +9 -1
- package/dist/cli/slash/parser.js +34 -9
- package/dist/cli/update.d.ts +3 -1
- package/dist/cli/update.js +16 -1
- package/dist/context/accounting.d.ts +6 -0
- package/dist/context/accounting.js +8 -2
- package/dist/context/compaction.d.ts +2 -1
- package/dist/context/compaction.js +39 -12
- package/dist/context/memory.d.ts +11 -0
- package/dist/context/memory.js +59 -4
- package/dist/eval/harness.d.ts +40 -5
- package/dist/eval/harness.js +103 -10
- package/dist/eval/judge.d.ts +32 -0
- package/dist/eval/judge.js +63 -0
- package/dist/eval/tasks.js +134 -0
- package/dist/index.js +239 -130
- package/dist/mcp/auth.d.ts +85 -0
- package/dist/mcp/auth.js +249 -0
- package/dist/mcp/client.d.ts +15 -0
- package/dist/mcp/client.js +42 -2
- package/dist/mcp/config.d.ts +31 -1
- package/dist/mcp/config.js +84 -1
- package/dist/mcp/registry.d.ts +19 -0
- package/dist/mcp/registry.js +118 -2
- package/dist/mcp/remote.d.ts +36 -0
- package/dist/mcp/remote.js +207 -0
- package/dist/mcp/sse.d.ts +42 -0
- package/dist/mcp/sse.js +310 -0
- package/dist/persistence/audit.d.ts +15 -3
- package/dist/persistence/audit.js +84 -13
- package/dist/persistence/store.d.ts +9 -0
- package/dist/persistence/store.js +17 -0
- package/dist/policy/approval.d.ts +15 -1
- package/dist/policy/approval.js +8 -0
- package/dist/policy/engine.js +9 -0
- package/dist/providers/endpoints.d.ts +43 -0
- package/dist/providers/endpoints.js +104 -0
- package/dist/providers.js +17 -14
- package/dist/tools/shell/shell-exec.d.ts +13 -0
- package/dist/tools/shell/shell-exec.js +64 -2
- package/dist/tui/app.js +172 -15
- package/dist/tui/app.test.js +27 -2
- package/dist/tui/approval.js +55 -1
- package/dist/tui/scroll-model.d.ts +2 -2
- package/dist/tui/scroll-model.js +9 -3
- package/dist/tui/tokens.d.ts +8 -11
- package/dist/tui/tokens.js +18 -11
- package/package.json +1 -1
|
@@ -4,8 +4,14 @@
|
|
|
4
4
|
*/
|
|
5
5
|
import { totalTokens } from './tokenizer.js';
|
|
6
6
|
import { getModelInfo } from '../providers/model-info.js';
|
|
7
|
+
/**
|
|
8
|
+
* Single output-reserve used by every budget in the harness (displayed
|
|
9
|
+
* accounting, runtime enforcement, compaction elide). One constant so the
|
|
10
|
+
* meter and the enforcer can never disagree.
|
|
11
|
+
*/
|
|
12
|
+
export const RESERVE_OUTPUT_TOKENS = 8000;
|
|
7
13
|
export function accounting(system, messages, opts = {}) {
|
|
8
|
-
const reserveOutput = opts.reserveOutput ??
|
|
14
|
+
const reserveOutput = opts.reserveOutput ?? RESERVE_OUTPUT_TOKENS;
|
|
9
15
|
// Window-aware default: the legacy 120k ceiling overflows small-window
|
|
10
16
|
// models (e.g. 8k local models) and wastes large ones — size to the model.
|
|
11
17
|
const cap = opts.cap ?? capForModel(opts.model, reserveOutput);
|
|
@@ -21,7 +27,7 @@ export function accounting(system, messages, opts = {}) {
|
|
|
21
27
|
* tiny windows still function. Unknown models use the registry fallback
|
|
22
28
|
* window (100k); a missing model name keeps the legacy 120k.
|
|
23
29
|
*/
|
|
24
|
-
export function capForModel(model, reserveOutput =
|
|
30
|
+
export function capForModel(model, reserveOutput = RESERVE_OUTPUT_TOKENS) {
|
|
25
31
|
if (!model)
|
|
26
32
|
return 120_000;
|
|
27
33
|
const window = getModelInfo(model).contextWindow;
|
|
@@ -2,13 +2,14 @@
|
|
|
2
2
|
* 8.3 — Auto-compaction: (a) elide → (b) summarize 60% with model.small → (c) keep last N verbatim
|
|
3
3
|
* Trigger at compactAt (80%) or /compact [focus]. Validates summary mentions every checkpointed file else fallback.
|
|
4
4
|
*/
|
|
5
|
-
import type { Message } from '../agent/message.js';
|
|
5
|
+
import type { Message, ContentBlock } from '../agent/message.js';
|
|
6
6
|
export interface CompactionResult {
|
|
7
7
|
messages: Message[];
|
|
8
8
|
summary: string;
|
|
9
9
|
dropped: number;
|
|
10
10
|
method: 'elide' | 'summarize' | 'fallback';
|
|
11
11
|
}
|
|
12
|
+
export declare function textBlock(s: string): ContentBlock;
|
|
12
13
|
export declare function compact(messages: Message[], opts: {
|
|
13
14
|
system?: string;
|
|
14
15
|
cap?: number;
|
|
@@ -1,14 +1,37 @@
|
|
|
1
1
|
import { compressTranscript } from './tokenizer.js';
|
|
2
|
-
import { capForModel } from './accounting.js';
|
|
2
|
+
import { capForModel, RESERVE_OUTPUT_TOKENS } from './accounting.js';
|
|
3
|
+
export function textBlock(s) {
|
|
4
|
+
return { kind: 'text', text: s };
|
|
5
|
+
}
|
|
6
|
+
/**
|
|
7
|
+
* Tightened checkpoint validation: every checkpointed file must be mentioned
|
|
8
|
+
* by a path segment, not a loose basename substring. `src/util.ts` is matched
|
|
9
|
+
* by "util.ts" only when it names a segment; `foo` never matches `foo.png`
|
|
10
|
+
* nor a path nested inside it.
|
|
11
|
+
*/
|
|
12
|
+
function mentionsFile(summary, filePath) {
|
|
13
|
+
const segs = filePath.split('/').filter(Boolean);
|
|
14
|
+
const tail = segs[segs.length - 1] ?? filePath;
|
|
15
|
+
// Match the full basename OR any path segment borne by the file.
|
|
16
|
+
// Prefer exact-path segments (src/util.ts or util.ts) to avoid the
|
|
17
|
+
// substring false-positive the old `includes(basename)` produced.
|
|
18
|
+
const quoted = filePath.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
19
|
+
const quotedTail = tail.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
20
|
+
return new RegExp(`(^|[\\W_])${quotedTail}([\\W_]|$)`).test(summary) || summary.includes(quoted);
|
|
21
|
+
}
|
|
3
22
|
export async function compact(messages, opts) {
|
|
4
23
|
const cap = opts.cap ?? capForModel(opts.model);
|
|
5
24
|
// (a) elide old tool results
|
|
6
|
-
const elided = compressTranscript(opts.system, messages, { total: cap, reservedOutput:
|
|
25
|
+
const elided = compressTranscript(opts.system, messages, { total: cap, reservedOutput: RESERVE_OUTPUT_TOKENS });
|
|
7
26
|
if (opts.checkpointedFiles && elided.dropped > 0) {
|
|
8
|
-
// (b) try summarize oldest 60% using strict template (mock small model)
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
27
|
+
// (b) try summarize oldest 60% using strict template (mock small model).
|
|
28
|
+
// Slice from the ELIDED sequence so the summarize pass operates on what
|
|
29
|
+
// actually survives elision (previously it sliced the raw `messages`,
|
|
30
|
+
// silently dropping the elide benefit on the oldest portion).
|
|
31
|
+
const elidedBest = elided.messages;
|
|
32
|
+
const n = Math.floor(elidedBest.length * 0.6);
|
|
33
|
+
const oldest = elidedBest.slice(0, n);
|
|
34
|
+
const newest = elidedBest.slice(n);
|
|
12
35
|
const template = `Summarize the following ${oldest.length} messages, mentioning every file: ${opts.checkpointedFiles.join(', ')}.\nFocus: ${opts.focus ?? 'general'}\n\n` + oldest.map((m) => JSON.stringify(m.content).slice(0, 500)).join('\n');
|
|
13
36
|
let summary = `Earlier in session: ${oldest.length} turns covering ${opts.checkpointedFiles.join(', ')}`;
|
|
14
37
|
if (opts.summarizeFn) {
|
|
@@ -17,24 +40,28 @@ export async function compact(messages, opts) {
|
|
|
17
40
|
}
|
|
18
41
|
catch { /* fallback */ }
|
|
19
42
|
}
|
|
20
|
-
// validate: every checkpointed file mentioned
|
|
21
|
-
const missing = opts.checkpointedFiles.filter((f) => !summary
|
|
43
|
+
// validate: every checkpointed file mentioned (path-segment match).
|
|
44
|
+
const missing = opts.checkpointedFiles.filter((f) => !mentionsFile(summary, f));
|
|
22
45
|
if (missing.length === 0) {
|
|
23
|
-
// (c) keep last N verbatim + summary as first message
|
|
24
|
-
const summaryMsg = { role: 'user', content: [
|
|
46
|
+
// (c) keep last N verbatim + summary as first message (typed block).
|
|
47
|
+
const summaryMsg = { role: 'user', content: [textBlock(summary)] };
|
|
25
48
|
return { messages: [summaryMsg, ...newest], summary, dropped: elided.dropped, method: 'summarize' };
|
|
26
49
|
}
|
|
27
50
|
// retry once then fallback to elide
|
|
28
51
|
if (opts.summarizeFn) {
|
|
29
52
|
try {
|
|
30
53
|
const retry = await opts.summarizeFn(template + '\nEnsure to mention: ' + missing.join(', '));
|
|
31
|
-
if (missing.every((f) => retry
|
|
32
|
-
const summaryMsg2 = { role: 'user', content: [
|
|
54
|
+
if (missing.every((f) => mentionsFile(retry, f))) {
|
|
55
|
+
const summaryMsg2 = { role: 'user', content: [textBlock(retry)] };
|
|
33
56
|
return { messages: [summaryMsg2, ...newest], summary: retry, dropped: elided.dropped, method: 'summarize' };
|
|
34
57
|
}
|
|
35
58
|
}
|
|
36
59
|
catch { /* fallback */ }
|
|
37
60
|
}
|
|
38
61
|
}
|
|
62
|
+
// Post-check the elide path too: if it still exceeds cap, keep eliding.
|
|
63
|
+
if (elided.dropped === 0) {
|
|
64
|
+
return { messages, summary: '', dropped: 0, method: 'fallback' };
|
|
65
|
+
}
|
|
39
66
|
return { messages: elided.messages, summary: elided.dropped > 0 ? `Elided ${elided.dropped} observations` : '', dropped: elided.dropped, method: elided.dropped > 0 ? 'elide' : 'fallback' };
|
|
40
67
|
}
|
package/dist/context/memory.d.ts
CHANGED
|
@@ -3,6 +3,17 @@ import type { PlanStep } from '../agent/runtime.js';
|
|
|
3
3
|
export declare const MEMORY_WRITE_LIMIT_CHARS = 8000;
|
|
4
4
|
/** Steady-state cap (~1k tokens ≈ 4k chars) kept via tail slice. */
|
|
5
5
|
export declare const MEMORY_STEADY_STATE_CHARS = 4000;
|
|
6
|
+
/** Archive index filename (recall without an LLM pass). */
|
|
7
|
+
export declare const MEMORY_ARCHIVE_INDEX = "archive-index.json";
|
|
8
|
+
/** Max retained archives. */
|
|
9
|
+
export declare const MEMORY_ARCHIVE_KEEP = 20;
|
|
10
|
+
export interface MemoryArchiveEntry {
|
|
11
|
+
file: string;
|
|
12
|
+
ts: number;
|
|
13
|
+
chars: number;
|
|
14
|
+
/** First line of the archived head — lets the model decide what to re-read. */
|
|
15
|
+
preview: string;
|
|
16
|
+
}
|
|
6
17
|
export declare function memoryWrite(cwd: string, content: string): Promise<string>;
|
|
7
18
|
export declare function loadMemory(cwd: string): Promise<string>;
|
|
8
19
|
/** Synchronous read for the system-prompt build path (run on every turn). */
|
package/dist/context/memory.js
CHANGED
|
@@ -10,22 +10,63 @@ import * as fs from 'node:fs/promises';
|
|
|
10
10
|
import { readFileSync } from 'node:fs';
|
|
11
11
|
import * as path from 'node:path';
|
|
12
12
|
import { redact } from '../policy/secret-redactor.js';
|
|
13
|
+
import { resolveAndFollowSymlinks, assertNotSymlink } from '../policy/path-guard.js';
|
|
13
14
|
/** Single-write ceiling — anything larger is rejected, not sliced. */
|
|
14
15
|
export const MEMORY_WRITE_LIMIT_CHARS = 8000;
|
|
15
16
|
/** Steady-state cap (~1k tokens ≈ 4k chars) kept via tail slice. */
|
|
16
17
|
export const MEMORY_STEADY_STATE_CHARS = 4000;
|
|
18
|
+
/** Archive index filename (recall without an LLM pass). */
|
|
19
|
+
export const MEMORY_ARCHIVE_INDEX = 'archive-index.json';
|
|
20
|
+
/** Max retained archives. */
|
|
21
|
+
export const MEMORY_ARCHIVE_KEEP = 20;
|
|
17
22
|
export async function memoryWrite(cwd, content) {
|
|
18
23
|
if (content.length > MEMORY_WRITE_LIMIT_CHARS) {
|
|
19
24
|
throw new Error(`memory budget exceeded: single write is ${content.length} chars (max ${MEMORY_WRITE_LIMIT_CHARS}) — split it into smaller notes`);
|
|
20
25
|
}
|
|
26
|
+
// B4 — symlink/TOCTOU guard: the memory dir must stay inside cwd even if
|
|
27
|
+
// `.klyro` or `.klyro/memory` is (or becomes) a symlink. Resolve+follow up
|
|
28
|
+
// the existing parent chain first (defeats an escaping parent symlink), then
|
|
29
|
+
// mkdir, then a final realpath + lstat: a swapped-in final symlink is
|
|
30
|
+
// refused before any write lands outside cwd.
|
|
21
31
|
const dir = path.join(cwd, '.klyro', 'memory');
|
|
32
|
+
await resolveAndFollowSymlinks(cwd, '.klyro'); // throws if `.klyro` escapes cwd
|
|
22
33
|
await fs.mkdir(dir, { recursive: true });
|
|
23
|
-
const
|
|
34
|
+
const { resolved } = await resolveAndFollowSymlinks(cwd, '.klyro/memory');
|
|
35
|
+
await assertNotSymlink(resolved);
|
|
36
|
+
const p = path.join(resolved, 'session-notes.md');
|
|
24
37
|
const prev = await fs.readFile(p, 'utf-8').catch(() => '');
|
|
25
38
|
// S4-at-rest: redact before appending — redact() only fires on secret
|
|
26
39
|
// shapes (key/token/password with [:=-]), so normal prose survives.
|
|
27
|
-
const
|
|
28
|
-
|
|
40
|
+
const full = prev + '\n' + redact(content);
|
|
41
|
+
let next = full;
|
|
42
|
+
if (full.length > MEMORY_STEADY_STATE_CHARS) {
|
|
43
|
+
// Rotation, not silent loss: the dropped head is archived to a dated
|
|
44
|
+
// file (pruned to the latest 20) instead of vanishing, and the archive
|
|
45
|
+
// index is refreshed so the model knows what exists to re-read.
|
|
46
|
+
const head = full.slice(0, full.length - MEMORY_STEADY_STATE_CHARS);
|
|
47
|
+
next = full.slice(-MEMORY_STEADY_STATE_CHARS);
|
|
48
|
+
try {
|
|
49
|
+
const stamp = new Date().toISOString().replace(/[:.]/g, '-');
|
|
50
|
+
const file = `archive-${stamp}.md`;
|
|
51
|
+
await fs.writeFile(path.join(resolved, file), head, 'utf-8');
|
|
52
|
+
const entries = (await fs.readdir(resolved)).filter((e) => e.startsWith('archive-') && e.endsWith('.md')).sort();
|
|
53
|
+
for (const old of entries.slice(0, Math.max(0, entries.length - MEMORY_ARCHIVE_KEEP))) {
|
|
54
|
+
await fs.unlink(path.join(resolved, old)).catch(() => undefined);
|
|
55
|
+
}
|
|
56
|
+
const index = [];
|
|
57
|
+
for (const e of (await fs.readdir(resolved)).filter((e) => e.startsWith('archive-') && e.endsWith('.md')).sort()) {
|
|
58
|
+
try {
|
|
59
|
+
const content = await fs.readFile(path.join(resolved, e), 'utf-8');
|
|
60
|
+
const stat = await fs.stat(path.join(resolved, e));
|
|
61
|
+
index.push({ file: e, ts: Math.round(stat.mtimeMs), chars: content.length, preview: content.split('\n')[0]?.slice(0, 160) ?? '' });
|
|
62
|
+
}
|
|
63
|
+
catch { /* skip unreadable entries */ }
|
|
64
|
+
}
|
|
65
|
+
await fs.writeFile(path.join(resolved, MEMORY_ARCHIVE_INDEX), JSON.stringify(index, null, 2), 'utf-8');
|
|
66
|
+
}
|
|
67
|
+
catch { /* archive is best-effort; the live notes still persist */ }
|
|
68
|
+
}
|
|
69
|
+
const tmp = path.join(resolved, `.session-notes.md.tmp-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2, 8)}`);
|
|
29
70
|
await fs.writeFile(tmp, next, 'utf-8');
|
|
30
71
|
try {
|
|
31
72
|
const fh = await fs.open(tmp, 'r+');
|
|
@@ -66,7 +107,21 @@ export function loadMemorySync(cwd) {
|
|
|
66
107
|
/** Wrap persisted notes into the injected prompt block; '' when empty. */
|
|
67
108
|
export function memoryBlock(cwd) {
|
|
68
109
|
const notes = loadMemorySync(cwd).trim();
|
|
69
|
-
|
|
110
|
+
if (!notes)
|
|
111
|
+
return '';
|
|
112
|
+
// Recall aid: tell the model which archives exist (with previews) so it
|
|
113
|
+
// can re-read them via read_file instead of losing rotated knowledge.
|
|
114
|
+
let archived = '';
|
|
115
|
+
try {
|
|
116
|
+
const raw = readFileSync(path.join(cwd, '.klyro', 'memory', MEMORY_ARCHIVE_INDEX), 'utf-8');
|
|
117
|
+
const entries = JSON.parse(raw);
|
|
118
|
+
if (Array.isArray(entries) && entries.length > 0) {
|
|
119
|
+
const lines = entries.slice(-5).map((e) => ` - .klyro/memory/${e.file} (${e.chars} chars): ${e.preview}`);
|
|
120
|
+
archived = `\nEarlier notes archived (read one with read_file for detail):\n${lines.join('\n')}`;
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
catch { /* no index — nothing archived yet */ }
|
|
124
|
+
return `\n\n<memory>\n${notes.slice(0, 4000)}${archived}\n</memory>`;
|
|
70
125
|
}
|
|
71
126
|
export function shouldRemind(turn, lastRemindTurn) {
|
|
72
127
|
return turn - lastRemindTurn >= 20;
|
package/dist/eval/harness.d.ts
CHANGED
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
* This is the MVP gate per Devolopment-plan.md / docs/plan.md §10: a
|
|
8
8
|
* reproducible suite of programmatic tasks. Real repo tasks come in v1.0.
|
|
9
9
|
*/
|
|
10
|
-
import type { StreamEvent } from '../agent/provider-adapter.js';
|
|
10
|
+
import type { ProviderAdapter, StreamEvent } from '../agent/provider-adapter.js';
|
|
11
11
|
export interface ScriptedTask {
|
|
12
12
|
id: string;
|
|
13
13
|
description: string;
|
|
@@ -20,6 +20,13 @@ export interface ScriptedTask {
|
|
|
20
20
|
expectStatus: 'complete' | 'max_steps' | 'aborted' | 'verify_failed' | 'no_final';
|
|
21
21
|
/** Expected tool-call count. */
|
|
22
22
|
expectToolCalls?: number;
|
|
23
|
+
/**
|
|
24
|
+
* Semantic rubric graded by a model judge (see judge.ts). Only runs when
|
|
25
|
+
* the caller supplies a judge adapter; otherwise recorded as skipped.
|
|
26
|
+
*/
|
|
27
|
+
judge?: {
|
|
28
|
+
rubric: string[];
|
|
29
|
+
};
|
|
23
30
|
}
|
|
24
31
|
export interface TaskResult {
|
|
25
32
|
id: string;
|
|
@@ -28,8 +35,17 @@ export interface TaskResult {
|
|
|
28
35
|
observedStatus?: string;
|
|
29
36
|
observedToolCalls?: number;
|
|
30
37
|
durationMs: number;
|
|
38
|
+
judge?: {
|
|
39
|
+
pass: boolean;
|
|
40
|
+
notes: string;
|
|
41
|
+
skipped: boolean;
|
|
42
|
+
};
|
|
31
43
|
}
|
|
32
|
-
export declare function runTask(t: ScriptedTask
|
|
44
|
+
export declare function runTask(t: ScriptedTask, opts?: {
|
|
45
|
+
judgeAdapter?: ProviderAdapter;
|
|
46
|
+
judgeModel?: string;
|
|
47
|
+
workDir?: string;
|
|
48
|
+
}): Promise<TaskResult>;
|
|
33
49
|
export interface HarnessSummary {
|
|
34
50
|
total: number;
|
|
35
51
|
passed: number;
|
|
@@ -38,7 +54,10 @@ export interface HarnessSummary {
|
|
|
38
54
|
results: TaskResult[];
|
|
39
55
|
durationMs: number;
|
|
40
56
|
}
|
|
41
|
-
export declare function runHarness(tasks: ScriptedTask[]
|
|
57
|
+
export declare function runHarness(tasks: ScriptedTask[], opts?: {
|
|
58
|
+
judgeAdapter?: ProviderAdapter;
|
|
59
|
+
judgeModel?: string;
|
|
60
|
+
}): Promise<HarnessSummary>;
|
|
42
61
|
/** Format a harness summary as a markdown report. */
|
|
43
62
|
export declare function formatReport(summary: HarnessSummary): string;
|
|
44
63
|
/** 5.4 — File-based fixture support: repo|repo.json, task.md, check.sh, meta.json */
|
|
@@ -48,14 +67,30 @@ export interface FileFixture {
|
|
|
48
67
|
checkSh: string;
|
|
49
68
|
meta: Record<string, unknown>;
|
|
50
69
|
repo?: string;
|
|
70
|
+
/**
|
|
71
|
+
* Optional canned agent turns (StreamEvent[][], same shape as
|
|
72
|
+
* ScriptedTask.script). When present the fixture is agent-driven: the
|
|
73
|
+
* runtime executes the script with real tools in a seeded tmp repo and
|
|
74
|
+
* check.sh then asserts the filesystem outcomes.
|
|
75
|
+
*/
|
|
76
|
+
script?: StreamEvent[][];
|
|
51
77
|
}
|
|
52
78
|
export declare function loadFileFixture(dir: string): Promise<FileFixture>;
|
|
53
79
|
export declare function runFileFixture(fixture: FileFixture, opts?: {
|
|
54
80
|
runs?: number;
|
|
55
81
|
parallel?: number;
|
|
56
82
|
}): Promise<TaskResult>;
|
|
57
|
-
/**
|
|
58
|
-
|
|
83
|
+
/**
|
|
84
|
+
* Agent-driven fixture: seed a tmp repo, run the fixture's canned script
|
|
85
|
+
* through the REAL runtime + tools, then assert outcomes with check.sh
|
|
86
|
+
* (plus optional meta judge rubric). This is what makes fixtures more than
|
|
87
|
+
* static `exit 0` stubs: the agent loop genuinely executes.
|
|
88
|
+
*/
|
|
89
|
+
export declare function runAgentFixture(fixture: FileFixture, opts?: {
|
|
90
|
+
judgeAdapter?: ProviderAdapter;
|
|
91
|
+
judgeModel?: string;
|
|
92
|
+
}): Promise<TaskResult>;
|
|
93
|
+
/** Hook the harness up to a session + audit log so task runs are durable. */ export declare function runHarnessWithPersistence(tasks: ScriptedTask[], opts: {
|
|
59
94
|
storeDir: string;
|
|
60
95
|
auditPath: string;
|
|
61
96
|
}): Promise<HarnessSummary>;
|
package/dist/eval/harness.js
CHANGED
|
@@ -42,9 +42,13 @@ function scriptedAdapter(script) {
|
|
|
42
42
|
},
|
|
43
43
|
};
|
|
44
44
|
}
|
|
45
|
-
export async function runTask(t) {
|
|
45
|
+
export async function runTask(t, opts = {}) {
|
|
46
46
|
const start = Date.now();
|
|
47
|
-
|
|
47
|
+
// workDir lets callers seed a repo and inspect outcomes afterwards
|
|
48
|
+
// (agent-driven fixtures); otherwise an isolated tmp dir is used and
|
|
49
|
+
// removed.
|
|
50
|
+
const owned = !opts.workDir;
|
|
51
|
+
const cwd = opts.workDir ?? path.join(os.tmpdir(), 'klyro-eval-' + t.id + '-' + Math.random().toString(36).slice(2));
|
|
48
52
|
await fs.mkdir(cwd, { recursive: true });
|
|
49
53
|
const reg = new ToolRegistry()
|
|
50
54
|
.register(readFileTool).register(writeFileTool).register(editFileTool)
|
|
@@ -72,6 +76,33 @@ export async function runTask(t) {
|
|
|
72
76
|
if (verifyFailure) {
|
|
73
77
|
details += ` verifyFailure=${verifyFailure};`;
|
|
74
78
|
}
|
|
79
|
+
// Model-graded semantic check (opt-in: needs a live judge adapter).
|
|
80
|
+
let judge;
|
|
81
|
+
if (t.judge && t.judge.rubric.length > 0) {
|
|
82
|
+
if (opts.judgeAdapter) {
|
|
83
|
+
const { runJudge } = await import('./judge.js');
|
|
84
|
+
const texts = result.transcript
|
|
85
|
+
.filter((m) => m.role === 'assistant')
|
|
86
|
+
.flatMap((m) => m.content)
|
|
87
|
+
.filter((b) => b.kind === 'text')
|
|
88
|
+
.map((b) => b.text)
|
|
89
|
+
.join('\n');
|
|
90
|
+
const v = await runJudge(opts.judgeAdapter, opts.judgeModel ?? 'mock-judge', {
|
|
91
|
+
task: t.task,
|
|
92
|
+
finalText: result.finalText,
|
|
93
|
+
toolCalls: result.toolCalls,
|
|
94
|
+
extra: `assistant transcript:\n${texts.slice(0, 3000)}`,
|
|
95
|
+
}, t.judge.rubric);
|
|
96
|
+
judge = { pass: v.pass, notes: v.notes, skipped: v.skipped };
|
|
97
|
+
if (!v.pass) {
|
|
98
|
+
details += ` judge=fail (${v.notes || 'rubric unmet'});`;
|
|
99
|
+
pass = false;
|
|
100
|
+
}
|
|
101
|
+
}
|
|
102
|
+
else {
|
|
103
|
+
judge = { pass: true, notes: 'no judge adapter — skipped', skipped: true };
|
|
104
|
+
}
|
|
105
|
+
}
|
|
75
106
|
return {
|
|
76
107
|
id: t.id,
|
|
77
108
|
status: pass ? 'pass' : 'fail',
|
|
@@ -79,20 +110,23 @@ export async function runTask(t) {
|
|
|
79
110
|
observedStatus,
|
|
80
111
|
observedToolCalls: result.toolCalls,
|
|
81
112
|
durationMs: Date.now() - start,
|
|
113
|
+
...(judge ? { judge } : {}),
|
|
82
114
|
};
|
|
83
115
|
}
|
|
84
116
|
finally {
|
|
85
|
-
|
|
86
|
-
|
|
117
|
+
if (owned) {
|
|
118
|
+
try {
|
|
119
|
+
await fs.rm(cwd, { recursive: true, force: true });
|
|
120
|
+
}
|
|
121
|
+
catch { }
|
|
87
122
|
}
|
|
88
|
-
catch { }
|
|
89
123
|
}
|
|
90
124
|
}
|
|
91
|
-
export async function runHarness(tasks) {
|
|
125
|
+
export async function runHarness(tasks, opts = {}) {
|
|
92
126
|
const start = Date.now();
|
|
93
127
|
const results = [];
|
|
94
128
|
for (const t of tasks)
|
|
95
|
-
results.push(await runTask(t));
|
|
129
|
+
results.push(await runTask(t, opts));
|
|
96
130
|
const passed = results.filter((r) => r.status === 'pass').length;
|
|
97
131
|
return {
|
|
98
132
|
total: results.length,
|
|
@@ -130,7 +164,14 @@ export async function loadFileFixture(dir) {
|
|
|
130
164
|
repo = await fs.readFile(path.join(dir, 'repo'), 'utf-8');
|
|
131
165
|
}
|
|
132
166
|
catch { /* ignore */ }
|
|
133
|
-
|
|
167
|
+
let script;
|
|
168
|
+
try {
|
|
169
|
+
const raw = JSON.parse(await fs.readFile(path.join(dir, 'script.json'), 'utf-8'));
|
|
170
|
+
if (Array.isArray(raw))
|
|
171
|
+
script = raw;
|
|
172
|
+
}
|
|
173
|
+
catch { /* no script → static fixture */ }
|
|
174
|
+
return { dir, task: task.trim(), checkSh, meta, ...(repo ? { repo } : {}), ...(script ? { script } : {}) };
|
|
134
175
|
}
|
|
135
176
|
export async function runFileFixture(fixture, opts = {}) {
|
|
136
177
|
const start = Date.now();
|
|
@@ -168,8 +209,60 @@ export async function runFileFixture(fixture, opts = {}) {
|
|
|
168
209
|
await fs.rm(tmp, { recursive: true, force: true }).catch(() => undefined);
|
|
169
210
|
return result;
|
|
170
211
|
}
|
|
171
|
-
/**
|
|
172
|
-
|
|
212
|
+
/**
|
|
213
|
+
* Agent-driven fixture: seed a tmp repo, run the fixture's canned script
|
|
214
|
+
* through the REAL runtime + tools, then assert outcomes with check.sh
|
|
215
|
+
* (plus optional meta judge rubric). This is what makes fixtures more than
|
|
216
|
+
* static `exit 0` stubs: the agent loop genuinely executes.
|
|
217
|
+
*/
|
|
218
|
+
export async function runAgentFixture(fixture, opts = {}) {
|
|
219
|
+
const start = Date.now();
|
|
220
|
+
const id = path.basename(fixture.dir);
|
|
221
|
+
if (!fixture.script) {
|
|
222
|
+
return { id, status: 'fail', details: 'agent fixture needs script.json', durationMs: Date.now() - start };
|
|
223
|
+
}
|
|
224
|
+
const tmp = path.join(os.tmpdir(), 'klyro-eval-agent-' + Math.random().toString(36).slice(2));
|
|
225
|
+
await fs.mkdir(tmp, { recursive: true });
|
|
226
|
+
try {
|
|
227
|
+
if (fixture.repo) {
|
|
228
|
+
const src = path.isAbsolute(fixture.repo) ? fixture.repo : path.join(fixture.dir, fixture.repo);
|
|
229
|
+
await fs.cp(src, tmp, { recursive: true }).catch(() => undefined);
|
|
230
|
+
}
|
|
231
|
+
const meta = fixture.meta;
|
|
232
|
+
const judgeRubric = meta['judge'] && typeof meta['judge'] === 'object'
|
|
233
|
+
? meta['judge'].rubric
|
|
234
|
+
: undefined;
|
|
235
|
+
const r = await runTask({
|
|
236
|
+
id,
|
|
237
|
+
description: fixture.task.slice(0, 120),
|
|
238
|
+
task: fixture.task,
|
|
239
|
+
script: fixture.script,
|
|
240
|
+
expectStatus: (typeof meta['expectStatus'] === 'string' ? meta['expectStatus'] : 'complete'),
|
|
241
|
+
...(typeof meta['expectToolCalls'] === 'number' ? { expectToolCalls: meta['expectToolCalls'] } : {}),
|
|
242
|
+
...(Array.isArray(judgeRubric) && judgeRubric.every((s) => typeof s === 'string') ? { judge: { rubric: judgeRubric } } : {}),
|
|
243
|
+
}, { workDir: tmp, ...(opts.judgeAdapter ? { judgeAdapter: opts.judgeAdapter, judgeModel: opts.judgeModel } : {}) });
|
|
244
|
+
if (r.status === 'fail')
|
|
245
|
+
return { ...r, durationMs: Date.now() - start };
|
|
246
|
+
// Structural pass — now assert real filesystem outcomes.
|
|
247
|
+
const { spawn } = await import('node:child_process');
|
|
248
|
+
const out = await new Promise((resolve) => {
|
|
249
|
+
const child = spawn('bash', ['-c', fixture.checkSh], { cwd: tmp, shell: false });
|
|
250
|
+
let buf = '';
|
|
251
|
+
child.stdout?.on('data', (b) => { buf += b.toString(); });
|
|
252
|
+
child.stderr?.on('data', (b) => { buf += b.toString(); });
|
|
253
|
+
child.on('close', (code) => resolve(`exit=${code ?? -1} ${buf.slice(0, 500)}`));
|
|
254
|
+
child.on('error', (err) => resolve(`spawn-error: ${String(err)}`));
|
|
255
|
+
});
|
|
256
|
+
if (!out.startsWith('exit=0')) {
|
|
257
|
+
return { ...r, status: 'fail', details: `${r.details} check.sh: ${out}`.trim(), durationMs: Date.now() - start };
|
|
258
|
+
}
|
|
259
|
+
return { ...r, durationMs: Date.now() - start };
|
|
260
|
+
}
|
|
261
|
+
finally {
|
|
262
|
+
await fs.rm(tmp, { recursive: true, force: true }).catch(() => undefined);
|
|
263
|
+
}
|
|
264
|
+
}
|
|
265
|
+
/** Hook the harness up to a session + audit log so task runs are durable. */ export async function runHarnessWithPersistence(tasks, opts) {
|
|
173
266
|
const store = new SessionStore(opts.storeDir);
|
|
174
267
|
const audit = new AuditLog(opts.auditPath);
|
|
175
268
|
const start = Date.now();
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Model-graded judge for evals: scores a finished run against a rubric.
|
|
3
|
+
*
|
|
4
|
+
* Structural asserts (status, tool counts) catch regressions; the judge
|
|
5
|
+
* catches semantic failures (wrong file, wrong content, ignored task).
|
|
6
|
+
* Runs on any ProviderAdapter — in CI, pass a live adapter + judge model;
|
|
7
|
+
* offline runs skip judging (recorded as `skipped`).
|
|
8
|
+
*/
|
|
9
|
+
import type { ProviderAdapter } from '../agent/provider-adapter.js';
|
|
10
|
+
export interface JudgeSpec {
|
|
11
|
+
/** Each item is one binary criterion, e.g. "note.txt contains exactly 'hi'". */
|
|
12
|
+
rubric: string[];
|
|
13
|
+
/** Model for grading (defaults to the caller's judge model). */
|
|
14
|
+
model?: string;
|
|
15
|
+
}
|
|
16
|
+
export interface JudgeVerdict {
|
|
17
|
+
pass: boolean;
|
|
18
|
+
scores: Record<string, number>;
|
|
19
|
+
notes: string;
|
|
20
|
+
skipped: boolean;
|
|
21
|
+
}
|
|
22
|
+
/**
|
|
23
|
+
* Grade a finished run. Returns `{pass:false}` (never throws) when the
|
|
24
|
+
* model output is unparsable or the call fails — an inconclusive judge
|
|
25
|
+
* must not silently pass.
|
|
26
|
+
*/
|
|
27
|
+
export declare function runJudge(adapter: ProviderAdapter, model: string, input: {
|
|
28
|
+
task: string;
|
|
29
|
+
finalText: string;
|
|
30
|
+
toolCalls: number;
|
|
31
|
+
extra?: string;
|
|
32
|
+
}, rubric: string[]): Promise<JudgeVerdict>;
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
function extractJson(text) {
|
|
2
|
+
const start = text.indexOf('{');
|
|
3
|
+
const end = text.lastIndexOf('}');
|
|
4
|
+
if (start === -1 || end <= start)
|
|
5
|
+
return null;
|
|
6
|
+
try {
|
|
7
|
+
return JSON.parse(text.slice(start, end + 1));
|
|
8
|
+
}
|
|
9
|
+
catch {
|
|
10
|
+
return null;
|
|
11
|
+
}
|
|
12
|
+
}
|
|
13
|
+
/**
|
|
14
|
+
* Grade a finished run. Returns `{pass:false}` (never throws) when the
|
|
15
|
+
* model output is unparsable or the call fails — an inconclusive judge
|
|
16
|
+
* must not silently pass.
|
|
17
|
+
*/
|
|
18
|
+
export async function runJudge(adapter, model, input, rubric) {
|
|
19
|
+
if (rubric.length === 0)
|
|
20
|
+
return { pass: true, scores: {}, notes: 'empty rubric', skipped: false };
|
|
21
|
+
const criteria = rubric.map((r, i) => ` c${i + 1}. ${r}`).join('\n');
|
|
22
|
+
const prompt = [
|
|
23
|
+
'You are an evaluator grading an AI coding agent run. Score ONLY the criteria below.',
|
|
24
|
+
`Task: ${input.task}`,
|
|
25
|
+
`Final answer: ${input.finalText.slice(0, 2000)}`,
|
|
26
|
+
`Tool calls made: ${input.toolCalls}`,
|
|
27
|
+
input.extra ? `Run facts:\n${input.extra.slice(0, 2000)}` : '',
|
|
28
|
+
'Criteria (score each 1 = met, 0 = not met):',
|
|
29
|
+
criteria,
|
|
30
|
+
'Respond with ONLY a JSON object: {"scores": {"c1": 1, ...}, "notes": "<one line>"}.',
|
|
31
|
+
].filter(Boolean).join('\n\n');
|
|
32
|
+
let text = '';
|
|
33
|
+
try {
|
|
34
|
+
for await (const ev of adapter.stream({
|
|
35
|
+
model,
|
|
36
|
+
system: 'You are a strict evaluator. Reply with only the requested JSON.',
|
|
37
|
+
messages: [{ role: 'user', content: [{ kind: 'text', text: prompt }] }],
|
|
38
|
+
tools: [],
|
|
39
|
+
})) {
|
|
40
|
+
if (ev.kind === 'text_delta')
|
|
41
|
+
text += ev.text;
|
|
42
|
+
else if (ev.kind === 'error')
|
|
43
|
+
return { pass: false, scores: {}, notes: `judge call failed: ${ev.message}`, skipped: false };
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
catch (err) {
|
|
47
|
+
return { pass: false, scores: {}, notes: `judge call threw: ${err instanceof Error ? err.message : String(err)}`, skipped: false };
|
|
48
|
+
}
|
|
49
|
+
const parsed = extractJson(text);
|
|
50
|
+
if (!parsed || typeof parsed.scores !== 'object' || parsed.scores === null) {
|
|
51
|
+
return { pass: false, scores: {}, notes: 'judge output unparsable', skipped: false };
|
|
52
|
+
}
|
|
53
|
+
const scores = {};
|
|
54
|
+
let pass = true;
|
|
55
|
+
rubric.forEach((_r, i) => {
|
|
56
|
+
const v = parsed.scores[`c${i + 1}`];
|
|
57
|
+
const n = v === 1 || v === '1' || v === true ? 1 : 0;
|
|
58
|
+
scores[`c${i + 1}`] = n;
|
|
59
|
+
if (n !== 1)
|
|
60
|
+
pass = false;
|
|
61
|
+
});
|
|
62
|
+
return { pass, scores, notes: typeof parsed.notes === 'string' ? parsed.notes.slice(0, 500) : '', skipped: false };
|
|
63
|
+
}
|