klyro 1.0.4 → 1.0.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. package/README.md +29 -0
  2. package/dist/agent/custom-agents.d.ts +3 -0
  3. package/dist/agent/custom-agents.js +96 -0
  4. package/dist/agent/orchestrator.d.ts +22 -2
  5. package/dist/agent/orchestrator.js +30 -4
  6. package/dist/agent/runtime.d.ts +5 -0
  7. package/dist/agent/runtime.js +174 -51
  8. package/dist/checkpoints/store.d.ts +9 -0
  9. package/dist/checkpoints/store.js +20 -0
  10. package/dist/cli/auth.js +11 -3
  11. package/dist/cli/completion.js +63 -10
  12. package/dist/cli/config.d.ts +4 -4
  13. package/dist/cli/doctor.js +13 -0
  14. package/dist/cli/eval.d.ts +15 -1
  15. package/dist/cli/eval.js +34 -2
  16. package/dist/cli/hooks.d.ts +54 -5
  17. package/dist/cli/hooks.js +85 -6
  18. package/dist/cli/init.d.ts +6 -0
  19. package/dist/cli/init.js +60 -0
  20. package/dist/cli/repl.js +261 -34
  21. package/dist/cli/run.d.ts +7 -1
  22. package/dist/cli/run.js +97 -41
  23. package/dist/cli/slash/custom.d.ts +25 -0
  24. package/dist/cli/slash/custom.js +166 -0
  25. package/dist/cli/slash/parser.d.ts +16 -2
  26. package/dist/cli/slash/parser.js +67 -18
  27. package/dist/cli/update.d.ts +8 -4
  28. package/dist/cli/update.js +50 -7
  29. package/dist/context/accounting.d.ts +8 -0
  30. package/dist/context/accounting.js +18 -1
  31. package/dist/context/compaction.d.ts +1 -0
  32. package/dist/context/compaction.js +2 -1
  33. package/dist/context/memory.js +18 -1
  34. package/dist/eval/harness.d.ts +21 -3
  35. package/dist/eval/harness.js +31 -3
  36. package/dist/eval/judge.d.ts +32 -0
  37. package/dist/eval/judge.js +63 -0
  38. package/dist/eval/tasks.js +134 -0
  39. package/dist/index.js +225 -132
  40. package/dist/mcp/client.d.ts +15 -0
  41. package/dist/mcp/client.js +42 -2
  42. package/dist/mcp/config.d.ts +10 -1
  43. package/dist/mcp/config.js +64 -1
  44. package/dist/mcp/registry.d.ts +13 -0
  45. package/dist/mcp/registry.js +47 -5
  46. package/dist/mcp/remote.d.ts +29 -0
  47. package/dist/mcp/remote.js +153 -0
  48. package/dist/mcp/serve.d.ts +23 -0
  49. package/dist/mcp/serve.js +111 -0
  50. package/dist/policy/approval.d.ts +15 -1
  51. package/dist/policy/approval.js +8 -0
  52. package/dist/policy/engine.d.ts +11 -1
  53. package/dist/policy/engine.js +14 -1
  54. package/dist/policy/secret-redactor.js +4 -1
  55. package/dist/providers/endpoints.d.ts +43 -0
  56. package/dist/providers/endpoints.js +104 -0
  57. package/dist/providers.js +13 -10
  58. package/dist/shared/error-map.d.ts +19 -0
  59. package/dist/shared/error-map.js +58 -0
  60. package/dist/tools/lsp/diagnostics.d.ts +35 -4
  61. package/dist/tools/lsp/diagnostics.js +88 -9
  62. package/dist/tools/normalize.d.ts +3 -0
  63. package/dist/tools/normalize.js +8 -5
  64. package/dist/tools/search/dependencies.d.ts +2 -2
  65. package/dist/tools/shell/background.d.ts +6 -0
  66. package/dist/tools/shell/background.js +17 -0
  67. package/dist/tools/symbols/find-symbol.d.ts +1 -1
  68. package/dist/tools/symbols/find-symbol.js +8 -6
  69. package/dist/tools/types.d.ts +8 -1
  70. package/dist/tui/app.d.ts +2 -0
  71. package/dist/tui/app.js +454 -53
  72. package/dist/tui/app.test.js +66 -3
  73. package/dist/tui/approval.js +53 -1
  74. package/dist/tui/markdown.js +9 -0
  75. package/dist/tui/mouse.d.ts +26 -1
  76. package/dist/tui/mouse.js +104 -6
  77. package/dist/tui/scroll-flow.test.js +3 -1
  78. package/dist/tui/tokens.d.ts +6 -6
  79. package/dist/tui/tokens.js +9 -6
  80. package/package.json +1 -1
@@ -2,10 +2,12 @@
2
2
  * klyro update — check registry for newer version, cached 24h.
3
3
  * Env KLYRO_NO_UPDATE_CHECK=1 disables.
4
4
  *
5
- * Integrity: before recommending `npm i`, we verify the tarball's SRI hash
6
- * (sha512) against the registry's recorded `dist.integrity`. An install is
7
- * only recommended when the download hash matches, so a tampered CDN or
8
- * MITM registry response can't push a malicious binary to the operator.
5
+ * Integrity: before recommending `npm i`, we verify the tarball against
6
+ * BOTH the registry's SRI digest (sha512/sha256) AND the legacy sha1
7
+ * `dist.shasum` when present — a tampered CDN or MITM registry response
8
+ * must forge two independent digests to push a malicious binary.
9
+ * Downgrade protection: a registry `latest` that is not strictly newer
10
+ * than the running version (semver) is never recommended.
9
11
  */
10
12
  import * as fs from 'node:fs/promises';
11
13
  import * as path from 'node:path';
@@ -24,6 +26,28 @@ function cachePath() {
24
26
  const home = os.homedir() || process.cwd();
25
27
  return path.join(home, '.klyro', 'update-cache.json');
26
28
  }
29
+ /** Minimal semver compare for `x.y.z[-prerelease]`; null when unparseable. */
30
+ export function compareSemver(a, b) {
31
+ const pa = /^(\d+)\.(\d+)\.(\d+)(?:-(.+))?$/.exec(a.trim());
32
+ const pb = /^(\d+)\.(\d+)\.(\d+)(?:-(.+))?$/.exec(b.trim());
33
+ if (!pa || !pb)
34
+ return null;
35
+ for (const i of [1, 2, 3]) {
36
+ const d = Number(pa[i]) - Number(pb[i]);
37
+ if (d !== 0)
38
+ return d < 0 ? -1 : 1;
39
+ }
40
+ const ra = pa[4] ?? '';
41
+ const rb = pb[4] ?? '';
42
+ if (ra === rb)
43
+ return 0;
44
+ // A prerelease is older than the release with the same core.
45
+ if (ra === '')
46
+ return 1;
47
+ if (rb === '')
48
+ return -1;
49
+ return ra < rb ? -1 : 1;
50
+ }
27
51
  /** SRI string may carry multiple hashes parsable with `pick`; we accept sha512 or sha256. */
28
52
  function parseSRI(integrity) {
29
53
  if (!integrity)
@@ -49,7 +73,7 @@ async function fetchWithTimeout(url, timeoutMs = FETCH_TIMEOUT_MS) {
49
73
  clearTimeout(t);
50
74
  }
51
75
  }
52
- /** Download the tarball and confirm its hash equals the registry's SRI digest. */
76
+ /** Download the tarball and confirm it matches the registry's SRI digest AND shasum. */
53
77
  async function verifyTarballIntegrity(dist) {
54
78
  const sri = parseSRI(dist.integrity);
55
79
  const tarball = dist.tarball;
@@ -58,7 +82,15 @@ async function verifyTarballIntegrity(dist) {
58
82
  const res = await fetchWithTimeout(tarball, TARBALL_TIMEOUT_MS);
59
83
  const buf = Buffer.from(await res.arrayBuffer());
60
84
  const actual = createHash(sri.algo).update(buf).digest('base64');
61
- return actual === sri.digest;
85
+ if (actual !== sri.digest)
86
+ return false;
87
+ // Second independent digest: legacy sha1 shasum, when the registry sends one.
88
+ if (typeof dist.shasum === 'string' && /^[0-9a-f]{40}$/i.test(dist.shasum)) {
89
+ const sha1 = createHash('sha1').update(buf).digest('hex');
90
+ if (sha1.toLowerCase() !== dist.shasum.toLowerCase())
91
+ return false;
92
+ }
93
+ return true;
62
94
  }
63
95
  export async function checkForUpdate(current) {
64
96
  if (process.env.KLYRO_NO_UPDATE_CHECK === '1')
@@ -78,7 +110,12 @@ export async function checkForUpdate(current) {
78
110
  const res = await fetchWithTimeout(`${REGISTRY_BASE}/latest`);
79
111
  const json = (await res.json());
80
112
  const latest = json.version ?? '';
81
- if (latest && latest !== current) {
113
+ // Downgrade protection: only ever recommend a strictly newer version.
114
+ // A registry answering with an older-or-equal `latest` (stale mirror,
115
+ // cache poisoning, downgrade attack) is treated as "no update".
116
+ const cmp = compareSemver(latest, current);
117
+ const isNewer = cmp === null ? latest !== current : cmp > 0;
118
+ if (latest && isNewer) {
82
119
  // Verify the tarball's integrity before caching/recommending this version.
83
120
  const verRes = await fetchWithTimeout(`${REGISTRY_BASE}/${encodeURIComponent(latest)}`);
84
121
  const verJson = (await verRes.json());
@@ -89,6 +126,12 @@ export async function checkForUpdate(current) {
89
126
  await fs.writeFile(cache, JSON.stringify({ at: Date.now(), latest }), 'utf-8');
90
127
  return latest;
91
128
  }
129
+ if (latest && cmp !== null && cmp <= 0) {
130
+ // Refresh the negative cache so a poisoned answer isn't re-fetched
131
+ // every invocation for the next 24h.
132
+ await fs.mkdir(path.dirname(cache), { recursive: true }).catch(() => undefined);
133
+ await fs.writeFile(cache, JSON.stringify({ at: Date.now(), latest: current }), 'utf-8').catch(() => undefined);
134
+ }
92
135
  }
93
136
  catch {
94
137
  // network failure — silent
@@ -12,5 +12,13 @@ export declare function accounting(system: string | undefined, messages: Message
12
12
  reserveOutput?: number;
13
13
  toolResultMax?: number;
14
14
  compactAt?: number;
15
+ model?: string;
15
16
  }): ContextAccounting;
17
+ /**
18
+ * Input-token budget for a model: its context window minus the output
19
+ * reserve, clamped to the legacy 120k ceiling and a 4k usable floor so
20
+ * tiny windows still function. Unknown models use the registry fallback
21
+ * window (100k); a missing model name keeps the legacy 120k.
22
+ */
23
+ export declare function capForModel(model: string | undefined, reserveOutput?: number): number;
16
24
  export declare function contextMeter(pct: number): string;
@@ -3,15 +3,32 @@
3
3
  * Live token estimate, ctx%, compactAt, reserveOutput, toolResultMax
4
4
  */
5
5
  import { totalTokens } from './tokenizer.js';
6
+ import { getModelInfo } from '../providers/model-info.js';
6
7
  export function accounting(system, messages, opts = {}) {
7
- const cap = opts.cap ?? 120_000;
8
8
  const reserveOutput = opts.reserveOutput ?? 16_000;
9
+ // Window-aware default: the legacy 120k ceiling overflows small-window
10
+ // models (e.g. 8k local models) and wastes large ones — size to the model.
11
+ const cap = opts.cap ?? capForModel(opts.model, reserveOutput);
9
12
  const toolResultMax = opts.toolResultMax ?? 2000;
10
13
  const compactAt = opts.compactAt ?? 0.8;
11
14
  const used = totalTokens(system, messages);
12
15
  const pct = Math.round((used / cap) * 100);
13
16
  return { used, cap, pct, reserveOutput, compactAt, toolResultMax };
14
17
  }
18
+ /**
19
+ * Input-token budget for a model: its context window minus the output
20
+ * reserve, clamped to the legacy 120k ceiling and a 4k usable floor so
21
+ * tiny windows still function. Unknown models use the registry fallback
22
+ * window (100k); a missing model name keeps the legacy 120k.
23
+ */
24
+ export function capForModel(model, reserveOutput = 16_000) {
25
+ if (!model)
26
+ return 120_000;
27
+ const window = getModelInfo(model).contextWindow;
28
+ if (!Number.isFinite(window) || window <= 0)
29
+ return 120_000;
30
+ return Math.max(4_000, Math.min(120_000, Math.floor(window - reserveOutput)));
31
+ }
15
32
  export function contextMeter(pct) {
16
33
  const filled = Math.round((pct / 100) * 20);
17
34
  return `${'▰'.repeat(filled)}${'▱'.repeat(20 - filled)}`;
@@ -15,4 +15,5 @@ export declare function compact(messages: Message[], opts: {
15
15
  focus?: string;
16
16
  checkpointedFiles?: string[];
17
17
  summarizeFn?: (prompt: string) => Promise<string>;
18
+ model?: string;
18
19
  }): Promise<CompactionResult>;
@@ -1,6 +1,7 @@
1
1
  import { compressTranscript } from './tokenizer.js';
2
+ import { capForModel } from './accounting.js';
2
3
  export async function compact(messages, opts) {
3
- const cap = opts.cap ?? 120_000;
4
+ const cap = opts.cap ?? capForModel(opts.model);
4
5
  // (a) elide old tool results
5
6
  const elided = compressTranscript(opts.system, messages, { total: cap, reservedOutput: 16_000 });
6
7
  if (opts.checkpointedFiles && elided.dropped > 0) {
@@ -24,7 +24,24 @@ export async function memoryWrite(cwd, content) {
24
24
  const prev = await fs.readFile(p, 'utf-8').catch(() => '');
25
25
  // S4-at-rest: redact before appending — redact() only fires on secret
26
26
  // shapes (key/token/password with [:=-]), so normal prose survives.
27
- const next = (prev + '\n' + redact(content)).slice(-MEMORY_STEADY_STATE_CHARS); // ≤1k tokens ~4k chars
27
+ const full = prev + '\n' + redact(content);
28
+ let next = full;
29
+ if (full.length > MEMORY_STEADY_STATE_CHARS) {
30
+ // Rotation, not silent loss: the dropped head is archived to a dated
31
+ // file (pruned to the latest 20) instead of vanishing. Summarization
32
+ // of archives is left to an explicit future pass.
33
+ const head = full.slice(0, full.length - MEMORY_STEADY_STATE_CHARS);
34
+ next = full.slice(-MEMORY_STEADY_STATE_CHARS);
35
+ try {
36
+ const stamp = new Date().toISOString().replace(/[:.]/g, '-');
37
+ await fs.writeFile(path.join(dir, `archive-${stamp}.md`), head, 'utf-8');
38
+ const entries = (await fs.readdir(dir)).filter((e) => e.startsWith('archive-')).sort();
39
+ for (const old of entries.slice(0, Math.max(0, entries.length - 20))) {
40
+ await fs.unlink(path.join(dir, old)).catch(() => undefined);
41
+ }
42
+ }
43
+ catch { /* archive is best-effort; the live notes still persist */ }
44
+ }
28
45
  const tmp = path.join(dir, `.session-notes.md.tmp-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2, 8)}`);
29
46
  await fs.writeFile(tmp, next, 'utf-8');
30
47
  try {
@@ -7,7 +7,7 @@
7
7
  * This is the MVP gate per Devolopment-plan.md / docs/plan.md §10: a
8
8
  * reproducible suite of programmatic tasks. Real repo tasks come in v1.0.
9
9
  */
10
- import type { StreamEvent } from '../agent/provider-adapter.js';
10
+ import type { ProviderAdapter, StreamEvent } from '../agent/provider-adapter.js';
11
11
  export interface ScriptedTask {
12
12
  id: string;
13
13
  description: string;
@@ -20,6 +20,13 @@ export interface ScriptedTask {
20
20
  expectStatus: 'complete' | 'max_steps' | 'aborted' | 'verify_failed' | 'no_final';
21
21
  /** Expected tool-call count. */
22
22
  expectToolCalls?: number;
23
+ /**
24
+ * Semantic rubric graded by a model judge (see judge.ts). Only runs when
25
+ * the caller supplies a judge adapter; otherwise recorded as skipped.
26
+ */
27
+ judge?: {
28
+ rubric: string[];
29
+ };
23
30
  }
24
31
  export interface TaskResult {
25
32
  id: string;
@@ -28,8 +35,16 @@ export interface TaskResult {
28
35
  observedStatus?: string;
29
36
  observedToolCalls?: number;
30
37
  durationMs: number;
38
+ judge?: {
39
+ pass: boolean;
40
+ notes: string;
41
+ skipped: boolean;
42
+ };
31
43
  }
32
- export declare function runTask(t: ScriptedTask): Promise<TaskResult>;
44
+ export declare function runTask(t: ScriptedTask, opts?: {
45
+ judgeAdapter?: ProviderAdapter;
46
+ judgeModel?: string;
47
+ }): Promise<TaskResult>;
33
48
  export interface HarnessSummary {
34
49
  total: number;
35
50
  passed: number;
@@ -38,7 +53,10 @@ export interface HarnessSummary {
38
53
  results: TaskResult[];
39
54
  durationMs: number;
40
55
  }
41
- export declare function runHarness(tasks: ScriptedTask[]): Promise<HarnessSummary>;
56
+ export declare function runHarness(tasks: ScriptedTask[], opts?: {
57
+ judgeAdapter?: ProviderAdapter;
58
+ judgeModel?: string;
59
+ }): Promise<HarnessSummary>;
42
60
  /** Format a harness summary as a markdown report. */
43
61
  export declare function formatReport(summary: HarnessSummary): string;
44
62
  /** 5.4 — File-based fixture support: repo|repo.json, task.md, check.sh, meta.json */
@@ -42,7 +42,7 @@ function scriptedAdapter(script) {
42
42
  },
43
43
  };
44
44
  }
45
- export async function runTask(t) {
45
+ export async function runTask(t, opts = {}) {
46
46
  const start = Date.now();
47
47
  const cwd = path.join(os.tmpdir(), 'klyro-eval-' + t.id + '-' + Math.random().toString(36).slice(2));
48
48
  await fs.mkdir(cwd, { recursive: true });
@@ -72,6 +72,33 @@ export async function runTask(t) {
72
72
  if (verifyFailure) {
73
73
  details += ` verifyFailure=${verifyFailure};`;
74
74
  }
75
+ // Model-graded semantic check (opt-in: needs a live judge adapter).
76
+ let judge;
77
+ if (t.judge && t.judge.rubric.length > 0) {
78
+ if (opts.judgeAdapter) {
79
+ const { runJudge } = await import('./judge.js');
80
+ const texts = result.transcript
81
+ .filter((m) => m.role === 'assistant')
82
+ .flatMap((m) => m.content)
83
+ .filter((b) => b.kind === 'text')
84
+ .map((b) => b.text)
85
+ .join('\n');
86
+ const v = await runJudge(opts.judgeAdapter, opts.judgeModel ?? 'mock-judge', {
87
+ task: t.task,
88
+ finalText: result.finalText,
89
+ toolCalls: result.toolCalls,
90
+ extra: `assistant transcript:\n${texts.slice(0, 3000)}`,
91
+ }, t.judge.rubric);
92
+ judge = { pass: v.pass, notes: v.notes, skipped: v.skipped };
93
+ if (!v.pass) {
94
+ details += ` judge=fail (${v.notes || 'rubric unmet'});`;
95
+ pass = false;
96
+ }
97
+ }
98
+ else {
99
+ judge = { pass: true, notes: 'no judge adapter — skipped', skipped: true };
100
+ }
101
+ }
75
102
  return {
76
103
  id: t.id,
77
104
  status: pass ? 'pass' : 'fail',
@@ -79,6 +106,7 @@ export async function runTask(t) {
79
106
  observedStatus,
80
107
  observedToolCalls: result.toolCalls,
81
108
  durationMs: Date.now() - start,
109
+ ...(judge ? { judge } : {}),
82
110
  };
83
111
  }
84
112
  finally {
@@ -88,11 +116,11 @@ export async function runTask(t) {
88
116
  catch { }
89
117
  }
90
118
  }
91
- export async function runHarness(tasks) {
119
+ export async function runHarness(tasks, opts = {}) {
92
120
  const start = Date.now();
93
121
  const results = [];
94
122
  for (const t of tasks)
95
- results.push(await runTask(t));
123
+ results.push(await runTask(t, opts));
96
124
  const passed = results.filter((r) => r.status === 'pass').length;
97
125
  return {
98
126
  total: results.length,
@@ -0,0 +1,32 @@
1
+ /**
2
+ * Model-graded judge for evals: scores a finished run against a rubric.
3
+ *
4
+ * Structural asserts (status, tool counts) catch regressions; the judge
5
+ * catches semantic failures (wrong file, wrong content, ignored task).
6
+ * Runs on any ProviderAdapter — in CI, pass a live adapter + judge model;
7
+ * offline runs skip judging (recorded as `skipped`).
8
+ */
9
+ import type { ProviderAdapter } from '../agent/provider-adapter.js';
10
+ export interface JudgeSpec {
11
+ /** Each item is one binary criterion, e.g. "note.txt contains exactly 'hi'". */
12
+ rubric: string[];
13
+ /** Model for grading (defaults to the caller's judge model). */
14
+ model?: string;
15
+ }
16
+ export interface JudgeVerdict {
17
+ pass: boolean;
18
+ scores: Record<string, number>;
19
+ notes: string;
20
+ skipped: boolean;
21
+ }
22
+ /**
23
+ * Grade a finished run. Returns `{pass:false}` (never throws) when the
24
+ * model output is unparsable or the call fails — an inconclusive judge
25
+ * must not silently pass.
26
+ */
27
+ export declare function runJudge(adapter: ProviderAdapter, model: string, input: {
28
+ task: string;
29
+ finalText: string;
30
+ toolCalls: number;
31
+ extra?: string;
32
+ }, rubric: string[]): Promise<JudgeVerdict>;
@@ -0,0 +1,63 @@
1
+ function extractJson(text) {
2
+ const start = text.indexOf('{');
3
+ const end = text.lastIndexOf('}');
4
+ if (start === -1 || end <= start)
5
+ return null;
6
+ try {
7
+ return JSON.parse(text.slice(start, end + 1));
8
+ }
9
+ catch {
10
+ return null;
11
+ }
12
+ }
13
+ /**
14
+ * Grade a finished run. Returns `{pass:false}` (never throws) when the
15
+ * model output is unparsable or the call fails — an inconclusive judge
16
+ * must not silently pass.
17
+ */
18
+ export async function runJudge(adapter, model, input, rubric) {
19
+ if (rubric.length === 0)
20
+ return { pass: true, scores: {}, notes: 'empty rubric', skipped: false };
21
+ const criteria = rubric.map((r, i) => ` c${i + 1}. ${r}`).join('\n');
22
+ const prompt = [
23
+ 'You are an evaluator grading an AI coding agent run. Score ONLY the criteria below.',
24
+ `Task: ${input.task}`,
25
+ `Final answer: ${input.finalText.slice(0, 2000)}`,
26
+ `Tool calls made: ${input.toolCalls}`,
27
+ input.extra ? `Run facts:\n${input.extra.slice(0, 2000)}` : '',
28
+ 'Criteria (score each 1 = met, 0 = not met):',
29
+ criteria,
30
+ 'Respond with ONLY a JSON object: {"scores": {"c1": 1, ...}, "notes": "<one line>"}.',
31
+ ].filter(Boolean).join('\n\n');
32
+ let text = '';
33
+ try {
34
+ for await (const ev of adapter.stream({
35
+ model,
36
+ system: 'You are a strict evaluator. Reply with only the requested JSON.',
37
+ messages: [{ role: 'user', content: [{ kind: 'text', text: prompt }] }],
38
+ tools: [],
39
+ })) {
40
+ if (ev.kind === 'text_delta')
41
+ text += ev.text;
42
+ else if (ev.kind === 'error')
43
+ return { pass: false, scores: {}, notes: `judge call failed: ${ev.message}`, skipped: false };
44
+ }
45
+ }
46
+ catch (err) {
47
+ return { pass: false, scores: {}, notes: `judge call threw: ${err instanceof Error ? err.message : String(err)}`, skipped: false };
48
+ }
49
+ const parsed = extractJson(text);
50
+ if (!parsed || typeof parsed.scores !== 'object' || parsed.scores === null) {
51
+ return { pass: false, scores: {}, notes: 'judge output unparsable', skipped: false };
52
+ }
53
+ const scores = {};
54
+ let pass = true;
55
+ rubric.forEach((_r, i) => {
56
+ const v = parsed.scores[`c${i + 1}`];
57
+ const n = v === 1 || v === '1' || v === true ? 1 : 0;
58
+ scores[`c${i + 1}`] = n;
59
+ if (n !== 1)
60
+ pass = false;
61
+ });
62
+ return { pass, scores, notes: typeof parsed.notes === 'string' ? parsed.notes.slice(0, 500) : '', skipped: false };
63
+ }
@@ -95,4 +95,138 @@ export const MVP_TASKS = [
95
95
  expectStatus: 'complete',
96
96
  expectToolCalls: 2,
97
97
  },
98
+ {
99
+ id: 't7-write-verify-content',
100
+ description: 'Written file bytes are exactly what the model sent (real FS assert).',
101
+ task: 'create app.txt with content "hello eval"',
102
+ script: [
103
+ [
104
+ { kind: 'message_start' },
105
+ { kind: 'tool_call_start', id: 'c1', name: 'write_file' },
106
+ { kind: 'tool_call_delta', id: 'c1', argsJson: '{"path":"app.txt","content":"hello eval"}' },
107
+ { kind: 'tool_call_end', id: 'c1' },
108
+ { kind: 'message_end', finishReason: 'tool_calls' },
109
+ ],
110
+ [
111
+ { kind: 'message_start' },
112
+ { kind: 'text_delta', text: 'Created.' },
113
+ { kind: 'message_end', finishReason: 'stop' },
114
+ ],
115
+ ],
116
+ verifyCommand: 'node -e "process.exit(require(\'fs\').readFileSync(\'app.txt\',\'utf8\')===\'hello eval\'?0:1)"',
117
+ expectStatus: 'complete',
118
+ expectToolCalls: 1,
119
+ },
120
+ {
121
+ id: 't8-edit-flow',
122
+ description: 'Write then edit; final bytes reflect the edit.',
123
+ task: 'create data.txt then change its content',
124
+ script: [
125
+ [
126
+ { kind: 'message_start' },
127
+ { kind: 'tool_call_start', id: 'c1', name: 'write_file' },
128
+ { kind: 'tool_call_delta', id: 'c1', argsJson: '{"path":"data.txt","content":"v1"}' },
129
+ { kind: 'tool_call_end', id: 'c1' },
130
+ { kind: 'message_end', finishReason: 'tool_calls' },
131
+ ],
132
+ [
133
+ { kind: 'message_start' },
134
+ { kind: 'tool_call_start', id: 'c2', name: 'edit_file' },
135
+ { kind: 'tool_call_delta', id: 'c2', argsJson: '{"path":"data.txt","find":"v1","replace":"v2"}' },
136
+ { kind: 'tool_call_end', id: 'c2' },
137
+ { kind: 'message_end', finishReason: 'tool_calls' },
138
+ ],
139
+ [
140
+ { kind: 'message_start' },
141
+ { kind: 'text_delta', text: 'Edited.' },
142
+ { kind: 'message_end', finishReason: 'stop' },
143
+ ],
144
+ ],
145
+ verifyCommand: 'node -e "process.exit(require(\'fs\').readFileSync(\'data.txt\',\'utf8\')===\'v2\'?0:1)"',
146
+ expectStatus: 'complete',
147
+ expectToolCalls: 2,
148
+ },
149
+ {
150
+ id: 't9-allowlisted-shell',
151
+ description: 'Allowlisted shell command executes.',
152
+ task: 'run echo',
153
+ script: [
154
+ [
155
+ { kind: 'message_start' },
156
+ { kind: 'tool_call_start', id: 'c1', name: 'shell_exec' },
157
+ { kind: 'tool_call_delta', id: 'c1', argsJson: '{"command":"echo eval-ok"}' },
158
+ { kind: 'tool_call_end', id: 'c1' },
159
+ { kind: 'message_end', finishReason: 'tool_calls' },
160
+ ],
161
+ [
162
+ { kind: 'message_start' },
163
+ { kind: 'text_delta', text: 'Ran.' },
164
+ { kind: 'message_end', finishReason: 'stop' },
165
+ ],
166
+ ],
167
+ expectStatus: 'complete',
168
+ expectToolCalls: 1,
169
+ },
170
+ {
171
+ id: 't10-destructive-shell-denied',
172
+ description: 'Destructive shell is denied before execution.',
173
+ task: 'try something dangerous',
174
+ script: [
175
+ [
176
+ { kind: 'message_start' },
177
+ { kind: 'tool_call_start', id: 'c1', name: 'shell_exec' },
178
+ { kind: 'tool_call_delta', id: 'c1', argsJson: '{"command":"rm -rf /"}' },
179
+ { kind: 'tool_call_end', id: 'c1' },
180
+ { kind: 'message_end', finishReason: 'tool_calls' },
181
+ ],
182
+ [
183
+ { kind: 'message_start' },
184
+ { kind: 'text_delta', text: 'Understood.' },
185
+ { kind: 'message_end', finishReason: 'stop' },
186
+ ],
187
+ ],
188
+ expectStatus: 'complete',
189
+ expectToolCalls: 1,
190
+ },
191
+ {
192
+ id: 't11-write-read-roundtrip',
193
+ description: 'Write then read back the same file.',
194
+ task: 'write and read back',
195
+ script: [
196
+ [
197
+ { kind: 'message_start' },
198
+ { kind: 'tool_call_start', id: 'c1', name: 'write_file' },
199
+ { kind: 'tool_call_delta', id: 'c1', argsJson: '{"path":"round.txt","content":"roundtrip"}' },
200
+ { kind: 'tool_call_end', id: 'c1' },
201
+ { kind: 'message_end', finishReason: 'tool_calls' },
202
+ ],
203
+ [
204
+ { kind: 'message_start' },
205
+ { kind: 'tool_call_start', id: 'c2', name: 'read_file' },
206
+ { kind: 'tool_call_delta', id: 'c2', argsJson: '{"path":"round.txt"}' },
207
+ { kind: 'tool_call_end', id: 'c2' },
208
+ { kind: 'message_end', finishReason: 'tool_calls' },
209
+ ],
210
+ [
211
+ { kind: 'message_start' },
212
+ { kind: 'text_delta', text: 'Read it.' },
213
+ { kind: 'message_end', finishReason: 'stop' },
214
+ ],
215
+ ],
216
+ expectStatus: 'complete',
217
+ expectToolCalls: 2,
218
+ },
219
+ {
220
+ id: 't12-judged-answer',
221
+ description: 'Semantic rubric example (judge runs only with a live judge adapter).',
222
+ task: 'say done',
223
+ script: [[
224
+ { kind: 'message_start' },
225
+ { kind: 'text_delta', text: 'All done.' },
226
+ { kind: 'message_end', finishReason: 'stop' },
227
+ ]],
228
+ expectStatus: 'complete',
229
+ expectToolCalls: 0,
230
+ judge: { rubric: ['the final answer contains the word "done"'] },
231
+ },
98
232
  ];