@hyperfixi/testing-framework 2.10.0 → 2.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@hyperfixi/testing-framework",
3
- "version": "2.10.0",
3
+ "version": "2.11.0",
4
4
  "description": "Cross-platform behavior testing suite for LokaScript applications",
5
5
  "main": "dist/index.js",
6
6
  "module": "dist/index.mjs",
@@ -29,7 +29,7 @@
29
29
  "scripts": {
30
30
  "build": "tsup",
31
31
  "dev": "tsup --watch",
32
- "pretest": "../../scripts/ensure-fresh.sh ../intent ../framework ../semantic ../patterns-reference ../core",
32
+ "pretest": "../../scripts/ensure-fresh.sh ../intent ../framework ../semantic ../patterns-reference ../core ../aot-compiler ../compilation-service",
33
33
  "test": "vitest run",
34
34
  "test:watch": "vitest",
35
35
  "test:coverage": "vitest run --coverage",
@@ -57,10 +57,11 @@
57
57
  "author": "LokaScript Contributors",
58
58
  "license": "MIT",
59
59
  "dependencies": {
60
- "@hyperfixi/core": "^2.10.0",
61
- "@hyperfixi/patterns-reference": "^2.10.0",
62
- "@lokascript/i18n": "^2.10.0",
63
- "@lokascript/semantic": "^2.10.0",
60
+ "@hyperfixi/core": "^2.11.0",
61
+ "@hyperfixi/patterns-reference": "^2.11.0",
62
+ "@lokascript/compilation-service": "^2.11.0",
63
+ "@lokascript/i18n": "^2.11.0",
64
+ "@lokascript/semantic": "^2.11.0",
64
65
  "diff": "^8.0.3",
65
66
  "esbuild": "^0.28.0",
66
67
  "happy-dom": "^20.10.6",
@@ -0,0 +1,162 @@
1
+ # Agent-loop benchmark
2
+
3
+ Measures what an LLM agent actually gets from emitting hyperscript through the
4
+ MCP validate/repair/compile loop, against twenty natural-language UI tasks.
5
+
6
+ Arc 3 of [AGENT_ERA_ROADMAP.md](../../../../docs-internal/AGENT_ERA_ROADMAP.md).
7
+
8
+ ```bash
9
+ cd packages/testing-framework
10
+ npx tsx src/agent-bench/cli.ts verify-references # every reference still usable
11
+ npx tsx src/agent-bench/cli.ts probe-variants # the deterministic finding
12
+ npx tsx src/agent-bench/cli.ts list # prompts, for a generator
13
+ npx tsx src/agent-bench/cli.ts score --run runs/my-run.json
14
+ ```
15
+
16
+ ## What it measures, and why in two halves
17
+
18
+ Two questions are scored separately, and keeping them apart is the point:
19
+
20
+ - **Does it parse?** `CompilationService.validate()` — the exact call the
21
+ `validate_and_compile` MCP tool makes.
22
+ - **Does it do the right thing?** The candidate is executed in jsdom against the
23
+ task's fixture and its DOM effect signature compared byte-for-byte to the
24
+ reference's. (Effect-signature primitives are shared with the R2 ratchet, so
25
+ the two cannot disagree about what a DOM effect is.)
26
+
27
+ A single blended "success rate" would hide the failure mode that matters most:
28
+ the parser **degrades rather than failing**, so a candidate can come back
29
+ `ok: true`, confidence 1.0, zero diagnostics — and target the wrong element, or
30
+ do nothing at all.
31
+
32
+ ### Half 1 — the plausible-phrasing probe (no generator needed)
33
+
34
+ `probe-variants` scores a fixed catalogue of phrasings a competent generator
35
+ plausibly reaches for ([variants.ts](./variants.ts), each annotated with why).
36
+ No LLM runs, so the result is a reproducible property of the parser — measured
37
+ byte-identical across runs — and the claim stays narrow and checkable: _these
38
+ phrasings behave thus_, not _a model emits them at rate R_.
39
+
40
+ Recorded in [`baselines/agent-bench-phrasings.json`](../../baselines/agent-bench-phrasings.json)
41
+ and ratcheted both directions at tolerance 0 by
42
+ [`agent-bench.test.ts`](./agent-bench.test.ts) — a regression fails, and so does
43
+ an unrecorded improvement (the numbers below are quoted in the docs, so a stale
44
+ baseline makes them lie).
45
+
46
+ ### Half 2 — the A/B run (needs a generator)
47
+
48
+ The loop being measured is the one a real integration runs, so simulating it
49
+ in-process would measure a simulation. Instead the harness is agent-driven:
50
+
51
+ 1. `cli.ts list --json` — the prompts.
52
+ 2. **One-shot condition.** The agent answers every prompt from the prompt alone,
53
+ with no validation. Record them; do not revise after seeing any score.
54
+ 3. **Loop condition.** The agent answers again, this time iterating through
55
+ `cli.ts feedback --task <id> --code "<src>"` until it parses or it gives up.
56
+ `feedback` returns **only** what the MCP loop returns — diagnostics and the
57
+ parsed IR. It never reveals the reference or the behavior verdict; leaking
58
+ either would turn the loop into an oracle and the delta into fiction.
59
+ 4. `cli.ts score --run <file>` — per-condition parse rate, behavior rate, and
60
+ the delta.
61
+
62
+ Run file:
63
+
64
+ ```json
65
+ {
66
+ "generator": { "model": "…", "date": "…", "context": "what the generator could see" },
67
+ "conditions": {
68
+ "one-shot": { "toggle-self-class": "on click toggle .active on me" },
69
+ "loop": { "toggle-self-class": "on click toggle .active on me" }
70
+ }
71
+ }
72
+ ```
73
+
74
+ **No A/B run is committed here, deliberately.** The tasks and their reference
75
+ implementations were authored in the same session that would have produced the
76
+ candidates, so any one-shot number from that session measures recall of
77
+ just-written answers, not generation. A meaningful run needs a generator that
78
+ has not seen this directory — until one exists, the honest position is a harness
79
+ with no number attached, not a flattering number with a caveat. `score` is fully
80
+ implemented and ready for that run.
81
+
82
+ Not a CI gate: LLM-in-the-loop is nondeterministic and this repo's gates stay
83
+ deterministic. Half 1 **is** gated, because it has no generator in it.
84
+
85
+ ## Findings
86
+
87
+ ### First probe, 2026-08-24 (pre-3b)
88
+
89
+ 37 plausible phrasings, 20 tasks: **97% parse, 49% behave correctly, 49% parse
90
+ clean but misbehave** — and exactly one of the failures produced a diagnostic.
91
+ Iterating `validate → fix → re-validate` could not move a single ☠ row, because
92
+ the loop was never told anything was wrong. The actionable conclusion was not
93
+ "polish the loop" but **make these failures loud** — which became Arc 3b.
94
+
95
+ ### Arc 3b, first diagnostic (unconsumed-input propagation)
96
+
97
+ The parser had been flagging dropped tokens all along — a `warning`-severity
98
+ `unconsumed-input` diagnostic on the node, hoisted from any depth, with a
99
+ confidence dock — and `CompilationService.normalize()` simply never read node
100
+ diagnostics, so `validate()` reported `ok` with an empty diagnostics array. The
101
+ fix is pure plumbing (no parser change, so the multilingual ratchets are
102
+ untouched): lift warning/error-severity node diagnostics into the response, as
103
+ `UNCONSUMED_INPUT` with a repair suggestion.
104
+
105
+ | | pre-3b | post-3b |
106
+ | ------------------------------------- | ----------- | ---------------------- |
107
+ | parse | 36/37 (97%) | 36/37 (97%) |
108
+ | behave correctly | 18/37 (49%) | 18/37 (49%) |
109
+ | wrong but **warned** (loop can react) | 1/37 | **12/37** (11 ⚠ + 1 ✗) |
110
+ | wrong and **silent** | 18/37 (49%) | **7/37 (19%)** |
111
+
112
+ One plumbing fix moved 11 of the 18 silent rows into the visible band: the
113
+ omitted-marker family, the whole attribute-write family, `to every .y`,
114
+ `the X of Y` properties, and `remove element`. Behavior is unchanged — these
115
+ phrasings are still wrong — but the loop can now see and repair them.
116
+
117
+ The remaining ☠ 7 split honestly in two:
118
+
119
+ - **Real diagnostic gaps (5)** — parses that consume everything yet provably
120
+ do nothing or bind the wrong target with no trace: `add .x to all .y` /
121
+ `remove .x from all .y` (note the asymmetry: `every` warns, `all` doesn't),
122
+ `set the text of #el`, `if #el has class .x`, `add .x to <body/>`. These are
123
+ the next 3b targets (a no-op-command diagnostic covers most).
124
+ - **Valid code, different intent (2)** — `add .hidden to #menu` (adds a class
125
+ named "hidden"; only wrong versus the _hide_ reference) and `on mouseover`
126
+ (a real handler for a neighbouring event). No parser diagnostic can catch
127
+ these; they are exactly what IR-vs-intent review (and Arc 4's equivalence
128
+ checking) is for.
129
+
130
+ ### Arc 3b, second diagnostic (inert-shape gate)
131
+
132
+ The five remaining real gaps all parsed fully-consumed at confidence 1.0 but
133
+ left distinctive fingerprints in the IR: `all .todo` becomes a property access
134
+ on the undefined identifier `all`; `has class` mis-tokenizes the predicate as a
135
+ class selector (`#box .has class .danger`) so the condition is always falsy;
136
+ `<body/>` survives as a raw non-CSS selector that `querySelector` rejects; and
137
+ `the text of #el` becomes an invisible expando write. Gate 4 of the
138
+ compilation-service validation pipeline (`validation/inert-shapes.ts`) matches
139
+ those fingerprints and warns (`INERT_QUANTIFIER_TARGET`,
140
+ `HALF_PARSED_CONDITION`, `UNSUPPORTED_QUERY_LITERAL`, `INERT_PROPERTY_WRITE`)
141
+ — warnings only, still no parser change.
142
+
143
+ | | pre-3b | after slice 1 | after slice 2 |
144
+ | ------------------------------------- | ----------- | ------------- | ---------------------- |
145
+ | wrong but **warned** (loop can react) | 1/37 | 12/37 | **17/37** (16 ⚠ + 1 ✗) |
146
+ | wrong and **silent** | 18/37 (49%) | 7/37 (19%) | **2/37 (5%)** |
147
+
148
+ The two remaining ☠ rows are the valid-code-different-intent pair
149
+ (`add .hidden to #menu`, `on mouseover`) — by design not diagnosable, and the
150
+ standing case for IR-vs-intent review and Arc 4's behavioral equivalence. **The
151
+ parser-gap silent band is now zero.**
152
+
153
+ Bands are computed by `harness.bandOf` — one function shared by the probe, the
154
+ committed baseline, and the ratchet test, so they cannot drift apart.
155
+
156
+ ## Adding a task
157
+
158
+ Append to [`tasks.ts`](./tasks.ts): a prompt with no hyperscript in it, a
159
+ reference, and the fixture markup it needs. Then run `verify-references` — a
160
+ reference that does not parse, or produces no DOM effect, is rejected (same
161
+ eligibility bar as R2's execution subset), because scoring against an empty
162
+ signature would make wrong answers look right.
@@ -0,0 +1,112 @@
1
+ /**
2
+ * Guards for the agent-loop benchmark.
3
+ *
4
+ * Two things rot silently and would make the benchmark lie:
5
+ *
6
+ * 1. **A reference that stops working.** Behavior correctness is defined as
7
+ * "same effect signature as the reference", so a reference that stops
8
+ * parsing — or starts producing NO effects — makes every candidate for
9
+ * that task score wrong (or, worse, makes an empty-effect candidate score
10
+ * right). Same eligibility bar as R2's execution subset.
11
+ *
12
+ * 2. **The silent band drifting.** `baselines/agent-bench-phrasings.json`
13
+ * records, per plausible phrasing, whether it is correct / rejected /
14
+ * silently wrong. This ratchets BOTH directions at tolerance 0: a
15
+ * regression (correct → silent) fails, and so does an improvement that
16
+ * wasn't re-baselined. The improvement direction matters as much here as
17
+ * in the R4 allowlist — a fixed parser gap that nobody re-records leaves
18
+ * the docs quoting a stale number, and the "half of plausible phrasings
19
+ * misbehave" claim is load-bearing for the roadmap.
20
+ *
21
+ * Deterministic and generator-free: no LLM runs here, so this IS a legitimate
22
+ * CI gate (the A/B run in README.md, which needs a generator, deliberately is
23
+ * not). Full sweep measures ~6s.
24
+ */
25
+
26
+ import { describe, it, expect, beforeAll } from 'vitest';
27
+ import { readFileSync } from 'node:fs';
28
+ import path from 'node:path';
29
+ import { fileURLToPath } from 'node:url';
30
+ import { TASKS, taskById } from './tasks.js';
31
+ import { VARIANTS } from './variants.js';
32
+ import {
33
+ bandOf,
34
+ executeCandidate,
35
+ initialize,
36
+ scoreCandidate,
37
+ validateCandidate,
38
+ } from './harness.js';
39
+
40
+ const BASELINE_PATH = path.resolve(
41
+ path.dirname(fileURLToPath(import.meta.url)),
42
+ '../../baselines/agent-bench-phrasings.json'
43
+ );
44
+
45
+ interface Baseline {
46
+ totals: Record<string, number>;
47
+ phrasings: Array<{ taskId: string; code: string; band: string }>;
48
+ }
49
+
50
+ describe('agent-bench: task references', () => {
51
+ beforeAll(async () => {
52
+ await initialize();
53
+ }, 60_000);
54
+
55
+ it('every task has a unique id', () => {
56
+ const ids = TASKS.map(t => t.id);
57
+ expect(new Set(ids).size).toBe(ids.length);
58
+ });
59
+
60
+ it('no prompt leaks hyperscript syntax to the generator', () => {
61
+ // A prompt containing the answer measures nothing. Selector/sigil
62
+ // characters and command keywords in the imperative position are the tells.
63
+ const leaks = TASKS.filter(t => /\bon click\b|=>|\s_=|@[a-z-]+\s|\*[a-z-]+\s/i.test(t.prompt));
64
+ expect(leaks.map(t => t.id)).toEqual([]);
65
+ });
66
+
67
+ it.each(TASKS.map(t => [t.id, t] as const))(
68
+ 'reference for %s parses and produces a usable effect signature',
69
+ async (_id, task) => {
70
+ const validation = await validateCandidate(task.reference);
71
+ expect(validation.ok, `reference does not parse: ${task.reference}`).toBe(true);
72
+ const { effects, error } = await executeCandidate(task, task.reference);
73
+ expect(error, `reference errored: ${error}`).toBeUndefined();
74
+ expect(effects.length, `reference has no DOM effect: ${task.reference}`).toBeGreaterThan(0);
75
+ },
76
+ 30_000
77
+ );
78
+ });
79
+
80
+ describe('agent-bench: plausible-phrasing ratchet', () => {
81
+ let baseline: Baseline;
82
+
83
+ beforeAll(async () => {
84
+ await initialize();
85
+ baseline = JSON.parse(readFileSync(BASELINE_PATH, 'utf8')) as Baseline;
86
+ }, 60_000);
87
+
88
+ it('baseline covers exactly the current variant set', () => {
89
+ const recorded = baseline.phrasings.map(p => p.code).sort();
90
+ const current = VARIANTS.map(v => v.code).sort();
91
+ expect(
92
+ recorded,
93
+ 'variants changed without regenerating: tsx src/agent-bench/cli.ts probe-variants --json > baselines/agent-bench-phrasings.json'
94
+ ).toEqual(current);
95
+ });
96
+
97
+ it('every phrasing still lands in its recorded band (both directions, tolerance 0)', async () => {
98
+ const drift: string[] = [];
99
+ for (const entry of baseline.phrasings) {
100
+ const task = taskById(entry.taskId);
101
+ expect(task, `baseline names unknown task ${entry.taskId}`).toBeDefined();
102
+ const s = await scoreCandidate(task!, entry.code);
103
+ const band = bandOf(s);
104
+ if (band !== entry.band) drift.push(`${entry.code}\n ${entry.band} → ${band}`);
105
+ }
106
+ expect(
107
+ drift,
108
+ 'phrasing behavior drifted. If this is an intentional improvement, regenerate:\n' +
109
+ ' tsx src/agent-bench/cli.ts probe-variants --json > baselines/agent-bench-phrasings.json'
110
+ ).toEqual([]);
111
+ }, 120_000);
112
+ });
@@ -0,0 +1,315 @@
1
+ #!/usr/bin/env tsx
2
+ /**
3
+ * Agent-loop benchmark CLI.
4
+ *
5
+ * verify-references every reference parses and has a usable signature
6
+ * list [--json] the task prompts, for a generator to answer
7
+ * feedback --task <id> --code <src> what the loop hands back for one candidate
8
+ * score --run <file> score a run file; prints per-condition rates + delta
9
+ *
10
+ * The generator is an AGENT, not this script: the loop being measured is the one
11
+ * a real integration runs, so simulating it in-process would measure a
12
+ * simulation. `list` emits the prompts, the agent answers them, `score` grades
13
+ * the answers. See README.md for the protocol.
14
+ *
15
+ * `feedback` deliberately returns ONLY what the MCP loop returns — diagnostics
16
+ * and the parsed IR. It never reveals the reference or whether the candidate
17
+ * behaves correctly. An agent iterating with `feedback` therefore has exactly
18
+ * the information the real loop gives it; leaking the behavior verdict would
19
+ * turn the loop condition into an oracle and the headline delta into fiction.
20
+ */
21
+
22
+ import { readFileSync } from 'node:fs';
23
+ import { TASKS, taskById } from './tasks.js';
24
+ import { VARIANTS } from './variants.js';
25
+ import {
26
+ bandOf,
27
+ executeCandidate,
28
+ scoreCandidate,
29
+ scoreCondition,
30
+ validateCandidate,
31
+ type Band,
32
+ type ConditionScore,
33
+ } from './harness.js';
34
+
35
+ interface RunFile {
36
+ generator?: Record<string, unknown>;
37
+ conditions: Record<string, Record<string, string>>;
38
+ }
39
+
40
+ function arg(name: string): string | undefined {
41
+ const i = process.argv.indexOf(`--${name}`);
42
+ return i === -1 ? undefined : process.argv[i + 1];
43
+ }
44
+
45
+ const pct = (n: number): string => `${(n * 100).toFixed(0)}%`;
46
+
47
+ // =============================================================================
48
+ // verify-references
49
+ // =============================================================================
50
+
51
+ async function verifyReferences(): Promise<number> {
52
+ let bad = 0;
53
+ const seen = new Map<string, string>();
54
+ for (const task of TASKS) {
55
+ const validation = await validateCandidate(task.reference);
56
+ const { effects, error } = await executeCandidate(task, task.reference);
57
+ const problems: string[] = [];
58
+ if (!validation.ok) problems.push('reference does not parse');
59
+ if (effects.length === 0) problems.push('reference produces NO effect signature');
60
+ if (error) problems.push(`error: ${error}`);
61
+ // Two tasks with identical signatures would let a candidate score right on
62
+ // the wrong task; not fatal (fixtures legitimately overlap) but worth saying.
63
+ const key = JSON.stringify(effects);
64
+ const twin = seen.get(key);
65
+ if (effects.length > 0 && twin) problems.push(`signature identical to ${twin}`);
66
+ if (effects.length > 0) seen.set(key, task.id);
67
+
68
+ if (problems.length > 0) {
69
+ bad++;
70
+ console.log(`✗ ${task.id}: ${problems.join('; ')}`);
71
+ console.log(` ${task.reference}`);
72
+ } else {
73
+ console.log(`✓ ${task.id} ${effects.length} effect(s)`);
74
+ }
75
+ }
76
+ console.log(
77
+ bad === 0
78
+ ? `\nAll ${TASKS.length} references usable.`
79
+ : `\n${bad}/${TASKS.length} references UNUSABLE — fix before scoring.`
80
+ );
81
+ return bad === 0 ? 0 : 1;
82
+ }
83
+
84
+ // =============================================================================
85
+ // list / feedback
86
+ // =============================================================================
87
+
88
+ function list(): number {
89
+ if (process.argv.includes('--json')) {
90
+ console.log(
91
+ JSON.stringify(
92
+ TASKS.map(t => ({ id: t.id, prompt: t.prompt })),
93
+ null,
94
+ 2
95
+ )
96
+ );
97
+ return 0;
98
+ }
99
+ for (const t of TASKS) console.log(`${t.id}\n ${t.prompt}\n`);
100
+ return 0;
101
+ }
102
+
103
+ async function feedback(): Promise<number> {
104
+ const id = arg('task');
105
+ const code = arg('code');
106
+ if (!id || code === undefined) {
107
+ console.error('usage: feedback --task <id> --code "<hyperscript>"');
108
+ return 2;
109
+ }
110
+ if (!taskById(id)) {
111
+ console.error(`unknown task: ${id}`);
112
+ return 2;
113
+ }
114
+ const v = await validateCandidate(code);
115
+ console.log(
116
+ JSON.stringify(
117
+ { ok: v.ok, confidence: v.confidence, parsed: v.summary, diagnostics: v.diagnostics },
118
+ null,
119
+ 2
120
+ )
121
+ );
122
+ if (!v.ok) {
123
+ console.log(
124
+ '\nNext step: apply the diagnostics above and re-run. get_code_fixes maps ' +
125
+ 'error codes to concrete fixes; get_command_docs lists per-command roles.'
126
+ );
127
+ }
128
+ return 0;
129
+ }
130
+
131
+ // =============================================================================
132
+ // score
133
+ // =============================================================================
134
+
135
+ function reportCondition(c: ConditionScore): void {
136
+ console.log(`\n── ${c.condition} ──`);
137
+ for (const s of c.scores) {
138
+ const mark = s.behaviorMatch ? '✓' : s.parsed ? '~' : '✗';
139
+ console.log(`${mark} ${s.taskId.padEnd(22)} ${s.code}`);
140
+ if (!s.behaviorMatch) {
141
+ if (!s.parsed) {
142
+ const first = s.validation.diagnostics.find(d => d.severity === 'error');
143
+ console.log(` did not parse: ${first?.message ?? 'unknown'}`);
144
+ } else {
145
+ console.log(` PARSED BUT WRONG — ${s.validation.summary ?? '?'}`);
146
+ console.log(` got ${JSON.stringify(s.execution.effects)}`);
147
+ console.log(` expected ${JSON.stringify(s.referenceEffects)}`);
148
+ if (s.execution.error) console.log(` error: ${s.execution.error}`);
149
+ }
150
+ }
151
+ }
152
+ for (const m of c.missing) console.log(`✗ ${m.padEnd(22)} (no candidate submitted)`);
153
+ console.log(
154
+ `\n parse rate ${pct(c.parseRate)}` +
155
+ `\n behavior rate ${pct(c.behaviorRate)}` +
156
+ `\n parsed-but-wrong ${c.silentlyWrongCount}`
157
+ );
158
+ }
159
+
160
+ async function score(): Promise<number> {
161
+ const file = arg('run');
162
+ if (!file) {
163
+ console.error('usage: score --run <file.json>');
164
+ return 2;
165
+ }
166
+ const run = JSON.parse(readFileSync(file, 'utf8')) as RunFile;
167
+ if (run.generator) console.log(`generator: ${JSON.stringify(run.generator)}`);
168
+
169
+ const results: ConditionScore[] = [];
170
+ for (const [condition, candidates] of Object.entries(run.conditions)) {
171
+ const c = await scoreCondition(condition, candidates, TASKS);
172
+ results.push(c);
173
+ reportCondition(c);
174
+ }
175
+
176
+ if (results.length >= 2) {
177
+ const [first, last] = [results[0]!, results[results.length - 1]!];
178
+ console.log(`\n══ ${first.condition} → ${last.condition} ══`);
179
+ console.log(` parse rate ${pct(first.parseRate)} → ${pct(last.parseRate)}`);
180
+ console.log(` behavior rate ${pct(first.behaviorRate)} → ${pct(last.behaviorRate)}`);
181
+ console.log(
182
+ ` parsed-but-wrong ${first.silentlyWrongCount} → ${last.silentlyWrongCount} (over ${TASKS.length} tasks)`
183
+ );
184
+ }
185
+ return 0;
186
+ }
187
+
188
+ // =============================================================================
189
+ // probe-variants
190
+ // =============================================================================
191
+
192
+ /**
193
+ * Score the plausible-phrasing catalogue. Needs no generator: each row is a
194
+ * fixed string, so the result is a reproducible property of the parser.
195
+ *
196
+ * The headline is the SILENT band — phrasings that parse clean, carry no
197
+ * warning, and misbehave. Those are invisible to the validate/repair loop by
198
+ * construction, so they bound how much the loop can ever deliver. Arc 3b work
199
+ * moves rows out of it into `warned-wrong` (wrong but visible), which the loop
200
+ * handles.
201
+ */
202
+ async function probeVariants(): Promise<number> {
203
+ const rows: Array<{
204
+ v: (typeof VARIANTS)[number];
205
+ parsed: boolean;
206
+ behaved: boolean;
207
+ band: Band;
208
+ detail: string;
209
+ }> = [];
210
+
211
+ for (const v of VARIANTS) {
212
+ const task = taskById(v.taskId);
213
+ if (!task) {
214
+ console.error(`unknown task in variants: ${v.taskId}`);
215
+ return 2;
216
+ }
217
+ const s = await scoreCandidate(task, v.code);
218
+ const band = bandOf(s);
219
+ const detail =
220
+ band === 'correct'
221
+ ? 'ok'
222
+ : band === 'rejected'
223
+ ? 'rejected'
224
+ : band === 'warned-wrong'
225
+ ? `WARNED (${s.validation.diagnostics.find(d => d.severity !== 'info')?.code ?? '?'}) — wrong, but the loop can see it`
226
+ : band === 'silent-noop'
227
+ ? 'SILENT NO-OP'
228
+ : `WRONG EFFECT ${JSON.stringify(s.execution.effects)}`;
229
+ rows.push({ v, parsed: s.parsed, behaved: s.behaviorMatch, band, detail });
230
+ }
231
+
232
+ const n = rows.length;
233
+
234
+ if (process.argv.includes('--json')) {
235
+ // Baseline shape: sorted by code so the file is diff-stable, and carrying
236
+ // only the deterministic verdict (never timings) so regeneration on another
237
+ // machine produces a byte-identical file. Bands come from harness.bandOf —
238
+ // the same function the ratchet test recomputes with.
239
+ console.log(
240
+ JSON.stringify(
241
+ {
242
+ note:
243
+ 'Deterministic parse-vs-behavior verdicts for plausible phrasings. ' +
244
+ 'Regenerate with: tsx src/agent-bench/cli.ts probe-variants --json',
245
+ totals: {
246
+ phrasings: n,
247
+ parse: rows.filter(r => r.parsed).length,
248
+ behaveCorrectly: rows.filter(r => r.behaved).length,
249
+ warnedWrong: rows.filter(r => r.band === 'warned-wrong').length,
250
+ silentBand: rows.filter(r => r.band.startsWith('silent')).length,
251
+ silentNoop: rows.filter(r => r.band === 'silent-noop').length,
252
+ },
253
+ phrasings: rows
254
+ .map(r => ({ taskId: r.v.taskId, code: r.v.code, band: r.band }))
255
+ .sort((a, b) => (a.code < b.code ? -1 : a.code > b.code ? 1 : 0)),
256
+ },
257
+ null,
258
+ 2
259
+ )
260
+ );
261
+ return 0;
262
+ }
263
+
264
+ for (const r of rows) {
265
+ const mark = r.behaved
266
+ ? '✓'
267
+ : r.band === 'warned-wrong'
268
+ ? '⚠'
269
+ : r.band === 'rejected'
270
+ ? '✗'
271
+ : '☠';
272
+ console.log(`${mark} ${r.v.taskId.padEnd(20)} ${r.v.code}`);
273
+ if (!r.behaved) console.log(` ${r.detail} — ${r.v.rationale}`);
274
+ }
275
+
276
+ const parsed = rows.filter(r => r.parsed).length;
277
+ const behaved = rows.filter(r => r.behaved).length;
278
+ const warned = rows.filter(r => r.band === 'warned-wrong').length;
279
+ const silent = rows.filter(r => r.band.startsWith('silent')).length;
280
+ const noop = rows.filter(r => r.band === 'silent-noop').length;
281
+ console.log(
282
+ `\n ${n} plausible phrasings` +
283
+ `\n parse ${parsed}/${n} (${pct(parsed / n)})` +
284
+ `\n behave correctly ${behaved}/${n} (${pct(behaved / n)})` +
285
+ `\n ⚠ wrong but WARNED (loop can react): ${warned}/${n}` +
286
+ `\n ☠ wrong and SILENT: ${silent}/${n} (${pct(silent / n)}) — of which ${noop} do NOTHING at all` +
287
+ `\n\n The ☠ band is what the validate/repair loop cannot see: no diagnostic,` +
288
+ `\n no error, nothing for an agent to react to.`
289
+ );
290
+ return 0;
291
+ }
292
+
293
+ // =============================================================================
294
+
295
+ const commands: Record<string, () => Promise<number> | number> = {
296
+ 'verify-references': verifyReferences,
297
+ 'probe-variants': probeVariants,
298
+ list,
299
+ feedback,
300
+ score,
301
+ };
302
+
303
+ async function main(): Promise<void> {
304
+ const cmd = process.argv[2] ?? '';
305
+ const run = commands[cmd];
306
+ if (!run) {
307
+ console.error(`usage: cli.ts <${Object.keys(commands).join('|')}> [options]`);
308
+ process.exit(2);
309
+ }
310
+ // Explicit exit: esbuild's daemon keeps the event loop alive (see CLAUDE.md),
311
+ // so a natural return would hang the process after the report is printed.
312
+ process.exit(await run());
313
+ }
314
+
315
+ void main();