@hyperfixi/testing-framework 2.10.0 → 2.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +7 -6
- package/src/agent-bench/README.md +162 -0
- package/src/agent-bench/agent-bench.test.ts +112 -0
- package/src/agent-bench/cli.ts +315 -0
- package/src/agent-bench/harness.ts +331 -0
- package/src/agent-bench/tasks.ts +214 -0
- package/src/agent-bench/variants.ts +241 -0
- package/src/multilingual/fidelity.ts +22 -376
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@hyperfixi/testing-framework",
|
|
3
|
-
"version": "2.
|
|
3
|
+
"version": "2.11.0",
|
|
4
4
|
"description": "Cross-platform behavior testing suite for LokaScript applications",
|
|
5
5
|
"main": "dist/index.js",
|
|
6
6
|
"module": "dist/index.mjs",
|
|
@@ -29,7 +29,7 @@
|
|
|
29
29
|
"scripts": {
|
|
30
30
|
"build": "tsup",
|
|
31
31
|
"dev": "tsup --watch",
|
|
32
|
-
"pretest": "../../scripts/ensure-fresh.sh ../intent ../framework ../semantic ../patterns-reference ../core",
|
|
32
|
+
"pretest": "../../scripts/ensure-fresh.sh ../intent ../framework ../semantic ../patterns-reference ../core ../aot-compiler ../compilation-service",
|
|
33
33
|
"test": "vitest run",
|
|
34
34
|
"test:watch": "vitest",
|
|
35
35
|
"test:coverage": "vitest run --coverage",
|
|
@@ -57,10 +57,11 @@
|
|
|
57
57
|
"author": "LokaScript Contributors",
|
|
58
58
|
"license": "MIT",
|
|
59
59
|
"dependencies": {
|
|
60
|
-
"@hyperfixi/core": "^2.
|
|
61
|
-
"@hyperfixi/patterns-reference": "^2.
|
|
62
|
-
"@lokascript/
|
|
63
|
-
"@lokascript/
|
|
60
|
+
"@hyperfixi/core": "^2.11.0",
|
|
61
|
+
"@hyperfixi/patterns-reference": "^2.11.0",
|
|
62
|
+
"@lokascript/compilation-service": "^2.11.0",
|
|
63
|
+
"@lokascript/i18n": "^2.11.0",
|
|
64
|
+
"@lokascript/semantic": "^2.11.0",
|
|
64
65
|
"diff": "^8.0.3",
|
|
65
66
|
"esbuild": "^0.28.0",
|
|
66
67
|
"happy-dom": "^20.10.6",
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
# Agent-loop benchmark
|
|
2
|
+
|
|
3
|
+
Measures what an LLM agent actually gets from emitting hyperscript through the
|
|
4
|
+
MCP validate/repair/compile loop, against twenty natural-language UI tasks.
|
|
5
|
+
|
|
6
|
+
Arc 3 of [AGENT_ERA_ROADMAP.md](../../../../docs-internal/AGENT_ERA_ROADMAP.md).
|
|
7
|
+
|
|
8
|
+
```bash
|
|
9
|
+
cd packages/testing-framework
|
|
10
|
+
npx tsx src/agent-bench/cli.ts verify-references # every reference still usable
|
|
11
|
+
npx tsx src/agent-bench/cli.ts probe-variants # the deterministic finding
|
|
12
|
+
npx tsx src/agent-bench/cli.ts list # prompts, for a generator
|
|
13
|
+
npx tsx src/agent-bench/cli.ts score --run runs/my-run.json
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
## What it measures, and why in two halves
|
|
17
|
+
|
|
18
|
+
Two questions are scored separately, and keeping them apart is the point:
|
|
19
|
+
|
|
20
|
+
- **Does it parse?** `CompilationService.validate()` — the exact call the
|
|
21
|
+
`validate_and_compile` MCP tool makes.
|
|
22
|
+
- **Does it do the right thing?** The candidate is executed in jsdom against the
|
|
23
|
+
task's fixture and its DOM effect signature compared byte-for-byte to the
|
|
24
|
+
reference's. (Effect-signature primitives are shared with the R2 ratchet, so
|
|
25
|
+
the two cannot disagree about what a DOM effect is.)
|
|
26
|
+
|
|
27
|
+
A single blended "success rate" would hide the failure mode that matters most:
|
|
28
|
+
the parser **degrades rather than failing**, so a candidate can come back
|
|
29
|
+
`ok: true`, confidence 1.0, zero diagnostics — and target the wrong element, or
|
|
30
|
+
do nothing at all.
|
|
31
|
+
|
|
32
|
+
### Half 1 — the plausible-phrasing probe (no generator needed)
|
|
33
|
+
|
|
34
|
+
`probe-variants` scores a fixed catalogue of phrasings a competent generator
|
|
35
|
+
plausibly reaches for ([variants.ts](./variants.ts), each annotated with why).
|
|
36
|
+
No LLM runs, so the result is a reproducible property of the parser — measured
|
|
37
|
+
byte-identical across runs — and the claim stays narrow and checkable: _these
|
|
38
|
+
phrasings behave thus_, not _a model emits them at rate R_.
|
|
39
|
+
|
|
40
|
+
Recorded in [`baselines/agent-bench-phrasings.json`](../../baselines/agent-bench-phrasings.json)
|
|
41
|
+
and ratcheted both directions at tolerance 0 by
|
|
42
|
+
[`agent-bench.test.ts`](./agent-bench.test.ts) — a regression fails, and so does
|
|
43
|
+
an unrecorded improvement (the numbers below are quoted in the docs, so a stale
|
|
44
|
+
baseline makes them lie).
|
|
45
|
+
|
|
46
|
+
### Half 2 — the A/B run (needs a generator)
|
|
47
|
+
|
|
48
|
+
The loop being measured is the one a real integration runs, so simulating it
|
|
49
|
+
in-process would measure a simulation. Instead the harness is agent-driven:
|
|
50
|
+
|
|
51
|
+
1. `cli.ts list --json` — the prompts.
|
|
52
|
+
2. **One-shot condition.** The agent answers every prompt from the prompt alone,
|
|
53
|
+
with no validation. Record them; do not revise after seeing any score.
|
|
54
|
+
3. **Loop condition.** The agent answers again, this time iterating through
|
|
55
|
+
`cli.ts feedback --task <id> --code "<src>"` until it parses or it gives up.
|
|
56
|
+
`feedback` returns **only** what the MCP loop returns — diagnostics and the
|
|
57
|
+
parsed IR. It never reveals the reference or the behavior verdict; leaking
|
|
58
|
+
either would turn the loop into an oracle and the delta into fiction.
|
|
59
|
+
4. `cli.ts score --run <file>` — per-condition parse rate, behavior rate, and
|
|
60
|
+
the delta.
|
|
61
|
+
|
|
62
|
+
Run file:
|
|
63
|
+
|
|
64
|
+
```json
|
|
65
|
+
{
|
|
66
|
+
"generator": { "model": "…", "date": "…", "context": "what the generator could see" },
|
|
67
|
+
"conditions": {
|
|
68
|
+
"one-shot": { "toggle-self-class": "on click toggle .active on me" },
|
|
69
|
+
"loop": { "toggle-self-class": "on click toggle .active on me" }
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
**No A/B run is committed here, deliberately.** The tasks and their reference
|
|
75
|
+
implementations were authored in the same session that would have produced the
|
|
76
|
+
candidates, so any one-shot number from that session measures recall of
|
|
77
|
+
just-written answers, not generation. A meaningful run needs a generator that
|
|
78
|
+
has not seen this directory — until one exists, the honest position is a harness
|
|
79
|
+
with no number attached, not a flattering number with a caveat. `score` is fully
|
|
80
|
+
implemented and ready for that run.
|
|
81
|
+
|
|
82
|
+
Not a CI gate: LLM-in-the-loop is nondeterministic and this repo's gates stay
|
|
83
|
+
deterministic. Half 1 **is** gated, because it has no generator in it.
|
|
84
|
+
|
|
85
|
+
## Findings
|
|
86
|
+
|
|
87
|
+
### First probe, 2026-08-24 (pre-3b)
|
|
88
|
+
|
|
89
|
+
37 plausible phrasings, 20 tasks: **97% parse, 49% behave correctly, 49% parse
|
|
90
|
+
clean but misbehave** — and exactly one of the failures produced a diagnostic.
|
|
91
|
+
Iterating `validate → fix → re-validate` could not move a single ☠ row, because
|
|
92
|
+
the loop was never told anything was wrong. The actionable conclusion was not
|
|
93
|
+
"polish the loop" but **make these failures loud** — which became Arc 3b.
|
|
94
|
+
|
|
95
|
+
### Arc 3b, first diagnostic (unconsumed-input propagation)
|
|
96
|
+
|
|
97
|
+
The parser had been flagging dropped tokens all along — a `warning`-severity
|
|
98
|
+
`unconsumed-input` diagnostic on the node, hoisted from any depth, with a
|
|
99
|
+
confidence dock — and `CompilationService.normalize()` simply never read node
|
|
100
|
+
diagnostics, so `validate()` reported `ok` with an empty diagnostics array. The
|
|
101
|
+
fix is pure plumbing (no parser change, so the multilingual ratchets are
|
|
102
|
+
untouched): lift warning/error-severity node diagnostics into the response, as
|
|
103
|
+
`UNCONSUMED_INPUT` with a repair suggestion.
|
|
104
|
+
|
|
105
|
+
| | pre-3b | post-3b |
|
|
106
|
+
| ------------------------------------- | ----------- | ---------------------- |
|
|
107
|
+
| parse | 36/37 (97%) | 36/37 (97%) |
|
|
108
|
+
| behave correctly | 18/37 (49%) | 18/37 (49%) |
|
|
109
|
+
| wrong but **warned** (loop can react) | 1/37 | **12/37** (11 ⚠ + 1 ✗) |
|
|
110
|
+
| wrong and **silent** | 18/37 (49%) | **7/37 (19%)** |
|
|
111
|
+
|
|
112
|
+
One plumbing fix moved 11 of the 18 silent rows into the visible band: the
|
|
113
|
+
omitted-marker family, the whole attribute-write family, `to every .y`,
|
|
114
|
+
`the X of Y` properties, and `remove element`. Behavior is unchanged — these
|
|
115
|
+
phrasings are still wrong — but the loop can now see and repair them.
|
|
116
|
+
|
|
117
|
+
The remaining ☠ 7 split honestly in two:
|
|
118
|
+
|
|
119
|
+
- **Real diagnostic gaps (5)** — parses that consume everything yet provably
|
|
120
|
+
do nothing or bind the wrong target with no trace: `add .x to all .y` /
|
|
121
|
+
`remove .x from all .y` (note the asymmetry: `every` warns, `all` doesn't),
|
|
122
|
+
`set the text of #el`, `if #el has class .x`, `add .x to <body/>`. These are
|
|
123
|
+
the next 3b targets (a no-op-command diagnostic covers most).
|
|
124
|
+
- **Valid code, different intent (2)** — `add .hidden to #menu` (adds a class
|
|
125
|
+
named "hidden"; only wrong versus the _hide_ reference) and `on mouseover`
|
|
126
|
+
(a real handler for a neighbouring event). No parser diagnostic can catch
|
|
127
|
+
these; they are exactly what IR-vs-intent review (and Arc 4's equivalence
|
|
128
|
+
checking) is for.
|
|
129
|
+
|
|
130
|
+
### Arc 3b, second diagnostic (inert-shape gate)
|
|
131
|
+
|
|
132
|
+
The five remaining real gaps all parsed fully-consumed at confidence 1.0 but
|
|
133
|
+
left distinctive fingerprints in the IR: `all .todo` becomes a property access
|
|
134
|
+
on the undefined identifier `all`; `has class` mis-tokenizes the predicate as a
|
|
135
|
+
class selector (`#box .has class .danger`) so the condition is always falsy;
|
|
136
|
+
`<body/>` survives as a raw non-CSS selector that `querySelector` rejects; and
|
|
137
|
+
`the text of #el` becomes an invisible expando write. Gate 4 of the
|
|
138
|
+
compilation-service validation pipeline (`validation/inert-shapes.ts`) matches
|
|
139
|
+
those fingerprints and warns (`INERT_QUANTIFIER_TARGET`,
|
|
140
|
+
`HALF_PARSED_CONDITION`, `UNSUPPORTED_QUERY_LITERAL`, `INERT_PROPERTY_WRITE`)
|
|
141
|
+
— warnings only, still no parser change.
|
|
142
|
+
|
|
143
|
+
| | pre-3b | after slice 1 | after slice 2 |
|
|
144
|
+
| ------------------------------------- | ----------- | ------------- | ---------------------- |
|
|
145
|
+
| wrong but **warned** (loop can react) | 1/37 | 12/37 | **17/37** (16 ⚠ + 1 ✗) |
|
|
146
|
+
| wrong and **silent** | 18/37 (49%) | 7/37 (19%) | **2/37 (5%)** |
|
|
147
|
+
|
|
148
|
+
The two remaining ☠ rows are the valid-code-different-intent pair
|
|
149
|
+
(`add .hidden to #menu`, `on mouseover`) — by design not diagnosable, and the
|
|
150
|
+
standing case for IR-vs-intent review and Arc 4's behavioral equivalence. **The
|
|
151
|
+
parser-gap silent band is now zero.**
|
|
152
|
+
|
|
153
|
+
Bands are computed by `harness.bandOf` — one function shared by the probe, the
|
|
154
|
+
committed baseline, and the ratchet test, so they cannot drift apart.
|
|
155
|
+
|
|
156
|
+
## Adding a task
|
|
157
|
+
|
|
158
|
+
Append to [`tasks.ts`](./tasks.ts): a prompt with no hyperscript in it, a
|
|
159
|
+
reference, and the fixture markup it needs. Then run `verify-references` — a
|
|
160
|
+
reference that does not parse, or produces no DOM effect, is rejected (same
|
|
161
|
+
eligibility bar as R2's execution subset), because scoring against an empty
|
|
162
|
+
signature would make wrong answers look right.
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Guards for the agent-loop benchmark.
|
|
3
|
+
*
|
|
4
|
+
* Two things rot silently and would make the benchmark lie:
|
|
5
|
+
*
|
|
6
|
+
* 1. **A reference that stops working.** Behavior correctness is defined as
|
|
7
|
+
* "same effect signature as the reference", so a reference that stops
|
|
8
|
+
* parsing — or starts producing NO effects — makes every candidate for
|
|
9
|
+
* that task score wrong (or, worse, makes an empty-effect candidate score
|
|
10
|
+
* right). Same eligibility bar as R2's execution subset.
|
|
11
|
+
*
|
|
12
|
+
* 2. **The silent band drifting.** `baselines/agent-bench-phrasings.json`
|
|
13
|
+
* records, per plausible phrasing, whether it is correct / rejected /
|
|
14
|
+
* silently wrong. This ratchets BOTH directions at tolerance 0: a
|
|
15
|
+
* regression (correct → silent) fails, and so does an improvement that
|
|
16
|
+
* wasn't re-baselined. The improvement direction matters as much here as
|
|
17
|
+
* in the R4 allowlist — a fixed parser gap that nobody re-records leaves
|
|
18
|
+
* the docs quoting a stale number, and the "half of plausible phrasings
|
|
19
|
+
* misbehave" claim is load-bearing for the roadmap.
|
|
20
|
+
*
|
|
21
|
+
* Deterministic and generator-free: no LLM runs here, so this IS a legitimate
|
|
22
|
+
* CI gate (the A/B run in README.md, which needs a generator, deliberately is
|
|
23
|
+
* not). Full sweep measures ~6s.
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
import { describe, it, expect, beforeAll } from 'vitest';
|
|
27
|
+
import { readFileSync } from 'node:fs';
|
|
28
|
+
import path from 'node:path';
|
|
29
|
+
import { fileURLToPath } from 'node:url';
|
|
30
|
+
import { TASKS, taskById } from './tasks.js';
|
|
31
|
+
import { VARIANTS } from './variants.js';
|
|
32
|
+
import {
|
|
33
|
+
bandOf,
|
|
34
|
+
executeCandidate,
|
|
35
|
+
initialize,
|
|
36
|
+
scoreCandidate,
|
|
37
|
+
validateCandidate,
|
|
38
|
+
} from './harness.js';
|
|
39
|
+
|
|
40
|
+
const BASELINE_PATH = path.resolve(
|
|
41
|
+
path.dirname(fileURLToPath(import.meta.url)),
|
|
42
|
+
'../../baselines/agent-bench-phrasings.json'
|
|
43
|
+
);
|
|
44
|
+
|
|
45
|
+
interface Baseline {
|
|
46
|
+
totals: Record<string, number>;
|
|
47
|
+
phrasings: Array<{ taskId: string; code: string; band: string }>;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
describe('agent-bench: task references', () => {
|
|
51
|
+
beforeAll(async () => {
|
|
52
|
+
await initialize();
|
|
53
|
+
}, 60_000);
|
|
54
|
+
|
|
55
|
+
it('every task has a unique id', () => {
|
|
56
|
+
const ids = TASKS.map(t => t.id);
|
|
57
|
+
expect(new Set(ids).size).toBe(ids.length);
|
|
58
|
+
});
|
|
59
|
+
|
|
60
|
+
it('no prompt leaks hyperscript syntax to the generator', () => {
|
|
61
|
+
// A prompt containing the answer measures nothing. Selector/sigil
|
|
62
|
+
// characters and command keywords in the imperative position are the tells.
|
|
63
|
+
const leaks = TASKS.filter(t => /\bon click\b|=>|\s_=|@[a-z-]+\s|\*[a-z-]+\s/i.test(t.prompt));
|
|
64
|
+
expect(leaks.map(t => t.id)).toEqual([]);
|
|
65
|
+
});
|
|
66
|
+
|
|
67
|
+
it.each(TASKS.map(t => [t.id, t] as const))(
|
|
68
|
+
'reference for %s parses and produces a usable effect signature',
|
|
69
|
+
async (_id, task) => {
|
|
70
|
+
const validation = await validateCandidate(task.reference);
|
|
71
|
+
expect(validation.ok, `reference does not parse: ${task.reference}`).toBe(true);
|
|
72
|
+
const { effects, error } = await executeCandidate(task, task.reference);
|
|
73
|
+
expect(error, `reference errored: ${error}`).toBeUndefined();
|
|
74
|
+
expect(effects.length, `reference has no DOM effect: ${task.reference}`).toBeGreaterThan(0);
|
|
75
|
+
},
|
|
76
|
+
30_000
|
|
77
|
+
);
|
|
78
|
+
});
|
|
79
|
+
|
|
80
|
+
describe('agent-bench: plausible-phrasing ratchet', () => {
|
|
81
|
+
let baseline: Baseline;
|
|
82
|
+
|
|
83
|
+
beforeAll(async () => {
|
|
84
|
+
await initialize();
|
|
85
|
+
baseline = JSON.parse(readFileSync(BASELINE_PATH, 'utf8')) as Baseline;
|
|
86
|
+
}, 60_000);
|
|
87
|
+
|
|
88
|
+
it('baseline covers exactly the current variant set', () => {
|
|
89
|
+
const recorded = baseline.phrasings.map(p => p.code).sort();
|
|
90
|
+
const current = VARIANTS.map(v => v.code).sort();
|
|
91
|
+
expect(
|
|
92
|
+
recorded,
|
|
93
|
+
'variants changed without regenerating: tsx src/agent-bench/cli.ts probe-variants --json > baselines/agent-bench-phrasings.json'
|
|
94
|
+
).toEqual(current);
|
|
95
|
+
});
|
|
96
|
+
|
|
97
|
+
it('every phrasing still lands in its recorded band (both directions, tolerance 0)', async () => {
|
|
98
|
+
const drift: string[] = [];
|
|
99
|
+
for (const entry of baseline.phrasings) {
|
|
100
|
+
const task = taskById(entry.taskId);
|
|
101
|
+
expect(task, `baseline names unknown task ${entry.taskId}`).toBeDefined();
|
|
102
|
+
const s = await scoreCandidate(task!, entry.code);
|
|
103
|
+
const band = bandOf(s);
|
|
104
|
+
if (band !== entry.band) drift.push(`${entry.code}\n ${entry.band} → ${band}`);
|
|
105
|
+
}
|
|
106
|
+
expect(
|
|
107
|
+
drift,
|
|
108
|
+
'phrasing behavior drifted. If this is an intentional improvement, regenerate:\n' +
|
|
109
|
+
' tsx src/agent-bench/cli.ts probe-variants --json > baselines/agent-bench-phrasings.json'
|
|
110
|
+
).toEqual([]);
|
|
111
|
+
}, 120_000);
|
|
112
|
+
});
|
|
@@ -0,0 +1,315 @@
|
|
|
1
|
+
#!/usr/bin/env tsx
|
|
2
|
+
/**
|
|
3
|
+
* Agent-loop benchmark CLI.
|
|
4
|
+
*
|
|
5
|
+
* verify-references every reference parses and has a usable signature
|
|
6
|
+
* list [--json] the task prompts, for a generator to answer
|
|
7
|
+
* feedback --task <id> --code <src> what the loop hands back for one candidate
|
|
8
|
+
* score --run <file> score a run file; prints per-condition rates + delta
|
|
9
|
+
*
|
|
10
|
+
* The generator is an AGENT, not this script: the loop being measured is the one
|
|
11
|
+
* a real integration runs, so simulating it in-process would measure a
|
|
12
|
+
* simulation. `list` emits the prompts, the agent answers them, `score` grades
|
|
13
|
+
* the answers. See README.md for the protocol.
|
|
14
|
+
*
|
|
15
|
+
* `feedback` deliberately returns ONLY what the MCP loop returns — diagnostics
|
|
16
|
+
* and the parsed IR. It never reveals the reference or whether the candidate
|
|
17
|
+
* behaves correctly. An agent iterating with `feedback` therefore has exactly
|
|
18
|
+
* the information the real loop gives it; leaking the behavior verdict would
|
|
19
|
+
* turn the loop condition into an oracle and the headline delta into fiction.
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
import { readFileSync } from 'node:fs';
|
|
23
|
+
import { TASKS, taskById } from './tasks.js';
|
|
24
|
+
import { VARIANTS } from './variants.js';
|
|
25
|
+
import {
|
|
26
|
+
bandOf,
|
|
27
|
+
executeCandidate,
|
|
28
|
+
scoreCandidate,
|
|
29
|
+
scoreCondition,
|
|
30
|
+
validateCandidate,
|
|
31
|
+
type Band,
|
|
32
|
+
type ConditionScore,
|
|
33
|
+
} from './harness.js';
|
|
34
|
+
|
|
35
|
+
interface RunFile {
|
|
36
|
+
generator?: Record<string, unknown>;
|
|
37
|
+
conditions: Record<string, Record<string, string>>;
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
function arg(name: string): string | undefined {
|
|
41
|
+
const i = process.argv.indexOf(`--${name}`);
|
|
42
|
+
return i === -1 ? undefined : process.argv[i + 1];
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
const pct = (n: number): string => `${(n * 100).toFixed(0)}%`;
|
|
46
|
+
|
|
47
|
+
// =============================================================================
|
|
48
|
+
// verify-references
|
|
49
|
+
// =============================================================================
|
|
50
|
+
|
|
51
|
+
async function verifyReferences(): Promise<number> {
|
|
52
|
+
let bad = 0;
|
|
53
|
+
const seen = new Map<string, string>();
|
|
54
|
+
for (const task of TASKS) {
|
|
55
|
+
const validation = await validateCandidate(task.reference);
|
|
56
|
+
const { effects, error } = await executeCandidate(task, task.reference);
|
|
57
|
+
const problems: string[] = [];
|
|
58
|
+
if (!validation.ok) problems.push('reference does not parse');
|
|
59
|
+
if (effects.length === 0) problems.push('reference produces NO effect signature');
|
|
60
|
+
if (error) problems.push(`error: ${error}`);
|
|
61
|
+
// Two tasks with identical signatures would let a candidate score right on
|
|
62
|
+
// the wrong task; not fatal (fixtures legitimately overlap) but worth saying.
|
|
63
|
+
const key = JSON.stringify(effects);
|
|
64
|
+
const twin = seen.get(key);
|
|
65
|
+
if (effects.length > 0 && twin) problems.push(`signature identical to ${twin}`);
|
|
66
|
+
if (effects.length > 0) seen.set(key, task.id);
|
|
67
|
+
|
|
68
|
+
if (problems.length > 0) {
|
|
69
|
+
bad++;
|
|
70
|
+
console.log(`✗ ${task.id}: ${problems.join('; ')}`);
|
|
71
|
+
console.log(` ${task.reference}`);
|
|
72
|
+
} else {
|
|
73
|
+
console.log(`✓ ${task.id} ${effects.length} effect(s)`);
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
console.log(
|
|
77
|
+
bad === 0
|
|
78
|
+
? `\nAll ${TASKS.length} references usable.`
|
|
79
|
+
: `\n${bad}/${TASKS.length} references UNUSABLE — fix before scoring.`
|
|
80
|
+
);
|
|
81
|
+
return bad === 0 ? 0 : 1;
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
// =============================================================================
|
|
85
|
+
// list / feedback
|
|
86
|
+
// =============================================================================
|
|
87
|
+
|
|
88
|
+
function list(): number {
|
|
89
|
+
if (process.argv.includes('--json')) {
|
|
90
|
+
console.log(
|
|
91
|
+
JSON.stringify(
|
|
92
|
+
TASKS.map(t => ({ id: t.id, prompt: t.prompt })),
|
|
93
|
+
null,
|
|
94
|
+
2
|
|
95
|
+
)
|
|
96
|
+
);
|
|
97
|
+
return 0;
|
|
98
|
+
}
|
|
99
|
+
for (const t of TASKS) console.log(`${t.id}\n ${t.prompt}\n`);
|
|
100
|
+
return 0;
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
async function feedback(): Promise<number> {
|
|
104
|
+
const id = arg('task');
|
|
105
|
+
const code = arg('code');
|
|
106
|
+
if (!id || code === undefined) {
|
|
107
|
+
console.error('usage: feedback --task <id> --code "<hyperscript>"');
|
|
108
|
+
return 2;
|
|
109
|
+
}
|
|
110
|
+
if (!taskById(id)) {
|
|
111
|
+
console.error(`unknown task: ${id}`);
|
|
112
|
+
return 2;
|
|
113
|
+
}
|
|
114
|
+
const v = await validateCandidate(code);
|
|
115
|
+
console.log(
|
|
116
|
+
JSON.stringify(
|
|
117
|
+
{ ok: v.ok, confidence: v.confidence, parsed: v.summary, diagnostics: v.diagnostics },
|
|
118
|
+
null,
|
|
119
|
+
2
|
|
120
|
+
)
|
|
121
|
+
);
|
|
122
|
+
if (!v.ok) {
|
|
123
|
+
console.log(
|
|
124
|
+
'\nNext step: apply the diagnostics above and re-run. get_code_fixes maps ' +
|
|
125
|
+
'error codes to concrete fixes; get_command_docs lists per-command roles.'
|
|
126
|
+
);
|
|
127
|
+
}
|
|
128
|
+
return 0;
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
// =============================================================================
|
|
132
|
+
// score
|
|
133
|
+
// =============================================================================
|
|
134
|
+
|
|
135
|
+
function reportCondition(c: ConditionScore): void {
|
|
136
|
+
console.log(`\n── ${c.condition} ──`);
|
|
137
|
+
for (const s of c.scores) {
|
|
138
|
+
const mark = s.behaviorMatch ? '✓' : s.parsed ? '~' : '✗';
|
|
139
|
+
console.log(`${mark} ${s.taskId.padEnd(22)} ${s.code}`);
|
|
140
|
+
if (!s.behaviorMatch) {
|
|
141
|
+
if (!s.parsed) {
|
|
142
|
+
const first = s.validation.diagnostics.find(d => d.severity === 'error');
|
|
143
|
+
console.log(` did not parse: ${first?.message ?? 'unknown'}`);
|
|
144
|
+
} else {
|
|
145
|
+
console.log(` PARSED BUT WRONG — ${s.validation.summary ?? '?'}`);
|
|
146
|
+
console.log(` got ${JSON.stringify(s.execution.effects)}`);
|
|
147
|
+
console.log(` expected ${JSON.stringify(s.referenceEffects)}`);
|
|
148
|
+
if (s.execution.error) console.log(` error: ${s.execution.error}`);
|
|
149
|
+
}
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
for (const m of c.missing) console.log(`✗ ${m.padEnd(22)} (no candidate submitted)`);
|
|
153
|
+
console.log(
|
|
154
|
+
`\n parse rate ${pct(c.parseRate)}` +
|
|
155
|
+
`\n behavior rate ${pct(c.behaviorRate)}` +
|
|
156
|
+
`\n parsed-but-wrong ${c.silentlyWrongCount}`
|
|
157
|
+
);
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
async function score(): Promise<number> {
|
|
161
|
+
const file = arg('run');
|
|
162
|
+
if (!file) {
|
|
163
|
+
console.error('usage: score --run <file.json>');
|
|
164
|
+
return 2;
|
|
165
|
+
}
|
|
166
|
+
const run = JSON.parse(readFileSync(file, 'utf8')) as RunFile;
|
|
167
|
+
if (run.generator) console.log(`generator: ${JSON.stringify(run.generator)}`);
|
|
168
|
+
|
|
169
|
+
const results: ConditionScore[] = [];
|
|
170
|
+
for (const [condition, candidates] of Object.entries(run.conditions)) {
|
|
171
|
+
const c = await scoreCondition(condition, candidates, TASKS);
|
|
172
|
+
results.push(c);
|
|
173
|
+
reportCondition(c);
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
if (results.length >= 2) {
|
|
177
|
+
const [first, last] = [results[0]!, results[results.length - 1]!];
|
|
178
|
+
console.log(`\n══ ${first.condition} → ${last.condition} ══`);
|
|
179
|
+
console.log(` parse rate ${pct(first.parseRate)} → ${pct(last.parseRate)}`);
|
|
180
|
+
console.log(` behavior rate ${pct(first.behaviorRate)} → ${pct(last.behaviorRate)}`);
|
|
181
|
+
console.log(
|
|
182
|
+
` parsed-but-wrong ${first.silentlyWrongCount} → ${last.silentlyWrongCount} (over ${TASKS.length} tasks)`
|
|
183
|
+
);
|
|
184
|
+
}
|
|
185
|
+
return 0;
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
// =============================================================================
|
|
189
|
+
// probe-variants
|
|
190
|
+
// =============================================================================
|
|
191
|
+
|
|
192
|
+
/**
|
|
193
|
+
* Score the plausible-phrasing catalogue. Needs no generator: each row is a
|
|
194
|
+
* fixed string, so the result is a reproducible property of the parser.
|
|
195
|
+
*
|
|
196
|
+
* The headline is the SILENT band — phrasings that parse clean, carry no
|
|
197
|
+
* warning, and misbehave. Those are invisible to the validate/repair loop by
|
|
198
|
+
* construction, so they bound how much the loop can ever deliver. Arc 3b work
|
|
199
|
+
* moves rows out of it into `warned-wrong` (wrong but visible), which the loop
|
|
200
|
+
* handles.
|
|
201
|
+
*/
|
|
202
|
+
async function probeVariants(): Promise<number> {
|
|
203
|
+
const rows: Array<{
|
|
204
|
+
v: (typeof VARIANTS)[number];
|
|
205
|
+
parsed: boolean;
|
|
206
|
+
behaved: boolean;
|
|
207
|
+
band: Band;
|
|
208
|
+
detail: string;
|
|
209
|
+
}> = [];
|
|
210
|
+
|
|
211
|
+
for (const v of VARIANTS) {
|
|
212
|
+
const task = taskById(v.taskId);
|
|
213
|
+
if (!task) {
|
|
214
|
+
console.error(`unknown task in variants: ${v.taskId}`);
|
|
215
|
+
return 2;
|
|
216
|
+
}
|
|
217
|
+
const s = await scoreCandidate(task, v.code);
|
|
218
|
+
const band = bandOf(s);
|
|
219
|
+
const detail =
|
|
220
|
+
band === 'correct'
|
|
221
|
+
? 'ok'
|
|
222
|
+
: band === 'rejected'
|
|
223
|
+
? 'rejected'
|
|
224
|
+
: band === 'warned-wrong'
|
|
225
|
+
? `WARNED (${s.validation.diagnostics.find(d => d.severity !== 'info')?.code ?? '?'}) — wrong, but the loop can see it`
|
|
226
|
+
: band === 'silent-noop'
|
|
227
|
+
? 'SILENT NO-OP'
|
|
228
|
+
: `WRONG EFFECT ${JSON.stringify(s.execution.effects)}`;
|
|
229
|
+
rows.push({ v, parsed: s.parsed, behaved: s.behaviorMatch, band, detail });
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
const n = rows.length;
|
|
233
|
+
|
|
234
|
+
if (process.argv.includes('--json')) {
|
|
235
|
+
// Baseline shape: sorted by code so the file is diff-stable, and carrying
|
|
236
|
+
// only the deterministic verdict (never timings) so regeneration on another
|
|
237
|
+
// machine produces a byte-identical file. Bands come from harness.bandOf —
|
|
238
|
+
// the same function the ratchet test recomputes with.
|
|
239
|
+
console.log(
|
|
240
|
+
JSON.stringify(
|
|
241
|
+
{
|
|
242
|
+
note:
|
|
243
|
+
'Deterministic parse-vs-behavior verdicts for plausible phrasings. ' +
|
|
244
|
+
'Regenerate with: tsx src/agent-bench/cli.ts probe-variants --json',
|
|
245
|
+
totals: {
|
|
246
|
+
phrasings: n,
|
|
247
|
+
parse: rows.filter(r => r.parsed).length,
|
|
248
|
+
behaveCorrectly: rows.filter(r => r.behaved).length,
|
|
249
|
+
warnedWrong: rows.filter(r => r.band === 'warned-wrong').length,
|
|
250
|
+
silentBand: rows.filter(r => r.band.startsWith('silent')).length,
|
|
251
|
+
silentNoop: rows.filter(r => r.band === 'silent-noop').length,
|
|
252
|
+
},
|
|
253
|
+
phrasings: rows
|
|
254
|
+
.map(r => ({ taskId: r.v.taskId, code: r.v.code, band: r.band }))
|
|
255
|
+
.sort((a, b) => (a.code < b.code ? -1 : a.code > b.code ? 1 : 0)),
|
|
256
|
+
},
|
|
257
|
+
null,
|
|
258
|
+
2
|
|
259
|
+
)
|
|
260
|
+
);
|
|
261
|
+
return 0;
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
for (const r of rows) {
|
|
265
|
+
const mark = r.behaved
|
|
266
|
+
? '✓'
|
|
267
|
+
: r.band === 'warned-wrong'
|
|
268
|
+
? '⚠'
|
|
269
|
+
: r.band === 'rejected'
|
|
270
|
+
? '✗'
|
|
271
|
+
: '☠';
|
|
272
|
+
console.log(`${mark} ${r.v.taskId.padEnd(20)} ${r.v.code}`);
|
|
273
|
+
if (!r.behaved) console.log(` ${r.detail} — ${r.v.rationale}`);
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
const parsed = rows.filter(r => r.parsed).length;
|
|
277
|
+
const behaved = rows.filter(r => r.behaved).length;
|
|
278
|
+
const warned = rows.filter(r => r.band === 'warned-wrong').length;
|
|
279
|
+
const silent = rows.filter(r => r.band.startsWith('silent')).length;
|
|
280
|
+
const noop = rows.filter(r => r.band === 'silent-noop').length;
|
|
281
|
+
console.log(
|
|
282
|
+
`\n ${n} plausible phrasings` +
|
|
283
|
+
`\n parse ${parsed}/${n} (${pct(parsed / n)})` +
|
|
284
|
+
`\n behave correctly ${behaved}/${n} (${pct(behaved / n)})` +
|
|
285
|
+
`\n ⚠ wrong but WARNED (loop can react): ${warned}/${n}` +
|
|
286
|
+
`\n ☠ wrong and SILENT: ${silent}/${n} (${pct(silent / n)}) — of which ${noop} do NOTHING at all` +
|
|
287
|
+
`\n\n The ☠ band is what the validate/repair loop cannot see: no diagnostic,` +
|
|
288
|
+
`\n no error, nothing for an agent to react to.`
|
|
289
|
+
);
|
|
290
|
+
return 0;
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
// =============================================================================
|
|
294
|
+
|
|
295
|
+
const commands: Record<string, () => Promise<number> | number> = {
|
|
296
|
+
'verify-references': verifyReferences,
|
|
297
|
+
'probe-variants': probeVariants,
|
|
298
|
+
list,
|
|
299
|
+
feedback,
|
|
300
|
+
score,
|
|
301
|
+
};
|
|
302
|
+
|
|
303
|
+
async function main(): Promise<void> {
|
|
304
|
+
const cmd = process.argv[2] ?? '';
|
|
305
|
+
const run = commands[cmd];
|
|
306
|
+
if (!run) {
|
|
307
|
+
console.error(`usage: cli.ts <${Object.keys(commands).join('|')}> [options]`);
|
|
308
|
+
process.exit(2);
|
|
309
|
+
}
|
|
310
|
+
// Explicit exit: esbuild's daemon keeps the event loop alive (see CLAUDE.md),
|
|
311
|
+
// so a natural return would hang the process after the report is printed.
|
|
312
|
+
process.exit(await run());
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
void main();
|