@hyperfixi/testing-framework 2.9.4 → 2.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +21 -1
- package/package.json +7 -6
- package/src/agent-bench/README.md +162 -0
- package/src/agent-bench/agent-bench.test.ts +112 -0
- package/src/agent-bench/cli.ts +315 -0
- package/src/agent-bench/harness.ts +331 -0
- package/src/agent-bench/tasks.ts +214 -0
- package/src/agent-bench/variants.ts +241 -0
- package/src/multilingual/fidelity.ts +22 -376
- package/src/multilingual/shipped-examples-execution.test.ts +10 -5
- package/src/multilingual/shipped-examples-execution.ts +32 -20
|
@@ -0,0 +1,331 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Agent-loop benchmark — the scoring engine.
|
|
3
|
+
*
|
|
4
|
+
* Answers two questions per candidate, deliberately separately:
|
|
5
|
+
*
|
|
6
|
+
* 1. **Does it parse?** `CompilationService.validate()` — the exact call the
|
|
7
|
+
* `validate_and_compile` MCP tool makes, so the number reflects what an
|
|
8
|
+
* agent actually sees.
|
|
9
|
+
* 2. **Does it DO the right thing?** The candidate is executed in jsdom
|
|
10
|
+
* against the task's fixture and its DOM effect signature compared to the
|
|
11
|
+
* reference's, byte for byte.
|
|
12
|
+
*
|
|
13
|
+
* Keeping them separate is the point of the benchmark rather than an
|
|
14
|
+
* implementation detail: hyperscript's parser degrades instead of failing, so a
|
|
15
|
+
* candidate can parse at confidence 1.0 with zero diagnostics and still target
|
|
16
|
+
* the wrong element (a dropped `on` silently rebinds the destination to `me`).
|
|
17
|
+
* A single "success rate" would hide exactly the failure mode the loop needs to
|
|
18
|
+
* catch, so `parseRate` and `behaviorRate` are reported side by side and their
|
|
19
|
+
* GAP is a headline number in its own right.
|
|
20
|
+
*
|
|
21
|
+
* Determinism: fresh JSDOM + fresh Runtime per execution, no network, no timers
|
|
22
|
+
* beyond a fixed settle window. Effect-signature primitives are imported from
|
|
23
|
+
* ../multilingual/effect-signature so this harness and the R2 ratchet can never
|
|
24
|
+
* disagree about what a DOM effect is.
|
|
25
|
+
*/
|
|
26
|
+
|
|
27
|
+
import { JSDOM } from 'jsdom';
|
|
28
|
+
import { parseSemantic, buildAST } from '@lokascript/semantic';
|
|
29
|
+
import { snapshot, diffSnapshots } from '../multilingual/effect-signature.js';
|
|
30
|
+
import { SHARED_FIXTURE, type BenchTask } from './tasks.js';
|
|
31
|
+
|
|
32
|
+
/** Settle window for the dispatched handler. No task waits/fetches. */
|
|
33
|
+
const SETTLE_MS = 20;
|
|
34
|
+
/** Per-execution hard timeout — a hung candidate must not hang the run. */
|
|
35
|
+
const EXECUTION_TIMEOUT_MS = 5000;
|
|
36
|
+
|
|
37
|
+
export interface Diagnostic {
|
|
38
|
+
severity?: string;
|
|
39
|
+
code?: string;
|
|
40
|
+
message?: string;
|
|
41
|
+
suggestion?: string;
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
export interface ValidationOutcome {
|
|
45
|
+
ok: boolean;
|
|
46
|
+
confidence?: number | undefined;
|
|
47
|
+
diagnostics: Diagnostic[];
|
|
48
|
+
/** Flattened `action.role=value` view of the parse, for eyeballing intent drift. */
|
|
49
|
+
summary?: string | undefined;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
export interface ExecutionOutcome {
|
|
53
|
+
effects: string[];
|
|
54
|
+
error?: string | undefined;
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
export interface TaskScore {
|
|
58
|
+
taskId: string;
|
|
59
|
+
code: string;
|
|
60
|
+
/** CompilationService said ok — what `validate_and_compile` reports. */
|
|
61
|
+
parsed: boolean;
|
|
62
|
+
/** Effect signature identical to the reference's. */
|
|
63
|
+
behaviorMatch: boolean;
|
|
64
|
+
/** parsed && !behaviorMatch && no visible diagnostic — the band the loop cannot see. */
|
|
65
|
+
silentlyWrong: boolean;
|
|
66
|
+
validation: ValidationOutcome;
|
|
67
|
+
execution: ExecutionOutcome;
|
|
68
|
+
referenceEffects: string[];
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* The five outcome bands, ordered best → worst. One function, used by the
|
|
73
|
+
* probe, the JSON baseline, and the ratchet test alike — a fork between them
|
|
74
|
+
* would let the ratchet pass against a baseline the probe can no longer
|
|
75
|
+
* produce.
|
|
76
|
+
*
|
|
77
|
+
* `warned-*`: wrong behavior, but validation carried a warning/error-severity
|
|
78
|
+
* diagnostic — VISIBLE to the loop, which can react (arc 3b's unconsumed-input
|
|
79
|
+
* propagation moves rows from silent-* to here). `silent-*`: wrong behavior
|
|
80
|
+
* and nothing to react to; the band that bounds the loop's ceiling.
|
|
81
|
+
*/
|
|
82
|
+
export type Band = 'correct' | 'rejected' | 'warned-wrong' | 'silent-wrong' | 'silent-noop';
|
|
83
|
+
|
|
84
|
+
export function bandOf(score: TaskScore): Band {
|
|
85
|
+
if (score.behaviorMatch) return 'correct';
|
|
86
|
+
if (!score.parsed) return 'rejected';
|
|
87
|
+
const visible = score.validation.diagnostics.some(
|
|
88
|
+
d => d.severity === 'warning' || d.severity === 'error'
|
|
89
|
+
);
|
|
90
|
+
if (visible) return 'warned-wrong';
|
|
91
|
+
return score.execution.effects.length === 0 ? 'silent-noop' : 'silent-wrong';
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
export interface ConditionScore {
|
|
95
|
+
condition: string;
|
|
96
|
+
scores: TaskScore[];
|
|
97
|
+
parseRate: number;
|
|
98
|
+
behaviorRate: number;
|
|
99
|
+
/** Tasks that parsed but behaved differently from the reference. */
|
|
100
|
+
silentlyWrongCount: number;
|
|
101
|
+
/** Tasks with no candidate submitted for this condition. */
|
|
102
|
+
missing: string[];
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
// =============================================================================
|
|
106
|
+
// Validation (the agent-visible surface)
|
|
107
|
+
// =============================================================================
|
|
108
|
+
|
|
109
|
+
let servicePromise: Promise<any> | null = null;
|
|
110
|
+
|
|
111
|
+
async function getService(): Promise<any> {
|
|
112
|
+
if (!servicePromise) {
|
|
113
|
+
servicePromise = import('@lokascript/compilation-service').then(m =>
|
|
114
|
+
m.CompilationService.create()
|
|
115
|
+
);
|
|
116
|
+
}
|
|
117
|
+
return servicePromise;
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/** Render a parse as `action(role=value, …)` so intent drift is readable. */
|
|
121
|
+
function summarize(semantic: unknown): string | undefined {
|
|
122
|
+
const node = semantic as { action?: string; roles?: Record<string, { value?: unknown }> };
|
|
123
|
+
if (!node?.action) return undefined;
|
|
124
|
+
const roles = Object.entries(node.roles ?? {})
|
|
125
|
+
.map(([k, v]) => `${k}=${String(v?.value ?? '')}`)
|
|
126
|
+
.sort()
|
|
127
|
+
.join(', ');
|
|
128
|
+
return `${node.action}(${roles})`;
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
export async function validateCandidate(code: string): Promise<ValidationOutcome> {
|
|
132
|
+
try {
|
|
133
|
+
const service = await getService();
|
|
134
|
+
const result = service.validate({ code, language: 'en' });
|
|
135
|
+
return {
|
|
136
|
+
ok: Boolean(result?.ok),
|
|
137
|
+
confidence: result?.confidence,
|
|
138
|
+
diagnostics: (result?.diagnostics ?? []) as Diagnostic[],
|
|
139
|
+
summary: summarize(result?.semantic),
|
|
140
|
+
};
|
|
141
|
+
} catch (e: unknown) {
|
|
142
|
+
return {
|
|
143
|
+
ok: false,
|
|
144
|
+
diagnostics: [
|
|
145
|
+
{ severity: 'error', code: 'HARNESS', message: e instanceof Error ? e.message : String(e) },
|
|
146
|
+
],
|
|
147
|
+
};
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
// =============================================================================
|
|
152
|
+
// Execution (what it actually does)
|
|
153
|
+
// =============================================================================
|
|
154
|
+
|
|
155
|
+
function buildDocument(task: BenchTask): JSDOM {
|
|
156
|
+
return new JSDOM(`<!DOCTYPE html><html><body>${SHARED_FIXTURE}${task.fixture}</body></html>`);
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
function installGlobals(dom: JSDOM): void {
|
|
160
|
+
const g = globalThis as Record<string, unknown>;
|
|
161
|
+
g.window = dom.window;
|
|
162
|
+
g.document = dom.window.document;
|
|
163
|
+
g.Event = dom.window.Event;
|
|
164
|
+
g.CustomEvent = dom.window.CustomEvent;
|
|
165
|
+
g.HTMLElement = dom.window.HTMLElement;
|
|
166
|
+
g.Element = dom.window.Element;
|
|
167
|
+
g.Node = dom.window.Node;
|
|
168
|
+
g.MutationObserver = dom.window.MutationObserver;
|
|
169
|
+
g.getComputedStyle = dom.window.getComputedStyle;
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
let core: {
|
|
173
|
+
Runtime: new () => { execute(ast: unknown, ctx: unknown): Promise<unknown> };
|
|
174
|
+
createContext: (el: HTMLElement) => unknown;
|
|
175
|
+
} | null = null;
|
|
176
|
+
|
|
177
|
+
let listenerErrors: string[] = [];
|
|
178
|
+
let trapInstalled = false;
|
|
179
|
+
|
|
180
|
+
/** Bootstraps jsdom globals BEFORE loading core (its dist evaluates `Element`). */
|
|
181
|
+
export async function initialize(): Promise<void> {
|
|
182
|
+
if (core) return;
|
|
183
|
+
installGlobals(new JSDOM('<!DOCTYPE html><html><body></body></html>'));
|
|
184
|
+
const mod = (await import('@hyperfixi/core')) as unknown as {
|
|
185
|
+
Runtime: new () => { execute(ast: unknown, ctx: unknown): Promise<unknown> };
|
|
186
|
+
createContext: (el: HTMLElement) => unknown;
|
|
187
|
+
};
|
|
188
|
+
core = { Runtime: mod.Runtime, createContext: mod.createContext };
|
|
189
|
+
if (!trapInstalled) {
|
|
190
|
+
// Handler bodies are async, so a throw inside one surfaces as an unhandled
|
|
191
|
+
// rejection rather than propagating to our await.
|
|
192
|
+
process.on('unhandledRejection', (reason: unknown) => {
|
|
193
|
+
listenerErrors.push(reason instanceof Error ? reason.message : String(reason));
|
|
194
|
+
});
|
|
195
|
+
trapInstalled = true;
|
|
196
|
+
}
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
async function executeInner(task: BenchTask, code: string): Promise<ExecutionOutcome> {
|
|
200
|
+
await initialize();
|
|
201
|
+
const dom = buildDocument(task);
|
|
202
|
+
installGlobals(dom);
|
|
203
|
+
const document = dom.window.document;
|
|
204
|
+
task.setup?.(document);
|
|
205
|
+
const btn = document.getElementById('btn')!;
|
|
206
|
+
listenerErrors = [];
|
|
207
|
+
|
|
208
|
+
// The runtime logs every failing command; across a full run that is noise.
|
|
209
|
+
const saved = { log: console.log, warn: console.warn, error: console.error };
|
|
210
|
+
console.log = console.warn = console.error = () => {};
|
|
211
|
+
try {
|
|
212
|
+
const parsed = parseSemantic(code, 'en') as { node?: unknown; confidence?: number };
|
|
213
|
+
if (!parsed.node || (parsed.confidence ?? 0) < 0.5) {
|
|
214
|
+
return { effects: [], error: `parse failed (confidence ${parsed.confidence ?? 0})` };
|
|
215
|
+
}
|
|
216
|
+
const built = buildAST(parsed.node as never) as { ast?: unknown };
|
|
217
|
+
if (!built.ast) return { effects: [], error: 'buildAST returned no AST' };
|
|
218
|
+
|
|
219
|
+
const runtime = new core!.Runtime();
|
|
220
|
+
const ctx = core!.createContext(btn as unknown as HTMLElement);
|
|
221
|
+
await runtime.execute(built.ast, ctx);
|
|
222
|
+
|
|
223
|
+
const before = snapshot(document);
|
|
224
|
+
const trig = task.trigger ?? { event: 'click' };
|
|
225
|
+
const target = trig.selector ? document.querySelector(trig.selector) : btn;
|
|
226
|
+
if (!target) return { effects: [], error: `trigger selector ${trig.selector} matched nothing` };
|
|
227
|
+
const event =
|
|
228
|
+
trig.detail !== undefined
|
|
229
|
+
? new dom.window.CustomEvent(trig.event, { bubbles: true, detail: trig.detail })
|
|
230
|
+
: new dom.window.Event(trig.event, { bubbles: true });
|
|
231
|
+
target.dispatchEvent(event);
|
|
232
|
+
await new Promise(r => setTimeout(r, SETTLE_MS));
|
|
233
|
+
|
|
234
|
+
const effects = diffSnapshots(before, snapshot(document));
|
|
235
|
+
return listenerErrors.length > 0
|
|
236
|
+
? { effects, error: `runtime: ${listenerErrors.join('; ')}` }
|
|
237
|
+
: { effects };
|
|
238
|
+
} catch (e: unknown) {
|
|
239
|
+
return { effects: [], error: e instanceof Error ? e.message : String(e) };
|
|
240
|
+
} finally {
|
|
241
|
+
console.log = saved.log;
|
|
242
|
+
console.warn = saved.warn;
|
|
243
|
+
console.error = saved.error;
|
|
244
|
+
dom.window.close();
|
|
245
|
+
}
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
/** Execute one candidate against a task's fixture. Never throws. */
|
|
249
|
+
export async function executeCandidate(task: BenchTask, code: string): Promise<ExecutionOutcome> {
|
|
250
|
+
return Promise.race([
|
|
251
|
+
executeInner(task, code),
|
|
252
|
+
new Promise<ExecutionOutcome>(resolve =>
|
|
253
|
+
setTimeout(
|
|
254
|
+
() => resolve({ effects: [], error: `execution timed out (${EXECUTION_TIMEOUT_MS}ms)` }),
|
|
255
|
+
EXECUTION_TIMEOUT_MS
|
|
256
|
+
)
|
|
257
|
+
),
|
|
258
|
+
]);
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
// =============================================================================
|
|
262
|
+
// Scoring
|
|
263
|
+
// =============================================================================
|
|
264
|
+
|
|
265
|
+
function sameEffects(a: string[], b: string[]): boolean {
|
|
266
|
+
return a.length === b.length && a.every((v, i) => v === b[i]);
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
/**
|
|
270
|
+
* Reference signatures are pure functions of (task, runtime) and every score
|
|
271
|
+
* needs one, so scoring N candidates for a task would otherwise execute its
|
|
272
|
+
* reference N times. Memoized per process; `initialize()` state is per-process
|
|
273
|
+
* too, so the cache can never outlive the runtime it describes.
|
|
274
|
+
*/
|
|
275
|
+
const referenceCache = new Map<string, string[]>();
|
|
276
|
+
|
|
277
|
+
export async function referenceEffectsFor(task: BenchTask): Promise<string[]> {
|
|
278
|
+
const hit = referenceCache.get(task.id);
|
|
279
|
+
if (hit) return hit;
|
|
280
|
+
const effects = (await executeCandidate(task, task.reference)).effects;
|
|
281
|
+
referenceCache.set(task.id, effects);
|
|
282
|
+
return effects;
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
export async function scoreCandidate(task: BenchTask, code: string): Promise<TaskScore> {
|
|
286
|
+
const referenceEffects = await referenceEffectsFor(task);
|
|
287
|
+
const validation = await validateCandidate(code);
|
|
288
|
+
const execution = await executeCandidate(task, code);
|
|
289
|
+
const behaviorMatch =
|
|
290
|
+
referenceEffects.length > 0 && sameEffects(execution.effects, referenceEffects);
|
|
291
|
+
const score: TaskScore = {
|
|
292
|
+
taskId: task.id,
|
|
293
|
+
code,
|
|
294
|
+
parsed: validation.ok,
|
|
295
|
+
behaviorMatch,
|
|
296
|
+
silentlyWrong: false,
|
|
297
|
+
validation,
|
|
298
|
+
execution,
|
|
299
|
+
referenceEffects,
|
|
300
|
+
};
|
|
301
|
+
score.silentlyWrong = bandOf(score).startsWith('silent');
|
|
302
|
+
return score;
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
export async function scoreCondition(
|
|
306
|
+
condition: string,
|
|
307
|
+
candidates: Record<string, string>,
|
|
308
|
+
tasks: readonly BenchTask[]
|
|
309
|
+
): Promise<ConditionScore> {
|
|
310
|
+
const scores: TaskScore[] = [];
|
|
311
|
+
const missing: string[] = [];
|
|
312
|
+
for (const task of tasks) {
|
|
313
|
+
const code = candidates[task.id];
|
|
314
|
+
if (code === undefined) {
|
|
315
|
+
missing.push(task.id);
|
|
316
|
+
continue;
|
|
317
|
+
}
|
|
318
|
+
scores.push(await scoreCandidate(task, code));
|
|
319
|
+
}
|
|
320
|
+
// Missing candidates count as failures in the denominator: a condition that
|
|
321
|
+
// simply declines to answer must not out-score one that tries and misses.
|
|
322
|
+
const denom = tasks.length;
|
|
323
|
+
return {
|
|
324
|
+
condition,
|
|
325
|
+
scores,
|
|
326
|
+
parseRate: scores.filter(s => s.parsed).length / denom,
|
|
327
|
+
behaviorRate: scores.filter(s => s.behaviorMatch).length / denom,
|
|
328
|
+
silentlyWrongCount: scores.filter(s => s.silentlyWrong).length,
|
|
329
|
+
missing,
|
|
330
|
+
};
|
|
331
|
+
}
|
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Agent-loop benchmark — the task corpus.
|
|
3
|
+
*
|
|
4
|
+
* Twenty natural-language UI requests, phrased the way a user would put them to
|
|
5
|
+
* an agent (no hyperscript syntax leaks into a prompt — that would hand the
|
|
6
|
+
* generator the answer). Each carries a reference implementation whose jsdom
|
|
7
|
+
* effect signature defines "behaviorally correct" for that task.
|
|
8
|
+
*
|
|
9
|
+
* Eligibility bar, same as R2's execution subset: a task is only usable if its
|
|
10
|
+
* reference PARSES and produces a NON-EMPTY effect signature in the current
|
|
11
|
+
* runtime. `cli.ts verify-references` enforces both, so a task whose reference
|
|
12
|
+
* rots (or whose fixture stops matching) fails loudly rather than scoring every
|
|
13
|
+
* candidate against an empty signature — which would make wrong answers look
|
|
14
|
+
* right.
|
|
15
|
+
*
|
|
16
|
+
* Seeded from the concepts in the gallery examples and the patterns corpus, but
|
|
17
|
+
* deliberately NOT copied from them verbatim: the corpus text is what the
|
|
18
|
+
* pattern database (and therefore `search_patterns`) already contains, so
|
|
19
|
+
* reusing it would measure recall of the corpus rather than generation.
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
/** Trigger to dispatch after the handler is installed. */
|
|
23
|
+
export interface BenchTrigger {
|
|
24
|
+
event: string;
|
|
25
|
+
/** Element to dispatch on (default `#btn`). */
|
|
26
|
+
selector?: string;
|
|
27
|
+
/** CustomEvent detail; when set, a CustomEvent is dispatched instead of Event. */
|
|
28
|
+
detail?: unknown;
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
export interface BenchTask {
|
|
32
|
+
id: string;
|
|
33
|
+
/** The request as a user would phrase it. No hyperscript in here. */
|
|
34
|
+
prompt: string;
|
|
35
|
+
/** Known-good hyperscript satisfying the prompt; defines the target behavior. */
|
|
36
|
+
reference: string;
|
|
37
|
+
/** Markup appended inside <body>, after the shared button. */
|
|
38
|
+
fixture: string;
|
|
39
|
+
/** Fixture preconditions applied before the handler is installed. */
|
|
40
|
+
setup?: (doc: Document) => void;
|
|
41
|
+
/** Defaults to a bubbling `click` on `#btn`. */
|
|
42
|
+
trigger?: BenchTrigger;
|
|
43
|
+
/** What the task probes — used to group the report. */
|
|
44
|
+
tags: readonly string[];
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/** Present in every fixture; the default trigger target. */
|
|
48
|
+
export const SHARED_FIXTURE = '<div class="card"><button id="btn">Click</button></div>';
|
|
49
|
+
|
|
50
|
+
export const TASKS: readonly BenchTask[] = [
|
|
51
|
+
{
|
|
52
|
+
id: 'toggle-self-class',
|
|
53
|
+
prompt: 'When the button is clicked, toggle the CSS class "active" on the button itself.',
|
|
54
|
+
reference: 'on click toggle .active on me',
|
|
55
|
+
fixture: '',
|
|
56
|
+
tags: ['toggle', 'self-reference'],
|
|
57
|
+
},
|
|
58
|
+
{
|
|
59
|
+
id: 'toggle-other-class',
|
|
60
|
+
prompt:
|
|
61
|
+
'When the button is clicked, toggle the CSS class "open" on the element whose id is "panel".',
|
|
62
|
+
reference: 'on click toggle .open on #panel',
|
|
63
|
+
fixture: '<div id="panel">panel</div>',
|
|
64
|
+
tags: ['toggle', 'destination'],
|
|
65
|
+
},
|
|
66
|
+
{
|
|
67
|
+
id: 'add-class-other',
|
|
68
|
+
prompt:
|
|
69
|
+
'When the button is clicked, add the CSS class "highlight" to the element whose id is "item".',
|
|
70
|
+
reference: 'on click add .highlight to #item',
|
|
71
|
+
fixture: '<div id="item">item</div>',
|
|
72
|
+
tags: ['add', 'destination'],
|
|
73
|
+
},
|
|
74
|
+
{
|
|
75
|
+
id: 'remove-class-all',
|
|
76
|
+
prompt:
|
|
77
|
+
'When the button is clicked, remove the CSS class "active" from every element that has the class "row".',
|
|
78
|
+
reference: 'on click remove .active from .row',
|
|
79
|
+
fixture: '<div class="row active">a</div><div class="row active">b</div>',
|
|
80
|
+
tags: ['remove', 'multi-target'],
|
|
81
|
+
},
|
|
82
|
+
{
|
|
83
|
+
id: 'put-text',
|
|
84
|
+
prompt:
|
|
85
|
+
'When the button is clicked, replace the contents of the element with id "output" with the text Saved.',
|
|
86
|
+
reference: 'on click put "Saved" into #output',
|
|
87
|
+
fixture: '<div id="output">old</div>',
|
|
88
|
+
tags: ['put', 'literal'],
|
|
89
|
+
},
|
|
90
|
+
{
|
|
91
|
+
id: 'hide-element',
|
|
92
|
+
prompt: 'When the button is clicked, hide the element whose id is "menu".',
|
|
93
|
+
reference: 'on click hide #menu',
|
|
94
|
+
fixture: '<div id="menu">menu</div>',
|
|
95
|
+
tags: ['hide'],
|
|
96
|
+
},
|
|
97
|
+
{
|
|
98
|
+
id: 'show-element',
|
|
99
|
+
prompt: 'When the button is clicked, make the hidden element with id "modal" visible again.',
|
|
100
|
+
reference: 'on click show #modal',
|
|
101
|
+
fixture: '<div id="modal" style="display: none">modal</div>',
|
|
102
|
+
tags: ['show'],
|
|
103
|
+
},
|
|
104
|
+
{
|
|
105
|
+
id: 'two-commands',
|
|
106
|
+
prompt:
|
|
107
|
+
'When the button is clicked, add the class "busy" to the button and also put the text Loading into the element with id "output".',
|
|
108
|
+
reference: 'on click add .busy to me then put "Loading" into #output',
|
|
109
|
+
fixture: '<div id="output">idle</div>',
|
|
110
|
+
tags: ['sequence', 'multi-command'],
|
|
111
|
+
},
|
|
112
|
+
{
|
|
113
|
+
id: 'closest-ancestor',
|
|
114
|
+
prompt:
|
|
115
|
+
'When the button is clicked, add the class "selected" to the nearest enclosing element that has the class "card".',
|
|
116
|
+
reference: 'on click add .selected to closest .card',
|
|
117
|
+
fixture: '',
|
|
118
|
+
tags: ['add', 'positional', 'closest'],
|
|
119
|
+
},
|
|
120
|
+
{
|
|
121
|
+
id: 'remove-element',
|
|
122
|
+
prompt: 'When the button is clicked, delete the element with id "item" from the page.',
|
|
123
|
+
reference: 'on click remove #item',
|
|
124
|
+
fixture: '<div id="item">doomed</div>',
|
|
125
|
+
tags: ['remove', 'element-removal'],
|
|
126
|
+
},
|
|
127
|
+
{
|
|
128
|
+
id: 'toggle-attribute',
|
|
129
|
+
prompt:
|
|
130
|
+
'When the button is clicked, toggle the "hidden" attribute on the element with id "message".',
|
|
131
|
+
reference: 'on click toggle @hidden on #message',
|
|
132
|
+
fixture: '<div id="message">msg</div>',
|
|
133
|
+
tags: ['toggle', 'attribute'],
|
|
134
|
+
},
|
|
135
|
+
{
|
|
136
|
+
id: 'set-attribute',
|
|
137
|
+
prompt:
|
|
138
|
+
'When the button is clicked, set the aria-expanded attribute of the element with id "panel" to true.',
|
|
139
|
+
// NB: `set @aria-expanded of #panel to "true"` — the phrasing an LLM reaches
|
|
140
|
+
// for first — parses clean and does NOTHING. The possessive form is the one
|
|
141
|
+
// that works. See runs/README findings.
|
|
142
|
+
reference: 'on click set #panel\'s @aria-expanded to "true"',
|
|
143
|
+
fixture: '<div id="panel" aria-expanded="false">panel</div>',
|
|
144
|
+
tags: ['set', 'attribute'],
|
|
145
|
+
},
|
|
146
|
+
{
|
|
147
|
+
id: 'set-style',
|
|
148
|
+
prompt:
|
|
149
|
+
'When the button is clicked, change the inline background colour of the element with id "swatch" to red.',
|
|
150
|
+
reference: 'on click set *background-color of #swatch to "red"',
|
|
151
|
+
fixture: '<div id="swatch">swatch</div>',
|
|
152
|
+
tags: ['set', 'style'],
|
|
153
|
+
},
|
|
154
|
+
{
|
|
155
|
+
id: 'set-inner-html',
|
|
156
|
+
prompt:
|
|
157
|
+
'When the button is clicked, set the innerHTML of the element with id "output" to the text Done.',
|
|
158
|
+
reference: 'on click set #output\'s innerHTML to "Done"',
|
|
159
|
+
fixture: '<div id="output">old</div>',
|
|
160
|
+
tags: ['set', 'possessive', 'property'],
|
|
161
|
+
},
|
|
162
|
+
{
|
|
163
|
+
id: 'add-class-multiple',
|
|
164
|
+
prompt:
|
|
165
|
+
'When the button is clicked, add the class "done" to every element that has the class "todo".',
|
|
166
|
+
reference: 'on click add .done to .todo',
|
|
167
|
+
fixture: '<li class="todo">a</li><li class="todo">b</li>',
|
|
168
|
+
tags: ['add', 'multi-target'],
|
|
169
|
+
},
|
|
170
|
+
{
|
|
171
|
+
id: 'mouseenter-hover',
|
|
172
|
+
prompt: 'When the mouse pointer enters the button, add the class "hover" to it.',
|
|
173
|
+
reference: 'on mouseenter add .hover to me',
|
|
174
|
+
fixture: '',
|
|
175
|
+
trigger: { event: 'mouseenter' },
|
|
176
|
+
tags: ['event', 'non-click-trigger'],
|
|
177
|
+
},
|
|
178
|
+
{
|
|
179
|
+
id: 'tabs-switch',
|
|
180
|
+
prompt:
|
|
181
|
+
'When the button is clicked, remove the class "active" from all elements with class "tab", then add that class to the element with id "tab2".',
|
|
182
|
+
reference: 'on click remove .active from .tab then add .active to #tab2',
|
|
183
|
+
fixture: '<div class="tab active" id="tab1">1</div><div class="tab" id="tab2">2</div>',
|
|
184
|
+
tags: ['sequence', 'multi-target', 'multi-command'],
|
|
185
|
+
},
|
|
186
|
+
{
|
|
187
|
+
id: 'conditional-class',
|
|
188
|
+
prompt:
|
|
189
|
+
'When the button is clicked, add the class "warned" to the element with id "box" only if that element already has the class "danger".',
|
|
190
|
+
reference: 'on click if #box matches .danger add .warned to #box end',
|
|
191
|
+
fixture: '<div id="box" class="danger">box</div>',
|
|
192
|
+
tags: ['conditional', 'if'],
|
|
193
|
+
},
|
|
194
|
+
{
|
|
195
|
+
id: 'custom-event',
|
|
196
|
+
prompt:
|
|
197
|
+
'When the button receives a custom event named "refresh", put the text Refreshed into the element with id "output".',
|
|
198
|
+
reference: 'on refresh put "Refreshed" into #output',
|
|
199
|
+
fixture: '<div id="output">stale</div>',
|
|
200
|
+
trigger: { event: 'refresh' },
|
|
201
|
+
tags: ['event', 'custom-event'],
|
|
202
|
+
},
|
|
203
|
+
{
|
|
204
|
+
id: 'add-to-body',
|
|
205
|
+
prompt: 'When the button is clicked, add the class "modal-open" to the page body.',
|
|
206
|
+
reference: 'on click add .modal-open to body',
|
|
207
|
+
fixture: '',
|
|
208
|
+
tags: ['add', 'body-target'],
|
|
209
|
+
},
|
|
210
|
+
] as const;
|
|
211
|
+
|
|
212
|
+
export function taskById(id: string): BenchTask | undefined {
|
|
213
|
+
return TASKS.find(t => t.id === id);
|
|
214
|
+
}
|