@hyperfixi/testing-framework 2.9.4 → 2.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +21 -1
- package/package.json +7 -6
- package/src/agent-bench/README.md +162 -0
- package/src/agent-bench/agent-bench.test.ts +112 -0
- package/src/agent-bench/cli.ts +315 -0
- package/src/agent-bench/harness.ts +331 -0
- package/src/agent-bench/tasks.ts +214 -0
- package/src/agent-bench/variants.ts +241 -0
- package/src/multilingual/fidelity.ts +22 -376
- package/src/multilingual/shipped-examples-execution.test.ts +10 -5
- package/src/multilingual/shipped-examples-execution.ts +32 -20
|
@@ -0,0 +1,241 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Plausible-phrasing probe — the generator-independent half of the benchmark.
|
|
3
|
+
*
|
|
4
|
+
* WHY THIS EXISTS. The A/B run (`score --run`) needs an agent to generate
|
|
5
|
+
* candidates, so its numbers are only as trustworthy as the generator's
|
|
6
|
+
* independence. This file needs no generator: every entry is a concrete
|
|
7
|
+
* phrasing, and whether it parses and what it DOES are deterministic properties
|
|
8
|
+
* of the parser. The claim is therefore narrow and checkable — "these phrasings
|
|
9
|
+
* behave thus" — not a statistical claim about how often a model emits them.
|
|
10
|
+
*
|
|
11
|
+
* WHAT IT MEASURES. The band the loop cannot currently see. `validate_and_compile`
|
|
12
|
+
* reports ok/diagnostics, so a phrasing that FAILS to parse is already handled:
|
|
13
|
+
* the agent gets an error and repairs. The dangerous phrasings are the ones that
|
|
14
|
+
* parse clean — confidence 1.0, zero diagnostics — and quietly do the wrong
|
|
15
|
+
* thing, or nothing. No amount of looping fixes those, because the loop is never
|
|
16
|
+
* told anything is wrong. Each such row is a candidate diagnostic.
|
|
17
|
+
*
|
|
18
|
+
* SELECTION. Each variant is a phrasing a competent generator plausibly reaches
|
|
19
|
+
* for: a neighbouring English preposition, the other of two documented spellings,
|
|
20
|
+
* a JS-flavoured construction, or a near-synonym event. Deliberate nonsense is
|
|
21
|
+
* excluded — it would inflate the failure count without teaching anything.
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
export interface Variant {
|
|
25
|
+
/** Task whose fixture/reference this is scored against. */
|
|
26
|
+
taskId: string;
|
|
27
|
+
code: string;
|
|
28
|
+
/** Why a generator plausibly emits this. */
|
|
29
|
+
rationale: string;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
export const VARIANTS: readonly Variant[] = [
|
|
33
|
+
// ── destination markers: the omitted-preposition family ───────────────────
|
|
34
|
+
{
|
|
35
|
+
taskId: 'toggle-other-class',
|
|
36
|
+
code: 'on click toggle .open #panel',
|
|
37
|
+
rationale: 'omits the destination marker "on" — reads naturally, mirrors CSS',
|
|
38
|
+
},
|
|
39
|
+
{
|
|
40
|
+
taskId: 'toggle-other-class',
|
|
41
|
+
code: 'on click toggle .open in #panel',
|
|
42
|
+
rationale: '"in" instead of "on" — plausible English for a containment target',
|
|
43
|
+
},
|
|
44
|
+
{
|
|
45
|
+
taskId: 'add-class-other',
|
|
46
|
+
code: 'on click add .highlight #item',
|
|
47
|
+
rationale: 'omits "to"',
|
|
48
|
+
},
|
|
49
|
+
{
|
|
50
|
+
taskId: 'add-class-other',
|
|
51
|
+
code: 'on click add .highlight on #item',
|
|
52
|
+
rationale: '"on" instead of "to" — the marker toggle uses',
|
|
53
|
+
},
|
|
54
|
+
{
|
|
55
|
+
taskId: 'remove-class-all',
|
|
56
|
+
code: 'on click remove .active from all .row',
|
|
57
|
+
rationale: '"from all" — English plural emphasis',
|
|
58
|
+
},
|
|
59
|
+
|
|
60
|
+
// ── attributes: three spellings, one of which works ───────────────────────
|
|
61
|
+
{
|
|
62
|
+
taskId: 'set-attribute',
|
|
63
|
+
code: 'on click set @aria-expanded of #panel to "true"',
|
|
64
|
+
rationale: '"of" form — reads best, and is what the docs use for style props',
|
|
65
|
+
},
|
|
66
|
+
{
|
|
67
|
+
taskId: 'set-attribute',
|
|
68
|
+
code: 'on click set @aria-expanded on #panel to "true"',
|
|
69
|
+
rationale: '"on" form',
|
|
70
|
+
},
|
|
71
|
+
{
|
|
72
|
+
taskId: 'set-attribute',
|
|
73
|
+
code: 'on click add @aria-expanded="true" to #panel',
|
|
74
|
+
rationale: 'HTML-flavoured: attribute literal added to a target',
|
|
75
|
+
},
|
|
76
|
+
{
|
|
77
|
+
taskId: 'toggle-attribute',
|
|
78
|
+
code: 'on click toggle @hidden of #message',
|
|
79
|
+
rationale: '"of" instead of "on" for an attribute target',
|
|
80
|
+
},
|
|
81
|
+
|
|
82
|
+
// ── properties: possessive vs dot ─────────────────────────────────────────
|
|
83
|
+
{
|
|
84
|
+
taskId: 'set-inner-html',
|
|
85
|
+
code: 'on click set #output.innerHTML to "Done"',
|
|
86
|
+
rationale: 'JS member access instead of the possessive',
|
|
87
|
+
},
|
|
88
|
+
{
|
|
89
|
+
taskId: 'set-inner-html',
|
|
90
|
+
code: 'on click set the innerHTML of #output to "Done"',
|
|
91
|
+
rationale: '"the X of Y" — the most natural English phrasing',
|
|
92
|
+
},
|
|
93
|
+
{
|
|
94
|
+
taskId: 'set-style',
|
|
95
|
+
code: 'on click set the style.backgroundColor of #swatch to "red"',
|
|
96
|
+
rationale: 'JS style-object path',
|
|
97
|
+
},
|
|
98
|
+
{
|
|
99
|
+
taskId: 'set-style',
|
|
100
|
+
code: 'on click set #swatch\'s *background-color to "red"',
|
|
101
|
+
rationale: 'possessive form of the style sigil',
|
|
102
|
+
},
|
|
103
|
+
|
|
104
|
+
// ── put/into ──────────────────────────────────────────────────────────────
|
|
105
|
+
{
|
|
106
|
+
taskId: 'put-text',
|
|
107
|
+
code: 'on click put "Saved" in #output',
|
|
108
|
+
rationale: '"in" instead of "into"',
|
|
109
|
+
},
|
|
110
|
+
{
|
|
111
|
+
taskId: 'put-text',
|
|
112
|
+
code: 'on click set the text of #output to "Saved"',
|
|
113
|
+
rationale: 'set-phrasing for a content write',
|
|
114
|
+
},
|
|
115
|
+
{
|
|
116
|
+
taskId: 'put-text',
|
|
117
|
+
code: 'on click put "Saved" into the #output',
|
|
118
|
+
rationale: 'stray article before the selector',
|
|
119
|
+
},
|
|
120
|
+
|
|
121
|
+
// ── sequencing ────────────────────────────────────────────────────────────
|
|
122
|
+
{
|
|
123
|
+
taskId: 'two-commands',
|
|
124
|
+
code: 'on click add .busy to me and put "Loading" into #output',
|
|
125
|
+
rationale: '"and" instead of "then" to join commands',
|
|
126
|
+
},
|
|
127
|
+
{
|
|
128
|
+
taskId: 'two-commands',
|
|
129
|
+
code: 'on click add .busy to me, put "Loading" into #output',
|
|
130
|
+
rationale: 'comma-separated commands',
|
|
131
|
+
},
|
|
132
|
+
{
|
|
133
|
+
taskId: 'tabs-switch',
|
|
134
|
+
code: 'on click remove .active from .tab and add .active to #tab2',
|
|
135
|
+
rationale: '"and" joining a two-step tab switch',
|
|
136
|
+
},
|
|
137
|
+
|
|
138
|
+
// ── conditionals ──────────────────────────────────────────────────────────
|
|
139
|
+
{
|
|
140
|
+
taskId: 'conditional-class',
|
|
141
|
+
code: 'on click if #box has class .danger add .warned to #box end',
|
|
142
|
+
rationale: '"has class" instead of "matches"',
|
|
143
|
+
},
|
|
144
|
+
{
|
|
145
|
+
taskId: 'conditional-class',
|
|
146
|
+
code: 'on click if #box matches .danger then add .warned to #box end',
|
|
147
|
+
rationale: 'explicit "then" after the condition',
|
|
148
|
+
},
|
|
149
|
+
|
|
150
|
+
// ── element removal vs class removal ──────────────────────────────────────
|
|
151
|
+
{
|
|
152
|
+
taskId: 'remove-element',
|
|
153
|
+
code: 'on click remove element #item',
|
|
154
|
+
rationale: 'disambiguating word "element" — remove is overloaded',
|
|
155
|
+
},
|
|
156
|
+
{
|
|
157
|
+
taskId: 'remove-element',
|
|
158
|
+
code: 'on click remove the #item',
|
|
159
|
+
rationale: 'stray article',
|
|
160
|
+
},
|
|
161
|
+
|
|
162
|
+
// ── show/hide ─────────────────────────────────────────────────────────────
|
|
163
|
+
{
|
|
164
|
+
taskId: 'hide-element',
|
|
165
|
+
code: 'on click hide the #menu',
|
|
166
|
+
rationale: 'stray article',
|
|
167
|
+
},
|
|
168
|
+
{
|
|
169
|
+
taskId: 'hide-element',
|
|
170
|
+
code: 'on click add .hidden to #menu',
|
|
171
|
+
rationale: 'class-based hiding — a common idiom, but not what hide does',
|
|
172
|
+
},
|
|
173
|
+
|
|
174
|
+
// ── events ────────────────────────────────────────────────────────────────
|
|
175
|
+
{
|
|
176
|
+
taskId: 'mouseenter-hover',
|
|
177
|
+
code: 'on mouseover add .hover to me',
|
|
178
|
+
rationale: 'mouseover is the more familiar event name',
|
|
179
|
+
},
|
|
180
|
+
{
|
|
181
|
+
taskId: 'custom-event',
|
|
182
|
+
code: 'on "refresh" put "Refreshed" into #output',
|
|
183
|
+
rationale: 'quoted event name',
|
|
184
|
+
},
|
|
185
|
+
|
|
186
|
+
// ── self-reference ────────────────────────────────────────────────────────
|
|
187
|
+
{
|
|
188
|
+
taskId: 'toggle-self-class',
|
|
189
|
+
code: 'on click toggle .active',
|
|
190
|
+
rationale: 'omits the target entirely, relying on an implicit self default',
|
|
191
|
+
},
|
|
192
|
+
{
|
|
193
|
+
taskId: 'toggle-self-class',
|
|
194
|
+
code: 'on click toggle .active on this',
|
|
195
|
+
rationale: '"this" instead of "me" — the JS spelling',
|
|
196
|
+
},
|
|
197
|
+
{
|
|
198
|
+
taskId: 'toggle-self-class',
|
|
199
|
+
code: 'on click toggle class .active on me',
|
|
200
|
+
rationale: 'the word "class" before the selector',
|
|
201
|
+
},
|
|
202
|
+
|
|
203
|
+
// ── multi-target / body ───────────────────────────────────────────────────
|
|
204
|
+
{
|
|
205
|
+
taskId: 'add-class-multiple',
|
|
206
|
+
code: 'on click add .done to all .todo',
|
|
207
|
+
rationale: '"to all" plural emphasis',
|
|
208
|
+
},
|
|
209
|
+
{
|
|
210
|
+
taskId: 'add-class-multiple',
|
|
211
|
+
code: 'on click add .done to every .todo',
|
|
212
|
+
rationale: '"every" — the wording used in the prompt itself',
|
|
213
|
+
},
|
|
214
|
+
{
|
|
215
|
+
taskId: 'add-to-body',
|
|
216
|
+
code: 'on click add .modal-open to the body',
|
|
217
|
+
rationale: 'stray article before body',
|
|
218
|
+
},
|
|
219
|
+
{
|
|
220
|
+
taskId: 'add-to-body',
|
|
221
|
+
code: 'on click add .modal-open to <body/>',
|
|
222
|
+
rationale: 'query literal for a tag target',
|
|
223
|
+
},
|
|
224
|
+
|
|
225
|
+
// ── positional ────────────────────────────────────────────────────────────
|
|
226
|
+
{
|
|
227
|
+
taskId: 'closest-ancestor',
|
|
228
|
+
code: 'on click add .selected to the closest .card',
|
|
229
|
+
rationale: 'stray article before a positional expression',
|
|
230
|
+
},
|
|
231
|
+
{
|
|
232
|
+
taskId: 'closest-ancestor',
|
|
233
|
+
code: 'on click add .selected to closest .card to me',
|
|
234
|
+
rationale: 'explicit anchor for closest',
|
|
235
|
+
},
|
|
236
|
+
{
|
|
237
|
+
taskId: 'show-element',
|
|
238
|
+
code: 'on click show the #modal',
|
|
239
|
+
rationale: 'stray article',
|
|
240
|
+
},
|
|
241
|
+
] as const;
|
|
@@ -1,377 +1,23 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Structural fidelity
|
|
3
|
-
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
const CHILD_FIELDS = [
|
|
25
|
-
'body',
|
|
26
|
-
'statements',
|
|
27
|
-
'thenBranch',
|
|
28
|
-
'elseBranch',
|
|
29
|
-
'branches',
|
|
30
|
-
'eventHandlers',
|
|
31
|
-
'initBlock',
|
|
32
|
-
] as const;
|
|
33
|
-
|
|
34
|
-
/**
|
|
35
|
-
* Collect the distinct command actions anywhere in a parsed semantic node tree
|
|
36
|
-
* (top-level action + nested body/statements/branches), excluding the structural
|
|
37
|
-
* `compound` wrapper. Returns a sorted array for stable comparison/serialization.
|
|
38
|
-
*/
|
|
39
|
-
export function collectActions(node: unknown): string[] {
|
|
40
|
-
const acc = new Set<string>();
|
|
41
|
-
walk(node, acc, 0);
|
|
42
|
-
return [...acc].sort();
|
|
43
|
-
}
|
|
44
|
-
|
|
45
|
-
function walk(node: unknown, acc: Set<string>, depth: number): void {
|
|
46
|
-
// Guard against pathological/cyclic structures.
|
|
47
|
-
if (depth > 64 || node === null || typeof node !== 'object') return;
|
|
48
|
-
|
|
49
|
-
const rec = node as Record<string, unknown>;
|
|
50
|
-
const action = rec.action;
|
|
51
|
-
if (typeof action === 'string' && !STRUCTURAL_ACTIONS.has(action)) {
|
|
52
|
-
acc.add(action);
|
|
53
|
-
}
|
|
54
|
-
|
|
55
|
-
for (const field of CHILD_FIELDS) {
|
|
56
|
-
const child = rec[field];
|
|
57
|
-
if (Array.isArray(child)) {
|
|
58
|
-
for (const c of child) walk(c, acc, depth + 1);
|
|
59
|
-
} else if (child && typeof child === 'object') {
|
|
60
|
-
walk(child, acc, depth + 1);
|
|
61
|
-
}
|
|
62
|
-
}
|
|
63
|
-
}
|
|
64
|
-
|
|
65
|
-
/**
|
|
66
|
-
* Like {@link collectActions} but **multiset**: duplicates are preserved (sorted,
|
|
67
|
-
* not deduped). Needed for precision — a *duplicate* spurious command (e.g. a
|
|
68
|
-
* renderer that injects a phantom `toggle` ahead of a real `toggle`, yielding
|
|
69
|
-
* `[toggle, toggle, put]`) is invisible to the Set-based `collectActions`.
|
|
70
|
-
*
|
|
71
|
-
* Kept as a parallel walk rather than refactoring `collectActions`: the latter
|
|
72
|
-
* feeds the committed regression baseline, so its traversal stays byte-identical.
|
|
73
|
-
*/
|
|
74
|
-
export function collectActionsMultiset(node: unknown): string[] {
|
|
75
|
-
const acc: string[] = [];
|
|
76
|
-
walkMultiset(node, acc, 0);
|
|
77
|
-
return acc.sort();
|
|
78
|
-
}
|
|
79
|
-
|
|
80
|
-
function walkMultiset(node: unknown, acc: string[], depth: number): void {
|
|
81
|
-
if (depth > 64 || node === null || typeof node !== 'object') return;
|
|
82
|
-
|
|
83
|
-
const rec = node as Record<string, unknown>;
|
|
84
|
-
const action = rec.action;
|
|
85
|
-
if (typeof action === 'string' && !STRUCTURAL_ACTIONS.has(action)) {
|
|
86
|
-
acc.push(action);
|
|
87
|
-
}
|
|
88
|
-
|
|
89
|
-
for (const field of CHILD_FIELDS) {
|
|
90
|
-
const child = rec[field];
|
|
91
|
-
if (Array.isArray(child)) {
|
|
92
|
-
for (const c of child) walkMultiset(c, acc, depth + 1);
|
|
93
|
-
} else if (child && typeof child === 'object') {
|
|
94
|
-
walkMultiset(child, acc, depth + 1);
|
|
95
|
-
}
|
|
96
|
-
}
|
|
97
|
-
}
|
|
98
|
-
|
|
99
|
-
/** Build a multiset count map from an action list. */
|
|
100
|
-
function actionCounts(items: readonly string[]): Map<string, number> {
|
|
101
|
-
const m = new Map<string, number>();
|
|
102
|
-
for (const it of items) m.set(it, (m.get(it) ?? 0) + 1);
|
|
103
|
-
return m;
|
|
104
|
-
}
|
|
105
|
-
|
|
106
|
-
/**
|
|
107
|
-
* Structural fidelity in [0, 1]: the fraction of the reference (English) actions
|
|
108
|
-
* also present in the candidate (recall). Returns `undefined` when the reference
|
|
109
|
-
* has no actions to compare against.
|
|
110
|
-
*/
|
|
111
|
-
export function computeFidelity(
|
|
112
|
-
reference: readonly string[],
|
|
113
|
-
candidate: readonly string[]
|
|
114
|
-
): number | undefined {
|
|
115
|
-
if (reference.length === 0) return undefined;
|
|
116
|
-
const cand = new Set(candidate);
|
|
117
|
-
let hits = 0;
|
|
118
|
-
for (const a of reference) {
|
|
119
|
-
if (cand.has(a)) hits++;
|
|
120
|
-
}
|
|
121
|
-
return hits / reference.length;
|
|
122
|
-
}
|
|
123
|
-
|
|
124
|
-
/**
|
|
125
|
-
* R0-recall on the **multiset** in [0, 1]: the fraction of the reference's
|
|
126
|
-
* actions — counting duplicates — also present in the candidate.
|
|
127
|
-
*
|
|
128
|
-
* {@link computeFidelity} scores the deduped Set signature, so a candidate that
|
|
129
|
-
* drops a REPEATED command scores 1.0: reference `[bind, bind]` collapses to
|
|
130
|
-
* `{bind}`, which `[bind]` satisfies in full. That is how `bind-two-way` sat at
|
|
131
|
-
* fidelity 1.0 across all 24 languages while every one of them parsed only the
|
|
132
|
-
* first of its two `bind`s. R1 (role signatures) is a Set too, and is equally
|
|
133
|
-
* blind. {@link computePrecision} catches the mirror case — a candidate that ADDS
|
|
134
|
-
* a duplicate — so before this signal existed the ratchet saw spurious commands
|
|
135
|
-
* but never dropped ones.
|
|
136
|
-
*
|
|
137
|
-
* Pass multisets (see {@link collectActionsMultiset}) on both sides.
|
|
138
|
-
* Returns `undefined` when the reference has no actions to compare against.
|
|
139
|
-
*/
|
|
140
|
-
export function computeMultisetRecall(
|
|
141
|
-
reference: readonly string[],
|
|
142
|
-
candidate: readonly string[]
|
|
143
|
-
): number | undefined {
|
|
144
|
-
if (reference.length === 0) return undefined;
|
|
145
|
-
const cand = actionCounts(candidate);
|
|
146
|
-
let matched = 0;
|
|
147
|
-
for (const a of reference) {
|
|
148
|
-
const remaining = cand.get(a) ?? 0;
|
|
149
|
-
if (remaining > 0) {
|
|
150
|
-
matched++;
|
|
151
|
-
cand.set(a, remaining - 1);
|
|
152
|
-
}
|
|
153
|
-
}
|
|
154
|
-
return matched / reference.length;
|
|
155
|
-
}
|
|
156
|
-
|
|
157
|
-
/**
|
|
158
|
-
* Structural **precision** in [0, 1]: the fraction of the *candidate's* actions
|
|
159
|
-
* that are justified by the reference (multiset-aware). The complement of
|
|
160
|
-
* {@link computeFidelity}'s recall — it falls below 1.0 when a parse/render adds
|
|
161
|
-
* commands the source never had.
|
|
162
|
-
*
|
|
163
|
-
* This is the signal recall + role-fidelity (R1) cannot see: a renderer that
|
|
164
|
-
* injects a phantom `toggle` (ja/ko/tr/ar/ru event-handler rendering) keeps recall
|
|
165
|
-
* 1.0 while precision drops. Pass multisets (see {@link collectActionsMultiset})
|
|
166
|
-
* so a *duplicated* spurious action is counted, not absorbed.
|
|
167
|
-
*
|
|
168
|
-
* Returns `undefined` when the candidate has no actions to score.
|
|
169
|
-
*/
|
|
170
|
-
export function computePrecision(
|
|
171
|
-
reference: readonly string[],
|
|
172
|
-
candidate: readonly string[]
|
|
173
|
-
): number | undefined {
|
|
174
|
-
if (candidate.length === 0) return undefined;
|
|
175
|
-
const ref = actionCounts(reference);
|
|
176
|
-
let matched = 0;
|
|
177
|
-
for (const a of candidate) {
|
|
178
|
-
const remaining = ref.get(a) ?? 0;
|
|
179
|
-
if (remaining > 0) {
|
|
180
|
-
matched++;
|
|
181
|
-
ref.set(a, remaining - 1);
|
|
182
|
-
}
|
|
183
|
-
}
|
|
184
|
-
return matched / candidate.length;
|
|
185
|
-
}
|
|
186
|
-
|
|
187
|
-
/**
|
|
188
|
-
* The candidate actions **not** justified by the reference (multiset difference,
|
|
189
|
-
* sorted) — i.e. the spurious/hallucinated commands a parse or render introduced.
|
|
190
|
-
* Empty when the candidate is a structural subset of the reference. Use for
|
|
191
|
-
* diagnostics ("which command was hallucinated?"), complementing the
|
|
192
|
-
* {@link computePrecision} score.
|
|
193
|
-
*/
|
|
194
|
-
export function spuriousActions(
|
|
195
|
-
reference: readonly string[],
|
|
196
|
-
candidate: readonly string[]
|
|
197
|
-
): string[] {
|
|
198
|
-
const ref = actionCounts(reference);
|
|
199
|
-
const extras: string[] = [];
|
|
200
|
-
for (const a of candidate) {
|
|
201
|
-
const remaining = ref.get(a) ?? 0;
|
|
202
|
-
if (remaining > 0) ref.set(a, remaining - 1);
|
|
203
|
-
else extras.push(a);
|
|
204
|
-
}
|
|
205
|
-
return extras.sort();
|
|
206
|
-
}
|
|
207
|
-
|
|
208
|
-
/**
|
|
209
|
-
* R1 — role-fidelity signature.
|
|
210
|
-
*
|
|
211
|
-
* Action-set fidelity cannot see a parse that finds the right commands with the
|
|
212
|
-
* WRONG roles: a swapped patient/destination executes wrongly while scoring 1.0.
|
|
213
|
-
* This signature captures, for every command node in the tree, which roles were
|
|
214
|
-
* filled and with what value *type* (`add.patient:selector`,
|
|
215
|
-
* `put.destination:reference`). Cross-language comparison is by role name +
|
|
216
|
-
* value type — never by value string, which is legitimately translated.
|
|
217
|
-
* The roles container is a ReadonlyMap on live nodes (serializes to {} in JSON),
|
|
218
|
-
* so the signature must be collected at validation time, not from results.json.
|
|
219
|
-
*/
|
|
220
|
-
export function collectRoleSignature(node: unknown): string[] {
|
|
221
|
-
const acc = new Set<string>();
|
|
222
|
-
walkRoles(node, acc, 0);
|
|
223
|
-
return [...acc].sort();
|
|
224
|
-
}
|
|
225
|
-
|
|
226
|
-
function walkRoles(node: unknown, acc: Set<string>, depth: number): void {
|
|
227
|
-
if (depth > 64 || node === null || typeof node !== 'object') return;
|
|
228
|
-
|
|
229
|
-
const rec = node as Record<string, unknown>;
|
|
230
|
-
const action = rec.action;
|
|
231
|
-
if (typeof action === 'string' && !STRUCTURAL_ACTIONS.has(action)) {
|
|
232
|
-
const roles = rec.roles;
|
|
233
|
-
const entries: Array<[unknown, unknown]> =
|
|
234
|
-
roles instanceof Map
|
|
235
|
-
? [...roles.entries()]
|
|
236
|
-
: roles && typeof roles === 'object'
|
|
237
|
-
? Object.entries(roles)
|
|
238
|
-
: [];
|
|
239
|
-
for (const [role, value] of entries) {
|
|
240
|
-
if (value === undefined || value === null) continue;
|
|
241
|
-
const kind =
|
|
242
|
-
typeof value === 'object' && typeof (value as { type?: unknown }).type === 'string'
|
|
243
|
-
? (value as { type: string }).type
|
|
244
|
-
: typeof value;
|
|
245
|
-
acc.add(`${action}.${String(role)}:${kind}`);
|
|
246
|
-
}
|
|
247
|
-
}
|
|
248
|
-
|
|
249
|
-
for (const field of CHILD_FIELDS) {
|
|
250
|
-
const child = rec[field];
|
|
251
|
-
if (Array.isArray(child)) {
|
|
252
|
-
for (const c of child) walkRoles(c, acc, depth + 1);
|
|
253
|
-
} else if (child && typeof child === 'object') {
|
|
254
|
-
walkRoles(child, acc, depth + 1);
|
|
255
|
-
}
|
|
256
|
-
}
|
|
257
|
-
}
|
|
258
|
-
|
|
259
|
-
/**
|
|
260
|
-
* Role-value kinds never emitted by the R3 walker. `reference` values (`me`,
|
|
261
|
-
* `it`, …) are mostly `fillSchemaDefaults` injections present identically on
|
|
262
|
-
* both sides — noise. `flag` names and `property-path` properties are bare
|
|
263
|
-
* identifiers, excluded by v1 (see {@link collectRoleValueSignature}).
|
|
264
|
-
*/
|
|
265
|
-
const VALUE_KIND_EXCLUSIONS = new Set(['reference', 'flag', 'property-path']);
|
|
266
|
-
|
|
267
|
-
/**
|
|
268
|
-
* A role value is compared cross-language only when its WHOLE surface form is
|
|
269
|
-
* code-shaped — language-invariant by construction, never legitimately
|
|
270
|
-
* translated. Conservative v1 whitelist; when a sweep firing turns out to be a
|
|
271
|
-
* legit translation difference, tighten here and document the exclusion.
|
|
272
|
-
*/
|
|
273
|
-
const INVARIANT_VALUE_PATTERNS: readonly RegExp[] = [
|
|
274
|
-
/^[#.[<@*]/, // selectors: #id .class [attr] <tag/> @attr *style
|
|
275
|
-
/^[:$^][A-Za-z_]\w*$/, // sigil refs: :local $global ^element
|
|
276
|
-
/^\d+(\.\d+)?(ms|s|m|h)?$/, // numbers and time literals: 2 1.5 200ms
|
|
277
|
-
/^[A-Za-z_][\w-]*:[\w-]+$/, // colon-qualified event names: draggable:start
|
|
278
|
-
/^(\.{0,2}\/|https?:)/, // URLs / paths: /api/data ./x ../y https://…
|
|
279
|
-
];
|
|
280
|
-
|
|
281
|
-
function isInvariantSurface(surface: string): boolean {
|
|
282
|
-
// Whole-surface rule, sweep-validated exclusions:
|
|
283
|
-
// - whitespace ⇒ mixed content. `if #modal exists` captures its condition as
|
|
284
|
-
// `#modal exists` — starts selector-shaped, but `exists` is prose that every
|
|
285
|
-
// language legitimately translates (16-language false firing without this).
|
|
286
|
-
// - `${` ⇒ template interpolation. `/api/search?q=${my value}` is an
|
|
287
|
-
// expression, and tokenizers split it at different points per language, so
|
|
288
|
-
// the captured surface isn't comparable verbatim.
|
|
289
|
-
if (/\s/.test(surface) || surface.includes('${')) return false;
|
|
290
|
-
return INVARIANT_VALUE_PATTERNS.some(re => re.test(surface));
|
|
291
|
-
}
|
|
292
|
-
|
|
293
|
-
/**
|
|
294
|
-
* The comparable surface form of a role value, or `undefined` when the value
|
|
295
|
-
* carries none: `.value` for `literal`/`selector` (string/number/boolean
|
|
296
|
-
* coerced), `.raw` for `expression`. Kinds in {@link VALUE_KIND_EXCLUSIONS}
|
|
297
|
-
* and values without a string `type` discriminator yield `undefined`.
|
|
298
|
-
*/
|
|
299
|
-
function roleValueSurface(value: unknown): string | undefined {
|
|
300
|
-
if (value === null || typeof value !== 'object') return undefined;
|
|
301
|
-
const rec = value as { type?: unknown; value?: unknown; raw?: unknown };
|
|
302
|
-
if (typeof rec.type !== 'string' || VALUE_KIND_EXCLUSIONS.has(rec.type)) return undefined;
|
|
303
|
-
if (rec.type === 'expression') {
|
|
304
|
-
return typeof rec.raw === 'string' ? rec.raw : undefined;
|
|
305
|
-
}
|
|
306
|
-
const v = rec.value;
|
|
307
|
-
return typeof v === 'string' || typeof v === 'number' || typeof v === 'boolean'
|
|
308
|
-
? String(v)
|
|
309
|
-
: undefined;
|
|
310
|
-
}
|
|
311
|
-
|
|
312
|
-
/**
|
|
313
|
-
* R3 — role-VALUE signature (invariant values only, multiset).
|
|
314
|
-
*
|
|
315
|
-
* R0/R1 compare actions and role *types*; values are never compared because
|
|
316
|
-
* they are legitimately translated. That leaves a live defect class with zero
|
|
317
|
-
* signal: right action counts, right role types, wrong role VALUE — the #633
|
|
318
|
-
* class, where ms captured `trigger` events named `draggable` instead of
|
|
319
|
-
* `draggable:start` (correct multiset, correct `trigger.event:literal`
|
|
320
|
-
* signature, silently wrong runtime behavior), and 18 other languages carried
|
|
321
|
-
* the same corruption with no side-effect at all.
|
|
322
|
-
*
|
|
323
|
-
* The subset of values compared here is language-invariant by construction —
|
|
324
|
-
* code, not prose (see {@link INVARIANT_VALUE_PATTERNS}). Emits a **multiset**
|
|
325
|
-
* of `` `action.role=value` `` entries (duplicates preserved, sorted); score
|
|
326
|
-
* with {@link computeMultisetRecall} so a dropped duplicate value is visible.
|
|
327
|
-
*
|
|
328
|
-
* Deliberately EXCLUDED in v1: bare-word identifiers (`startX` — usually
|
|
329
|
-
* invariant, but property/variable names occasionally get localized in seeds),
|
|
330
|
-
* string literals (message strings are legitimately translated), expression
|
|
331
|
-
* raws mixing native words + code (`次 .item`), and `reference` values
|
|
332
|
-
* (`me`/`it` — mostly `fillSchemaDefaults` injections on both sides).
|
|
333
|
-
*
|
|
334
|
-
* Blind spot: recall fires when a *translation* loses/corrupts an invariant
|
|
335
|
-
* value. If the **en reference itself** corrupts a value, every language flags
|
|
336
|
-
* at once — a 24-language R3 firestorm on one pattern means "suspect the en
|
|
337
|
-
* parse first" (unlike R0, where en corruption moves nothing).
|
|
338
|
-
*
|
|
339
|
-
* The roles container is a ReadonlyMap on live nodes (serializes to {} in
|
|
340
|
-
* results.json), so collect at validation time, never from the JSON.
|
|
341
|
-
*/
|
|
342
|
-
export function collectRoleValueSignature(node: unknown): string[] {
|
|
343
|
-
const acc: string[] = [];
|
|
344
|
-
walkRoleValues(node, acc, 0);
|
|
345
|
-
return acc.sort();
|
|
346
|
-
}
|
|
347
|
-
|
|
348
|
-
function walkRoleValues(node: unknown, acc: string[], depth: number): void {
|
|
349
|
-
if (depth > 64 || node === null || typeof node !== 'object') return;
|
|
350
|
-
|
|
351
|
-
const rec = node as Record<string, unknown>;
|
|
352
|
-
const action = rec.action;
|
|
353
|
-
if (typeof action === 'string' && !STRUCTURAL_ACTIONS.has(action)) {
|
|
354
|
-
const roles = rec.roles;
|
|
355
|
-
const entries: Array<[unknown, unknown]> =
|
|
356
|
-
roles instanceof Map
|
|
357
|
-
? [...roles.entries()]
|
|
358
|
-
: roles && typeof roles === 'object'
|
|
359
|
-
? Object.entries(roles)
|
|
360
|
-
: [];
|
|
361
|
-
for (const [role, value] of entries) {
|
|
362
|
-
if (value === undefined || value === null) continue;
|
|
363
|
-
const surface = roleValueSurface(value);
|
|
364
|
-
if (surface === undefined || !isInvariantSurface(surface)) continue;
|
|
365
|
-
acc.push(`${action}.${String(role)}=${surface}`);
|
|
366
|
-
}
|
|
367
|
-
}
|
|
368
|
-
|
|
369
|
-
for (const field of CHILD_FIELDS) {
|
|
370
|
-
const child = rec[field];
|
|
371
|
-
if (Array.isArray(child)) {
|
|
372
|
-
for (const c of child) walkRoleValues(c, acc, depth + 1);
|
|
373
|
-
} else if (child && typeof child === 'object') {
|
|
374
|
-
walkRoleValues(child, acc, depth + 1);
|
|
375
|
-
}
|
|
376
|
-
}
|
|
377
|
-
}
|
|
2
|
+
* Structural fidelity scorers — canonical home is `@lokascript/semantic/fidelity`
|
|
3
|
+
* (moved there in agent-era arc 4 so any consumer can score a candidate parse
|
|
4
|
+
* against a reference without pulling in this test framework; see
|
|
5
|
+
* docs/FIDELITY.md and docs-internal/AGENT_ERA_ROADMAP.md).
|
|
6
|
+
*
|
|
7
|
+
* This module re-exports the full surface so every in-repo consumer
|
|
8
|
+
* (orchestrator, parse-validator, triage tools, the fidelity test suite) keeps
|
|
9
|
+
* its import path — and the test suite now exercises the extracted module
|
|
10
|
+
* through this shim, which keeps the two from drifting.
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
export {
|
|
14
|
+
FIDELITY_THRESHOLD,
|
|
15
|
+
collectActions,
|
|
16
|
+
collectActionsMultiset,
|
|
17
|
+
computeFidelity,
|
|
18
|
+
computeMultisetRecall,
|
|
19
|
+
computePrecision,
|
|
20
|
+
spuriousActions,
|
|
21
|
+
collectRoleSignature,
|
|
22
|
+
collectRoleValueSignature,
|
|
23
|
+
} from '@lokascript/semantic/fidelity';
|
|
@@ -87,16 +87,21 @@ describe('shipped-examples execution gate', () => {
|
|
|
87
87
|
}, 240_000);
|
|
88
88
|
|
|
89
89
|
it('walks pages and compares handlers (sanity: extraction and both engines working)', () => {
|
|
90
|
-
// Floors well below current values (
|
|
90
|
+
// Floors well below current values (48 / 261 / 122 / 52) but far above
|
|
91
91
|
// zero: a broken walk, extractor, or engine bootstrap fails loudly here
|
|
92
92
|
// instead of making assertions 2-3 vacuously pass.
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
93
|
+
//
|
|
94
|
+
// Calibrated against the git-TRACKED corpus — the sweep ignores untracked
|
|
95
|
+
// examples/ dirs, so these numbers are the same on every machine and in
|
|
96
|
+
// CI. (The original floors were measured on a working tree with four
|
|
97
|
+
// gitignored dirs present and failed every clean checkout — #862.)
|
|
98
|
+
expect(result.pages).toBeGreaterThan(35);
|
|
99
|
+
expect(result.handlers).toBeGreaterThan(200);
|
|
100
|
+
expect(result.compared.length).toBeGreaterThan(90);
|
|
96
101
|
// Vacuous (empty-vs-empty) pairs are NOT parity evidence — the floor is on
|
|
97
102
|
// real, non-empty signature matches.
|
|
98
103
|
const realMatches = result.compared.filter(c => c.match && !c.vacuous).length;
|
|
99
|
-
expect(realMatches).toBeGreaterThan(
|
|
104
|
+
expect(realMatches).toBeGreaterThan(40);
|
|
100
105
|
});
|
|
101
106
|
|
|
102
107
|
it('has no NEW divergence from upstream outside the allowlist', () => {
|