@hyperfixi/testing-framework 2.9.4 → 2.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,241 @@
1
+ /**
2
+ * Plausible-phrasing probe — the generator-independent half of the benchmark.
3
+ *
4
+ * WHY THIS EXISTS. The A/B run (`score --run`) needs an agent to generate
5
+ * candidates, so its numbers are only as trustworthy as the generator's
6
+ * independence. This file needs no generator: every entry is a concrete
7
+ * phrasing, and whether it parses and what it DOES are deterministic properties
8
+ * of the parser. The claim is therefore narrow and checkable — "these phrasings
9
+ * behave thus" — not a statistical claim about how often a model emits them.
10
+ *
11
+ * WHAT IT MEASURES. The band the loop cannot currently see. `validate_and_compile`
12
+ * reports ok/diagnostics, so a phrasing that FAILS to parse is already handled:
13
+ * the agent gets an error and repairs. The dangerous phrasings are the ones that
14
+ * parse clean — confidence 1.0, zero diagnostics — and quietly do the wrong
15
+ * thing, or nothing. No amount of looping fixes those, because the loop is never
16
+ * told anything is wrong. Each such row is a candidate diagnostic.
17
+ *
18
+ * SELECTION. Each variant is a phrasing a competent generator plausibly reaches
19
+ * for: a neighbouring English preposition, the other of two documented spellings,
20
+ * a JS-flavoured construction, or a near-synonym event. Deliberate nonsense is
21
+ * excluded — it would inflate the failure count without teaching anything.
22
+ */
23
+
24
+ export interface Variant {
25
+ /** Task whose fixture/reference this is scored against. */
26
+ taskId: string;
27
+ code: string;
28
+ /** Why a generator plausibly emits this. */
29
+ rationale: string;
30
+ }
31
+
32
+ export const VARIANTS: readonly Variant[] = [
33
+ // ── destination markers: the omitted-preposition family ───────────────────
34
+ {
35
+ taskId: 'toggle-other-class',
36
+ code: 'on click toggle .open #panel',
37
+ rationale: 'omits the destination marker "on" — reads naturally, mirrors CSS',
38
+ },
39
+ {
40
+ taskId: 'toggle-other-class',
41
+ code: 'on click toggle .open in #panel',
42
+ rationale: '"in" instead of "on" — plausible English for a containment target',
43
+ },
44
+ {
45
+ taskId: 'add-class-other',
46
+ code: 'on click add .highlight #item',
47
+ rationale: 'omits "to"',
48
+ },
49
+ {
50
+ taskId: 'add-class-other',
51
+ code: 'on click add .highlight on #item',
52
+ rationale: '"on" instead of "to" — the marker toggle uses',
53
+ },
54
+ {
55
+ taskId: 'remove-class-all',
56
+ code: 'on click remove .active from all .row',
57
+ rationale: '"from all" — English plural emphasis',
58
+ },
59
+
60
+ // ── attributes: three spellings, one of which works ───────────────────────
61
+ {
62
+ taskId: 'set-attribute',
63
+ code: 'on click set @aria-expanded of #panel to "true"',
64
+ rationale: '"of" form — reads best, and is what the docs use for style props',
65
+ },
66
+ {
67
+ taskId: 'set-attribute',
68
+ code: 'on click set @aria-expanded on #panel to "true"',
69
+ rationale: '"on" form',
70
+ },
71
+ {
72
+ taskId: 'set-attribute',
73
+ code: 'on click add @aria-expanded="true" to #panel',
74
+ rationale: 'HTML-flavoured: attribute literal added to a target',
75
+ },
76
+ {
77
+ taskId: 'toggle-attribute',
78
+ code: 'on click toggle @hidden of #message',
79
+ rationale: '"of" instead of "on" for an attribute target',
80
+ },
81
+
82
+ // ── properties: possessive vs dot ─────────────────────────────────────────
83
+ {
84
+ taskId: 'set-inner-html',
85
+ code: 'on click set #output.innerHTML to "Done"',
86
+ rationale: 'JS member access instead of the possessive',
87
+ },
88
+ {
89
+ taskId: 'set-inner-html',
90
+ code: 'on click set the innerHTML of #output to "Done"',
91
+ rationale: '"the X of Y" — the most natural English phrasing',
92
+ },
93
+ {
94
+ taskId: 'set-style',
95
+ code: 'on click set the style.backgroundColor of #swatch to "red"',
96
+ rationale: 'JS style-object path',
97
+ },
98
+ {
99
+ taskId: 'set-style',
100
+ code: 'on click set #swatch\'s *background-color to "red"',
101
+ rationale: 'possessive form of the style sigil',
102
+ },
103
+
104
+ // ── put/into ──────────────────────────────────────────────────────────────
105
+ {
106
+ taskId: 'put-text',
107
+ code: 'on click put "Saved" in #output',
108
+ rationale: '"in" instead of "into"',
109
+ },
110
+ {
111
+ taskId: 'put-text',
112
+ code: 'on click set the text of #output to "Saved"',
113
+ rationale: 'set-phrasing for a content write',
114
+ },
115
+ {
116
+ taskId: 'put-text',
117
+ code: 'on click put "Saved" into the #output',
118
+ rationale: 'stray article before the selector',
119
+ },
120
+
121
+ // ── sequencing ────────────────────────────────────────────────────────────
122
+ {
123
+ taskId: 'two-commands',
124
+ code: 'on click add .busy to me and put "Loading" into #output',
125
+ rationale: '"and" instead of "then" to join commands',
126
+ },
127
+ {
128
+ taskId: 'two-commands',
129
+ code: 'on click add .busy to me, put "Loading" into #output',
130
+ rationale: 'comma-separated commands',
131
+ },
132
+ {
133
+ taskId: 'tabs-switch',
134
+ code: 'on click remove .active from .tab and add .active to #tab2',
135
+ rationale: '"and" joining a two-step tab switch',
136
+ },
137
+
138
+ // ── conditionals ──────────────────────────────────────────────────────────
139
+ {
140
+ taskId: 'conditional-class',
141
+ code: 'on click if #box has class .danger add .warned to #box end',
142
+ rationale: '"has class" instead of "matches"',
143
+ },
144
+ {
145
+ taskId: 'conditional-class',
146
+ code: 'on click if #box matches .danger then add .warned to #box end',
147
+ rationale: 'explicit "then" after the condition',
148
+ },
149
+
150
+ // ── element removal vs class removal ──────────────────────────────────────
151
+ {
152
+ taskId: 'remove-element',
153
+ code: 'on click remove element #item',
154
+ rationale: 'disambiguating word "element" — remove is overloaded',
155
+ },
156
+ {
157
+ taskId: 'remove-element',
158
+ code: 'on click remove the #item',
159
+ rationale: 'stray article',
160
+ },
161
+
162
+ // ── show/hide ─────────────────────────────────────────────────────────────
163
+ {
164
+ taskId: 'hide-element',
165
+ code: 'on click hide the #menu',
166
+ rationale: 'stray article',
167
+ },
168
+ {
169
+ taskId: 'hide-element',
170
+ code: 'on click add .hidden to #menu',
171
+ rationale: 'class-based hiding — a common idiom, but not what hide does',
172
+ },
173
+
174
+ // ── events ────────────────────────────────────────────────────────────────
175
+ {
176
+ taskId: 'mouseenter-hover',
177
+ code: 'on mouseover add .hover to me',
178
+ rationale: 'mouseover is the more familiar event name',
179
+ },
180
+ {
181
+ taskId: 'custom-event',
182
+ code: 'on "refresh" put "Refreshed" into #output',
183
+ rationale: 'quoted event name',
184
+ },
185
+
186
+ // ── self-reference ────────────────────────────────────────────────────────
187
+ {
188
+ taskId: 'toggle-self-class',
189
+ code: 'on click toggle .active',
190
+ rationale: 'omits the target entirely, relying on an implicit self default',
191
+ },
192
+ {
193
+ taskId: 'toggle-self-class',
194
+ code: 'on click toggle .active on this',
195
+ rationale: '"this" instead of "me" — the JS spelling',
196
+ },
197
+ {
198
+ taskId: 'toggle-self-class',
199
+ code: 'on click toggle class .active on me',
200
+ rationale: 'the word "class" before the selector',
201
+ },
202
+
203
+ // ── multi-target / body ───────────────────────────────────────────────────
204
+ {
205
+ taskId: 'add-class-multiple',
206
+ code: 'on click add .done to all .todo',
207
+ rationale: '"to all" plural emphasis',
208
+ },
209
+ {
210
+ taskId: 'add-class-multiple',
211
+ code: 'on click add .done to every .todo',
212
+ rationale: '"every" — the wording used in the prompt itself',
213
+ },
214
+ {
215
+ taskId: 'add-to-body',
216
+ code: 'on click add .modal-open to the body',
217
+ rationale: 'stray article before body',
218
+ },
219
+ {
220
+ taskId: 'add-to-body',
221
+ code: 'on click add .modal-open to <body/>',
222
+ rationale: 'query literal for a tag target',
223
+ },
224
+
225
+ // ── positional ────────────────────────────────────────────────────────────
226
+ {
227
+ taskId: 'closest-ancestor',
228
+ code: 'on click add .selected to the closest .card',
229
+ rationale: 'stray article before a positional expression',
230
+ },
231
+ {
232
+ taskId: 'closest-ancestor',
233
+ code: 'on click add .selected to closest .card to me',
234
+ rationale: 'explicit anchor for closest',
235
+ },
236
+ {
237
+ taskId: 'show-element',
238
+ code: 'on click show the #modal',
239
+ rationale: 'stray article',
240
+ },
241
+ ] as const;
@@ -1,377 +1,23 @@
1
1
  /**
2
- * Structural fidelity for multilingual parses.
3
- *
4
- * The parse-validator's success metric is "the parser returned a non-null node".
5
- * That conflates a faithful parse with a *degenerate* one — a translated pattern
6
- * can parse non-null while silently dropping most of the source's commands (e.g.
7
- * `focus-trap` parses as a bare `if` or a stray `from` in several languages, with
8
- * the `focus`/`halt`/condition lost). This module derives a lightweight
9
- * structural signature from a parsed node and scores a translation's parse against
10
- * the English reference parse, so those degenerate passes are visible.
11
- *
12
- * The signature is intentionally word-order agnostic: it is the *set* of command
13
- * actions in the node tree, so a faithful SOV/VSO reorder scores 1.0 while a parse
14
- * that loses commands scores low.
15
- */
16
-
17
- /** Passes scoring below this are flagged as degenerate (lost >half the structure). */
18
- export const FIDELITY_THRESHOLD = 0.5;
19
-
20
- /** The structural `compound` wrapper is not a command; never counts as an action. */
21
- const STRUCTURAL_ACTIONS = new Set(['compound']);
22
-
23
- /** Node-array fields the walk recurses into (event/loop/conditional/behavior bodies). */
24
- const CHILD_FIELDS = [
25
- 'body',
26
- 'statements',
27
- 'thenBranch',
28
- 'elseBranch',
29
- 'branches',
30
- 'eventHandlers',
31
- 'initBlock',
32
- ] as const;
33
-
34
- /**
35
- * Collect the distinct command actions anywhere in a parsed semantic node tree
36
- * (top-level action + nested body/statements/branches), excluding the structural
37
- * `compound` wrapper. Returns a sorted array for stable comparison/serialization.
38
- */
39
- export function collectActions(node: unknown): string[] {
40
- const acc = new Set<string>();
41
- walk(node, acc, 0);
42
- return [...acc].sort();
43
- }
44
-
45
- function walk(node: unknown, acc: Set<string>, depth: number): void {
46
- // Guard against pathological/cyclic structures.
47
- if (depth > 64 || node === null || typeof node !== 'object') return;
48
-
49
- const rec = node as Record<string, unknown>;
50
- const action = rec.action;
51
- if (typeof action === 'string' && !STRUCTURAL_ACTIONS.has(action)) {
52
- acc.add(action);
53
- }
54
-
55
- for (const field of CHILD_FIELDS) {
56
- const child = rec[field];
57
- if (Array.isArray(child)) {
58
- for (const c of child) walk(c, acc, depth + 1);
59
- } else if (child && typeof child === 'object') {
60
- walk(child, acc, depth + 1);
61
- }
62
- }
63
- }
64
-
65
- /**
66
- * Like {@link collectActions} but **multiset**: duplicates are preserved (sorted,
67
- * not deduped). Needed for precision — a *duplicate* spurious command (e.g. a
68
- * renderer that injects a phantom `toggle` ahead of a real `toggle`, yielding
69
- * `[toggle, toggle, put]`) is invisible to the Set-based `collectActions`.
70
- *
71
- * Kept as a parallel walk rather than refactoring `collectActions`: the latter
72
- * feeds the committed regression baseline, so its traversal stays byte-identical.
73
- */
74
- export function collectActionsMultiset(node: unknown): string[] {
75
- const acc: string[] = [];
76
- walkMultiset(node, acc, 0);
77
- return acc.sort();
78
- }
79
-
80
- function walkMultiset(node: unknown, acc: string[], depth: number): void {
81
- if (depth > 64 || node === null || typeof node !== 'object') return;
82
-
83
- const rec = node as Record<string, unknown>;
84
- const action = rec.action;
85
- if (typeof action === 'string' && !STRUCTURAL_ACTIONS.has(action)) {
86
- acc.push(action);
87
- }
88
-
89
- for (const field of CHILD_FIELDS) {
90
- const child = rec[field];
91
- if (Array.isArray(child)) {
92
- for (const c of child) walkMultiset(c, acc, depth + 1);
93
- } else if (child && typeof child === 'object') {
94
- walkMultiset(child, acc, depth + 1);
95
- }
96
- }
97
- }
98
-
99
- /** Build a multiset count map from an action list. */
100
- function actionCounts(items: readonly string[]): Map<string, number> {
101
- const m = new Map<string, number>();
102
- for (const it of items) m.set(it, (m.get(it) ?? 0) + 1);
103
- return m;
104
- }
105
-
106
- /**
107
- * Structural fidelity in [0, 1]: the fraction of the reference (English) actions
108
- * also present in the candidate (recall). Returns `undefined` when the reference
109
- * has no actions to compare against.
110
- */
111
- export function computeFidelity(
112
- reference: readonly string[],
113
- candidate: readonly string[]
114
- ): number | undefined {
115
- if (reference.length === 0) return undefined;
116
- const cand = new Set(candidate);
117
- let hits = 0;
118
- for (const a of reference) {
119
- if (cand.has(a)) hits++;
120
- }
121
- return hits / reference.length;
122
- }
123
-
124
- /**
125
- * R0-recall on the **multiset** in [0, 1]: the fraction of the reference's
126
- * actions — counting duplicates — also present in the candidate.
127
- *
128
- * {@link computeFidelity} scores the deduped Set signature, so a candidate that
129
- * drops a REPEATED command scores 1.0: reference `[bind, bind]` collapses to
130
- * `{bind}`, which `[bind]` satisfies in full. That is how `bind-two-way` sat at
131
- * fidelity 1.0 across all 24 languages while every one of them parsed only the
132
- * first of its two `bind`s. R1 (role signatures) is a Set too, and is equally
133
- * blind. {@link computePrecision} catches the mirror case — a candidate that ADDS
134
- * a duplicate — so before this signal existed the ratchet saw spurious commands
135
- * but never dropped ones.
136
- *
137
- * Pass multisets (see {@link collectActionsMultiset}) on both sides.
138
- * Returns `undefined` when the reference has no actions to compare against.
139
- */
140
- export function computeMultisetRecall(
141
- reference: readonly string[],
142
- candidate: readonly string[]
143
- ): number | undefined {
144
- if (reference.length === 0) return undefined;
145
- const cand = actionCounts(candidate);
146
- let matched = 0;
147
- for (const a of reference) {
148
- const remaining = cand.get(a) ?? 0;
149
- if (remaining > 0) {
150
- matched++;
151
- cand.set(a, remaining - 1);
152
- }
153
- }
154
- return matched / reference.length;
155
- }
156
-
157
- /**
158
- * Structural **precision** in [0, 1]: the fraction of the *candidate's* actions
159
- * that are justified by the reference (multiset-aware). The complement of
160
- * {@link computeFidelity}'s recall — it falls below 1.0 when a parse/render adds
161
- * commands the source never had.
162
- *
163
- * This is the signal recall + role-fidelity (R1) cannot see: a renderer that
164
- * injects a phantom `toggle` (ja/ko/tr/ar/ru event-handler rendering) keeps recall
165
- * 1.0 while precision drops. Pass multisets (see {@link collectActionsMultiset})
166
- * so a *duplicated* spurious action is counted, not absorbed.
167
- *
168
- * Returns `undefined` when the candidate has no actions to score.
169
- */
170
- export function computePrecision(
171
- reference: readonly string[],
172
- candidate: readonly string[]
173
- ): number | undefined {
174
- if (candidate.length === 0) return undefined;
175
- const ref = actionCounts(reference);
176
- let matched = 0;
177
- for (const a of candidate) {
178
- const remaining = ref.get(a) ?? 0;
179
- if (remaining > 0) {
180
- matched++;
181
- ref.set(a, remaining - 1);
182
- }
183
- }
184
- return matched / candidate.length;
185
- }
186
-
187
- /**
188
- * The candidate actions **not** justified by the reference (multiset difference,
189
- * sorted) — i.e. the spurious/hallucinated commands a parse or render introduced.
190
- * Empty when the candidate is a structural subset of the reference. Use for
191
- * diagnostics ("which command was hallucinated?"), complementing the
192
- * {@link computePrecision} score.
193
- */
194
- export function spuriousActions(
195
- reference: readonly string[],
196
- candidate: readonly string[]
197
- ): string[] {
198
- const ref = actionCounts(reference);
199
- const extras: string[] = [];
200
- for (const a of candidate) {
201
- const remaining = ref.get(a) ?? 0;
202
- if (remaining > 0) ref.set(a, remaining - 1);
203
- else extras.push(a);
204
- }
205
- return extras.sort();
206
- }
207
-
208
- /**
209
- * R1 — role-fidelity signature.
210
- *
211
- * Action-set fidelity cannot see a parse that finds the right commands with the
212
- * WRONG roles: a swapped patient/destination executes wrongly while scoring 1.0.
213
- * This signature captures, for every command node in the tree, which roles were
214
- * filled and with what value *type* (`add.patient:selector`,
215
- * `put.destination:reference`). Cross-language comparison is by role name +
216
- * value type — never by value string, which is legitimately translated.
217
- * The roles container is a ReadonlyMap on live nodes (serializes to {} in JSON),
218
- * so the signature must be collected at validation time, not from results.json.
219
- */
220
- export function collectRoleSignature(node: unknown): string[] {
221
- const acc = new Set<string>();
222
- walkRoles(node, acc, 0);
223
- return [...acc].sort();
224
- }
225
-
226
- function walkRoles(node: unknown, acc: Set<string>, depth: number): void {
227
- if (depth > 64 || node === null || typeof node !== 'object') return;
228
-
229
- const rec = node as Record<string, unknown>;
230
- const action = rec.action;
231
- if (typeof action === 'string' && !STRUCTURAL_ACTIONS.has(action)) {
232
- const roles = rec.roles;
233
- const entries: Array<[unknown, unknown]> =
234
- roles instanceof Map
235
- ? [...roles.entries()]
236
- : roles && typeof roles === 'object'
237
- ? Object.entries(roles)
238
- : [];
239
- for (const [role, value] of entries) {
240
- if (value === undefined || value === null) continue;
241
- const kind =
242
- typeof value === 'object' && typeof (value as { type?: unknown }).type === 'string'
243
- ? (value as { type: string }).type
244
- : typeof value;
245
- acc.add(`${action}.${String(role)}:${kind}`);
246
- }
247
- }
248
-
249
- for (const field of CHILD_FIELDS) {
250
- const child = rec[field];
251
- if (Array.isArray(child)) {
252
- for (const c of child) walkRoles(c, acc, depth + 1);
253
- } else if (child && typeof child === 'object') {
254
- walkRoles(child, acc, depth + 1);
255
- }
256
- }
257
- }
258
-
259
- /**
260
- * Role-value kinds never emitted by the R3 walker. `reference` values (`me`,
261
- * `it`, …) are mostly `fillSchemaDefaults` injections present identically on
262
- * both sides — noise. `flag` names and `property-path` properties are bare
263
- * identifiers, excluded by v1 (see {@link collectRoleValueSignature}).
264
- */
265
- const VALUE_KIND_EXCLUSIONS = new Set(['reference', 'flag', 'property-path']);
266
-
267
- /**
268
- * A role value is compared cross-language only when its WHOLE surface form is
269
- * code-shaped — language-invariant by construction, never legitimately
270
- * translated. Conservative v1 whitelist; when a sweep firing turns out to be a
271
- * legit translation difference, tighten here and document the exclusion.
272
- */
273
- const INVARIANT_VALUE_PATTERNS: readonly RegExp[] = [
274
- /^[#.[<@*]/, // selectors: #id .class [attr] <tag/> @attr *style
275
- /^[:$^][A-Za-z_]\w*$/, // sigil refs: :local $global ^element
276
- /^\d+(\.\d+)?(ms|s|m|h)?$/, // numbers and time literals: 2 1.5 200ms
277
- /^[A-Za-z_][\w-]*:[\w-]+$/, // colon-qualified event names: draggable:start
278
- /^(\.{0,2}\/|https?:)/, // URLs / paths: /api/data ./x ../y https://…
279
- ];
280
-
281
- function isInvariantSurface(surface: string): boolean {
282
- // Whole-surface rule, sweep-validated exclusions:
283
- // - whitespace ⇒ mixed content. `if #modal exists` captures its condition as
284
- // `#modal exists` — starts selector-shaped, but `exists` is prose that every
285
- // language legitimately translates (16-language false firing without this).
286
- // - `${` ⇒ template interpolation. `/api/search?q=${my value}` is an
287
- // expression, and tokenizers split it at different points per language, so
288
- // the captured surface isn't comparable verbatim.
289
- if (/\s/.test(surface) || surface.includes('${')) return false;
290
- return INVARIANT_VALUE_PATTERNS.some(re => re.test(surface));
291
- }
292
-
293
- /**
294
- * The comparable surface form of a role value, or `undefined` when the value
295
- * carries none: `.value` for `literal`/`selector` (string/number/boolean
296
- * coerced), `.raw` for `expression`. Kinds in {@link VALUE_KIND_EXCLUSIONS}
297
- * and values without a string `type` discriminator yield `undefined`.
298
- */
299
- function roleValueSurface(value: unknown): string | undefined {
300
- if (value === null || typeof value !== 'object') return undefined;
301
- const rec = value as { type?: unknown; value?: unknown; raw?: unknown };
302
- if (typeof rec.type !== 'string' || VALUE_KIND_EXCLUSIONS.has(rec.type)) return undefined;
303
- if (rec.type === 'expression') {
304
- return typeof rec.raw === 'string' ? rec.raw : undefined;
305
- }
306
- const v = rec.value;
307
- return typeof v === 'string' || typeof v === 'number' || typeof v === 'boolean'
308
- ? String(v)
309
- : undefined;
310
- }
311
-
312
- /**
313
- * R3 — role-VALUE signature (invariant values only, multiset).
314
- *
315
- * R0/R1 compare actions and role *types*; values are never compared because
316
- * they are legitimately translated. That leaves a live defect class with zero
317
- * signal: right action counts, right role types, wrong role VALUE — the #633
318
- * class, where ms captured `trigger` events named `draggable` instead of
319
- * `draggable:start` (correct multiset, correct `trigger.event:literal`
320
- * signature, silently wrong runtime behavior), and 18 other languages carried
321
- * the same corruption with no side-effect at all.
322
- *
323
- * The subset of values compared here is language-invariant by construction —
324
- * code, not prose (see {@link INVARIANT_VALUE_PATTERNS}). Emits a **multiset**
325
- * of `` `action.role=value` `` entries (duplicates preserved, sorted); score
326
- * with {@link computeMultisetRecall} so a dropped duplicate value is visible.
327
- *
328
- * Deliberately EXCLUDED in v1: bare-word identifiers (`startX` — usually
329
- * invariant, but property/variable names occasionally get localized in seeds),
330
- * string literals (message strings are legitimately translated), expression
331
- * raws mixing native words + code (`次 .item`), and `reference` values
332
- * (`me`/`it` — mostly `fillSchemaDefaults` injections on both sides).
333
- *
334
- * Blind spot: recall fires when a *translation* loses/corrupts an invariant
335
- * value. If the **en reference itself** corrupts a value, every language flags
336
- * at once — a 24-language R3 firestorm on one pattern means "suspect the en
337
- * parse first" (unlike R0, where en corruption moves nothing).
338
- *
339
- * The roles container is a ReadonlyMap on live nodes (serializes to {} in
340
- * results.json), so collect at validation time, never from the JSON.
341
- */
342
- export function collectRoleValueSignature(node: unknown): string[] {
343
- const acc: string[] = [];
344
- walkRoleValues(node, acc, 0);
345
- return acc.sort();
346
- }
347
-
348
- function walkRoleValues(node: unknown, acc: string[], depth: number): void {
349
- if (depth > 64 || node === null || typeof node !== 'object') return;
350
-
351
- const rec = node as Record<string, unknown>;
352
- const action = rec.action;
353
- if (typeof action === 'string' && !STRUCTURAL_ACTIONS.has(action)) {
354
- const roles = rec.roles;
355
- const entries: Array<[unknown, unknown]> =
356
- roles instanceof Map
357
- ? [...roles.entries()]
358
- : roles && typeof roles === 'object'
359
- ? Object.entries(roles)
360
- : [];
361
- for (const [role, value] of entries) {
362
- if (value === undefined || value === null) continue;
363
- const surface = roleValueSurface(value);
364
- if (surface === undefined || !isInvariantSurface(surface)) continue;
365
- acc.push(`${action}.${String(role)}=${surface}`);
366
- }
367
- }
368
-
369
- for (const field of CHILD_FIELDS) {
370
- const child = rec[field];
371
- if (Array.isArray(child)) {
372
- for (const c of child) walkRoleValues(c, acc, depth + 1);
373
- } else if (child && typeof child === 'object') {
374
- walkRoleValues(child, acc, depth + 1);
375
- }
376
- }
377
- }
2
+ * Structural fidelity scorers — canonical home is `@lokascript/semantic/fidelity`
3
+ * (moved there in agent-era arc 4 so any consumer can score a candidate parse
4
+ * against a reference without pulling in this test framework; see
5
+ * docs/FIDELITY.md and docs-internal/AGENT_ERA_ROADMAP.md).
6
+ *
7
+ * This module re-exports the full surface so every in-repo consumer
8
+ * (orchestrator, parse-validator, triage tools, the fidelity test suite) keeps
9
+ * its import path — and the test suite now exercises the extracted module
10
+ * through this shim, which keeps the two from drifting.
11
+ */
12
+
13
+ export {
14
+ FIDELITY_THRESHOLD,
15
+ collectActions,
16
+ collectActionsMultiset,
17
+ computeFidelity,
18
+ computeMultisetRecall,
19
+ computePrecision,
20
+ spuriousActions,
21
+ collectRoleSignature,
22
+ collectRoleValueSignature,
23
+ } from '@lokascript/semantic/fidelity';
@@ -87,16 +87,21 @@ describe('shipped-examples execution gate', () => {
87
87
  }, 240_000);
88
88
 
89
89
  it('walks pages and compares handlers (sanity: extraction and both engines working)', () => {
90
- // Floors well below current values (55 / 333 / 162 / 74) but far above
90
+ // Floors well below current values (48 / 261 / 122 / 52) but far above
91
91
  // zero: a broken walk, extractor, or engine bootstrap fails loudly here
92
92
  // instead of making assertions 2-3 vacuously pass.
93
- expect(result.pages).toBeGreaterThan(40);
94
- expect(result.handlers).toBeGreaterThan(250);
95
- expect(result.compared.length).toBeGreaterThan(120);
93
+ //
94
+ // Calibrated against the git-TRACKED corpus — the sweep ignores untracked
95
+ // examples/ dirs, so these numbers are the same on every machine and in
96
+ // CI. (The original floors were measured on a working tree with four
97
+ // gitignored dirs present and failed every clean checkout — #862.)
98
+ expect(result.pages).toBeGreaterThan(35);
99
+ expect(result.handlers).toBeGreaterThan(200);
100
+ expect(result.compared.length).toBeGreaterThan(90);
96
101
  // Vacuous (empty-vs-empty) pairs are NOT parity evidence — the floor is on
97
102
  // real, non-empty signature matches.
98
103
  const realMatches = result.compared.filter(c => c.match && !c.vacuous).length;
99
- expect(realMatches).toBeGreaterThan(60);
104
+ expect(realMatches).toBeGreaterThan(40);
100
105
  });
101
106
 
102
107
  it('has no NEW divergence from upstream outside the allowlist', () => {