@hyperfixi/testing-framework 2.5.1 → 2.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,136 @@
1
+ import { describe, it, expect } from 'vitest';
2
+ import {
3
+ collectActions,
4
+ collectActionsMultiset,
5
+ computeFidelity,
6
+ computePrecision,
7
+ spuriousActions,
8
+ FIDELITY_THRESHOLD,
9
+ } from './fidelity';
10
+
11
+ describe('collectActions', () => {
12
+ it('collects distinct actions across a nested event-handler tree', () => {
13
+ // Shape of the English `focus-trap` parse: on { if ; focus ; halt }.
14
+ const node = {
15
+ kind: 'event-handler',
16
+ action: 'on',
17
+ body: [
18
+ {
19
+ kind: 'compound',
20
+ action: 'compound',
21
+ statements: [
22
+ { kind: 'command', action: 'if' },
23
+ { kind: 'command', action: 'focus' },
24
+ { kind: 'command', action: 'halt' },
25
+ ],
26
+ },
27
+ ],
28
+ };
29
+ expect(collectActions(node)).toEqual(['focus', 'halt', 'if', 'on']);
30
+ });
31
+
32
+ it('excludes the structural `compound` wrapper', () => {
33
+ const node = { action: 'compound', statements: [{ action: 'toggle' }] };
34
+ expect(collectActions(node)).toEqual(['toggle']);
35
+ });
36
+
37
+ it('handles a flat command and non-object input', () => {
38
+ expect(collectActions({ kind: 'command', action: 'toggle' })).toEqual(['toggle']);
39
+ expect(collectActions(null)).toEqual([]);
40
+ expect(collectActions(undefined)).toEqual([]);
41
+ });
42
+ });
43
+
44
+ describe('computeFidelity', () => {
45
+ const en = ['focus', 'halt', 'if', 'on'];
46
+
47
+ it('scores a faithful (reordered) parse 1.0', () => {
48
+ // SOV/VSO reorder keeps the same command set → full fidelity.
49
+ expect(computeFidelity(en, ['on', 'if', 'halt', 'focus'])).toBe(1);
50
+ });
51
+
52
+ it('scores a degenerate parse low (focus-trap dropping commands)', () => {
53
+ // ja `focus-trap` parsed as `on { from }` — only `on` survives of 4.
54
+ const score = computeFidelity(en, ['on', 'from']);
55
+ expect(score).toBeCloseTo(0.25);
56
+ expect(score! < FIDELITY_THRESHOLD).toBe(true);
57
+ });
58
+
59
+ it('is recall, not precision — extra candidate actions do not lower it', () => {
60
+ expect(computeFidelity(['toggle'], ['toggle', 'add', 'remove'])).toBe(1);
61
+ });
62
+
63
+ it('returns undefined when there is no reference to score against', () => {
64
+ expect(computeFidelity([], ['on'])).toBeUndefined();
65
+ });
66
+ });
67
+
68
+ describe('collectActionsMultiset', () => {
69
+ it('preserves duplicate actions the Set-based collector drops', () => {
70
+ // The ja/ko/tr round-trip of `on click toggle .open then put "hi" into #out`
71
+ // renders a phantom `toggle` ahead of the real one: [on, toggle, toggle, put].
72
+ const node = {
73
+ kind: 'event-handler',
74
+ action: 'on',
75
+ body: [
76
+ { kind: 'command', action: 'toggle' }, // phantom, injected by the renderer
77
+ { kind: 'command', action: 'toggle' }, // the real one
78
+ { kind: 'command', action: 'put' },
79
+ ],
80
+ };
81
+ expect(collectActionsMultiset(node)).toEqual(['on', 'put', 'toggle', 'toggle']);
82
+ // The existing Set-based collector cannot see the duplicate:
83
+ expect(collectActions(node)).toEqual(['on', 'put', 'toggle']);
84
+ });
85
+
86
+ it('still excludes the structural `compound` wrapper', () => {
87
+ const node = { action: 'compound', statements: [{ action: 'toggle' }, { action: 'toggle' }] };
88
+ expect(collectActionsMultiset(node)).toEqual(['toggle', 'toggle']);
89
+ });
90
+ });
91
+
92
+ describe('computePrecision', () => {
93
+ it('scores a faithful (reordered) parse 1.0', () => {
94
+ const en = ['focus', 'halt', 'if', 'on'];
95
+ expect(computePrecision(en, ['on', 'if', 'halt', 'focus'])).toBe(1);
96
+ });
97
+
98
+ it('catches the phantom `toggle` recall is blind to (the renderer bug)', () => {
99
+ // Real shape: EN `on click add .x then remove .y` round-tripped through ja
100
+ // renders as [on, toggle, add, remove] — recall stays 1.0, precision drops.
101
+ const en = ['add', 'on', 'remove'];
102
+ const ja = ['add', 'on', 'remove', 'toggle'];
103
+ expect(computeFidelity(en, ja)).toBe(1); // recall is fooled
104
+ expect(computePrecision(en, ja)).toBeCloseTo(3 / 4); // precision is not
105
+ });
106
+
107
+ it('penalizes a duplicated spurious action (multiset)', () => {
108
+ // [toggle, toggle, put] vs reference [put, toggle]: one toggle is spurious.
109
+ expect(computePrecision(['put', 'toggle'], ['put', 'toggle', 'toggle'])).toBeCloseTo(2 / 3);
110
+ });
111
+
112
+ it('scores 0 when every candidate action is spurious (ar/ru substitutive case)', () => {
113
+ // ar `on click if … end` collapsed to just a phantom [toggle].
114
+ expect(computePrecision(['if', 'on'], ['toggle'])).toBe(0);
115
+ });
116
+
117
+ it('returns undefined when there is no candidate to score', () => {
118
+ expect(computePrecision(['on'], [])).toBeUndefined();
119
+ });
120
+ });
121
+
122
+ describe('spuriousActions', () => {
123
+ it('lists the hallucinated commands a render/parse introduced', () => {
124
+ expect(spuriousActions(['add', 'on', 'remove'], ['add', 'on', 'remove', 'toggle'])).toEqual([
125
+ 'toggle',
126
+ ]);
127
+ });
128
+
129
+ it('counts duplicates as spurious beyond the reference multiset', () => {
130
+ expect(spuriousActions(['put', 'toggle'], ['put', 'toggle', 'toggle'])).toEqual(['toggle']);
131
+ });
132
+
133
+ it('is empty for a faithful subset/reorder', () => {
134
+ expect(spuriousActions(['focus', 'halt', 'if', 'on'], ['on', 'if', 'focus'])).toEqual([]);
135
+ });
136
+ });
@@ -0,0 +1,224 @@
1
+ /**
2
+ * Structural fidelity for multilingual parses.
3
+ *
4
+ * The parse-validator's success metric is "the parser returned a non-null node".
5
+ * That conflates a faithful parse with a *degenerate* one — a translated pattern
6
+ * can parse non-null while silently dropping most of the source's commands (e.g.
7
+ * `focus-trap` parses as a bare `if` or a stray `from` in several languages, with
8
+ * the `focus`/`halt`/condition lost). This module derives a lightweight
9
+ * structural signature from a parsed node and scores a translation's parse against
10
+ * the English reference parse, so those degenerate passes are visible.
11
+ *
12
+ * The signature is intentionally word-order agnostic: it is the *set* of command
13
+ * actions in the node tree, so a faithful SOV/VSO reorder scores 1.0 while a parse
14
+ * that loses commands scores low.
15
+ */
16
+
17
+ /** Passes scoring below this are flagged as degenerate (lost >half the structure). */
18
+ export const FIDELITY_THRESHOLD = 0.5;
19
+
20
+ /** The structural `compound` wrapper is not a command; never counts as an action. */
21
+ const STRUCTURAL_ACTIONS = new Set(['compound']);
22
+
23
+ /** Node-array fields the walk recurses into (event/loop/conditional/behavior bodies). */
24
+ const CHILD_FIELDS = [
25
+ 'body',
26
+ 'statements',
27
+ 'thenBranch',
28
+ 'elseBranch',
29
+ 'branches',
30
+ 'eventHandlers',
31
+ 'initBlock',
32
+ ] as const;
33
+
34
+ /**
35
+ * Collect the distinct command actions anywhere in a parsed semantic node tree
36
+ * (top-level action + nested body/statements/branches), excluding the structural
37
+ * `compound` wrapper. Returns a sorted array for stable comparison/serialization.
38
+ */
39
+ export function collectActions(node: unknown): string[] {
40
+ const acc = new Set<string>();
41
+ walk(node, acc, 0);
42
+ return [...acc].sort();
43
+ }
44
+
45
+ function walk(node: unknown, acc: Set<string>, depth: number): void {
46
+ // Guard against pathological/cyclic structures.
47
+ if (depth > 64 || node === null || typeof node !== 'object') return;
48
+
49
+ const rec = node as Record<string, unknown>;
50
+ const action = rec.action;
51
+ if (typeof action === 'string' && !STRUCTURAL_ACTIONS.has(action)) {
52
+ acc.add(action);
53
+ }
54
+
55
+ for (const field of CHILD_FIELDS) {
56
+ const child = rec[field];
57
+ if (Array.isArray(child)) {
58
+ for (const c of child) walk(c, acc, depth + 1);
59
+ } else if (child && typeof child === 'object') {
60
+ walk(child, acc, depth + 1);
61
+ }
62
+ }
63
+ }
64
+
65
+ /**
66
+ * Like {@link collectActions} but **multiset**: duplicates are preserved (sorted,
67
+ * not deduped). Needed for precision — a *duplicate* spurious command (e.g. a
68
+ * renderer that injects a phantom `toggle` ahead of a real `toggle`, yielding
69
+ * `[toggle, toggle, put]`) is invisible to the Set-based `collectActions`.
70
+ *
71
+ * Kept as a parallel walk rather than refactoring `collectActions`: the latter
72
+ * feeds the committed regression baseline, so its traversal stays byte-identical.
73
+ */
74
+ export function collectActionsMultiset(node: unknown): string[] {
75
+ const acc: string[] = [];
76
+ walkMultiset(node, acc, 0);
77
+ return acc.sort();
78
+ }
79
+
80
+ function walkMultiset(node: unknown, acc: string[], depth: number): void {
81
+ if (depth > 64 || node === null || typeof node !== 'object') return;
82
+
83
+ const rec = node as Record<string, unknown>;
84
+ const action = rec.action;
85
+ if (typeof action === 'string' && !STRUCTURAL_ACTIONS.has(action)) {
86
+ acc.push(action);
87
+ }
88
+
89
+ for (const field of CHILD_FIELDS) {
90
+ const child = rec[field];
91
+ if (Array.isArray(child)) {
92
+ for (const c of child) walkMultiset(c, acc, depth + 1);
93
+ } else if (child && typeof child === 'object') {
94
+ walkMultiset(child, acc, depth + 1);
95
+ }
96
+ }
97
+ }
98
+
99
+ /** Build a multiset count map from an action list. */
100
+ function actionCounts(items: readonly string[]): Map<string, number> {
101
+ const m = new Map<string, number>();
102
+ for (const it of items) m.set(it, (m.get(it) ?? 0) + 1);
103
+ return m;
104
+ }
105
+
106
+ /**
107
+ * Structural fidelity in [0, 1]: the fraction of the reference (English) actions
108
+ * also present in the candidate (recall). Returns `undefined` when the reference
109
+ * has no actions to compare against.
110
+ */
111
+ export function computeFidelity(
112
+ reference: readonly string[],
113
+ candidate: readonly string[]
114
+ ): number | undefined {
115
+ if (reference.length === 0) return undefined;
116
+ const cand = new Set(candidate);
117
+ let hits = 0;
118
+ for (const a of reference) {
119
+ if (cand.has(a)) hits++;
120
+ }
121
+ return hits / reference.length;
122
+ }
123
+
124
+ /**
125
+ * Structural **precision** in [0, 1]: the fraction of the *candidate's* actions
126
+ * that are justified by the reference (multiset-aware). The complement of
127
+ * {@link computeFidelity}'s recall — it falls below 1.0 when a parse/render adds
128
+ * commands the source never had.
129
+ *
130
+ * This is the signal recall + role-fidelity (R1) cannot see: a renderer that
131
+ * injects a phantom `toggle` (ja/ko/tr/ar/ru event-handler rendering) keeps recall
132
+ * 1.0 while precision drops. Pass multisets (see {@link collectActionsMultiset})
133
+ * so a *duplicated* spurious action is counted, not absorbed.
134
+ *
135
+ * Returns `undefined` when the candidate has no actions to score.
136
+ */
137
+ export function computePrecision(
138
+ reference: readonly string[],
139
+ candidate: readonly string[]
140
+ ): number | undefined {
141
+ if (candidate.length === 0) return undefined;
142
+ const ref = actionCounts(reference);
143
+ let matched = 0;
144
+ for (const a of candidate) {
145
+ const remaining = ref.get(a) ?? 0;
146
+ if (remaining > 0) {
147
+ matched++;
148
+ ref.set(a, remaining - 1);
149
+ }
150
+ }
151
+ return matched / candidate.length;
152
+ }
153
+
154
+ /**
155
+ * The candidate actions **not** justified by the reference (multiset difference,
156
+ * sorted) — i.e. the spurious/hallucinated commands a parse or render introduced.
157
+ * Empty when the candidate is a structural subset of the reference. Use for
158
+ * diagnostics ("which command was hallucinated?"), complementing the
159
+ * {@link computePrecision} score.
160
+ */
161
+ export function spuriousActions(
162
+ reference: readonly string[],
163
+ candidate: readonly string[]
164
+ ): string[] {
165
+ const ref = actionCounts(reference);
166
+ const extras: string[] = [];
167
+ for (const a of candidate) {
168
+ const remaining = ref.get(a) ?? 0;
169
+ if (remaining > 0) ref.set(a, remaining - 1);
170
+ else extras.push(a);
171
+ }
172
+ return extras.sort();
173
+ }
174
+
175
+ /**
176
+ * R1 — role-fidelity signature.
177
+ *
178
+ * Action-set fidelity cannot see a parse that finds the right commands with the
179
+ * WRONG roles: a swapped patient/destination executes wrongly while scoring 1.0.
180
+ * This signature captures, for every command node in the tree, which roles were
181
+ * filled and with what value *type* (`add.patient:selector`,
182
+ * `put.destination:reference`). Cross-language comparison is by role name +
183
+ * value type — never by value string, which is legitimately translated.
184
+ * The roles container is a ReadonlyMap on live nodes (serializes to {} in JSON),
185
+ * so the signature must be collected at validation time, not from results.json.
186
+ */
187
+ export function collectRoleSignature(node: unknown): string[] {
188
+ const acc = new Set<string>();
189
+ walkRoles(node, acc, 0);
190
+ return [...acc].sort();
191
+ }
192
+
193
+ function walkRoles(node: unknown, acc: Set<string>, depth: number): void {
194
+ if (depth > 64 || node === null || typeof node !== 'object') return;
195
+
196
+ const rec = node as Record<string, unknown>;
197
+ const action = rec.action;
198
+ if (typeof action === 'string' && !STRUCTURAL_ACTIONS.has(action)) {
199
+ const roles = rec.roles;
200
+ const entries: Array<[unknown, unknown]> =
201
+ roles instanceof Map
202
+ ? [...roles.entries()]
203
+ : roles && typeof roles === 'object'
204
+ ? Object.entries(roles)
205
+ : [];
206
+ for (const [role, value] of entries) {
207
+ if (value === undefined || value === null) continue;
208
+ const kind =
209
+ typeof value === 'object' && typeof (value as { type?: unknown }).type === 'string'
210
+ ? (value as { type: string }).type
211
+ : typeof value;
212
+ acc.add(`${action}.${String(role)}:${kind}`);
213
+ }
214
+ }
215
+
216
+ for (const field of CHILD_FIELDS) {
217
+ const child = rec[field];
218
+ if (Array.isArray(child)) {
219
+ for (const c of child) walkRoles(c, acc, depth + 1);
220
+ } else if (child && typeof child === 'object') {
221
+ walkRoles(child, acc, depth + 1);
222
+ }
223
+ }
224
+ }
@@ -0,0 +1,39 @@
1
+ /**
2
+ * Pattern Loader Tests
3
+ *
4
+ * Covers isHtmlMarkupPattern: the guard that keeps HTML-markup patterns out of
5
+ * the semantic-parse denominator (they're validated by the DOM/Playwright
6
+ * suites, not the hyperscript text parser).
7
+ */
8
+ import { describe, it, expect } from 'vitest';
9
+ import { isHtmlMarkupPattern } from './html-pattern';
10
+
11
+ describe('isHtmlMarkupPattern', () => {
12
+ it('flags HTML-markup patterns (excluded from semantic parsing)', () => {
13
+ const htmlPatterns = [
14
+ '<div hx-live="put $count into me"></div>',
15
+ '<div sse-connect="/events" sse-swap="tick" hx-target="#feed"></div>',
16
+ '<div ws-connect="wss://example/api"><form ws-send></form></div>',
17
+ '<script type="text/hyperscript-template" component="hello-world"><span>Hi</span></script>',
18
+ ' <button _="on click increment ^count">+</button>', // leading whitespace
19
+ ];
20
+ for (const code of htmlPatterns) {
21
+ expect(isHtmlMarkupPattern(code), code).toBe(true);
22
+ }
23
+ });
24
+
25
+ it('does not flag real hyperscript source', () => {
26
+ const hyperscript = [
27
+ 'toggle .active on #button',
28
+ 'on click increment $count',
29
+ 'put :x into me',
30
+ 'bind $greeting to #name-input',
31
+ 'live put `Count: ${$count}` into me end',
32
+ 'fetch /api/data as json then put it into #out',
33
+ '#count を 増加', // SOV, non-Latin
34
+ ];
35
+ for (const code of hyperscript) {
36
+ expect(isHtmlMarkupPattern(code), code).toBe(false);
37
+ }
38
+ });
39
+ });
@@ -0,0 +1,24 @@
1
+ /**
2
+ * HTML-markup pattern detection.
3
+ *
4
+ * A handful of DB patterns (`hx-live`, `sse-connect`/`ws-connect`, and the
5
+ * `<script type="text/hyperscript-template">` component patterns) are authored
6
+ * as HTML markup — the hyperscript they exercise lives inside attributes
7
+ * (`hx-live="…"`, `_="…"`) and is processed at runtime by the htmx-compat /
8
+ * component layer, which is covered by the DOM + Playwright suites.
9
+ *
10
+ * The multilingual harness validates patterns with the hyperscript *text*
11
+ * parser, which structurally cannot parse HTML — grading these is a category
12
+ * error that understates the real parse rate, so they are excluded from the
13
+ * semantic-parse denominator.
14
+ *
15
+ * Top-level hyperscript never begins with `<` (commands start with a
16
+ * verb/event keyword or a `.`/`#`/`:`/`$` token), so a leading `<` is a
17
+ * reliable, self-maintaining signal — new HTML patterns are excluded
18
+ * automatically without a manual allow/deny list.
19
+ *
20
+ * Kept dependency-free so it is unit-testable without the patterns DB.
21
+ */
22
+ export function isHtmlMarkupPattern(code: string): boolean {
23
+ return /^\s*</.test(code);
24
+ }
@@ -7,11 +7,20 @@ import { promisify } from 'node:util';
7
7
  import { loadPatterns } from './pattern-loader';
8
8
  import { selectBundle, getBundleInfo } from './bundle-builder';
9
9
  import { ParseValidator } from './validators/parse-validator';
10
+ import { ExecutionValidator, loadExecutionSubset } from './validators/execution-validator';
10
11
  import { SizeValidator } from './validators/size-validator';
11
12
  import { ConsoleReporter } from './reporters/console-reporter';
12
13
  import { JSONReporter } from './reporters/json-reporter';
13
14
  import { RegressionReporter } from './reporters/regression-reporter';
14
- import type { TestConfig, TestResults, LanguageResults, LanguageCode, Reporter } from './types';
15
+ import type {
16
+ TestConfig,
17
+ TestResults,
18
+ LanguageResults,
19
+ LanguageCode,
20
+ Reporter,
21
+ BundleInfo,
22
+ } from './types';
23
+ import { computeFidelity, computePrecision, FIDELITY_THRESHOLD } from './fidelity';
15
24
 
16
25
  const execAsync = promisify(exec);
17
26
 
@@ -48,9 +57,12 @@ export class TestOrchestrator {
48
57
  // Add JSON reporter for structured output
49
58
  this.reporters.push(new JSONReporter('./test-results/results.json'));
50
59
 
51
- // Add regression reporter if requested
60
+ // Add regression reporter if requested. Default to the COMMITTED baseline
61
+ // (test-results/ is gitignored), overridable via --baseline.
52
62
  if (this.config.regression) {
53
- this.reporters.push(new RegressionReporter('./test-results/baseline.json'));
63
+ this.reporters.push(
64
+ new RegressionReporter(this.config.baselinePath ?? './baselines/multilingual-priority.json')
65
+ );
54
66
  }
55
67
  }
56
68
 
@@ -88,6 +100,15 @@ export class TestOrchestrator {
88
100
  languageResults.push(result);
89
101
  }
90
102
 
103
+ // Score structural fidelity against the English reference parse. This
104
+ // surfaces *degenerate* passes — patterns that parse non-null but drop most
105
+ // of the source's commands — which the parse-rate metric alone can't see.
106
+ this.scoreFidelity(languageResults);
107
+
108
+ // R2 — execution smoke: run the curated subset in jsdom and compare DOM
109
+ // effects against the en reference's (see execution-validator.ts).
110
+ await this.scoreExecution(languageResults);
111
+
91
112
  // Collect bundle information
92
113
  const bundles: TestResults['bundles'] = {};
93
114
  for (const langResult of languageResults) {
@@ -158,13 +179,25 @@ export class TestOrchestrator {
158
179
  private async testLanguage(language: LanguageCode, patterns: any[]): Promise<LanguageResults> {
159
180
  const startTime = performance.now();
160
181
 
161
- // Select bundle for this language
162
- const bundle = this.config.bundle
182
+ // Select bundle for this language. The bundle is consumed only for size
183
+ // reporting — parse validation below runs in-process via parseSemantic — so a
184
+ // missing display bundle is a warning, not a fatal error that aborts the sweep.
185
+ const selected = this.config.bundle
163
186
  ? await getBundleInfo(this.config.bundle)
164
187
  : await selectBundle([language], this.config.build || false);
165
188
 
166
- if (!bundle || !bundle.exists) {
167
- throw new Error(`Bundle not found for language: ${language}`);
189
+ const bundle: BundleInfo = selected ?? {
190
+ name: `browser-${language}`,
191
+ path: '',
192
+ languages: [language],
193
+ size: 0,
194
+ exists: false,
195
+ };
196
+
197
+ if (!bundle.exists) {
198
+ console.warn(
199
+ `⚠ No display bundle for '${language}' (${bundle.name}); reporting size 0 and continuing with in-process parse.`
200
+ );
168
201
  }
169
202
 
170
203
  // Notify reporters
@@ -209,6 +242,143 @@ export class TestOrchestrator {
209
242
  return result;
210
243
  }
211
244
 
245
+ /**
246
+ * Score structural fidelity of every parse against the English reference parse
247
+ * of the same pattern, in-place. Fills `ParseResult.fidelity` plus per-language
248
+ * `avgFidelity` / `degeneratePasses`. English is the reference, so its own
249
+ * results are left unscored. No-op when English wasn't part of the run.
250
+ */
251
+ private scoreFidelity(languageResults: LanguageResults[]): void {
252
+ const en = languageResults.find(r => r.language === 'en');
253
+ if (!en) return;
254
+
255
+ // codeExampleId -> English action signature (only successful en parses).
256
+ const reference = new Map<string, string[]>();
257
+ // R0-precision: codeExampleId -> English action MULTISET (duplicates kept).
258
+ const multisetReference = new Map<string, string[]>();
259
+ // R1: codeExampleId -> English role signature (action.role:valueType set).
260
+ const roleReference = new Map<string, string[]>();
261
+ for (const r of en.parseResults) {
262
+ if (r.success && r.actionSignature && r.actionSignature.length > 0) {
263
+ reference.set(r.pattern.codeExampleId, r.actionSignature);
264
+ }
265
+ if (r.success && r.actionMultisetSignature && r.actionMultisetSignature.length > 0) {
266
+ multisetReference.set(r.pattern.codeExampleId, r.actionMultisetSignature);
267
+ }
268
+ if (r.success && r.roleSignature && r.roleSignature.length > 0) {
269
+ roleReference.set(r.pattern.codeExampleId, r.roleSignature);
270
+ }
271
+ }
272
+
273
+ for (const lang of languageResults) {
274
+ if (lang.language === 'en') continue;
275
+
276
+ const degenerate: string[] = [];
277
+ const lossy: string[] = [];
278
+ const scores: number[] = [];
279
+ const precisionScores: number[] = [];
280
+ const roleScores: number[] = [];
281
+
282
+ for (const result of lang.parseResults) {
283
+ if (!result.success || !result.actionSignature) continue;
284
+ const ref = reference.get(result.pattern.codeExampleId);
285
+ if (!ref) continue;
286
+
287
+ const fidelity = computeFidelity(ref, result.actionSignature);
288
+ if (fidelity === undefined) continue;
289
+
290
+ result.fidelity = fidelity;
291
+ scores.push(fidelity);
292
+ if (fidelity < FIDELITY_THRESHOLD) degenerate.push(result.pattern.codeExampleId);
293
+ else if (fidelity < 1) lossy.push(result.pattern.codeExampleId);
294
+
295
+ // R0-precision — fraction of THIS parse's actions justified by the en
296
+ // multiset reference (catches phantom/spurious commands recall misses).
297
+ const multisetRef = multisetReference.get(result.pattern.codeExampleId);
298
+ if (multisetRef && result.actionMultisetSignature) {
299
+ const precision = computePrecision(multisetRef, result.actionMultisetSignature);
300
+ if (precision !== undefined) {
301
+ result.precision = precision;
302
+ precisionScores.push(precision);
303
+ }
304
+ }
305
+
306
+ // R1 — role recall vs the en role signature (role name + value type).
307
+ const roleRef = roleReference.get(result.pattern.codeExampleId);
308
+ if (roleRef && result.roleSignature) {
309
+ const roleFidelity = computeFidelity(roleRef, result.roleSignature);
310
+ if (roleFidelity !== undefined) {
311
+ result.roleFidelity = roleFidelity;
312
+ roleScores.push(roleFidelity);
313
+ }
314
+ }
315
+ }
316
+
317
+ lang.avgFidelity =
318
+ scores.length > 0 ? scores.reduce((a, b) => a + b, 0) / scores.length : undefined;
319
+ lang.avgPrecision =
320
+ precisionScores.length > 0
321
+ ? precisionScores.reduce((a, b) => a + b, 0) / precisionScores.length
322
+ : undefined;
323
+ lang.avgRoleFidelity =
324
+ roleScores.length > 0
325
+ ? roleScores.reduce((a, b) => a + b, 0) / roleScores.length
326
+ : undefined;
327
+ lang.degeneratePasses = degenerate.sort();
328
+ lang.lossyPasses = lossy.sort();
329
+ }
330
+ }
331
+
332
+ /**
333
+ * R2 — execute the curated execution subset for every language in the run
334
+ * and score each translation's DOM-effect signature against the en
335
+ * reference's, in-place (`avgExecutionFidelity` / `executionFailures`).
336
+ * English is the reference and is left unscored; patterns whose en
337
+ * reference errors or produces no effects are excluded (no usable
338
+ * reference). No-op when English wasn't part of the run.
339
+ */
340
+ private async scoreExecution(languageResults: LanguageResults[]): Promise<void> {
341
+ const en = languageResults.find(r => r.language === 'en');
342
+ if (!en) return;
343
+
344
+ const sources = await loadExecutionSubset(languageResults.map(r => r.language));
345
+ const validator = new ExecutionValidator();
346
+ await validator.initialize();
347
+
348
+ // codeExampleId -> en effect signature (only clean, effectful references).
349
+ const reference = new Map<string, string[]>();
350
+ for (const [id, code] of sources.get('en') ?? []) {
351
+ const res = await validator.execute(id, code, 'en');
352
+ if (!res.error && res.effects.length > 0) reference.set(id, res.effects);
353
+ }
354
+ if (reference.size === 0) return;
355
+
356
+ for (const lang of languageResults) {
357
+ if (lang.language === 'en') continue;
358
+ const langSources = sources.get(lang.language);
359
+ if (!langSources) continue;
360
+
361
+ const failures: string[] = [];
362
+ let scored = 0;
363
+ let matched = 0;
364
+ for (const [id, refEffects] of reference) {
365
+ const code = langSources.get(id);
366
+ if (!code) continue; // no translation — excluded, not failed
367
+ const res = await validator.execute(id, code, lang.language);
368
+ // Match on DOM effects ONLY. Trapped runtime errors are diagnostic:
369
+ // their attribution rides on unhandled-rejection timing (racy), while
370
+ // the effect snapshot is synchronous and deterministic — and an AST
371
+ // mis-build that damages behavior shows up as differing effects anyway.
372
+ const match = JSON.stringify(res.effects) === JSON.stringify(refEffects);
373
+ scored++;
374
+ if (match) matched++;
375
+ else failures.push(id);
376
+ }
377
+ lang.avgExecutionFidelity = scored > 0 ? matched / scored : undefined;
378
+ lang.executionFailures = failures.sort();
379
+ }
380
+ }
381
+
212
382
  /**
213
383
  * Group patterns by language
214
384
  */
@@ -12,6 +12,9 @@ import {
12
12
  type Translation,
13
13
  } from '@hyperfixi/patterns-reference';
14
14
  import type { LanguageCode, PatternTranslation, TestConfig, SamplingStrategy } from './types';
15
+ import { isHtmlMarkupPattern } from './html-pattern';
16
+
17
+ export { isHtmlMarkupPattern };
15
18
 
16
19
  /**
17
20
  * Load patterns for testing based on configuration
@@ -25,13 +28,19 @@ export async function loadPatterns(config: TestConfig): Promise<PatternTranslati
25
28
  results.push(...translations);
26
29
  }
27
30
 
31
+ // Exclude HTML-markup patterns from the semantic-parse set (see
32
+ // isHtmlMarkupPattern): the hyperscript text parser can't grade HTML, so
33
+ // counting them would understate the real parse rate. They are validated by
34
+ // the DOM/Playwright + i18n-htmx suites instead.
35
+ const semanticPatterns = results.filter(p => !isHtmlMarkupPattern(p.hyperscript));
36
+
28
37
  // Apply sampling if in quick mode
29
38
  if (config.mode === 'quick') {
30
39
  const limit = config.quickModeLimit || 10;
31
- return samplePatterns(results, { type: 'stratified', perCategory: limit });
40
+ return samplePatterns(semanticPatterns, { type: 'stratified', perCategory: limit });
32
41
  }
33
42
 
34
- return results;
43
+ return semanticPatterns;
35
44
  }
36
45
 
37
46
  /**