@hyperfixi/testing-framework 2.5.1 → 2.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{assertions-CthVynH6.d.mts → assertions-CsGP61iW.d.mts} +1 -1
- package/dist/{assertions-CthVynH6.d.ts → assertions-CsGP61iW.d.ts} +1 -1
- package/dist/assertions.d.mts +1 -1
- package/dist/assertions.d.ts +1 -1
- package/dist/index.d.mts +2 -2
- package/dist/index.d.ts +2 -2
- package/package.json +10 -6
- package/src/multilingual/bundle-builder.ts +34 -16
- package/src/multilingual/cli.ts +357 -37
- package/src/multilingual/fidelity.test.ts +136 -0
- package/src/multilingual/fidelity.ts +224 -0
- package/src/multilingual/html-pattern.test.ts +39 -0
- package/src/multilingual/html-pattern.ts +24 -0
- package/src/multilingual/orchestrator.ts +177 -7
- package/src/multilingual/pattern-loader.ts +11 -2
- package/src/multilingual/reporters/console-reporter.ts +101 -0
- package/src/multilingual/reporters/regression-reporter.test.ts +406 -0
- package/src/multilingual/reporters/regression-reporter.ts +136 -1
- package/src/multilingual/types.ts +146 -0
- package/src/multilingual/validators/execution-validator.test.ts +412 -0
- package/src/multilingual/validators/execution-validator.ts +510 -0
- package/src/multilingual/validators/parse-validator.ts +22 -0
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
import { describe, it, expect } from 'vitest';
|
|
2
|
+
import {
|
|
3
|
+
collectActions,
|
|
4
|
+
collectActionsMultiset,
|
|
5
|
+
computeFidelity,
|
|
6
|
+
computePrecision,
|
|
7
|
+
spuriousActions,
|
|
8
|
+
FIDELITY_THRESHOLD,
|
|
9
|
+
} from './fidelity';
|
|
10
|
+
|
|
11
|
+
describe('collectActions', () => {
|
|
12
|
+
it('collects distinct actions across a nested event-handler tree', () => {
|
|
13
|
+
// Shape of the English `focus-trap` parse: on { if ; focus ; halt }.
|
|
14
|
+
const node = {
|
|
15
|
+
kind: 'event-handler',
|
|
16
|
+
action: 'on',
|
|
17
|
+
body: [
|
|
18
|
+
{
|
|
19
|
+
kind: 'compound',
|
|
20
|
+
action: 'compound',
|
|
21
|
+
statements: [
|
|
22
|
+
{ kind: 'command', action: 'if' },
|
|
23
|
+
{ kind: 'command', action: 'focus' },
|
|
24
|
+
{ kind: 'command', action: 'halt' },
|
|
25
|
+
],
|
|
26
|
+
},
|
|
27
|
+
],
|
|
28
|
+
};
|
|
29
|
+
expect(collectActions(node)).toEqual(['focus', 'halt', 'if', 'on']);
|
|
30
|
+
});
|
|
31
|
+
|
|
32
|
+
it('excludes the structural `compound` wrapper', () => {
|
|
33
|
+
const node = { action: 'compound', statements: [{ action: 'toggle' }] };
|
|
34
|
+
expect(collectActions(node)).toEqual(['toggle']);
|
|
35
|
+
});
|
|
36
|
+
|
|
37
|
+
it('handles a flat command and non-object input', () => {
|
|
38
|
+
expect(collectActions({ kind: 'command', action: 'toggle' })).toEqual(['toggle']);
|
|
39
|
+
expect(collectActions(null)).toEqual([]);
|
|
40
|
+
expect(collectActions(undefined)).toEqual([]);
|
|
41
|
+
});
|
|
42
|
+
});
|
|
43
|
+
|
|
44
|
+
describe('computeFidelity', () => {
|
|
45
|
+
const en = ['focus', 'halt', 'if', 'on'];
|
|
46
|
+
|
|
47
|
+
it('scores a faithful (reordered) parse 1.0', () => {
|
|
48
|
+
// SOV/VSO reorder keeps the same command set → full fidelity.
|
|
49
|
+
expect(computeFidelity(en, ['on', 'if', 'halt', 'focus'])).toBe(1);
|
|
50
|
+
});
|
|
51
|
+
|
|
52
|
+
it('scores a degenerate parse low (focus-trap dropping commands)', () => {
|
|
53
|
+
// ja `focus-trap` parsed as `on { from }` — only `on` survives of 4.
|
|
54
|
+
const score = computeFidelity(en, ['on', 'from']);
|
|
55
|
+
expect(score).toBeCloseTo(0.25);
|
|
56
|
+
expect(score! < FIDELITY_THRESHOLD).toBe(true);
|
|
57
|
+
});
|
|
58
|
+
|
|
59
|
+
it('is recall, not precision — extra candidate actions do not lower it', () => {
|
|
60
|
+
expect(computeFidelity(['toggle'], ['toggle', 'add', 'remove'])).toBe(1);
|
|
61
|
+
});
|
|
62
|
+
|
|
63
|
+
it('returns undefined when there is no reference to score against', () => {
|
|
64
|
+
expect(computeFidelity([], ['on'])).toBeUndefined();
|
|
65
|
+
});
|
|
66
|
+
});
|
|
67
|
+
|
|
68
|
+
describe('collectActionsMultiset', () => {
|
|
69
|
+
it('preserves duplicate actions the Set-based collector drops', () => {
|
|
70
|
+
// The ja/ko/tr round-trip of `on click toggle .open then put "hi" into #out`
|
|
71
|
+
// renders a phantom `toggle` ahead of the real one: [on, toggle, toggle, put].
|
|
72
|
+
const node = {
|
|
73
|
+
kind: 'event-handler',
|
|
74
|
+
action: 'on',
|
|
75
|
+
body: [
|
|
76
|
+
{ kind: 'command', action: 'toggle' }, // phantom, injected by the renderer
|
|
77
|
+
{ kind: 'command', action: 'toggle' }, // the real one
|
|
78
|
+
{ kind: 'command', action: 'put' },
|
|
79
|
+
],
|
|
80
|
+
};
|
|
81
|
+
expect(collectActionsMultiset(node)).toEqual(['on', 'put', 'toggle', 'toggle']);
|
|
82
|
+
// The existing Set-based collector cannot see the duplicate:
|
|
83
|
+
expect(collectActions(node)).toEqual(['on', 'put', 'toggle']);
|
|
84
|
+
});
|
|
85
|
+
|
|
86
|
+
it('still excludes the structural `compound` wrapper', () => {
|
|
87
|
+
const node = { action: 'compound', statements: [{ action: 'toggle' }, { action: 'toggle' }] };
|
|
88
|
+
expect(collectActionsMultiset(node)).toEqual(['toggle', 'toggle']);
|
|
89
|
+
});
|
|
90
|
+
});
|
|
91
|
+
|
|
92
|
+
describe('computePrecision', () => {
|
|
93
|
+
it('scores a faithful (reordered) parse 1.0', () => {
|
|
94
|
+
const en = ['focus', 'halt', 'if', 'on'];
|
|
95
|
+
expect(computePrecision(en, ['on', 'if', 'halt', 'focus'])).toBe(1);
|
|
96
|
+
});
|
|
97
|
+
|
|
98
|
+
it('catches the phantom `toggle` recall is blind to (the renderer bug)', () => {
|
|
99
|
+
// Real shape: EN `on click add .x then remove .y` round-tripped through ja
|
|
100
|
+
// renders as [on, toggle, add, remove] — recall stays 1.0, precision drops.
|
|
101
|
+
const en = ['add', 'on', 'remove'];
|
|
102
|
+
const ja = ['add', 'on', 'remove', 'toggle'];
|
|
103
|
+
expect(computeFidelity(en, ja)).toBe(1); // recall is fooled
|
|
104
|
+
expect(computePrecision(en, ja)).toBeCloseTo(3 / 4); // precision is not
|
|
105
|
+
});
|
|
106
|
+
|
|
107
|
+
it('penalizes a duplicated spurious action (multiset)', () => {
|
|
108
|
+
// [toggle, toggle, put] vs reference [put, toggle]: one toggle is spurious.
|
|
109
|
+
expect(computePrecision(['put', 'toggle'], ['put', 'toggle', 'toggle'])).toBeCloseTo(2 / 3);
|
|
110
|
+
});
|
|
111
|
+
|
|
112
|
+
it('scores 0 when every candidate action is spurious (ar/ru substitutive case)', () => {
|
|
113
|
+
// ar `on click if … end` collapsed to just a phantom [toggle].
|
|
114
|
+
expect(computePrecision(['if', 'on'], ['toggle'])).toBe(0);
|
|
115
|
+
});
|
|
116
|
+
|
|
117
|
+
it('returns undefined when there is no candidate to score', () => {
|
|
118
|
+
expect(computePrecision(['on'], [])).toBeUndefined();
|
|
119
|
+
});
|
|
120
|
+
});
|
|
121
|
+
|
|
122
|
+
describe('spuriousActions', () => {
|
|
123
|
+
it('lists the hallucinated commands a render/parse introduced', () => {
|
|
124
|
+
expect(spuriousActions(['add', 'on', 'remove'], ['add', 'on', 'remove', 'toggle'])).toEqual([
|
|
125
|
+
'toggle',
|
|
126
|
+
]);
|
|
127
|
+
});
|
|
128
|
+
|
|
129
|
+
it('counts duplicates as spurious beyond the reference multiset', () => {
|
|
130
|
+
expect(spuriousActions(['put', 'toggle'], ['put', 'toggle', 'toggle'])).toEqual(['toggle']);
|
|
131
|
+
});
|
|
132
|
+
|
|
133
|
+
it('is empty for a faithful subset/reorder', () => {
|
|
134
|
+
expect(spuriousActions(['focus', 'halt', 'if', 'on'], ['on', 'if', 'focus'])).toEqual([]);
|
|
135
|
+
});
|
|
136
|
+
});
|
|
@@ -0,0 +1,224 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Structural fidelity for multilingual parses.
|
|
3
|
+
*
|
|
4
|
+
* The parse-validator's success metric is "the parser returned a non-null node".
|
|
5
|
+
* That conflates a faithful parse with a *degenerate* one — a translated pattern
|
|
6
|
+
* can parse non-null while silently dropping most of the source's commands (e.g.
|
|
7
|
+
* `focus-trap` parses as a bare `if` or a stray `from` in several languages, with
|
|
8
|
+
* the `focus`/`halt`/condition lost). This module derives a lightweight
|
|
9
|
+
* structural signature from a parsed node and scores a translation's parse against
|
|
10
|
+
* the English reference parse, so those degenerate passes are visible.
|
|
11
|
+
*
|
|
12
|
+
* The signature is intentionally word-order agnostic: it is the *set* of command
|
|
13
|
+
* actions in the node tree, so a faithful SOV/VSO reorder scores 1.0 while a parse
|
|
14
|
+
* that loses commands scores low.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
/** Passes scoring below this are flagged as degenerate (lost >half the structure). */
|
|
18
|
+
export const FIDELITY_THRESHOLD = 0.5;
|
|
19
|
+
|
|
20
|
+
/** The structural `compound` wrapper is not a command; never counts as an action. */
|
|
21
|
+
const STRUCTURAL_ACTIONS = new Set(['compound']);
|
|
22
|
+
|
|
23
|
+
/** Node-array fields the walk recurses into (event/loop/conditional/behavior bodies). */
|
|
24
|
+
const CHILD_FIELDS = [
|
|
25
|
+
'body',
|
|
26
|
+
'statements',
|
|
27
|
+
'thenBranch',
|
|
28
|
+
'elseBranch',
|
|
29
|
+
'branches',
|
|
30
|
+
'eventHandlers',
|
|
31
|
+
'initBlock',
|
|
32
|
+
] as const;
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* Collect the distinct command actions anywhere in a parsed semantic node tree
|
|
36
|
+
* (top-level action + nested body/statements/branches), excluding the structural
|
|
37
|
+
* `compound` wrapper. Returns a sorted array for stable comparison/serialization.
|
|
38
|
+
*/
|
|
39
|
+
export function collectActions(node: unknown): string[] {
|
|
40
|
+
const acc = new Set<string>();
|
|
41
|
+
walk(node, acc, 0);
|
|
42
|
+
return [...acc].sort();
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
function walk(node: unknown, acc: Set<string>, depth: number): void {
|
|
46
|
+
// Guard against pathological/cyclic structures.
|
|
47
|
+
if (depth > 64 || node === null || typeof node !== 'object') return;
|
|
48
|
+
|
|
49
|
+
const rec = node as Record<string, unknown>;
|
|
50
|
+
const action = rec.action;
|
|
51
|
+
if (typeof action === 'string' && !STRUCTURAL_ACTIONS.has(action)) {
|
|
52
|
+
acc.add(action);
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
for (const field of CHILD_FIELDS) {
|
|
56
|
+
const child = rec[field];
|
|
57
|
+
if (Array.isArray(child)) {
|
|
58
|
+
for (const c of child) walk(c, acc, depth + 1);
|
|
59
|
+
} else if (child && typeof child === 'object') {
|
|
60
|
+
walk(child, acc, depth + 1);
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* Like {@link collectActions} but **multiset**: duplicates are preserved (sorted,
|
|
67
|
+
* not deduped). Needed for precision — a *duplicate* spurious command (e.g. a
|
|
68
|
+
* renderer that injects a phantom `toggle` ahead of a real `toggle`, yielding
|
|
69
|
+
* `[toggle, toggle, put]`) is invisible to the Set-based `collectActions`.
|
|
70
|
+
*
|
|
71
|
+
* Kept as a parallel walk rather than refactoring `collectActions`: the latter
|
|
72
|
+
* feeds the committed regression baseline, so its traversal stays byte-identical.
|
|
73
|
+
*/
|
|
74
|
+
export function collectActionsMultiset(node: unknown): string[] {
|
|
75
|
+
const acc: string[] = [];
|
|
76
|
+
walkMultiset(node, acc, 0);
|
|
77
|
+
return acc.sort();
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
function walkMultiset(node: unknown, acc: string[], depth: number): void {
|
|
81
|
+
if (depth > 64 || node === null || typeof node !== 'object') return;
|
|
82
|
+
|
|
83
|
+
const rec = node as Record<string, unknown>;
|
|
84
|
+
const action = rec.action;
|
|
85
|
+
if (typeof action === 'string' && !STRUCTURAL_ACTIONS.has(action)) {
|
|
86
|
+
acc.push(action);
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
for (const field of CHILD_FIELDS) {
|
|
90
|
+
const child = rec[field];
|
|
91
|
+
if (Array.isArray(child)) {
|
|
92
|
+
for (const c of child) walkMultiset(c, acc, depth + 1);
|
|
93
|
+
} else if (child && typeof child === 'object') {
|
|
94
|
+
walkMultiset(child, acc, depth + 1);
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/** Build a multiset count map from an action list. */
|
|
100
|
+
function actionCounts(items: readonly string[]): Map<string, number> {
|
|
101
|
+
const m = new Map<string, number>();
|
|
102
|
+
for (const it of items) m.set(it, (m.get(it) ?? 0) + 1);
|
|
103
|
+
return m;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* Structural fidelity in [0, 1]: the fraction of the reference (English) actions
|
|
108
|
+
* also present in the candidate (recall). Returns `undefined` when the reference
|
|
109
|
+
* has no actions to compare against.
|
|
110
|
+
*/
|
|
111
|
+
export function computeFidelity(
|
|
112
|
+
reference: readonly string[],
|
|
113
|
+
candidate: readonly string[]
|
|
114
|
+
): number | undefined {
|
|
115
|
+
if (reference.length === 0) return undefined;
|
|
116
|
+
const cand = new Set(candidate);
|
|
117
|
+
let hits = 0;
|
|
118
|
+
for (const a of reference) {
|
|
119
|
+
if (cand.has(a)) hits++;
|
|
120
|
+
}
|
|
121
|
+
return hits / reference.length;
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/**
|
|
125
|
+
* Structural **precision** in [0, 1]: the fraction of the *candidate's* actions
|
|
126
|
+
* that are justified by the reference (multiset-aware). The complement of
|
|
127
|
+
* {@link computeFidelity}'s recall — it falls below 1.0 when a parse/render adds
|
|
128
|
+
* commands the source never had.
|
|
129
|
+
*
|
|
130
|
+
* This is the signal recall + role-fidelity (R1) cannot see: a renderer that
|
|
131
|
+
* injects a phantom `toggle` (ja/ko/tr/ar/ru event-handler rendering) keeps recall
|
|
132
|
+
* 1.0 while precision drops. Pass multisets (see {@link collectActionsMultiset})
|
|
133
|
+
* so a *duplicated* spurious action is counted, not absorbed.
|
|
134
|
+
*
|
|
135
|
+
* Returns `undefined` when the candidate has no actions to score.
|
|
136
|
+
*/
|
|
137
|
+
export function computePrecision(
|
|
138
|
+
reference: readonly string[],
|
|
139
|
+
candidate: readonly string[]
|
|
140
|
+
): number | undefined {
|
|
141
|
+
if (candidate.length === 0) return undefined;
|
|
142
|
+
const ref = actionCounts(reference);
|
|
143
|
+
let matched = 0;
|
|
144
|
+
for (const a of candidate) {
|
|
145
|
+
const remaining = ref.get(a) ?? 0;
|
|
146
|
+
if (remaining > 0) {
|
|
147
|
+
matched++;
|
|
148
|
+
ref.set(a, remaining - 1);
|
|
149
|
+
}
|
|
150
|
+
}
|
|
151
|
+
return matched / candidate.length;
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/**
|
|
155
|
+
* The candidate actions **not** justified by the reference (multiset difference,
|
|
156
|
+
* sorted) — i.e. the spurious/hallucinated commands a parse or render introduced.
|
|
157
|
+
* Empty when the candidate is a structural subset of the reference. Use for
|
|
158
|
+
* diagnostics ("which command was hallucinated?"), complementing the
|
|
159
|
+
* {@link computePrecision} score.
|
|
160
|
+
*/
|
|
161
|
+
export function spuriousActions(
|
|
162
|
+
reference: readonly string[],
|
|
163
|
+
candidate: readonly string[]
|
|
164
|
+
): string[] {
|
|
165
|
+
const ref = actionCounts(reference);
|
|
166
|
+
const extras: string[] = [];
|
|
167
|
+
for (const a of candidate) {
|
|
168
|
+
const remaining = ref.get(a) ?? 0;
|
|
169
|
+
if (remaining > 0) ref.set(a, remaining - 1);
|
|
170
|
+
else extras.push(a);
|
|
171
|
+
}
|
|
172
|
+
return extras.sort();
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
/**
|
|
176
|
+
* R1 — role-fidelity signature.
|
|
177
|
+
*
|
|
178
|
+
* Action-set fidelity cannot see a parse that finds the right commands with the
|
|
179
|
+
* WRONG roles: a swapped patient/destination executes wrongly while scoring 1.0.
|
|
180
|
+
* This signature captures, for every command node in the tree, which roles were
|
|
181
|
+
* filled and with what value *type* (`add.patient:selector`,
|
|
182
|
+
* `put.destination:reference`). Cross-language comparison is by role name +
|
|
183
|
+
* value type — never by value string, which is legitimately translated.
|
|
184
|
+
* The roles container is a ReadonlyMap on live nodes (serializes to {} in JSON),
|
|
185
|
+
* so the signature must be collected at validation time, not from results.json.
|
|
186
|
+
*/
|
|
187
|
+
export function collectRoleSignature(node: unknown): string[] {
|
|
188
|
+
const acc = new Set<string>();
|
|
189
|
+
walkRoles(node, acc, 0);
|
|
190
|
+
return [...acc].sort();
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
function walkRoles(node: unknown, acc: Set<string>, depth: number): void {
|
|
194
|
+
if (depth > 64 || node === null || typeof node !== 'object') return;
|
|
195
|
+
|
|
196
|
+
const rec = node as Record<string, unknown>;
|
|
197
|
+
const action = rec.action;
|
|
198
|
+
if (typeof action === 'string' && !STRUCTURAL_ACTIONS.has(action)) {
|
|
199
|
+
const roles = rec.roles;
|
|
200
|
+
const entries: Array<[unknown, unknown]> =
|
|
201
|
+
roles instanceof Map
|
|
202
|
+
? [...roles.entries()]
|
|
203
|
+
: roles && typeof roles === 'object'
|
|
204
|
+
? Object.entries(roles)
|
|
205
|
+
: [];
|
|
206
|
+
for (const [role, value] of entries) {
|
|
207
|
+
if (value === undefined || value === null) continue;
|
|
208
|
+
const kind =
|
|
209
|
+
typeof value === 'object' && typeof (value as { type?: unknown }).type === 'string'
|
|
210
|
+
? (value as { type: string }).type
|
|
211
|
+
: typeof value;
|
|
212
|
+
acc.add(`${action}.${String(role)}:${kind}`);
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
for (const field of CHILD_FIELDS) {
|
|
217
|
+
const child = rec[field];
|
|
218
|
+
if (Array.isArray(child)) {
|
|
219
|
+
for (const c of child) walkRoles(c, acc, depth + 1);
|
|
220
|
+
} else if (child && typeof child === 'object') {
|
|
221
|
+
walkRoles(child, acc, depth + 1);
|
|
222
|
+
}
|
|
223
|
+
}
|
|
224
|
+
}
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Pattern Loader Tests
|
|
3
|
+
*
|
|
4
|
+
* Covers isHtmlMarkupPattern: the guard that keeps HTML-markup patterns out of
|
|
5
|
+
* the semantic-parse denominator (they're validated by the DOM/Playwright
|
|
6
|
+
* suites, not the hyperscript text parser).
|
|
7
|
+
*/
|
|
8
|
+
import { describe, it, expect } from 'vitest';
|
|
9
|
+
import { isHtmlMarkupPattern } from './html-pattern';
|
|
10
|
+
|
|
11
|
+
describe('isHtmlMarkupPattern', () => {
|
|
12
|
+
it('flags HTML-markup patterns (excluded from semantic parsing)', () => {
|
|
13
|
+
const htmlPatterns = [
|
|
14
|
+
'<div hx-live="put $count into me"></div>',
|
|
15
|
+
'<div sse-connect="/events" sse-swap="tick" hx-target="#feed"></div>',
|
|
16
|
+
'<div ws-connect="wss://example/api"><form ws-send></form></div>',
|
|
17
|
+
'<script type="text/hyperscript-template" component="hello-world"><span>Hi</span></script>',
|
|
18
|
+
' <button _="on click increment ^count">+</button>', // leading whitespace
|
|
19
|
+
];
|
|
20
|
+
for (const code of htmlPatterns) {
|
|
21
|
+
expect(isHtmlMarkupPattern(code), code).toBe(true);
|
|
22
|
+
}
|
|
23
|
+
});
|
|
24
|
+
|
|
25
|
+
it('does not flag real hyperscript source', () => {
|
|
26
|
+
const hyperscript = [
|
|
27
|
+
'toggle .active on #button',
|
|
28
|
+
'on click increment $count',
|
|
29
|
+
'put :x into me',
|
|
30
|
+
'bind $greeting to #name-input',
|
|
31
|
+
'live put `Count: ${$count}` into me end',
|
|
32
|
+
'fetch /api/data as json then put it into #out',
|
|
33
|
+
'#count を 増加', // SOV, non-Latin
|
|
34
|
+
];
|
|
35
|
+
for (const code of hyperscript) {
|
|
36
|
+
expect(isHtmlMarkupPattern(code), code).toBe(false);
|
|
37
|
+
}
|
|
38
|
+
});
|
|
39
|
+
});
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* HTML-markup pattern detection.
|
|
3
|
+
*
|
|
4
|
+
* A handful of DB patterns (`hx-live`, `sse-connect`/`ws-connect`, and the
|
|
5
|
+
* `<script type="text/hyperscript-template">` component patterns) are authored
|
|
6
|
+
* as HTML markup — the hyperscript they exercise lives inside attributes
|
|
7
|
+
* (`hx-live="…"`, `_="…"`) and is processed at runtime by the htmx-compat /
|
|
8
|
+
* component layer, which is covered by the DOM + Playwright suites.
|
|
9
|
+
*
|
|
10
|
+
* The multilingual harness validates patterns with the hyperscript *text*
|
|
11
|
+
* parser, which structurally cannot parse HTML — grading these is a category
|
|
12
|
+
* error that understates the real parse rate, so they are excluded from the
|
|
13
|
+
* semantic-parse denominator.
|
|
14
|
+
*
|
|
15
|
+
* Top-level hyperscript never begins with `<` (commands start with a
|
|
16
|
+
* verb/event keyword or a `.`/`#`/`:`/`$` token), so a leading `<` is a
|
|
17
|
+
* reliable, self-maintaining signal — new HTML patterns are excluded
|
|
18
|
+
* automatically without a manual allow/deny list.
|
|
19
|
+
*
|
|
20
|
+
* Kept dependency-free so it is unit-testable without the patterns DB.
|
|
21
|
+
*/
|
|
22
|
+
export function isHtmlMarkupPattern(code: string): boolean {
|
|
23
|
+
return /^\s*</.test(code);
|
|
24
|
+
}
|
|
@@ -7,11 +7,20 @@ import { promisify } from 'node:util';
|
|
|
7
7
|
import { loadPatterns } from './pattern-loader';
|
|
8
8
|
import { selectBundle, getBundleInfo } from './bundle-builder';
|
|
9
9
|
import { ParseValidator } from './validators/parse-validator';
|
|
10
|
+
import { ExecutionValidator, loadExecutionSubset } from './validators/execution-validator';
|
|
10
11
|
import { SizeValidator } from './validators/size-validator';
|
|
11
12
|
import { ConsoleReporter } from './reporters/console-reporter';
|
|
12
13
|
import { JSONReporter } from './reporters/json-reporter';
|
|
13
14
|
import { RegressionReporter } from './reporters/regression-reporter';
|
|
14
|
-
import type {
|
|
15
|
+
import type {
|
|
16
|
+
TestConfig,
|
|
17
|
+
TestResults,
|
|
18
|
+
LanguageResults,
|
|
19
|
+
LanguageCode,
|
|
20
|
+
Reporter,
|
|
21
|
+
BundleInfo,
|
|
22
|
+
} from './types';
|
|
23
|
+
import { computeFidelity, computePrecision, FIDELITY_THRESHOLD } from './fidelity';
|
|
15
24
|
|
|
16
25
|
const execAsync = promisify(exec);
|
|
17
26
|
|
|
@@ -48,9 +57,12 @@ export class TestOrchestrator {
|
|
|
48
57
|
// Add JSON reporter for structured output
|
|
49
58
|
this.reporters.push(new JSONReporter('./test-results/results.json'));
|
|
50
59
|
|
|
51
|
-
// Add regression reporter if requested
|
|
60
|
+
// Add regression reporter if requested. Default to the COMMITTED baseline
|
|
61
|
+
// (test-results/ is gitignored), overridable via --baseline.
|
|
52
62
|
if (this.config.regression) {
|
|
53
|
-
this.reporters.push(
|
|
63
|
+
this.reporters.push(
|
|
64
|
+
new RegressionReporter(this.config.baselinePath ?? './baselines/multilingual-priority.json')
|
|
65
|
+
);
|
|
54
66
|
}
|
|
55
67
|
}
|
|
56
68
|
|
|
@@ -88,6 +100,15 @@ export class TestOrchestrator {
|
|
|
88
100
|
languageResults.push(result);
|
|
89
101
|
}
|
|
90
102
|
|
|
103
|
+
// Score structural fidelity against the English reference parse. This
|
|
104
|
+
// surfaces *degenerate* passes — patterns that parse non-null but drop most
|
|
105
|
+
// of the source's commands — which the parse-rate metric alone can't see.
|
|
106
|
+
this.scoreFidelity(languageResults);
|
|
107
|
+
|
|
108
|
+
// R2 — execution smoke: run the curated subset in jsdom and compare DOM
|
|
109
|
+
// effects against the en reference's (see execution-validator.ts).
|
|
110
|
+
await this.scoreExecution(languageResults);
|
|
111
|
+
|
|
91
112
|
// Collect bundle information
|
|
92
113
|
const bundles: TestResults['bundles'] = {};
|
|
93
114
|
for (const langResult of languageResults) {
|
|
@@ -158,13 +179,25 @@ export class TestOrchestrator {
|
|
|
158
179
|
private async testLanguage(language: LanguageCode, patterns: any[]): Promise<LanguageResults> {
|
|
159
180
|
const startTime = performance.now();
|
|
160
181
|
|
|
161
|
-
// Select bundle for this language
|
|
162
|
-
|
|
182
|
+
// Select bundle for this language. The bundle is consumed only for size
|
|
183
|
+
// reporting — parse validation below runs in-process via parseSemantic — so a
|
|
184
|
+
// missing display bundle is a warning, not a fatal error that aborts the sweep.
|
|
185
|
+
const selected = this.config.bundle
|
|
163
186
|
? await getBundleInfo(this.config.bundle)
|
|
164
187
|
: await selectBundle([language], this.config.build || false);
|
|
165
188
|
|
|
166
|
-
|
|
167
|
-
|
|
189
|
+
const bundle: BundleInfo = selected ?? {
|
|
190
|
+
name: `browser-${language}`,
|
|
191
|
+
path: '',
|
|
192
|
+
languages: [language],
|
|
193
|
+
size: 0,
|
|
194
|
+
exists: false,
|
|
195
|
+
};
|
|
196
|
+
|
|
197
|
+
if (!bundle.exists) {
|
|
198
|
+
console.warn(
|
|
199
|
+
`⚠ No display bundle for '${language}' (${bundle.name}); reporting size 0 and continuing with in-process parse.`
|
|
200
|
+
);
|
|
168
201
|
}
|
|
169
202
|
|
|
170
203
|
// Notify reporters
|
|
@@ -209,6 +242,143 @@ export class TestOrchestrator {
|
|
|
209
242
|
return result;
|
|
210
243
|
}
|
|
211
244
|
|
|
245
|
+
/**
|
|
246
|
+
* Score structural fidelity of every parse against the English reference parse
|
|
247
|
+
* of the same pattern, in-place. Fills `ParseResult.fidelity` plus per-language
|
|
248
|
+
* `avgFidelity` / `degeneratePasses`. English is the reference, so its own
|
|
249
|
+
* results are left unscored. No-op when English wasn't part of the run.
|
|
250
|
+
*/
|
|
251
|
+
private scoreFidelity(languageResults: LanguageResults[]): void {
|
|
252
|
+
const en = languageResults.find(r => r.language === 'en');
|
|
253
|
+
if (!en) return;
|
|
254
|
+
|
|
255
|
+
// codeExampleId -> English action signature (only successful en parses).
|
|
256
|
+
const reference = new Map<string, string[]>();
|
|
257
|
+
// R0-precision: codeExampleId -> English action MULTISET (duplicates kept).
|
|
258
|
+
const multisetReference = new Map<string, string[]>();
|
|
259
|
+
// R1: codeExampleId -> English role signature (action.role:valueType set).
|
|
260
|
+
const roleReference = new Map<string, string[]>();
|
|
261
|
+
for (const r of en.parseResults) {
|
|
262
|
+
if (r.success && r.actionSignature && r.actionSignature.length > 0) {
|
|
263
|
+
reference.set(r.pattern.codeExampleId, r.actionSignature);
|
|
264
|
+
}
|
|
265
|
+
if (r.success && r.actionMultisetSignature && r.actionMultisetSignature.length > 0) {
|
|
266
|
+
multisetReference.set(r.pattern.codeExampleId, r.actionMultisetSignature);
|
|
267
|
+
}
|
|
268
|
+
if (r.success && r.roleSignature && r.roleSignature.length > 0) {
|
|
269
|
+
roleReference.set(r.pattern.codeExampleId, r.roleSignature);
|
|
270
|
+
}
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
for (const lang of languageResults) {
|
|
274
|
+
if (lang.language === 'en') continue;
|
|
275
|
+
|
|
276
|
+
const degenerate: string[] = [];
|
|
277
|
+
const lossy: string[] = [];
|
|
278
|
+
const scores: number[] = [];
|
|
279
|
+
const precisionScores: number[] = [];
|
|
280
|
+
const roleScores: number[] = [];
|
|
281
|
+
|
|
282
|
+
for (const result of lang.parseResults) {
|
|
283
|
+
if (!result.success || !result.actionSignature) continue;
|
|
284
|
+
const ref = reference.get(result.pattern.codeExampleId);
|
|
285
|
+
if (!ref) continue;
|
|
286
|
+
|
|
287
|
+
const fidelity = computeFidelity(ref, result.actionSignature);
|
|
288
|
+
if (fidelity === undefined) continue;
|
|
289
|
+
|
|
290
|
+
result.fidelity = fidelity;
|
|
291
|
+
scores.push(fidelity);
|
|
292
|
+
if (fidelity < FIDELITY_THRESHOLD) degenerate.push(result.pattern.codeExampleId);
|
|
293
|
+
else if (fidelity < 1) lossy.push(result.pattern.codeExampleId);
|
|
294
|
+
|
|
295
|
+
// R0-precision — fraction of THIS parse's actions justified by the en
|
|
296
|
+
// multiset reference (catches phantom/spurious commands recall misses).
|
|
297
|
+
const multisetRef = multisetReference.get(result.pattern.codeExampleId);
|
|
298
|
+
if (multisetRef && result.actionMultisetSignature) {
|
|
299
|
+
const precision = computePrecision(multisetRef, result.actionMultisetSignature);
|
|
300
|
+
if (precision !== undefined) {
|
|
301
|
+
result.precision = precision;
|
|
302
|
+
precisionScores.push(precision);
|
|
303
|
+
}
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
// R1 — role recall vs the en role signature (role name + value type).
|
|
307
|
+
const roleRef = roleReference.get(result.pattern.codeExampleId);
|
|
308
|
+
if (roleRef && result.roleSignature) {
|
|
309
|
+
const roleFidelity = computeFidelity(roleRef, result.roleSignature);
|
|
310
|
+
if (roleFidelity !== undefined) {
|
|
311
|
+
result.roleFidelity = roleFidelity;
|
|
312
|
+
roleScores.push(roleFidelity);
|
|
313
|
+
}
|
|
314
|
+
}
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
lang.avgFidelity =
|
|
318
|
+
scores.length > 0 ? scores.reduce((a, b) => a + b, 0) / scores.length : undefined;
|
|
319
|
+
lang.avgPrecision =
|
|
320
|
+
precisionScores.length > 0
|
|
321
|
+
? precisionScores.reduce((a, b) => a + b, 0) / precisionScores.length
|
|
322
|
+
: undefined;
|
|
323
|
+
lang.avgRoleFidelity =
|
|
324
|
+
roleScores.length > 0
|
|
325
|
+
? roleScores.reduce((a, b) => a + b, 0) / roleScores.length
|
|
326
|
+
: undefined;
|
|
327
|
+
lang.degeneratePasses = degenerate.sort();
|
|
328
|
+
lang.lossyPasses = lossy.sort();
|
|
329
|
+
}
|
|
330
|
+
}
|
|
331
|
+
|
|
332
|
+
/**
|
|
333
|
+
* R2 — execute the curated execution subset for every language in the run
|
|
334
|
+
* and score each translation's DOM-effect signature against the en
|
|
335
|
+
* reference's, in-place (`avgExecutionFidelity` / `executionFailures`).
|
|
336
|
+
* English is the reference and is left unscored; patterns whose en
|
|
337
|
+
* reference errors or produces no effects are excluded (no usable
|
|
338
|
+
* reference). No-op when English wasn't part of the run.
|
|
339
|
+
*/
|
|
340
|
+
private async scoreExecution(languageResults: LanguageResults[]): Promise<void> {
|
|
341
|
+
const en = languageResults.find(r => r.language === 'en');
|
|
342
|
+
if (!en) return;
|
|
343
|
+
|
|
344
|
+
const sources = await loadExecutionSubset(languageResults.map(r => r.language));
|
|
345
|
+
const validator = new ExecutionValidator();
|
|
346
|
+
await validator.initialize();
|
|
347
|
+
|
|
348
|
+
// codeExampleId -> en effect signature (only clean, effectful references).
|
|
349
|
+
const reference = new Map<string, string[]>();
|
|
350
|
+
for (const [id, code] of sources.get('en') ?? []) {
|
|
351
|
+
const res = await validator.execute(id, code, 'en');
|
|
352
|
+
if (!res.error && res.effects.length > 0) reference.set(id, res.effects);
|
|
353
|
+
}
|
|
354
|
+
if (reference.size === 0) return;
|
|
355
|
+
|
|
356
|
+
for (const lang of languageResults) {
|
|
357
|
+
if (lang.language === 'en') continue;
|
|
358
|
+
const langSources = sources.get(lang.language);
|
|
359
|
+
if (!langSources) continue;
|
|
360
|
+
|
|
361
|
+
const failures: string[] = [];
|
|
362
|
+
let scored = 0;
|
|
363
|
+
let matched = 0;
|
|
364
|
+
for (const [id, refEffects] of reference) {
|
|
365
|
+
const code = langSources.get(id);
|
|
366
|
+
if (!code) continue; // no translation — excluded, not failed
|
|
367
|
+
const res = await validator.execute(id, code, lang.language);
|
|
368
|
+
// Match on DOM effects ONLY. Trapped runtime errors are diagnostic:
|
|
369
|
+
// their attribution rides on unhandled-rejection timing (racy), while
|
|
370
|
+
// the effect snapshot is synchronous and deterministic — and an AST
|
|
371
|
+
// mis-build that damages behavior shows up as differing effects anyway.
|
|
372
|
+
const match = JSON.stringify(res.effects) === JSON.stringify(refEffects);
|
|
373
|
+
scored++;
|
|
374
|
+
if (match) matched++;
|
|
375
|
+
else failures.push(id);
|
|
376
|
+
}
|
|
377
|
+
lang.avgExecutionFidelity = scored > 0 ? matched / scored : undefined;
|
|
378
|
+
lang.executionFailures = failures.sort();
|
|
379
|
+
}
|
|
380
|
+
}
|
|
381
|
+
|
|
212
382
|
/**
|
|
213
383
|
* Group patterns by language
|
|
214
384
|
*/
|
|
@@ -12,6 +12,9 @@ import {
|
|
|
12
12
|
type Translation,
|
|
13
13
|
} from '@hyperfixi/patterns-reference';
|
|
14
14
|
import type { LanguageCode, PatternTranslation, TestConfig, SamplingStrategy } from './types';
|
|
15
|
+
import { isHtmlMarkupPattern } from './html-pattern';
|
|
16
|
+
|
|
17
|
+
export { isHtmlMarkupPattern };
|
|
15
18
|
|
|
16
19
|
/**
|
|
17
20
|
* Load patterns for testing based on configuration
|
|
@@ -25,13 +28,19 @@ export async function loadPatterns(config: TestConfig): Promise<PatternTranslati
|
|
|
25
28
|
results.push(...translations);
|
|
26
29
|
}
|
|
27
30
|
|
|
31
|
+
// Exclude HTML-markup patterns from the semantic-parse set (see
|
|
32
|
+
// isHtmlMarkupPattern): the hyperscript text parser can't grade HTML, so
|
|
33
|
+
// counting them would understate the real parse rate. They are validated by
|
|
34
|
+
// the DOM/Playwright + i18n-htmx suites instead.
|
|
35
|
+
const semanticPatterns = results.filter(p => !isHtmlMarkupPattern(p.hyperscript));
|
|
36
|
+
|
|
28
37
|
// Apply sampling if in quick mode
|
|
29
38
|
if (config.mode === 'quick') {
|
|
30
39
|
const limit = config.quickModeLimit || 10;
|
|
31
|
-
return samplePatterns(
|
|
40
|
+
return samplePatterns(semanticPatterns, { type: 'stratified', perCategory: limit });
|
|
32
41
|
}
|
|
33
42
|
|
|
34
|
-
return
|
|
43
|
+
return semanticPatterns;
|
|
35
44
|
}
|
|
36
45
|
|
|
37
46
|
/**
|