@hyperfixi/testing-framework 2.7.2 → 2.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/dist/assertions.d.mts +26 -1
  2. package/dist/assertions.d.ts +26 -1
  3. package/dist/index.d.mts +5 -112
  4. package/dist/index.d.ts +5 -112
  5. package/dist/runner.d.mts +112 -0
  6. package/dist/runner.d.ts +112 -0
  7. package/dist/runner.js +1102 -0
  8. package/dist/runner.js.map +1 -0
  9. package/dist/runner.mjs +1097 -0
  10. package/dist/runner.mjs.map +1 -0
  11. package/dist/{assertions-CsGP61iW.d.mts → types-D-rCVkf3.d.mts} +1 -24
  12. package/dist/{assertions-CsGP61iW.d.ts → types-D-rCVkf3.d.ts} +1 -24
  13. package/package.json +11 -26
  14. package/src/multilingual/canonical-validity.test.ts +69 -0
  15. package/src/multilingual/canonical-validity.ts +132 -0
  16. package/src/multilingual/cli.ts +185 -1
  17. package/src/multilingual/fidelity.test.ts +192 -0
  18. package/src/multilingual/fidelity.ts +153 -0
  19. package/src/multilingual/foreign-canonical-validity.test.ts +0 -0
  20. package/src/multilingual/foreign-canonical-validity.ts +158 -0
  21. package/src/multilingual/orchestrator.ts +47 -1
  22. package/src/multilingual/reporters/console-reporter.ts +51 -0
  23. package/src/multilingual/reporters/regression-reporter.ts +23 -0
  24. package/src/multilingual/tools/diagnose-coverage.ts +118 -0
  25. package/src/multilingual/tools/triage-r1.ts +149 -0
  26. package/src/multilingual/types.ts +74 -0
  27. package/src/multilingual/validators/parse-validator.ts +9 -1
  28. package/src/runner.test.ts +7 -2
  29. package/src/vocab/batch3-roundtrip.test.ts +184 -0
  30. package/src/vocab/checks.test.ts +362 -0
  31. package/src/vocab/checks.ts +311 -0
  32. package/src/vocab/cli.ts +196 -0
  33. package/src/vocab/dump.ts +80 -0
  34. package/src/vocab/model.ts +110 -0
  35. package/src/vocab/report.ts +120 -0
  36. package/src/vocab/types.ts +100 -0
@@ -100,6 +100,19 @@ export interface TestConfig {
100
100
 
101
101
  /** Number of patterns per language in quick mode */
102
102
  quickModeLimit?: number;
103
+
104
+ /**
105
+ * Report the semantic parser's `unconsumed-input` firing rate per language,
106
+ * then exit. Read-only: never gates, never writes a baseline.
107
+ */
108
+ diagnoseCoverage?: boolean;
109
+
110
+ /**
111
+ * Itemize R1 role-fidelity misses (missing `action.role:type` entries vs the
112
+ * en reference, clustered) per language, then exit. Read-only: never gates,
113
+ * never writes a baseline. Scope with `languages`.
114
+ */
115
+ triageR1?: boolean;
103
116
  }
104
117
 
105
118
  /**
@@ -138,6 +151,17 @@ export interface ParseResult {
138
151
  */
139
152
  roleSignature?: string[];
140
153
 
154
+ /**
155
+ * R3 — role-VALUE signature: `action.role=value` multiset, filtered to
156
+ * language-invariant surface forms (selectors, sigil refs, numbers/time
157
+ * literals, colon-qualified event names, URLs — code, not prose). The one
158
+ * value class that CAN be compared verbatim across languages; catches
159
+ * right-action/right-type/wrong-VALUE corruption (`draggable` captured for
160
+ * `draggable:start`) that every type-based signal misses. Set by the
161
+ * validator (roles are a ReadonlyMap — serialize to {} in results.json).
162
+ */
163
+ roleValueSignature?: string[];
164
+
141
165
  /**
142
166
  * Structural fidelity vs the English reference parse, in [0, 1]: the fraction
143
167
  * of the English parse's distinct actions also present in this language's
@@ -157,6 +181,15 @@ export interface ParseResult {
157
181
  */
158
182
  precision?: number;
159
183
 
184
+ /**
185
+ * R0-recall on the MULTISET vs the English reference, in [0, 1]. `fidelity`
186
+ * above scores the deduped Set, so a parse that drops a REPEATED command
187
+ * (`[bind, bind]` → `[bind]`) still scores 1.0. This counts duplicates, and is
188
+ * the mirror of `precision`: together they see both a dropped and an added
189
+ * copy of a command. `undefined` when there is no usable English reference.
190
+ */
191
+ multisetRecall?: number;
192
+
160
193
  /**
161
194
  * R1 — role fidelity vs the English reference, in [0, 1]: the fraction of the
162
195
  * English parse's role-signature entries also present here. Catches a parse
@@ -164,6 +197,23 @@ export interface ParseResult {
164
197
  * patient/destination executes wrongly while action-fidelity scores 1.0).
165
198
  */
166
199
  roleFidelity?: number;
200
+
201
+ /**
202
+ * R3 — invariant role-VALUE recall vs the English reference, in [0, 1]: the
203
+ * fraction of the English parse's `roleValueSignature` entries (multiset —
204
+ * duplicates counted) also present here. Falls below 1.0 when a translation
205
+ * loses or corrupts a language-invariant value (selector, sigil ref, time
206
+ * literal, colon-qualified event name, URL) while actions and role types
207
+ * still match. `undefined` when the en reference has no invariant values.
208
+ */
209
+ valueRecall?: number;
210
+
211
+ /**
212
+ * R3 — the en reference's `action.role=value` entries missing from this
213
+ * parse (multiset difference). Populated only when `valueRecall` < 1, for
214
+ * diagnostics ("which value was lost/corrupted?").
215
+ */
216
+ valueRecallMissing?: string[];
167
217
  }
168
218
 
169
219
  /**
@@ -205,6 +255,14 @@ export interface LanguageResults {
205
255
  */
206
256
  avgPrecision?: number | undefined;
207
257
 
258
+ /**
259
+ * R0-recall-multiset — mean `multisetRecall` over successful parses with an
260
+ * English reference. Recorded + ratcheted alongside avgFidelity: a drop means a
261
+ * parser change started dropping a REPEATED command, which the Set-based
262
+ * avgFidelity and avgRoleFidelity cannot see.
263
+ */
264
+ avgMultisetRecall?: number | undefined;
265
+
208
266
  /**
209
267
  * R1 — mean `roleFidelity` over successful parses with an English reference.
210
268
  * Recorded + ratcheted in the baseline; the burn-down is deliberately NOT part
@@ -212,6 +270,14 @@ export interface LanguageResults {
212
270
  */
213
271
  avgRoleFidelity?: number | undefined;
214
272
 
273
+ /**
274
+ * R3 — mean `valueRecall` over successful parses with an English reference.
275
+ * Recorded + ratcheted alongside avgRoleFidelity: a drop means a translation
276
+ * started losing/corrupting a language-invariant role VALUE (the #633 class),
277
+ * which every action/type-based signal above scores as perfect.
278
+ */
279
+ avgValueRecall?: number | undefined;
280
+
215
281
  /**
216
282
  * R2 — mean executionFidelity over the curated execution subset (see
217
283
  * validators/execution-validator.ts): 1 per pattern whose jsdom DOM-effect
@@ -298,8 +364,12 @@ export interface Baseline {
298
364
  avgFidelity?: number | undefined;
299
365
  /** R0-precision — mean precision vs the English reference (see ParseResult.precision). */
300
366
  avgPrecision?: number | undefined;
367
+ /** R0-recall-multiset — mean multisetRecall vs the English reference (see ParseResult.multisetRecall). */
368
+ avgMultisetRecall?: number | undefined;
301
369
  /** R1 — mean role fidelity vs the English reference (see ParseResult.roleFidelity). */
302
370
  avgRoleFidelity?: number | undefined;
371
+ /** R3 — mean invariant role-VALUE recall vs the English reference (see ParseResult.valueRecall). */
372
+ avgValueRecall?: number | undefined;
303
373
  /** R2 — mean execution fidelity over the curated execution subset (see LanguageResults.avgExecutionFidelity). */
304
374
  avgExecutionFidelity?: number | undefined;
305
375
  /** R2 — curated-subset pattern IDs whose execution diverged from the en reference. */
@@ -335,8 +405,12 @@ export interface RegressionResult {
335
405
  avgFidelityDelta: number;
336
406
  /** R0-precision — absolute change in avgPrecision (current − baseline). 0 when either side lacks data. Negative = phantom commands introduced. */
337
407
  avgPrecisionDelta: number;
408
+ /** R0-recall-multiset — absolute change in avgMultisetRecall (current − baseline). 0 when either side lacks data. Negative = a repeated command is being dropped. */
409
+ avgMultisetRecallDelta: number;
338
410
  /** R1 — absolute change in avgRoleFidelity (current − baseline). 0 when either side lacks data. */
339
411
  avgRoleFidelityDelta: number;
412
+ /** R3 — absolute change in avgValueRecall (current − baseline). 0 when either side lacks data. Negative = an invariant role VALUE is being lost/corrupted. */
413
+ avgValueRecallDelta: number;
340
414
  /** R2 — absolute change in avgExecutionFidelity (current − baseline). 0 when either side lacks data. */
341
415
  avgExecutionFidelityDelta: number;
342
416
  /**
@@ -6,7 +6,12 @@ import { MultilingualHyperscript } from '@hyperfixi/core/multilingual';
6
6
  import type { SemanticNode } from '@lokascript/semantic';
7
7
  import { fillSchemaDefaults } from '@lokascript/semantic';
8
8
  import type { PatternTranslation, ParseResult, Validator } from '../types';
9
- import { collectActions, collectActionsMultiset, collectRoleSignature } from '../fidelity';
9
+ import {
10
+ collectActions,
11
+ collectActionsMultiset,
12
+ collectRoleSignature,
13
+ collectRoleValueSignature,
14
+ } from '../fidelity';
10
15
 
11
16
  /**
12
17
  * Parse Validator
@@ -106,6 +111,9 @@ export class ParseValidator implements Validator<ParseResult[]> {
106
111
  // R1 role signature (role name + value type per command) — collected here
107
112
  // because live nodes carry roles as a ReadonlyMap that serializes to {}.
108
113
  roleSignature: collectRoleSignature(semanticNode),
114
+ // R3 role-VALUE signature (invariant-shaped values only, multiset) —
115
+ // catches right-action/right-type/wrong-VALUE corruption (the #633 class).
116
+ roleValueSignature: collectRoleValueSignature(semanticNode),
109
117
  duration: performance.now() - startTime,
110
118
  };
111
119
  } catch (error) {
@@ -438,7 +438,11 @@ describe('CoreTestRunner', () => {
438
438
 
439
439
  const result = await runner.runTest(test, context);
440
440
 
441
- expect(result.duration).toBeGreaterThanOrEqual(50);
441
+ // Tolerance below the 50ms sleep: setTimeout clamping can fire the
442
+ // timer ~1-2ms early on CI runners (observed: 49ms in pre-publish run
443
+ // 29800757714), and the assertion is about duration RECORDING, not
444
+ // timer precision.
445
+ expect(result.duration).toBeGreaterThanOrEqual(45);
442
446
  });
443
447
  });
444
448
  });
@@ -686,7 +690,8 @@ describe('measurePerformance', () => {
686
690
 
687
691
  const metrics = await measurePerformance(mockPage, testFn);
688
692
 
689
- expect(metrics.loadTime).toBeGreaterThanOrEqual(50);
693
+ // Same timer-clamping tolerance as the duration test above.
694
+ expect(metrics.loadTime).toBeGreaterThanOrEqual(45);
690
695
  });
691
696
  });
692
697
 
@@ -0,0 +1,184 @@
1
+ /**
2
+ * Batch 3 red→green proofs — one representative render→parse round-trip per
3
+ * V1 dict-fix class (docs-internal/MULTILINGUAL_NEXT_STEPS.md § "V1 probe
4
+ * conclusion (Batch 3)"). Each of these was a live render→parse break before
5
+ * the Batch 3 dictionary fixes:
6
+ *
7
+ * - select class: the dict word doubled as the pick keyword — the bare select
8
+ * render parsed null (de `auswählen #note`).
9
+ * - wrong-verb class: the dict word was another command's verb — the render
10
+ * parsed as THAT action (bn clone → copy, id close → hide, sw copy → clone,
11
+ * vi prepend → add, qu change-event → toggle).
12
+ * - reset class: the dict event word captured on.event as an expression (a
13
+ * broken listener) or a wrong event; the profile verb round-trips.
14
+ * - submit class: the dict event word doubled as the send verb — corpus
15
+ * on-submit rows captured event "send" (es/pl/tr/vi live).
16
+ * - blur class (it): the noun form dropped blur.patient in command position.
17
+ * - empty class (ko/qu profile-alternatives): the dict adjective renders the
18
+ * empty COMMAND but only the profile verb parsed. (ja is waived: bare 空
19
+ * phantoms the hot `is empty` rows if registered.)
20
+ *
21
+ * Assertions are on captured ACTION and role VALUES, never "it parses".
22
+ */
23
+
24
+ import { describe, expect, it } from 'vitest';
25
+ import { parseSemantic } from '@lokascript/semantic';
26
+ import { GrammarTransformer } from '@lokascript/i18n';
27
+
28
+ /** Verbatim (action, role, value) triples from a parse tree. */
29
+ function triples(node: unknown, acc: string[] = [], depth = 0): string[] {
30
+ if (depth > 64 || node === null || typeof node !== 'object') return acc;
31
+ const rec = node as Record<string, unknown>;
32
+ if (typeof rec.action === 'string') {
33
+ const roles = rec.roles;
34
+ const entries: Array<[unknown, unknown]> =
35
+ roles instanceof Map
36
+ ? [...roles.entries()]
37
+ : roles && typeof roles === 'object'
38
+ ? Object.entries(roles)
39
+ : [];
40
+ for (const [role, v] of entries) {
41
+ if (v === null || v === undefined) continue;
42
+ const val = (v as { value?: unknown }).value ?? (v as { raw?: unknown }).raw;
43
+ acc.push(`${rec.action}.${String(role)}=${String(val)}`);
44
+ }
45
+ }
46
+ for (const f of ['body', 'commands', 'children', 'thenBranch', 'elseBranch']) {
47
+ const c = rec[f];
48
+ if (Array.isArray(c)) for (const x of c) triples(x, acc, depth + 1);
49
+ else if (c && typeof c === 'object') triples(c, acc, depth + 1);
50
+ }
51
+ return acc;
52
+ }
53
+
54
+ function renderAndParse(en: string, lang: string): { render: string; triples: string[] } {
55
+ const render = new GrammarTransformer('en', lang).transform(en);
56
+ const result = parseSemantic(render, lang);
57
+ return { render, triples: result.node ? triples(result.node) : [] };
58
+ }
59
+
60
+ describe('Batch 3 — select class (dict word was the pick keyword)', () => {
61
+ it('de: `select #note` renders markieren and parses back as select', () => {
62
+ const { render, triples: t } = renderAndParse('select #note', 'de');
63
+ expect(render).toContain('markieren');
64
+ expect(t).toContain('select.patient=#note');
65
+ });
66
+
67
+ it('ar: `select #note` renders ظلل and parses back as select', () => {
68
+ const { render, triples: t } = renderAndParse('select #note', 'ar');
69
+ expect(render).toContain('ظلل');
70
+ expect(t).toContain('select.patient=#note');
71
+ });
72
+ });
73
+
74
+ describe('Batch 3 — wrong-verb class (dict word was another command)', () => {
75
+ it('bn: `clone #card` no longer parses as copy', () => {
76
+ const { render, triples: t } = renderAndParse('clone #card', 'bn');
77
+ expect(render).toContain('ক্লোন');
78
+ expect(t).toContain('clone.patient=#card');
79
+ });
80
+
81
+ it('id: `close #modal` no longer parses as hide', () => {
82
+ const { render, triples: t } = renderAndParse('close #modal', 'id');
83
+ expect(render).toContain('tutupkan');
84
+ expect(t).toContain('close.patient=#modal');
85
+ });
86
+
87
+ it('sw: `copy #text` no longer parses as clone', () => {
88
+ const { render, triples: t } = renderAndParse('copy #text', 'sw');
89
+ expect(render).toContain('nakala');
90
+ expect(t).toContain('copy.patient=#text');
91
+ });
92
+
93
+ it('vi: `prepend "x" to #list` no longer parses as add', () => {
94
+ const { render, triples: t } = renderAndParse('prepend "x" to #list', 'vi');
95
+ expect(render).toContain('thêm vào đầu');
96
+ expect(t.some(x => x.startsWith('prepend.'))).toBe(true);
97
+ });
98
+ });
99
+
100
+ describe('Batch 3 — reset class (broken/wrong-event listener)', () => {
101
+ it.each([
102
+ ['it', 'reimpostare'],
103
+ ['ko', '재설정'],
104
+ ['pl', 'zresetuj'],
105
+ ['pt', 'redefinir'],
106
+ ['ru', 'сбросить'],
107
+ ['uk', 'скинути'],
108
+ ['qu', 'musuqchay'],
109
+ ])('%s: on-reset render captures on.event="reset" (canonical)', (langCode, word) => {
110
+ const { render, triples: t } = renderAndParse('on reset log "done"', langCode);
111
+ expect(render).toContain(word);
112
+ expect(t).toContain('on.event=reset');
113
+ });
114
+ });
115
+
116
+ describe('Batch 3 — submit class (dict word was the send verb)', () => {
117
+ it.each([
118
+ ['es', 'envío'],
119
+ ['pl', 'wysłaniu'],
120
+ ['tr', 'gönderme'],
121
+ ['vi', 'nộp'],
122
+ ['qu', 'apaykachay'],
123
+ ])(
124
+ '%s: corpus-shaped on-submit render captures on.event="submit", not "send"',
125
+ (langCode, word) => {
126
+ const { render, triples: t } = renderAndParse(
127
+ 'on submit add @disabled to <button/> in me put "Submitting..." into <button/> in me',
128
+ langCode
129
+ );
130
+ expect(render).toContain(word);
131
+ expect(t).toContain('on.event=submit');
132
+ expect(t).not.toContain('on.event=send');
133
+ }
134
+ );
135
+ });
136
+
137
+ describe('Batch 3 — qu change (dict word was the toggle verb)', () => {
138
+ it('qu: on-change render captures on.event="change", not "toggle"', () => {
139
+ const { render, triples: t } = renderAndParse('on change log "x"', 'qu');
140
+ expect(render).toContain('kambiay');
141
+ expect(t).toContain('on.event=change');
142
+ });
143
+ });
144
+
145
+ describe('Batch 3 — it blur (noun form dropped the command patient)', () => {
146
+ it('it: blur command render captures blur.patient', () => {
147
+ const { render, triples: t } = renderAndParse('on keydown[key=="Escape"] blur me', 'it');
148
+ expect(render).toContain('sfuocare');
149
+ expect(t).toContain('blur.patient=me');
150
+ });
151
+
152
+ it('it: blur event position still captures on.event="blur"', () => {
153
+ const { triples: t } = renderAndParse('on blur log "x"', 'it');
154
+ expect(t).toContain('on.event=blur');
155
+ });
156
+ });
157
+
158
+ describe('Batch 3 — empty class (ko/qu profile alternatives)', () => {
159
+ it('ko: the dict adjective now parses the empty command', () => {
160
+ const t = triples(parseSemantic('#list 를 비어있는', 'ko').node);
161
+ expect(t).toContain('empty.patient=#list');
162
+ });
163
+
164
+ it('qu: apostrophe-less chusaq now parses the empty command', () => {
165
+ const t = triples(parseSemantic('#list ta chusaq', 'qu').node);
166
+ expect(t).toContain('empty.patient=#list');
167
+ });
168
+
169
+ it('ko/qu: the hot `is empty` expression rows keep parsing without a phantom empty action', () => {
170
+ for (const [text, langCode] of [
171
+ [
172
+ '블러 할 때 만약 내 값 이다 비어있는 .error 를 추가 나 에 아니면 .error 를 제거 나 에서 끝',
173
+ 'ko',
174
+ ],
175
+ [
176
+ 'paqariy pi sichus noqaq chanin kanqa chusaq .error ta noqa man yapay manachus .error ta noqa manta qichuy tukuy',
177
+ 'qu',
178
+ ],
179
+ ] as const) {
180
+ const t = triples(parseSemantic(text, langCode).node);
181
+ expect(t.some(x => x.startsWith('empty.'))).toBe(false);
182
+ }
183
+ });
184
+ });