@hyperfixi/testing-framework 2.7.2 → 2.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. package/CHANGELOG.md +393 -0
  2. package/dist/assertions.d.mts +26 -1
  3. package/dist/assertions.d.ts +26 -1
  4. package/dist/index.d.mts +5 -112
  5. package/dist/index.d.ts +5 -112
  6. package/dist/runner.d.mts +112 -0
  7. package/dist/runner.d.ts +112 -0
  8. package/dist/runner.js +1102 -0
  9. package/dist/runner.js.map +1 -0
  10. package/dist/runner.mjs +1097 -0
  11. package/dist/runner.mjs.map +1 -0
  12. package/dist/{assertions-CsGP61iW.d.mts → types-D-rCVkf3.d.mts} +1 -24
  13. package/dist/{assertions-CsGP61iW.d.ts → types-D-rCVkf3.d.ts} +1 -24
  14. package/package.json +13 -27
  15. package/src/multilingual/canonical-validity.test.ts +69 -0
  16. package/src/multilingual/canonical-validity.ts +132 -0
  17. package/src/multilingual/cli.ts +247 -14
  18. package/src/multilingual/fidelity.test.ts +192 -0
  19. package/src/multilingual/fidelity.ts +153 -0
  20. package/src/multilingual/foreign-canonical-validity.test.ts +0 -0
  21. package/src/multilingual/foreign-canonical-validity.ts +158 -0
  22. package/src/multilingual/orchestrator.ts +47 -1
  23. package/src/multilingual/reporters/console-reporter.ts +51 -0
  24. package/src/multilingual/reporters/regression-reporter.test.ts +78 -0
  25. package/src/multilingual/reporters/regression-reporter.ts +28 -1
  26. package/src/multilingual/tools/diagnose-coverage.ts +118 -0
  27. package/src/multilingual/tools/triage-r1.ts +149 -0
  28. package/src/multilingual/types.ts +74 -0
  29. package/src/multilingual/validators/parse-validator.ts +9 -1
  30. package/src/runner.test.ts +7 -2
  31. package/src/vocab/batch3-roundtrip.test.ts +184 -0
  32. package/src/vocab/checks.test.ts +362 -0
  33. package/src/vocab/checks.ts +311 -0
  34. package/src/vocab/cli.ts +196 -0
  35. package/src/vocab/dump.ts +80 -0
  36. package/src/vocab/model.ts +110 -0
  37. package/src/vocab/report.ts +120 -0
  38. package/src/vocab/types.ts +100 -0
@@ -121,6 +121,39 @@ export function computeFidelity(
121
121
  return hits / reference.length;
122
122
  }
123
123
 
124
+ /**
125
+ * R0-recall on the **multiset** in [0, 1]: the fraction of the reference's
126
+ * actions — counting duplicates — also present in the candidate.
127
+ *
128
+ * {@link computeFidelity} scores the deduped Set signature, so a candidate that
129
+ * drops a REPEATED command scores 1.0: reference `[bind, bind]` collapses to
130
+ * `{bind}`, which `[bind]` satisfies in full. That is how `bind-two-way` sat at
131
+ * fidelity 1.0 across all 24 languages while every one of them parsed only the
132
+ * first of its two `bind`s. R1 (role signatures) is a Set too, and is equally
133
+ * blind. {@link computePrecision} catches the mirror case — a candidate that ADDS
134
+ * a duplicate — so before this signal existed the ratchet saw spurious commands
135
+ * but never dropped ones.
136
+ *
137
+ * Pass multisets (see {@link collectActionsMultiset}) on both sides.
138
+ * Returns `undefined` when the reference has no actions to compare against.
139
+ */
140
+ export function computeMultisetRecall(
141
+ reference: readonly string[],
142
+ candidate: readonly string[]
143
+ ): number | undefined {
144
+ if (reference.length === 0) return undefined;
145
+ const cand = actionCounts(candidate);
146
+ let matched = 0;
147
+ for (const a of reference) {
148
+ const remaining = cand.get(a) ?? 0;
149
+ if (remaining > 0) {
150
+ matched++;
151
+ cand.set(a, remaining - 1);
152
+ }
153
+ }
154
+ return matched / reference.length;
155
+ }
156
+
124
157
  /**
125
158
  * Structural **precision** in [0, 1]: the fraction of the *candidate's* actions
126
159
  * that are justified by the reference (multiset-aware). The complement of
@@ -222,3 +255,123 @@ function walkRoles(node: unknown, acc: Set<string>, depth: number): void {
222
255
  }
223
256
  }
224
257
  }
258
+
259
+ /**
260
+ * Role-value kinds never emitted by the R3 walker. `reference` values (`me`,
261
+ * `it`, …) are mostly `fillSchemaDefaults` injections present identically on
262
+ * both sides — noise. `flag` names and `property-path` properties are bare
263
+ * identifiers, excluded by v1 (see {@link collectRoleValueSignature}).
264
+ */
265
+ const VALUE_KIND_EXCLUSIONS = new Set(['reference', 'flag', 'property-path']);
266
+
267
+ /**
268
+ * A role value is compared cross-language only when its WHOLE surface form is
269
+ * code-shaped — language-invariant by construction, never legitimately
270
+ * translated. Conservative v1 whitelist; when a sweep firing turns out to be a
271
+ * legit translation difference, tighten here and document the exclusion.
272
+ */
273
+ const INVARIANT_VALUE_PATTERNS: readonly RegExp[] = [
274
+ /^[#.[<@*]/, // selectors: #id .class [attr] <tag/> @attr *style
275
+ /^[:$^][A-Za-z_]\w*$/, // sigil refs: :local $global ^element
276
+ /^\d+(\.\d+)?(ms|s|m|h)?$/, // numbers and time literals: 2 1.5 200ms
277
+ /^[A-Za-z_][\w-]*:[\w-]+$/, // colon-qualified event names: draggable:start
278
+ /^(\.{0,2}\/|https?:)/, // URLs / paths: /api/data ./x ../y https://…
279
+ ];
280
+
281
+ function isInvariantSurface(surface: string): boolean {
282
+ // Whole-surface rule, sweep-validated exclusions:
283
+ // - whitespace ⇒ mixed content. `if #modal exists` captures its condition as
284
+ // `#modal exists` — starts selector-shaped, but `exists` is prose that every
285
+ // language legitimately translates (16-language false firing without this).
286
+ // - `${` ⇒ template interpolation. `/api/search?q=${my value}` is an
287
+ // expression, and tokenizers split it at different points per language, so
288
+ // the captured surface isn't comparable verbatim.
289
+ if (/\s/.test(surface) || surface.includes('${')) return false;
290
+ return INVARIANT_VALUE_PATTERNS.some(re => re.test(surface));
291
+ }
292
+
293
+ /**
294
+ * The comparable surface form of a role value, or `undefined` when the value
295
+ * carries none: `.value` for `literal`/`selector` (string/number/boolean
296
+ * coerced), `.raw` for `expression`. Kinds in {@link VALUE_KIND_EXCLUSIONS}
297
+ * and values without a string `type` discriminator yield `undefined`.
298
+ */
299
+ function roleValueSurface(value: unknown): string | undefined {
300
+ if (value === null || typeof value !== 'object') return undefined;
301
+ const rec = value as { type?: unknown; value?: unknown; raw?: unknown };
302
+ if (typeof rec.type !== 'string' || VALUE_KIND_EXCLUSIONS.has(rec.type)) return undefined;
303
+ if (rec.type === 'expression') {
304
+ return typeof rec.raw === 'string' ? rec.raw : undefined;
305
+ }
306
+ const v = rec.value;
307
+ return typeof v === 'string' || typeof v === 'number' || typeof v === 'boolean'
308
+ ? String(v)
309
+ : undefined;
310
+ }
311
+
312
+ /**
313
+ * R3 — role-VALUE signature (invariant values only, multiset).
314
+ *
315
+ * R0/R1 compare actions and role *types*; values are never compared because
316
+ * they are legitimately translated. That leaves a live defect class with zero
317
+ * signal: right action counts, right role types, wrong role VALUE — the #633
318
+ * class, where ms captured `trigger` events named `draggable` instead of
319
+ * `draggable:start` (correct multiset, correct `trigger.event:literal`
320
+ * signature, silently wrong runtime behavior), and 18 other languages carried
321
+ * the same corruption with no side-effect at all.
322
+ *
323
+ * The subset of values compared here is language-invariant by construction —
324
+ * code, not prose (see {@link INVARIANT_VALUE_PATTERNS}). Emits a **multiset**
325
+ * of `` `action.role=value` `` entries (duplicates preserved, sorted); score
326
+ * with {@link computeMultisetRecall} so a dropped duplicate value is visible.
327
+ *
328
+ * Deliberately EXCLUDED in v1: bare-word identifiers (`startX` — usually
329
+ * invariant, but property/variable names occasionally get localized in seeds),
330
+ * string literals (message strings are legitimately translated), expression
331
+ * raws mixing native words + code (`次 .item`), and `reference` values
332
+ * (`me`/`it` — mostly `fillSchemaDefaults` injections on both sides).
333
+ *
334
+ * Blind spot: recall fires when a *translation* loses/corrupts an invariant
335
+ * value. If the **en reference itself** corrupts a value, every language flags
336
+ * at once — a 24-language R3 firestorm on one pattern means "suspect the en
337
+ * parse first" (unlike R0, where en corruption moves nothing).
338
+ *
339
+ * The roles container is a ReadonlyMap on live nodes (serializes to {} in
340
+ * results.json), so collect at validation time, never from the JSON.
341
+ */
342
+ export function collectRoleValueSignature(node: unknown): string[] {
343
+ const acc: string[] = [];
344
+ walkRoleValues(node, acc, 0);
345
+ return acc.sort();
346
+ }
347
+
348
+ function walkRoleValues(node: unknown, acc: string[], depth: number): void {
349
+ if (depth > 64 || node === null || typeof node !== 'object') return;
350
+
351
+ const rec = node as Record<string, unknown>;
352
+ const action = rec.action;
353
+ if (typeof action === 'string' && !STRUCTURAL_ACTIONS.has(action)) {
354
+ const roles = rec.roles;
355
+ const entries: Array<[unknown, unknown]> =
356
+ roles instanceof Map
357
+ ? [...roles.entries()]
358
+ : roles && typeof roles === 'object'
359
+ ? Object.entries(roles)
360
+ : [];
361
+ for (const [role, value] of entries) {
362
+ if (value === undefined || value === null) continue;
363
+ const surface = roleValueSurface(value);
364
+ if (surface === undefined || !isInvariantSurface(surface)) continue;
365
+ acc.push(`${action}.${String(role)}=${surface}`);
366
+ }
367
+ }
368
+
369
+ for (const field of CHILD_FIELDS) {
370
+ const child = rec[field];
371
+ if (Array.isArray(child)) {
372
+ for (const c of child) walkRoleValues(c, acc, depth + 1);
373
+ } else if (child && typeof child === 'object') {
374
+ walkRoleValues(child, acc, depth + 1);
375
+ }
376
+ }
377
+ }
@@ -0,0 +1,158 @@
1
+ /**
2
+ * Foreign→English canonical-validity gate
3
+ * ---------------------------------------
4
+ * The sibling en-render gate (canonical-validity.ts) renders every corpus English
5
+ * reference and parses it on the real `hyperscript.org` engine. But the PRODUCTION
6
+ * path is foreign→English: an authored non-English source is parsed and rendered to
7
+ * English (`preprocessToEnglish` → `render`). A parse can be role-faithful (the
8
+ * fidelity ratchet scores ~1.0) yet still render English the canonical parser rejects.
9
+ * (The build-time `@hyperscript-tools/i18n` transpiler now parse-gates its ENGLISH
10
+ * side with the same loader recipe; faithful foreign-output gating there still awaits
11
+ * the v2 semantic-engine transpiler, since its GrammarTransformer is lossy in reverse.)
12
+ *
13
+ * This gate closes that blind spot for the multilingual path: for every language, it
14
+ * renders each authored `pattern_translation` to English and parses the result on
15
+ * the canonical engine, failing on any invalid (pattern, language) pair that is not
16
+ * in the committed allowlist. The allowlist is keyed by pattern id → the languages
17
+ * that currently fail, so a fix that clears a family across languages shrinks (or
18
+ * removes) its entry — the list only ever ratchets down.
19
+ *
20
+ * Denominator: only (pattern, language) pairs whose EN reference the canonical parser
21
+ * already accepts are scored, so a handful of inherently non-canonical corpus rows
22
+ * never distort the signal (same fairness rule as the en gate).
23
+ *
24
+ * DB dependency: reads authored translations from `pattern_translations`, which only
25
+ * exist after `npm run populate`. Generate the baseline and run the gate against a
26
+ * freshly populated DB (CI's multilingual-validation job populates; see the
27
+ * provenance-stamp discipline in packages/patterns-reference/CLAUDE.md).
28
+ */
29
+
30
+ import { getAllPatterns, getTranslationsByLanguage } from '@hyperfixi/patterns-reference';
31
+ import { parseSemantic, render } from '@lokascript/semantic';
32
+ import { loadCanonicalParser, type CanonicalValidate } from './canonical-validity';
33
+
34
+ /**
35
+ * The 23 non-English priority languages (the `browser-priority` corpus set). English
36
+ * is the reference, scored by the sibling en gate.
37
+ */
38
+ export const FOREIGN_LANGUAGES = [
39
+ 'es',
40
+ 'fr',
41
+ 'pt',
42
+ 'it',
43
+ 'id',
44
+ 'ms',
45
+ 'sw',
46
+ 'zh',
47
+ 'vi',
48
+ 'tl',
49
+ 'ja',
50
+ 'ko',
51
+ 'tr',
52
+ 'qu',
53
+ 'hi',
54
+ 'bn',
55
+ 'ar',
56
+ 'de',
57
+ 'ru',
58
+ 'uk',
59
+ 'pl',
60
+ 'th',
61
+ 'he',
62
+ ] as const;
63
+
64
+ export interface ForeignValidityFailure {
65
+ /** Corpus pattern id (`code_example` id the translation belongs to). */
66
+ id: string;
67
+ language: string;
68
+ /** The authored foreign source that was rendered. */
69
+ foreign: string;
70
+ /** The English the renderer produced from the foreign parse. */
71
+ rendered: string;
72
+ error: string;
73
+ }
74
+
75
+ export interface ForeignValidityResult {
76
+ /** (pattern, language) pairs whose EN reference the canonical parser accepts. */
77
+ checked: number;
78
+ valid: number;
79
+ failures: ForeignValidityFailure[];
80
+ }
81
+
82
+ /**
83
+ * Render every authored foreign `pattern_translation` to English and parse it on the
84
+ * canonical engine. Only pairs whose EN reference is itself canonical-valid are scored.
85
+ */
86
+ export async function checkForeignRenderValidity(opts?: {
87
+ validate?: CanonicalValidate;
88
+ languages?: readonly string[];
89
+ /** Translations fetched per language (defaults comfortably above the corpus size). */
90
+ perLanguageLimit?: number;
91
+ }): Promise<ForeignValidityResult> {
92
+ const validate = opts?.validate ?? (await loadCanonicalParser());
93
+ const languages = opts?.languages ?? FOREIGN_LANGUAGES;
94
+ const limit = opts?.perLanguageLimit ?? 500;
95
+
96
+ const patterns = await getAllPatterns();
97
+ const enAccepts = new Map(patterns.map(p => [p.id, validate(p.rawCode).length === 0]));
98
+
99
+ const failures: ForeignValidityFailure[] = [];
100
+ let checked = 0;
101
+ let valid = 0;
102
+
103
+ for (const language of languages) {
104
+ const translations = await getTranslationsByLanguage(language, limit);
105
+ for (const t of translations) {
106
+ if (!enAccepts.get(t.codeExampleId)) continue; // fair denominator: EN accepts the reference
107
+ checked++;
108
+
109
+ let rendered: string;
110
+ let errors: string[];
111
+ try {
112
+ const node = parseSemantic(t.hyperscript, language).node;
113
+ rendered = node ? render(node, 'en') : '(no node)';
114
+ errors = validate(rendered);
115
+ } catch (e) {
116
+ // parseSemantic/render only — validate never throws (its tokenizer-level
117
+ // throws fold into the returned array), so an invalid render is reported
118
+ // WITH the render that caused it, not discarded as '(threw)'.
119
+ rendered = '(threw)';
120
+ errors = ['threw: ' + (e as Error).message.split('\n')[0]];
121
+ }
122
+
123
+ if (errors.length === 0) {
124
+ valid++;
125
+ } else {
126
+ failures.push({
127
+ id: t.codeExampleId,
128
+ language,
129
+ foreign: t.hyperscript,
130
+ rendered,
131
+ error: errors[0] ?? 'unknown error',
132
+ });
133
+ }
134
+ }
135
+ }
136
+
137
+ return { checked, valid, failures };
138
+ }
139
+
140
+ /**
141
+ * Group failures into the committed allowlist shape: `{ patternId: [langs…] }`
142
+ * (languages sorted for a stable diff). Used by the baseline generator and by the
143
+ * gate's stale-entry check.
144
+ */
145
+ export function groupFailuresByPattern(
146
+ failures: readonly ForeignValidityFailure[]
147
+ ): Record<string, string[]> {
148
+ const byPattern = new Map<string, Set<string>>();
149
+ for (const f of failures) {
150
+ if (!byPattern.has(f.id)) byPattern.set(f.id, new Set());
151
+ byPattern.get(f.id)!.add(f.language);
152
+ }
153
+ const out: Record<string, string[]> = {};
154
+ for (const id of [...byPattern.keys()].sort()) {
155
+ out[id] = [...byPattern.get(id)!].sort();
156
+ }
157
+ return out;
158
+ }
@@ -20,7 +20,13 @@ import type {
20
20
  Reporter,
21
21
  BundleInfo,
22
22
  } from './types';
23
- import { computeFidelity, computePrecision, FIDELITY_THRESHOLD } from './fidelity';
23
+ import {
24
+ computeFidelity,
25
+ computePrecision,
26
+ computeMultisetRecall,
27
+ spuriousActions,
28
+ FIDELITY_THRESHOLD,
29
+ } from './fidelity';
24
30
 
25
31
  const execAsync = promisify(exec);
26
32
 
@@ -258,6 +264,8 @@ export class TestOrchestrator {
258
264
  const multisetReference = new Map<string, string[]>();
259
265
  // R1: codeExampleId -> English role signature (action.role:valueType set).
260
266
  const roleReference = new Map<string, string[]>();
267
+ // R3: codeExampleId -> English role-VALUE signature (invariant values, multiset).
268
+ const roleValueReference = new Map<string, string[]>();
261
269
  for (const r of en.parseResults) {
262
270
  if (r.success && r.actionSignature && r.actionSignature.length > 0) {
263
271
  reference.set(r.pattern.codeExampleId, r.actionSignature);
@@ -268,6 +276,9 @@ export class TestOrchestrator {
268
276
  if (r.success && r.roleSignature && r.roleSignature.length > 0) {
269
277
  roleReference.set(r.pattern.codeExampleId, r.roleSignature);
270
278
  }
279
+ if (r.success && r.roleValueSignature && r.roleValueSignature.length > 0) {
280
+ roleValueReference.set(r.pattern.codeExampleId, r.roleValueSignature);
281
+ }
271
282
  }
272
283
 
273
284
  for (const lang of languageResults) {
@@ -277,7 +288,9 @@ export class TestOrchestrator {
277
288
  const lossy: string[] = [];
278
289
  const scores: number[] = [];
279
290
  const precisionScores: number[] = [];
291
+ const multisetRecallScores: number[] = [];
280
292
  const roleScores: number[] = [];
293
+ const valueRecallScores: number[] = [];
281
294
 
282
295
  for (const result of lang.parseResults) {
283
296
  if (!result.success || !result.actionSignature) continue;
@@ -294,6 +307,9 @@ export class TestOrchestrator {
294
307
 
295
308
  // R0-precision — fraction of THIS parse's actions justified by the en
296
309
  // multiset reference (catches phantom/spurious commands recall misses).
310
+ // R0-recall-multiset — the mirror: fraction of the en reference's actions,
311
+ // counting duplicates, present here (catches a DROPPED repeated command,
312
+ // which the Set-based fidelity/roleFidelity above cannot see).
297
313
  const multisetRef = multisetReference.get(result.pattern.codeExampleId);
298
314
  if (multisetRef && result.actionMultisetSignature) {
299
315
  const precision = computePrecision(multisetRef, result.actionMultisetSignature);
@@ -301,6 +317,11 @@ export class TestOrchestrator {
301
317
  result.precision = precision;
302
318
  precisionScores.push(precision);
303
319
  }
320
+ const multisetRecall = computeMultisetRecall(multisetRef, result.actionMultisetSignature);
321
+ if (multisetRecall !== undefined) {
322
+ result.multisetRecall = multisetRecall;
323
+ multisetRecallScores.push(multisetRecall);
324
+ }
304
325
  }
305
326
 
306
327
  // R1 — role recall vs the en role signature (role name + value type).
@@ -312,6 +333,23 @@ export class TestOrchestrator {
312
333
  roleScores.push(roleFidelity);
313
334
  }
314
335
  }
336
+
337
+ // R3 — invariant role-VALUE recall vs the en value signature (multiset).
338
+ // Values are compared verbatim: the filtered subset (selectors, sigil
339
+ // refs, time literals, colon-qualified events, URLs) is code, not prose.
340
+ const valueRef = roleValueReference.get(result.pattern.codeExampleId);
341
+ if (valueRef && result.roleValueSignature) {
342
+ const valueRecall = computeMultisetRecall(valueRef, result.roleValueSignature);
343
+ if (valueRecall !== undefined) {
344
+ result.valueRecall = valueRecall;
345
+ valueRecallScores.push(valueRecall);
346
+ if (valueRecall < 1) {
347
+ // Missing = reference entries not covered here (multiset diff;
348
+ // spuriousActions with the arguments swapped).
349
+ result.valueRecallMissing = spuriousActions(result.roleValueSignature, valueRef);
350
+ }
351
+ }
352
+ }
315
353
  }
316
354
 
317
355
  lang.avgFidelity =
@@ -320,10 +358,18 @@ export class TestOrchestrator {
320
358
  precisionScores.length > 0
321
359
  ? precisionScores.reduce((a, b) => a + b, 0) / precisionScores.length
322
360
  : undefined;
361
+ lang.avgMultisetRecall =
362
+ multisetRecallScores.length > 0
363
+ ? multisetRecallScores.reduce((a, b) => a + b, 0) / multisetRecallScores.length
364
+ : undefined;
323
365
  lang.avgRoleFidelity =
324
366
  roleScores.length > 0
325
367
  ? roleScores.reduce((a, b) => a + b, 0) / roleScores.length
326
368
  : undefined;
369
+ lang.avgValueRecall =
370
+ valueRecallScores.length > 0
371
+ ? valueRecallScores.reduce((a, b) => a + b, 0) / valueRecallScores.length
372
+ : undefined;
327
373
  lang.degeneratePasses = degenerate.sort();
328
374
  lang.lossyPasses = lossy.sort();
329
375
  }
@@ -158,11 +158,62 @@ export class ConsoleReporter implements Reporter {
158
158
  }
159
159
 
160
160
  this.reportDegeneratePasses(results);
161
+ this.reportValueRecall(results);
161
162
  this.reportExecutionFailures(results);
162
163
 
163
164
  this.log('');
164
165
  }
165
166
 
167
+ /**
168
+ * R3 — surface parses that lost/corrupted a language-invariant role VALUE
169
+ * (selector, sigil ref, time literal, colon-qualified event name, URL) vs
170
+ * the en reference. These score 1.0 on every action/type-based signal yet
171
+ * behave differently at runtime, so they're reported separately. Report-only:
172
+ * the ratchet gate lives in the --regression path.
173
+ */
174
+ private reportValueRecall(results: TestResults): void {
175
+ const scored = results.languageResults.filter(l => l.avgValueRecall !== undefined);
176
+ if (scored.length === 0) return;
177
+
178
+ const avg = scored.reduce((sum, l) => sum + (l.avgValueRecall ?? 0), 0) / scored.length;
179
+ this.log('');
180
+ this.log(
181
+ this.bright(`Role values (R3): avgValueRecall ${avg.toFixed(4)} over invariant values`)
182
+ );
183
+
184
+ // pattern id -> "lang: missing entries" rows
185
+ const byPattern = new Map<string, string[]>();
186
+ let instances = 0;
187
+ for (const lang of scored) {
188
+ for (const r of lang.parseResults) {
189
+ if (r.valueRecall === undefined || r.valueRecall >= 1) continue;
190
+ instances++;
191
+ let rows = byPattern.get(r.pattern.codeExampleId);
192
+ if (!rows) {
193
+ rows = [];
194
+ byPattern.set(r.pattern.codeExampleId, rows);
195
+ }
196
+ rows.push(`${lang.language}: missing ${(r.valueRecallMissing ?? []).join(', ')}`);
197
+ }
198
+ }
199
+ if (byPattern.size === 0) return;
200
+
201
+ this.log(
202
+ this.yellow(
203
+ `⚠ Invariant value loss: ${instances} instance(s) across ${byPattern.size} pattern(s)`
204
+ )
205
+ );
206
+ this.log(
207
+ this.dim(
208
+ ' (an action.role=value present in the en reference is absent — not parse failures)'
209
+ )
210
+ );
211
+ for (const [id, rows] of [...byPattern.entries()].sort((a, b) => b[1].length - a[1].length)) {
212
+ this.log(` ${this.dim('-')} ${id}`);
213
+ for (const row of rows.sort()) this.log(` ${this.dim(row)}`);
214
+ }
215
+ }
216
+
166
217
  /**
167
218
  * R2 — surface curated-subset patterns whose jsdom execution diverged from
168
219
  * the en reference's DOM effects. These can score 1.0 on parse fidelity yet
@@ -404,3 +404,81 @@ describe('RegressionReporter precision ratchet — R0-precision (trust floor)',
404
404
  expect(saved.languages.ja!.avgPrecision).toBeCloseTo(0.95, 5);
405
405
  });
406
406
  });
407
+
408
+ describe('RegressionReporter per-pattern parse ratchet — R5', () => {
409
+ let dir: string;
410
+ afterEach(() => {
411
+ if (dir) rmSync(dir, { recursive: true, force: true });
412
+ });
413
+
414
+ function reporterWith(baseline: Baseline): RegressionReporter {
415
+ dir = mkdtempSync(join(tmpdir(), 'parse-ratchet-'));
416
+ const path = join(dir, 'baseline.json');
417
+ writeFileSync(path, JSON.stringify(baseline));
418
+ return new RegressionReporter(path);
419
+ }
420
+
421
+ /** Minimal ParseResult for a pattern that no longer parses. */
422
+ function fail(id: string): ParseResult {
423
+ return { ...pass(id), success: false };
424
+ }
425
+
426
+ function baselineOf(ids: string[], failing: string[] = []): Baseline {
427
+ const patterns: Record<string, { success: boolean; confidence: number | undefined }> = {};
428
+ for (const id of ids) {
429
+ patterns[id] = { success: !failing.includes(id), confidence: 1 };
430
+ }
431
+ return {
432
+ timestamp: '',
433
+ commit: 'base',
434
+ languages: {
435
+ ja: {
436
+ parseSuccess: ids.length - failing.length,
437
+ parseFailure: failing.length,
438
+ parseRate: (ids.length - failing.length) / ids.length,
439
+ avgConfidence: 1,
440
+ avgFidelity: 1,
441
+ degeneratePasses: [],
442
+ lossyPasses: [],
443
+ bundleSize: undefined,
444
+ patterns,
445
+ },
446
+ },
447
+ bundles: {},
448
+ };
449
+ }
450
+
451
+ it('flags a baseline pass that no longer parses', () => {
452
+ const reporter = reporterWith(baselineOf(['a', 'b', 'c']));
453
+ reporter.reportComplete(results(lang([pass('a'), fail('b'), pass('c')], [])));
454
+ const r = reporter.getRegressionResults().find(x => x.language === 'ja')!;
455
+ expect(r.newFailures).toEqual(['b']);
456
+ });
457
+
458
+ it('does not flag a pattern that was already failing in the baseline', () => {
459
+ const reporter = reporterWith(baselineOf(['a', 'b'], ['b']));
460
+ reporter.reportComplete(results(lang([pass('a'), fail('b')], [])));
461
+ const r = reporter.getRegressionResults().find(x => x.language === 'ja')!;
462
+ expect(r.newFailures).toEqual([]);
463
+ });
464
+
465
+ it('never retro-flags when the baseline has no per-pattern data', () => {
466
+ const noPatterns = baselineOf(['a', 'b']);
467
+ delete (noPatterns.languages.ja as { patterns?: unknown }).patterns;
468
+ const reporter = reporterWith(noPatterns);
469
+ reporter.reportComplete(results(lang([fail('a'), fail('b')], [])));
470
+ const r = reporter.getRegressionResults().find(x => x.language === 'ja')!;
471
+ expect(r.newFailures).toEqual([]);
472
+ });
473
+
474
+ // The list is the R5 gate's evidence, not just a reporting nicety: it used to
475
+ // be `.slice(0, 10)`, which would have under-reported how much broke — and, if
476
+ // the gate ever counted it, capped the count at 10.
477
+ it('returns every new failure, uncapped', () => {
478
+ const ids = Array.from({ length: 25 }, (_, i) => `p${i}`);
479
+ const reporter = reporterWith(baselineOf(ids));
480
+ reporter.reportComplete(results(lang(ids.map(fail), [])));
481
+ const r = reporter.getRegressionResults().find(x => x.language === 'ja')!;
482
+ expect(r.newFailures).toHaveLength(25);
483
+ });
484
+ });
@@ -118,12 +118,25 @@ export class RegressionReporter implements Reporter {
118
118
  langResult.avgPrecision !== undefined && baselineLang.avgPrecision !== undefined
119
119
  ? langResult.avgPrecision - baselineLang.avgPrecision
120
120
  : 0;
121
+ // R0-recall-multiset — same both-sides guard. Negative = a repeated command
122
+ // is now being dropped, which avgFidelity (a Set) cannot see.
123
+ const avgMultisetRecallDelta =
124
+ langResult.avgMultisetRecall !== undefined && baselineLang.avgMultisetRecall !== undefined
125
+ ? langResult.avgMultisetRecall - baselineLang.avgMultisetRecall
126
+ : 0;
121
127
  // R1 — only meaningful when BOTH sides carry role data; an un-regenerated
122
128
  // baseline (no avgRoleFidelity yet) must never retro-flag.
123
129
  const avgRoleFidelityDelta =
124
130
  langResult.avgRoleFidelity !== undefined && baselineLang.avgRoleFidelity !== undefined
125
131
  ? langResult.avgRoleFidelity - baselineLang.avgRoleFidelity
126
132
  : 0;
133
+ // R3 — same both-sides guard. Negative = a language-invariant role VALUE
134
+ // (selector/sigil/time/colon-event/URL) is now lost or corrupted, which
135
+ // every action/type-based signal scores as perfect.
136
+ const avgValueRecallDelta =
137
+ langResult.avgValueRecall !== undefined && baselineLang.avgValueRecall !== undefined
138
+ ? langResult.avgValueRecall - baselineLang.avgValueRecall
139
+ : 0;
127
140
  // R2 — same both-sides guard: an un-regenerated baseline (no execution
128
141
  // data yet) must never retro-flag.
129
142
  const avgExecutionFidelityDelta =
@@ -163,7 +176,9 @@ export class RegressionReporter implements Reporter {
163
176
  avgConfidenceDelta,
164
177
  avgFidelityDelta,
165
178
  avgPrecisionDelta,
179
+ avgMultisetRecallDelta,
166
180
  avgRoleFidelityDelta,
181
+ avgValueRecallDelta,
167
182
  avgExecutionFidelityDelta,
168
183
  bundleSizeDelta: bundleSizeDelta !== undefined ? bundleSizeDelta : undefined,
169
184
  newFailures,
@@ -211,7 +226,11 @@ export class RegressionReporter implements Reporter {
211
226
  }
212
227
  }
213
228
 
214
- return newFailures.slice(0, 10); // Limit to 10 for reporting
229
+ // Returned UNCAPPED: this list is the R5 parse ratchet's evidence (cli.ts),
230
+ // not just a reporting nicety, and a gate that silently truncated itself
231
+ // would under-report how much broke. Display sites do their own capping
232
+ // (console-reporter shows 5; the CLI shows 20).
233
+ return newFailures;
215
234
  }
216
235
 
217
236
  /**
@@ -364,9 +383,17 @@ export class RegressionReporter implements Reporter {
364
383
  // R0-precision — fraction of each parse's actions justified by the en
365
384
  // reference (phantom-command signal recall can't see). Recorded + ratcheted.
366
385
  avgPrecision: langResult.avgPrecision ?? undefined,
386
+ // R0-recall-multiset — the mirror of precision: a DROPPED repeated command
387
+ // (`[bind, bind]` parsed as `[bind]`) is invisible to the Set-based
388
+ // avgFidelity and avgRoleFidelity. Recorded + ratcheted.
389
+ avgMultisetRecall: langResult.avgMultisetRecall ?? undefined,
367
390
  // R1 — role fidelity (role name + value type vs the en reference).
368
391
  // Recorded + ratcheted; burn-down is NOT part of the parsing-track goal.
369
392
  avgRoleFidelity: langResult.avgRoleFidelity ?? undefined,
393
+ // R3 — invariant role-VALUE recall (verbatim value comparison over the
394
+ // code-shaped subset: selectors, sigil refs, time literals,
395
+ // colon-qualified event names, URLs). Recorded + ratcheted.
396
+ avgValueRecall: langResult.avgValueRecall ?? undefined,
370
397
  // R2 — execution fidelity over the curated subset (DOM effects vs the
371
398
  // en reference in jsdom). Recorded + ratcheted; burn-down deferred.
372
399
  avgExecutionFidelity: langResult.avgExecutionFidelity ?? undefined,