@hyperfixi/testing-framework 2.7.2 → 2.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. package/CHANGELOG.md +393 -0
  2. package/dist/assertions.d.mts +26 -1
  3. package/dist/assertions.d.ts +26 -1
  4. package/dist/index.d.mts +5 -112
  5. package/dist/index.d.ts +5 -112
  6. package/dist/runner.d.mts +112 -0
  7. package/dist/runner.d.ts +112 -0
  8. package/dist/runner.js +1102 -0
  9. package/dist/runner.js.map +1 -0
  10. package/dist/runner.mjs +1097 -0
  11. package/dist/runner.mjs.map +1 -0
  12. package/dist/{assertions-CsGP61iW.d.mts → types-D-rCVkf3.d.mts} +1 -24
  13. package/dist/{assertions-CsGP61iW.d.ts → types-D-rCVkf3.d.ts} +1 -24
  14. package/package.json +13 -27
  15. package/src/multilingual/canonical-validity.test.ts +69 -0
  16. package/src/multilingual/canonical-validity.ts +132 -0
  17. package/src/multilingual/cli.ts +247 -14
  18. package/src/multilingual/fidelity.test.ts +192 -0
  19. package/src/multilingual/fidelity.ts +153 -0
  20. package/src/multilingual/foreign-canonical-validity.test.ts +0 -0
  21. package/src/multilingual/foreign-canonical-validity.ts +158 -0
  22. package/src/multilingual/orchestrator.ts +47 -1
  23. package/src/multilingual/reporters/console-reporter.ts +51 -0
  24. package/src/multilingual/reporters/regression-reporter.test.ts +78 -0
  25. package/src/multilingual/reporters/regression-reporter.ts +28 -1
  26. package/src/multilingual/tools/diagnose-coverage.ts +118 -0
  27. package/src/multilingual/tools/triage-r1.ts +149 -0
  28. package/src/multilingual/types.ts +74 -0
  29. package/src/multilingual/validators/parse-validator.ts +9 -1
  30. package/src/runner.test.ts +7 -2
  31. package/src/vocab/batch3-roundtrip.test.ts +184 -0
  32. package/src/vocab/checks.test.ts +362 -0
  33. package/src/vocab/checks.ts +311 -0
  34. package/src/vocab/cli.ts +196 -0
  35. package/src/vocab/dump.ts +80 -0
  36. package/src/vocab/model.ts +110 -0
  37. package/src/vocab/report.ts +120 -0
  38. package/src/vocab/types.ts +100 -0
@@ -10,6 +10,8 @@ import * as path from 'node:path';
10
10
  import { fileURLToPath } from 'node:url';
11
11
  import { checkDbStamp, getDefaultDbPath } from '@hyperfixi/patterns-reference';
12
12
  import { TestOrchestrator } from './orchestrator';
13
+ import { diagnoseCoverage } from './tools/diagnose-coverage';
14
+ import { triageR1 } from './tools/triage-r1';
13
15
  import type { TestConfig, LanguageCode } from './types';
14
16
 
15
17
  /**
@@ -172,6 +174,17 @@ function parseArgs(): TestConfig {
172
174
  config.saveBaseline = true;
173
175
  break;
174
176
 
177
+ case '--diagnose-coverage':
178
+ // Read-only measurement pass; short-circuits main() before any gate.
179
+ config.diagnoseCoverage = true;
180
+ break;
181
+
182
+ case '--triage-r1':
183
+ // Read-only measurement pass; short-circuits main() before any gate.
184
+ // Combine with --languages to scope (en is force-loaded as reference).
185
+ config.triageR1 = true;
186
+ break;
187
+
175
188
  default:
176
189
  if (arg && arg.startsWith('-')) {
177
190
  console.error(`Unknown option: ${arg}`);
@@ -227,6 +240,15 @@ OPTIONS:
227
240
  --categories <cats> Filter by categories (comma-separated)
228
241
  --limit <n> Patterns per language in quick mode (default: 10)
229
242
  --save-baseline Save current results as new baseline
243
+ --triage-r1 Itemize R1 role-fidelity misses per language
244
+ (missing action.role:type entries vs the en
245
+ reference, clustered; scope with --languages)
246
+ --diagnose-coverage Report how often the semantic parser matched a
247
+ pattern that ignored part of its input, per
248
+ language, with sample dropped spans. Read-only:
249
+ never gates and never writes a baseline. Run it
250
+ before considering an input-coverage penalty in
251
+ the confidence model.
230
252
  -h, --help Show this help message
231
253
 
232
254
  EXAMPLES:
@@ -257,6 +279,39 @@ async function main(): Promise<void> {
257
279
  // --save-baseline needs the regression reporter wired so it can persist.
258
280
  if (config.saveBaseline) config.regression = true;
259
281
 
282
+ // --diagnose-coverage is a measurement pass, not a gate: it reads the corpus,
283
+ // parses each row, and reports the `unconsumed-input` firing rate. It compares
284
+ // nothing against the baseline and writes nothing, so it runs before (and
285
+ // instead of) the regression machinery. It DOES execute dist/, so a stale
286
+ // build would mis-measure — warn rather than refuse, since no baseline is at
287
+ // risk.
288
+ if (config.diagnoseCoverage) {
289
+ const staleDists = findStaleDists();
290
+ if (staleDists.length > 0) {
291
+ console.warn(
292
+ `⚠ Stale dist/ in: ${staleDists.join(', ')} — these numbers describe the built\n` +
293
+ ' output, not your checkout. Rebuild first: npm run test:multilingual:build-deps\n'
294
+ );
295
+ }
296
+ await diagnoseCoverage(config);
297
+ process.exit(0);
298
+ }
299
+
300
+ // --triage-r1 is the same kind of measurement pass for R1 role fidelity:
301
+ // itemizes each language's missing `action.role:type` entries vs the en
302
+ // reference, clustered by entry (with the mistype pairing). Read-only.
303
+ if (config.triageR1) {
304
+ const staleDists = findStaleDists();
305
+ if (staleDists.length > 0) {
306
+ console.warn(
307
+ `⚠ Stale dist/ in: ${staleDists.join(', ')} — these numbers describe the built\n` +
308
+ ' output, not your checkout. Rebuild first: npm run test:multilingual:build-deps\n'
309
+ );
310
+ }
311
+ await triageR1(config);
312
+ process.exit(0);
313
+ }
314
+
260
315
  // DB freshness guard: refuse to run a regression/baseline compare against a
261
316
  // patterns.db generated from *different* source than is currently checked out
262
317
  // (the cross-branch "phantom regression" footgun). The committed baseline only
@@ -321,12 +376,11 @@ async function main(): Promise<void> {
321
376
  // gap (newer feature patterns lack complete non-English coverage), so a
322
377
  // 100%-or-fail gate would never go green. A language fails the gate when
323
378
  // its parse rate drops more than REGRESSION_TOLERANCE_PTS below baseline.
324
- // We deliberately gate on the RATE, not on per-pattern flips: a handful of
325
- // borderline-confidence patterns can flip pass/fail between builds without
326
- // a real regression (the per-pattern `newFailures` is kept for reporting
327
- // only). The baseline must be generated against a freshly `populate`d
328
- // patterns.db (CI re-populates), or every language reads as shifted.
329
- // Missing baseline → fail.
379
+ // The rate is a COARSE signal by design; the per-pattern R5 ratchet below
380
+ // is what catches a small number of patterns breaking (the rate alone
381
+ // cannot — see its comment). The baseline must be generated against a
382
+ // freshly `populate`d patterns.db (CI re-populates), or every language
383
+ // reads as shifted. Missing baseline → fail.
330
384
  // - otherwise: the normal absolute pass/fail (used by the quick-mode gate,
331
385
  // which is genuinely expected at 100%).
332
386
  const REGRESSION_TOLERANCE_PTS = 2;
@@ -363,7 +417,20 @@ async function main(): Promise<void> {
363
417
  // un-regenerated baseline never retro-flags. avgFidelity is deterministic
364
418
  // (parse-derived, independent of the patterns.db confidence column), so the
365
419
  // tolerance can be tight.
366
- const LOSSY_REGRESSION_TOLERANCE = 3;
420
+ // Tolerance 0 since 2026-07-25. It was 3, and that cushion was measured
421
+ // to be the gate's real blind spot: removing `את` from go.destination's
422
+ // markerLegacy.he — the literal #763 regression, which two vitest cases
423
+ // caught and the gate did not — produces exactly ONE faithful→lossy flip
424
+ // and nothing else. Even a nonsense he marker override produced only two.
425
+ // The parser degrades rather than returning null, so the parse-based
426
+ // signals (rate, R5) never fire for this class; the lossy flip is the
427
+ // only place it shows, and a cushion of 3 swallowed it.
428
+ // A faithful→lossy flip is binary and structural — the same argument that
429
+ // already puts R2/R4 at 0 — so unlike the avg* deltas it has no float or
430
+ // collation noise to absorb. The degenerate ratchet keeps its 3: it is a
431
+ // strictly coarser view of the same event, and a >50% structure loss
432
+ // never occurs without the lossy flip firing first.
433
+ const LOSSY_REGRESSION_TOLERANCE = 0;
367
434
  // 0.02 ≈ six single-pattern drops in a ~154-pattern language — absorbs any
368
435
  // rare populate jitter while still catching real per-language cluster
369
436
  // regressions (typically ≥0.03). The per-pattern lossy ratchet above is the
@@ -387,6 +454,19 @@ async function main(): Promise<void> {
387
454
  r => r.avgPrecisionDelta < -AVG_PRECISION_DROP_TOLERANCE
388
455
  );
389
456
 
457
+ // R0-recall-multiset ratchet: the mirror of the precision ratchet. Every
458
+ // signal above is computed on a deduped SET, so a parse that drops a
459
+ // REPEATED command scores a perfect 1.0 — reference `[bind, bind]`
460
+ // collapses to `{bind}`, which `[bind]` satisfies in full. That is exactly
461
+ // how `bind-two-way` recorded fidelity 1.0 in all 24 languages while every
462
+ // one of them parsed only the first of its two binds. Counting duplicates
463
+ // makes the drop visible. Deltas are 0 unless the baseline carries
464
+ // avgMultisetRecall, so an un-regenerated baseline never retro-flags.
465
+ const AVG_MULTISET_RECALL_DROP_TOLERANCE = 0.02;
466
+ const multisetRecallDrops = allResults.filter(
467
+ r => r.avgMultisetRecallDelta < -AVG_MULTISET_RECALL_DROP_TOLERANCE
468
+ );
469
+
390
470
  // R1 — role-fidelity ratchet (§8): same semantics as the avgFidelity
391
471
  // ratchet, on the role-recall signal (action.role:valueType vs the en
392
472
  // reference). Deltas are 0 unless the baseline carries avgRoleFidelity,
@@ -398,6 +478,20 @@ async function main(): Promise<void> {
398
478
  r => r.avgRoleFidelityDelta < -AVG_ROLE_FIDELITY_DROP_TOLERANCE
399
479
  );
400
480
 
481
+ // R3 — role-VALUE ratchet: same semantics as the role-fidelity ratchet,
482
+ // on the invariant-value signal (`action.role=value` multiset over the
483
+ // code-shaped subset: selectors, sigil refs, time literals,
484
+ // colon-qualified event names, URLs — compared VERBATIM vs the en
485
+ // reference). A drop means a translation started losing/corrupting an
486
+ // invariant value (`draggable` captured for `draggable:start`, the #633
487
+ // class) — invisible to every action/type-based signal above. Deltas
488
+ // are 0 unless the baseline carries avgValueRecall, so an
489
+ // un-regenerated baseline never retro-flags.
490
+ const AVG_VALUE_RECALL_DROP_TOLERANCE = 0.02;
491
+ const valueRecallDrops = allResults.filter(
492
+ r => r.avgValueRecallDelta < -AVG_VALUE_RECALL_DROP_TOLERANCE
493
+ );
494
+
401
495
  // R2 — execution ratchet (§8): curated-subset patterns whose jsdom DOM
402
496
  // effects matched the en reference in the baseline but diverge now.
403
497
  // Tolerance 0: execution is binary and the harness is deterministic
@@ -408,6 +502,80 @@ async function main(): Promise<void> {
408
502
  r.newExecutionFailures.map(id => `${r.language}/${id}`)
409
503
  );
410
504
 
505
+ // R5 — per-pattern parse ratchet: a pattern that parsed in the baseline
506
+ // no longer parses at all. Tolerance 0, and it is NOT redundant with the
507
+ // parse-rate signal above — every other signal is structurally incapable
508
+ // of seeing a small number of patterns break:
509
+ // * parse rate is per-language, so at ~154 patterns one flip is a
510
+ // 0.65pt drop against a 2pt tolerance — you need ≥4 in ONE language.
511
+ // * the five avg* ratchets never see it at all: orchestrator.ts skips
512
+ // failed parses BEFORE scoring, so a failure leaves both numerator
513
+ // and denominator. The delta is exactly 0.0000, not merely small.
514
+ // (Perversely, a *lossy* pattern that degrades to not-parsing raises
515
+ // avgFidelity, because its sub-1.0 score leaves the denominator.)
516
+ // * the degenerate/lossy ratchets iterate patterns that DID parse.
517
+ // * R2 covers 47 curated ids; R4's denominator excludes ~14% of the
518
+ // corpus and is full-mode only.
519
+ // That gap is not hypothetical: #763's `markerOverride.he` change stopped
520
+ // `לך את back` (go-back, he) parsing, and the gate would have gone green.
521
+ // The tolerances elsewhere are cross-machine float/collation headroom for
522
+ // AVERAGES; a binary pass→fail flip has no such noise to absorb, so this
523
+ // one is 0. Guarded by the baseline carrying per-pattern `patterns` data
524
+ // (findNewFailures returns [] without it), so an un-regenerated baseline
525
+ // never retro-flags.
526
+ const parseRegressions = allResults.flatMap(r =>
527
+ r.newFailures.map(id => `${r.language}/${id}`)
528
+ );
529
+
530
+ // R4 — canonical-validity ratchet: render every authored foreign
531
+ // translation to English and parse the result on the real
532
+ // hyperscript.org engine, diffing the invalid (pattern, language)
533
+ // pairs against the committed allowlist
534
+ // (baselines/foreign-canonical-validity.json). Both directions fail:
535
+ // a NEW invalid pair is a validity regression; a stale entry (an
536
+ // allowlisted pair that now renders valid) must be pruned in the same
537
+ // change that cleared it (tools/regen-foreign-baseline.ts). Tolerance
538
+ // 0 — the render+parse is deterministic against a fresh DB, and the
539
+ // DB/dist freshness guards above already refused stale inputs. Reuses
540
+ // checkForeignRenderValidity + the allowlist VERBATIM, so this and the
541
+ // vitest gate (foreign-canonical-validity.test.ts) cannot disagree.
542
+ // Full-mode only (quick mode keeps its speed contract) and skipped
543
+ // with a warning when the allowlist is absent, so a checkout without
544
+ // the baseline never retro-flags.
545
+ let validityNewInvalid: string[] = [];
546
+ let validityStale: string[] = [];
547
+ let validityChecked = false;
548
+ if (config.mode !== 'quick') {
549
+ const validityBaselinePath = path.resolve(
550
+ path.dirname(fileURLToPath(import.meta.url)),
551
+ '../../baselines/foreign-canonical-validity.json'
552
+ );
553
+ if (!fs.existsSync(validityBaselinePath)) {
554
+ console.warn(`⚠ R4 validity ratchet skipped: no allowlist at ${validityBaselinePath}.`);
555
+ } else {
556
+ const { checkForeignRenderValidity, groupFailuresByPattern } =
557
+ await import('./foreign-canonical-validity');
558
+ const allowlist = JSON.parse(fs.readFileSync(validityBaselinePath, 'utf8')) as {
559
+ allowedInvalid: Record<string, string[]>;
560
+ };
561
+ const pairKey = (id: string, lang: string) => `${id}/${lang}`;
562
+ const allowed = new Set(
563
+ Object.entries(allowlist.allowedInvalid).flatMap(([id, langs]) =>
564
+ langs.map(l => pairKey(id, l))
565
+ )
566
+ );
567
+ const validity = await checkForeignRenderValidity();
568
+ const current = new Set(
569
+ Object.entries(groupFailuresByPattern(validity.failures)).flatMap(([id, langs]) =>
570
+ langs.map(l => pairKey(id, l))
571
+ )
572
+ );
573
+ validityNewInvalid = [...current].filter(p => !allowed.has(p)).sort();
574
+ validityStale = [...allowed].filter(p => !current.has(p)).sort();
575
+ validityChecked = true;
576
+ }
577
+ }
578
+
411
579
  let failed = false;
412
580
 
413
581
  if (regressed.length > 0) {
@@ -417,13 +585,29 @@ async function main(): Promise<void> {
417
585
  );
418
586
  for (const r of regressed) {
419
587
  const fails = r.newFailures.length
420
- ? ` — newly failing: ${r.newFailures.join(', ')}`
588
+ ? ` — newly failing: ${r.newFailures.slice(0, 20).join(', ')}`
421
589
  : '';
422
590
  console.error(` ${r.language}: ΔparseRate ${r.parseRateDelta.toFixed(1)}pts${fails}`);
423
591
  }
424
592
  failed = true;
425
593
  }
426
594
 
595
+ if (parseRegressions.length > 0) {
596
+ console.error(
597
+ `\n✗ Parse regression vs baseline (R5): ${parseRegressions.length} pattern(s) ` +
598
+ `parsed in the baseline and no longer parse at all:`
599
+ );
600
+ for (const p of parseRegressions.slice(0, 20)) console.error(` ${p}`);
601
+ if (parseRegressions.length > 20) {
602
+ console.error(` … and ${parseRegressions.length - 20} more`);
603
+ }
604
+ console.error(
605
+ ` (tolerance 0 — the percentage-based signals cannot see a handful of ` +
606
+ `broken patterns; if intentional, regenerate the baseline with --save-baseline)`
607
+ );
608
+ failed = true;
609
+ }
610
+
427
611
  if (fidelityRegressions.length > FIDELITY_REGRESSION_TOLERANCE) {
428
612
  console.error(
429
613
  `\n✗ Fidelity regression vs baseline: ${fidelityRegressions.length} faithful pass(es) ` +
@@ -453,11 +637,6 @@ async function main(): Promise<void> {
453
637
  `if intentional, regenerate the baseline with --save-baseline)`
454
638
  );
455
639
  failed = true;
456
- } else if (lossyRegressions.length > 0) {
457
- console.warn(
458
- `\n⚠ ${lossyRegressions.length} correctness regression(s) within tolerance ` +
459
- `(${LOSSY_REGRESSION_TOLERANCE}): ${lossyRegressions.join(', ')}`
460
- );
461
640
  }
462
641
 
463
642
  if (fidelityDrops.length > 0) {
@@ -485,6 +664,21 @@ async function main(): Promise<void> {
485
664
  failed = true;
486
665
  }
487
666
 
667
+ if (multisetRecallDrops.length > 0) {
668
+ console.error(
669
+ `\n✗ avgMultisetRecall dropped > ${AVG_MULTISET_RECALL_DROP_TOLERANCE} in ` +
670
+ `${multisetRecallDrops.length} language(s) — a parse started dropping a ` +
671
+ `REPEATED command (invisible to the Set-based fidelity/roleFidelity):`
672
+ );
673
+ for (const r of multisetRecallDrops) {
674
+ console.error(
675
+ ` ${r.language}: ΔavgMultisetRecall ${r.avgMultisetRecallDelta.toFixed(4)}`
676
+ );
677
+ }
678
+ console.error(` (if intentional, regenerate the baseline with --save-baseline)`);
679
+ failed = true;
680
+ }
681
+
488
682
  if (roleFidelityDrops.length > 0) {
489
683
  console.error(
490
684
  `\n✗ avgRoleFidelity dropped > ${AVG_ROLE_FIDELITY_DROP_TOLERANCE} in ` +
@@ -499,6 +693,19 @@ async function main(): Promise<void> {
499
693
  failed = true;
500
694
  }
501
695
 
696
+ if (valueRecallDrops.length > 0) {
697
+ console.error(
698
+ `\n✗ avgValueRecall dropped > ${AVG_VALUE_RECALL_DROP_TOLERANCE} in ` +
699
+ `${valueRecallDrops.length} language(s) — a parse started losing/corrupting ` +
700
+ `a language-invariant role VALUE (invisible to the action/type-based signals):`
701
+ );
702
+ for (const r of valueRecallDrops) {
703
+ console.error(` ${r.language}: ΔavgValueRecall ${r.avgValueRecallDelta.toFixed(4)}`);
704
+ }
705
+ console.error(` (if intentional, regenerate the baseline with --save-baseline)`);
706
+ failed = true;
707
+ }
708
+
502
709
  if (executionRegressions.length > 0) {
503
710
  console.error(
504
711
  `\n✗ Execution regression vs baseline (R2): ${executionRegressions.length} ` +
@@ -512,12 +719,38 @@ async function main(): Promise<void> {
512
719
  failed = true;
513
720
  }
514
721
 
722
+ if (validityNewInvalid.length > 0) {
723
+ console.error(
724
+ `\n✗ Canonical-validity regression (R4): ${validityNewInvalid.length} foreign ` +
725
+ `translation(s) now render English the hyperscript.org parser rejects:`
726
+ );
727
+ for (const p of validityNewInvalid) console.error(` ${p}`);
728
+ console.error(
729
+ ` (triage with tools/triage-foreign-residual.ts; if the invalidity is ` +
730
+ `expected, allowlist it via tools/regen-foreign-baseline.ts)`
731
+ );
732
+ failed = true;
733
+ }
734
+
735
+ if (validityStale.length > 0) {
736
+ console.error(
737
+ `\n✗ Stale validity allowlist (R4): ${validityStale.length} allowlisted ` +
738
+ `pair(s) now render VALID — prune with tools/regen-foreign-baseline.ts:`
739
+ );
740
+ for (const p of validityStale) console.error(` ${p}`);
741
+ failed = true;
742
+ }
743
+
515
744
  if (failed) {
516
745
  exitCode = 1;
517
746
  } else {
518
747
  console.log(
519
748
  `\n✓ No regression vs baseline ` +
520
- `(parse-rate ${REGRESSION_TOLERANCE_PTS}pts, fidelity + correctness + execution ratchets).`
749
+ `(parse-rate ${REGRESSION_TOLERANCE_PTS}pts, per-pattern parse (R5) + ` +
750
+ `fidelity + correctness + ` +
751
+ `precision + multiset-recall + role + value + execution ratchets` +
752
+ (validityChecked ? ` + canonical validity (R4)` : '') +
753
+ `).`
521
754
  );
522
755
  exitCode = 0;
523
756
  }
@@ -2,7 +2,9 @@ import { describe, it, expect } from 'vitest';
2
2
  import {
3
3
  collectActions,
4
4
  collectActionsMultiset,
5
+ collectRoleValueSignature,
5
6
  computeFidelity,
7
+ computeMultisetRecall,
6
8
  computePrecision,
7
9
  spuriousActions,
8
10
  FIDELITY_THRESHOLD,
@@ -119,6 +121,196 @@ describe('computePrecision', () => {
119
121
  });
120
122
  });
121
123
 
124
+ describe('computeMultisetRecall', () => {
125
+ it('catches the dropped duplicate that Set-based recall is blind to', () => {
126
+ // The `bind-two-way` shape. EN `bind $n to #a bind $n to #b` is [bind, bind];
127
+ // a parse that truncates to the first command yields [bind]. Every Set-based
128
+ // signal reads this as perfect — which is why the row recorded fidelity 1.0
129
+ // across all 24 languages while every one of them dropped half its body.
130
+ const en = ['bind', 'bind'];
131
+ const truncated = ['bind'];
132
+
133
+ expect(computeFidelity(collectSet(en), collectSet(truncated))).toBe(1); // recall is fooled
134
+ expect(computePrecision(en, truncated)).toBe(1); // precision is fooled too
135
+ expect(computeMultisetRecall(en, truncated)).toBe(0.5); // this is not
136
+ });
137
+
138
+ it('scores a faithful (reordered) parse 1.0', () => {
139
+ expect(computeMultisetRecall(['bind', 'bind'], ['bind', 'bind'])).toBe(1);
140
+ expect(computeMultisetRecall(['add', 'on', 'remove'], ['on', 'remove', 'add'])).toBe(1);
141
+ });
142
+
143
+ it('is not fooled by an added duplicate (that is precision’s job)', () => {
144
+ // A candidate with a phantom extra still has full recall; precision catches it.
145
+ expect(computeMultisetRecall(['bind'], ['bind', 'bind'])).toBe(1);
146
+ expect(computePrecision(['bind'], ['bind', 'bind'])).toBe(0.5);
147
+ });
148
+
149
+ it('returns undefined when there is no reference to score against', () => {
150
+ expect(computeMultisetRecall([], ['bind'])).toBeUndefined();
151
+ });
152
+ });
153
+
154
+ /** The deduped Set signature `collectActions` produces, from a multiset. */
155
+ function collectSet(actions: readonly string[]): string[] {
156
+ return [...new Set(actions)].sort();
157
+ }
158
+
159
+ describe('collectRoleValueSignature', () => {
160
+ it('emits every invariant-shaped value class, from a roles Map', () => {
161
+ const node = {
162
+ kind: 'event-handler',
163
+ action: 'on',
164
+ roles: new Map<string, unknown>([
165
+ ['event', { type: 'expression', raw: 'draggable:start' }], // colon-qualified (the #633 class)
166
+ ]),
167
+ body: [
168
+ {
169
+ kind: 'command',
170
+ action: 'toggle',
171
+ roles: new Map<string, unknown>([
172
+ ['patient', { type: 'selector', value: '.active', selectorKind: 'class' }],
173
+ ]),
174
+ },
175
+ {
176
+ kind: 'command',
177
+ action: 'set',
178
+ roles: new Map<string, unknown>([
179
+ ['destination', { type: 'expression', raw: ':x' }], // sigil ref
180
+ ]),
181
+ },
182
+ {
183
+ kind: 'command',
184
+ action: 'wait',
185
+ roles: new Map<string, unknown>([
186
+ ['duration', { type: 'literal', value: '200ms', dataType: 'duration' }],
187
+ ]),
188
+ },
189
+ {
190
+ kind: 'command',
191
+ action: 'fetch',
192
+ roles: new Map<string, unknown>([['source', { type: 'literal', value: '/api/data' }]]),
193
+ },
194
+ ],
195
+ };
196
+ expect(collectRoleValueSignature(node)).toEqual([
197
+ 'fetch.source=/api/data',
198
+ 'on.event=draggable:start',
199
+ 'set.destination=:x',
200
+ 'toggle.patient=.active',
201
+ 'wait.duration=200ms',
202
+ ]);
203
+ });
204
+
205
+ it('excludes non-invariant values: references, prose, bare words, mixed raws, flags, property-paths', () => {
206
+ const node = {
207
+ kind: 'command',
208
+ action: 'put',
209
+ roles: new Map<string, unknown>([
210
+ ['destination', { type: 'reference', value: 'me' }], // fillSchemaDefaults noise
211
+ ['patient', { type: 'literal', value: 'Hello world', dataType: 'string' }], // legitimately translated
212
+ ['source', { type: 'expression', raw: 'startX' }], // bare identifier — v1 exclusion
213
+ ['modifier', { type: 'expression', raw: '次 .item' }], // native words + code mixed
214
+ ['condition', { type: 'expression', raw: '#modal exists' }], // selector-prefixed prose (if-exists class)
215
+ ['url', { type: 'literal', value: '/api/search?q=${my' }], // truncated template interpolation
216
+ ['flagRole', { type: 'flag', name: 'async', enabled: true }],
217
+ [
218
+ 'pathRole',
219
+ { type: 'property-path', object: { type: 'reference', value: 'me' }, property: 'value' },
220
+ ],
221
+ ]),
222
+ };
223
+ expect(collectRoleValueSignature(node)).toEqual([]);
224
+ });
225
+
226
+ it('is a multiset — a dropped duplicate value is visible via computeMultisetRecall', () => {
227
+ // The bind-two-way shape: two `bind`s to two different sigil refs; a
228
+ // truncating parse keeps only the first. Both entries share NO key with a
229
+ // Set — but if both bound the SAME ref, dedup would hide the drop:
230
+ const twoBinds = {
231
+ action: 'compound',
232
+ statements: [
233
+ {
234
+ action: 'bind',
235
+ roles: new Map<string, unknown>([['source', { type: 'expression', raw: '$name' }]]),
236
+ },
237
+ {
238
+ action: 'bind',
239
+ roles: new Map<string, unknown>([['source', { type: 'expression', raw: '$name' }]]),
240
+ },
241
+ ],
242
+ };
243
+ const oneBind = {
244
+ action: 'bind',
245
+ roles: new Map<string, unknown>([['source', { type: 'expression', raw: '$name' }]]),
246
+ };
247
+ const ref = collectRoleValueSignature(twoBinds);
248
+ const cand = collectRoleValueSignature(oneBind);
249
+ expect(ref).toEqual(['bind.source=$name', 'bind.source=$name']);
250
+ expect(cand).toEqual(['bind.source=$name']);
251
+ expect(computeMultisetRecall(ref, cand)).toBe(0.5);
252
+ });
253
+
254
+ it('recurses into behavior-shaped nodes (eventHandlers + initBlock)', () => {
255
+ // No other walker test exercises these two CHILD_FIELDS; ad-hoc walkers
256
+ // that omit them see behavior bodies as empty.
257
+ const behavior = {
258
+ kind: 'behavior',
259
+ action: 'behavior',
260
+ roles: new Map<string, unknown>([['name', { type: 'expression', raw: 'Draggable' }]]),
261
+ eventHandlers: [
262
+ {
263
+ kind: 'event-handler',
264
+ action: 'on',
265
+ roles: new Map<string, unknown>([
266
+ ['event', { type: 'literal', value: 'draggable:start' }],
267
+ ]),
268
+ body: [
269
+ {
270
+ kind: 'command',
271
+ action: 'add',
272
+ roles: new Map<string, unknown>([
273
+ ['patient', { type: 'selector', value: '.dragging' }],
274
+ ]),
275
+ },
276
+ ],
277
+ },
278
+ ],
279
+ initBlock: [
280
+ {
281
+ kind: 'command',
282
+ action: 'set',
283
+ roles: new Map<string, unknown>([['destination', { type: 'expression', raw: '*width' }]]),
284
+ },
285
+ ],
286
+ };
287
+ expect(collectRoleValueSignature(behavior)).toEqual([
288
+ 'add.patient=.dragging',
289
+ 'on.event=draggable:start',
290
+ 'set.destination=*width',
291
+ ]);
292
+ });
293
+
294
+ it('reads plain-object roles (synthetic/JSON-shaped nodes) too', () => {
295
+ const node = {
296
+ action: 'toggle',
297
+ roles: { patient: { type: 'selector', value: '#count' } },
298
+ };
299
+ expect(collectRoleValueSignature(node)).toEqual(['toggle.patient=#count']);
300
+ });
301
+
302
+ it('skips the structural compound wrapper and handles non-object input', () => {
303
+ const node = {
304
+ action: 'compound',
305
+ roles: new Map<string, unknown>([['patient', { type: 'selector', value: '.x' }]]),
306
+ statements: [{ action: 'toggle', roles: { patient: { type: 'selector', value: '.x' } } }],
307
+ };
308
+ expect(collectRoleValueSignature(node)).toEqual(['toggle.patient=.x']);
309
+ expect(collectRoleValueSignature(null)).toEqual([]);
310
+ expect(collectRoleValueSignature(undefined)).toEqual([]);
311
+ });
312
+ });
313
+
122
314
  describe('spuriousActions', () => {
123
315
  it('lists the hallucinated commands a render/parse introduced', () => {
124
316
  expect(spuriousActions(['add', 'on', 'remove'], ['add', 'on', 'remove', 'toggle'])).toEqual([