@hyperfixi/testing-framework 2.7.2 → 2.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +393 -0
- package/dist/assertions.d.mts +26 -1
- package/dist/assertions.d.ts +26 -1
- package/dist/index.d.mts +5 -112
- package/dist/index.d.ts +5 -112
- package/dist/runner.d.mts +112 -0
- package/dist/runner.d.ts +112 -0
- package/dist/runner.js +1102 -0
- package/dist/runner.js.map +1 -0
- package/dist/runner.mjs +1097 -0
- package/dist/runner.mjs.map +1 -0
- package/dist/{assertions-CsGP61iW.d.mts → types-D-rCVkf3.d.mts} +1 -24
- package/dist/{assertions-CsGP61iW.d.ts → types-D-rCVkf3.d.ts} +1 -24
- package/package.json +13 -27
- package/src/multilingual/canonical-validity.test.ts +69 -0
- package/src/multilingual/canonical-validity.ts +132 -0
- package/src/multilingual/cli.ts +247 -14
- package/src/multilingual/fidelity.test.ts +192 -0
- package/src/multilingual/fidelity.ts +153 -0
- package/src/multilingual/foreign-canonical-validity.test.ts +0 -0
- package/src/multilingual/foreign-canonical-validity.ts +158 -0
- package/src/multilingual/orchestrator.ts +47 -1
- package/src/multilingual/reporters/console-reporter.ts +51 -0
- package/src/multilingual/reporters/regression-reporter.test.ts +78 -0
- package/src/multilingual/reporters/regression-reporter.ts +28 -1
- package/src/multilingual/tools/diagnose-coverage.ts +118 -0
- package/src/multilingual/tools/triage-r1.ts +149 -0
- package/src/multilingual/types.ts +74 -0
- package/src/multilingual/validators/parse-validator.ts +9 -1
- package/src/runner.test.ts +7 -2
- package/src/vocab/batch3-roundtrip.test.ts +184 -0
- package/src/vocab/checks.test.ts +362 -0
- package/src/vocab/checks.ts +311 -0
- package/src/vocab/cli.ts +196 -0
- package/src/vocab/dump.ts +80 -0
- package/src/vocab/model.ts +110 -0
- package/src/vocab/report.ts +120 -0
- package/src/vocab/types.ts +100 -0
package/src/multilingual/cli.ts
CHANGED
|
@@ -10,6 +10,8 @@ import * as path from 'node:path';
|
|
|
10
10
|
import { fileURLToPath } from 'node:url';
|
|
11
11
|
import { checkDbStamp, getDefaultDbPath } from '@hyperfixi/patterns-reference';
|
|
12
12
|
import { TestOrchestrator } from './orchestrator';
|
|
13
|
+
import { diagnoseCoverage } from './tools/diagnose-coverage';
|
|
14
|
+
import { triageR1 } from './tools/triage-r1';
|
|
13
15
|
import type { TestConfig, LanguageCode } from './types';
|
|
14
16
|
|
|
15
17
|
/**
|
|
@@ -172,6 +174,17 @@ function parseArgs(): TestConfig {
|
|
|
172
174
|
config.saveBaseline = true;
|
|
173
175
|
break;
|
|
174
176
|
|
|
177
|
+
case '--diagnose-coverage':
|
|
178
|
+
// Read-only measurement pass; short-circuits main() before any gate.
|
|
179
|
+
config.diagnoseCoverage = true;
|
|
180
|
+
break;
|
|
181
|
+
|
|
182
|
+
case '--triage-r1':
|
|
183
|
+
// Read-only measurement pass; short-circuits main() before any gate.
|
|
184
|
+
// Combine with --languages to scope (en is force-loaded as reference).
|
|
185
|
+
config.triageR1 = true;
|
|
186
|
+
break;
|
|
187
|
+
|
|
175
188
|
default:
|
|
176
189
|
if (arg && arg.startsWith('-')) {
|
|
177
190
|
console.error(`Unknown option: ${arg}`);
|
|
@@ -227,6 +240,15 @@ OPTIONS:
|
|
|
227
240
|
--categories <cats> Filter by categories (comma-separated)
|
|
228
241
|
--limit <n> Patterns per language in quick mode (default: 10)
|
|
229
242
|
--save-baseline Save current results as new baseline
|
|
243
|
+
--triage-r1 Itemize R1 role-fidelity misses per language
|
|
244
|
+
(missing action.role:type entries vs the en
|
|
245
|
+
reference, clustered; scope with --languages)
|
|
246
|
+
--diagnose-coverage Report how often the semantic parser matched a
|
|
247
|
+
pattern that ignored part of its input, per
|
|
248
|
+
language, with sample dropped spans. Read-only:
|
|
249
|
+
never gates and never writes a baseline. Run it
|
|
250
|
+
before considering an input-coverage penalty in
|
|
251
|
+
the confidence model.
|
|
230
252
|
-h, --help Show this help message
|
|
231
253
|
|
|
232
254
|
EXAMPLES:
|
|
@@ -257,6 +279,39 @@ async function main(): Promise<void> {
|
|
|
257
279
|
// --save-baseline needs the regression reporter wired so it can persist.
|
|
258
280
|
if (config.saveBaseline) config.regression = true;
|
|
259
281
|
|
|
282
|
+
// --diagnose-coverage is a measurement pass, not a gate: it reads the corpus,
|
|
283
|
+
// parses each row, and reports the `unconsumed-input` firing rate. It compares
|
|
284
|
+
// nothing against the baseline and writes nothing, so it runs before (and
|
|
285
|
+
// instead of) the regression machinery. It DOES execute dist/, so a stale
|
|
286
|
+
// build would mis-measure — warn rather than refuse, since no baseline is at
|
|
287
|
+
// risk.
|
|
288
|
+
if (config.diagnoseCoverage) {
|
|
289
|
+
const staleDists = findStaleDists();
|
|
290
|
+
if (staleDists.length > 0) {
|
|
291
|
+
console.warn(
|
|
292
|
+
`⚠ Stale dist/ in: ${staleDists.join(', ')} — these numbers describe the built\n` +
|
|
293
|
+
' output, not your checkout. Rebuild first: npm run test:multilingual:build-deps\n'
|
|
294
|
+
);
|
|
295
|
+
}
|
|
296
|
+
await diagnoseCoverage(config);
|
|
297
|
+
process.exit(0);
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
// --triage-r1 is the same kind of measurement pass for R1 role fidelity:
|
|
301
|
+
// itemizes each language's missing `action.role:type` entries vs the en
|
|
302
|
+
// reference, clustered by entry (with the mistype pairing). Read-only.
|
|
303
|
+
if (config.triageR1) {
|
|
304
|
+
const staleDists = findStaleDists();
|
|
305
|
+
if (staleDists.length > 0) {
|
|
306
|
+
console.warn(
|
|
307
|
+
`⚠ Stale dist/ in: ${staleDists.join(', ')} — these numbers describe the built\n` +
|
|
308
|
+
' output, not your checkout. Rebuild first: npm run test:multilingual:build-deps\n'
|
|
309
|
+
);
|
|
310
|
+
}
|
|
311
|
+
await triageR1(config);
|
|
312
|
+
process.exit(0);
|
|
313
|
+
}
|
|
314
|
+
|
|
260
315
|
// DB freshness guard: refuse to run a regression/baseline compare against a
|
|
261
316
|
// patterns.db generated from *different* source than is currently checked out
|
|
262
317
|
// (the cross-branch "phantom regression" footgun). The committed baseline only
|
|
@@ -321,12 +376,11 @@ async function main(): Promise<void> {
|
|
|
321
376
|
// gap (newer feature patterns lack complete non-English coverage), so a
|
|
322
377
|
// 100%-or-fail gate would never go green. A language fails the gate when
|
|
323
378
|
// its parse rate drops more than REGRESSION_TOLERANCE_PTS below baseline.
|
|
324
|
-
//
|
|
325
|
-
//
|
|
326
|
-
//
|
|
327
|
-
//
|
|
328
|
-
//
|
|
329
|
-
// Missing baseline → fail.
|
|
379
|
+
// The rate is a COARSE signal by design; the per-pattern R5 ratchet below
|
|
380
|
+
// is what catches a small number of patterns breaking (the rate alone
|
|
381
|
+
// cannot — see its comment). The baseline must be generated against a
|
|
382
|
+
// freshly `populate`d patterns.db (CI re-populates), or every language
|
|
383
|
+
// reads as shifted. Missing baseline → fail.
|
|
330
384
|
// - otherwise: the normal absolute pass/fail (used by the quick-mode gate,
|
|
331
385
|
// which is genuinely expected at 100%).
|
|
332
386
|
const REGRESSION_TOLERANCE_PTS = 2;
|
|
@@ -363,7 +417,20 @@ async function main(): Promise<void> {
|
|
|
363
417
|
// un-regenerated baseline never retro-flags. avgFidelity is deterministic
|
|
364
418
|
// (parse-derived, independent of the patterns.db confidence column), so the
|
|
365
419
|
// tolerance can be tight.
|
|
366
|
-
|
|
420
|
+
// Tolerance 0 since 2026-07-25. It was 3, and that cushion was measured
|
|
421
|
+
// to be the gate's real blind spot: removing `את` from go.destination's
|
|
422
|
+
// markerLegacy.he — the literal #763 regression, which two vitest cases
|
|
423
|
+
// caught and the gate did not — produces exactly ONE faithful→lossy flip
|
|
424
|
+
// and nothing else. Even a nonsense he marker override produced only two.
|
|
425
|
+
// The parser degrades rather than returning null, so the parse-based
|
|
426
|
+
// signals (rate, R5) never fire for this class; the lossy flip is the
|
|
427
|
+
// only place it shows, and a cushion of 3 swallowed it.
|
|
428
|
+
// A faithful→lossy flip is binary and structural — the same argument that
|
|
429
|
+
// already puts R2/R4 at 0 — so unlike the avg* deltas it has no float or
|
|
430
|
+
// collation noise to absorb. The degenerate ratchet keeps its 3: it is a
|
|
431
|
+
// strictly coarser view of the same event, and a >50% structure loss
|
|
432
|
+
// never occurs without the lossy flip firing first.
|
|
433
|
+
const LOSSY_REGRESSION_TOLERANCE = 0;
|
|
367
434
|
// 0.02 ≈ six single-pattern drops in a ~154-pattern language — absorbs any
|
|
368
435
|
// rare populate jitter while still catching real per-language cluster
|
|
369
436
|
// regressions (typically ≥0.03). The per-pattern lossy ratchet above is the
|
|
@@ -387,6 +454,19 @@ async function main(): Promise<void> {
|
|
|
387
454
|
r => r.avgPrecisionDelta < -AVG_PRECISION_DROP_TOLERANCE
|
|
388
455
|
);
|
|
389
456
|
|
|
457
|
+
// R0-recall-multiset ratchet: the mirror of the precision ratchet. Every
|
|
458
|
+
// signal above is computed on a deduped SET, so a parse that drops a
|
|
459
|
+
// REPEATED command scores a perfect 1.0 — reference `[bind, bind]`
|
|
460
|
+
// collapses to `{bind}`, which `[bind]` satisfies in full. That is exactly
|
|
461
|
+
// how `bind-two-way` recorded fidelity 1.0 in all 24 languages while every
|
|
462
|
+
// one of them parsed only the first of its two binds. Counting duplicates
|
|
463
|
+
// makes the drop visible. Deltas are 0 unless the baseline carries
|
|
464
|
+
// avgMultisetRecall, so an un-regenerated baseline never retro-flags.
|
|
465
|
+
const AVG_MULTISET_RECALL_DROP_TOLERANCE = 0.02;
|
|
466
|
+
const multisetRecallDrops = allResults.filter(
|
|
467
|
+
r => r.avgMultisetRecallDelta < -AVG_MULTISET_RECALL_DROP_TOLERANCE
|
|
468
|
+
);
|
|
469
|
+
|
|
390
470
|
// R1 — role-fidelity ratchet (§8): same semantics as the avgFidelity
|
|
391
471
|
// ratchet, on the role-recall signal (action.role:valueType vs the en
|
|
392
472
|
// reference). Deltas are 0 unless the baseline carries avgRoleFidelity,
|
|
@@ -398,6 +478,20 @@ async function main(): Promise<void> {
|
|
|
398
478
|
r => r.avgRoleFidelityDelta < -AVG_ROLE_FIDELITY_DROP_TOLERANCE
|
|
399
479
|
);
|
|
400
480
|
|
|
481
|
+
// R3 — role-VALUE ratchet: same semantics as the role-fidelity ratchet,
|
|
482
|
+
// on the invariant-value signal (`action.role=value` multiset over the
|
|
483
|
+
// code-shaped subset: selectors, sigil refs, time literals,
|
|
484
|
+
// colon-qualified event names, URLs — compared VERBATIM vs the en
|
|
485
|
+
// reference). A drop means a translation started losing/corrupting an
|
|
486
|
+
// invariant value (`draggable` captured for `draggable:start`, the #633
|
|
487
|
+
// class) — invisible to every action/type-based signal above. Deltas
|
|
488
|
+
// are 0 unless the baseline carries avgValueRecall, so an
|
|
489
|
+
// un-regenerated baseline never retro-flags.
|
|
490
|
+
const AVG_VALUE_RECALL_DROP_TOLERANCE = 0.02;
|
|
491
|
+
const valueRecallDrops = allResults.filter(
|
|
492
|
+
r => r.avgValueRecallDelta < -AVG_VALUE_RECALL_DROP_TOLERANCE
|
|
493
|
+
);
|
|
494
|
+
|
|
401
495
|
// R2 — execution ratchet (§8): curated-subset patterns whose jsdom DOM
|
|
402
496
|
// effects matched the en reference in the baseline but diverge now.
|
|
403
497
|
// Tolerance 0: execution is binary and the harness is deterministic
|
|
@@ -408,6 +502,80 @@ async function main(): Promise<void> {
|
|
|
408
502
|
r.newExecutionFailures.map(id => `${r.language}/${id}`)
|
|
409
503
|
);
|
|
410
504
|
|
|
505
|
+
// R5 — per-pattern parse ratchet: a pattern that parsed in the baseline
|
|
506
|
+
// no longer parses at all. Tolerance 0, and it is NOT redundant with the
|
|
507
|
+
// parse-rate signal above — every other signal is structurally incapable
|
|
508
|
+
// of seeing a small number of patterns break:
|
|
509
|
+
// * parse rate is per-language, so at ~154 patterns one flip is a
|
|
510
|
+
// 0.65pt drop against a 2pt tolerance — you need ≥4 in ONE language.
|
|
511
|
+
// * the five avg* ratchets never see it at all: orchestrator.ts skips
|
|
512
|
+
// failed parses BEFORE scoring, so a failure leaves both numerator
|
|
513
|
+
// and denominator. The delta is exactly 0.0000, not merely small.
|
|
514
|
+
// (Perversely, a *lossy* pattern that degrades to not-parsing raises
|
|
515
|
+
// avgFidelity, because its sub-1.0 score leaves the denominator.)
|
|
516
|
+
// * the degenerate/lossy ratchets iterate patterns that DID parse.
|
|
517
|
+
// * R2 covers 47 curated ids; R4's denominator excludes ~14% of the
|
|
518
|
+
// corpus and is full-mode only.
|
|
519
|
+
// That gap is not hypothetical: #763's `markerOverride.he` change stopped
|
|
520
|
+
// `לך את back` (go-back, he) parsing, and the gate would have gone green.
|
|
521
|
+
// The tolerances elsewhere are cross-machine float/collation headroom for
|
|
522
|
+
// AVERAGES; a binary pass→fail flip has no such noise to absorb, so this
|
|
523
|
+
// one is 0. Guarded by the baseline carrying per-pattern `patterns` data
|
|
524
|
+
// (findNewFailures returns [] without it), so an un-regenerated baseline
|
|
525
|
+
// never retro-flags.
|
|
526
|
+
const parseRegressions = allResults.flatMap(r =>
|
|
527
|
+
r.newFailures.map(id => `${r.language}/${id}`)
|
|
528
|
+
);
|
|
529
|
+
|
|
530
|
+
// R4 — canonical-validity ratchet: render every authored foreign
|
|
531
|
+
// translation to English and parse the result on the real
|
|
532
|
+
// hyperscript.org engine, diffing the invalid (pattern, language)
|
|
533
|
+
// pairs against the committed allowlist
|
|
534
|
+
// (baselines/foreign-canonical-validity.json). Both directions fail:
|
|
535
|
+
// a NEW invalid pair is a validity regression; a stale entry (an
|
|
536
|
+
// allowlisted pair that now renders valid) must be pruned in the same
|
|
537
|
+
// change that cleared it (tools/regen-foreign-baseline.ts). Tolerance
|
|
538
|
+
// 0 — the render+parse is deterministic against a fresh DB, and the
|
|
539
|
+
// DB/dist freshness guards above already refused stale inputs. Reuses
|
|
540
|
+
// checkForeignRenderValidity + the allowlist VERBATIM, so this and the
|
|
541
|
+
// vitest gate (foreign-canonical-validity.test.ts) cannot disagree.
|
|
542
|
+
// Full-mode only (quick mode keeps its speed contract) and skipped
|
|
543
|
+
// with a warning when the allowlist is absent, so a checkout without
|
|
544
|
+
// the baseline never retro-flags.
|
|
545
|
+
let validityNewInvalid: string[] = [];
|
|
546
|
+
let validityStale: string[] = [];
|
|
547
|
+
let validityChecked = false;
|
|
548
|
+
if (config.mode !== 'quick') {
|
|
549
|
+
const validityBaselinePath = path.resolve(
|
|
550
|
+
path.dirname(fileURLToPath(import.meta.url)),
|
|
551
|
+
'../../baselines/foreign-canonical-validity.json'
|
|
552
|
+
);
|
|
553
|
+
if (!fs.existsSync(validityBaselinePath)) {
|
|
554
|
+
console.warn(`⚠ R4 validity ratchet skipped: no allowlist at ${validityBaselinePath}.`);
|
|
555
|
+
} else {
|
|
556
|
+
const { checkForeignRenderValidity, groupFailuresByPattern } =
|
|
557
|
+
await import('./foreign-canonical-validity');
|
|
558
|
+
const allowlist = JSON.parse(fs.readFileSync(validityBaselinePath, 'utf8')) as {
|
|
559
|
+
allowedInvalid: Record<string, string[]>;
|
|
560
|
+
};
|
|
561
|
+
const pairKey = (id: string, lang: string) => `${id}/${lang}`;
|
|
562
|
+
const allowed = new Set(
|
|
563
|
+
Object.entries(allowlist.allowedInvalid).flatMap(([id, langs]) =>
|
|
564
|
+
langs.map(l => pairKey(id, l))
|
|
565
|
+
)
|
|
566
|
+
);
|
|
567
|
+
const validity = await checkForeignRenderValidity();
|
|
568
|
+
const current = new Set(
|
|
569
|
+
Object.entries(groupFailuresByPattern(validity.failures)).flatMap(([id, langs]) =>
|
|
570
|
+
langs.map(l => pairKey(id, l))
|
|
571
|
+
)
|
|
572
|
+
);
|
|
573
|
+
validityNewInvalid = [...current].filter(p => !allowed.has(p)).sort();
|
|
574
|
+
validityStale = [...allowed].filter(p => !current.has(p)).sort();
|
|
575
|
+
validityChecked = true;
|
|
576
|
+
}
|
|
577
|
+
}
|
|
578
|
+
|
|
411
579
|
let failed = false;
|
|
412
580
|
|
|
413
581
|
if (regressed.length > 0) {
|
|
@@ -417,13 +585,29 @@ async function main(): Promise<void> {
|
|
|
417
585
|
);
|
|
418
586
|
for (const r of regressed) {
|
|
419
587
|
const fails = r.newFailures.length
|
|
420
|
-
? ` — newly failing: ${r.newFailures.join(', ')}`
|
|
588
|
+
? ` — newly failing: ${r.newFailures.slice(0, 20).join(', ')}`
|
|
421
589
|
: '';
|
|
422
590
|
console.error(` ${r.language}: ΔparseRate ${r.parseRateDelta.toFixed(1)}pts${fails}`);
|
|
423
591
|
}
|
|
424
592
|
failed = true;
|
|
425
593
|
}
|
|
426
594
|
|
|
595
|
+
if (parseRegressions.length > 0) {
|
|
596
|
+
console.error(
|
|
597
|
+
`\n✗ Parse regression vs baseline (R5): ${parseRegressions.length} pattern(s) ` +
|
|
598
|
+
`parsed in the baseline and no longer parse at all:`
|
|
599
|
+
);
|
|
600
|
+
for (const p of parseRegressions.slice(0, 20)) console.error(` ${p}`);
|
|
601
|
+
if (parseRegressions.length > 20) {
|
|
602
|
+
console.error(` … and ${parseRegressions.length - 20} more`);
|
|
603
|
+
}
|
|
604
|
+
console.error(
|
|
605
|
+
` (tolerance 0 — the percentage-based signals cannot see a handful of ` +
|
|
606
|
+
`broken patterns; if intentional, regenerate the baseline with --save-baseline)`
|
|
607
|
+
);
|
|
608
|
+
failed = true;
|
|
609
|
+
}
|
|
610
|
+
|
|
427
611
|
if (fidelityRegressions.length > FIDELITY_REGRESSION_TOLERANCE) {
|
|
428
612
|
console.error(
|
|
429
613
|
`\n✗ Fidelity regression vs baseline: ${fidelityRegressions.length} faithful pass(es) ` +
|
|
@@ -453,11 +637,6 @@ async function main(): Promise<void> {
|
|
|
453
637
|
`if intentional, regenerate the baseline with --save-baseline)`
|
|
454
638
|
);
|
|
455
639
|
failed = true;
|
|
456
|
-
} else if (lossyRegressions.length > 0) {
|
|
457
|
-
console.warn(
|
|
458
|
-
`\n⚠ ${lossyRegressions.length} correctness regression(s) within tolerance ` +
|
|
459
|
-
`(${LOSSY_REGRESSION_TOLERANCE}): ${lossyRegressions.join(', ')}`
|
|
460
|
-
);
|
|
461
640
|
}
|
|
462
641
|
|
|
463
642
|
if (fidelityDrops.length > 0) {
|
|
@@ -485,6 +664,21 @@ async function main(): Promise<void> {
|
|
|
485
664
|
failed = true;
|
|
486
665
|
}
|
|
487
666
|
|
|
667
|
+
if (multisetRecallDrops.length > 0) {
|
|
668
|
+
console.error(
|
|
669
|
+
`\n✗ avgMultisetRecall dropped > ${AVG_MULTISET_RECALL_DROP_TOLERANCE} in ` +
|
|
670
|
+
`${multisetRecallDrops.length} language(s) — a parse started dropping a ` +
|
|
671
|
+
`REPEATED command (invisible to the Set-based fidelity/roleFidelity):`
|
|
672
|
+
);
|
|
673
|
+
for (const r of multisetRecallDrops) {
|
|
674
|
+
console.error(
|
|
675
|
+
` ${r.language}: ΔavgMultisetRecall ${r.avgMultisetRecallDelta.toFixed(4)}`
|
|
676
|
+
);
|
|
677
|
+
}
|
|
678
|
+
console.error(` (if intentional, regenerate the baseline with --save-baseline)`);
|
|
679
|
+
failed = true;
|
|
680
|
+
}
|
|
681
|
+
|
|
488
682
|
if (roleFidelityDrops.length > 0) {
|
|
489
683
|
console.error(
|
|
490
684
|
`\n✗ avgRoleFidelity dropped > ${AVG_ROLE_FIDELITY_DROP_TOLERANCE} in ` +
|
|
@@ -499,6 +693,19 @@ async function main(): Promise<void> {
|
|
|
499
693
|
failed = true;
|
|
500
694
|
}
|
|
501
695
|
|
|
696
|
+
if (valueRecallDrops.length > 0) {
|
|
697
|
+
console.error(
|
|
698
|
+
`\n✗ avgValueRecall dropped > ${AVG_VALUE_RECALL_DROP_TOLERANCE} in ` +
|
|
699
|
+
`${valueRecallDrops.length} language(s) — a parse started losing/corrupting ` +
|
|
700
|
+
`a language-invariant role VALUE (invisible to the action/type-based signals):`
|
|
701
|
+
);
|
|
702
|
+
for (const r of valueRecallDrops) {
|
|
703
|
+
console.error(` ${r.language}: ΔavgValueRecall ${r.avgValueRecallDelta.toFixed(4)}`);
|
|
704
|
+
}
|
|
705
|
+
console.error(` (if intentional, regenerate the baseline with --save-baseline)`);
|
|
706
|
+
failed = true;
|
|
707
|
+
}
|
|
708
|
+
|
|
502
709
|
if (executionRegressions.length > 0) {
|
|
503
710
|
console.error(
|
|
504
711
|
`\n✗ Execution regression vs baseline (R2): ${executionRegressions.length} ` +
|
|
@@ -512,12 +719,38 @@ async function main(): Promise<void> {
|
|
|
512
719
|
failed = true;
|
|
513
720
|
}
|
|
514
721
|
|
|
722
|
+
if (validityNewInvalid.length > 0) {
|
|
723
|
+
console.error(
|
|
724
|
+
`\n✗ Canonical-validity regression (R4): ${validityNewInvalid.length} foreign ` +
|
|
725
|
+
`translation(s) now render English the hyperscript.org parser rejects:`
|
|
726
|
+
);
|
|
727
|
+
for (const p of validityNewInvalid) console.error(` ${p}`);
|
|
728
|
+
console.error(
|
|
729
|
+
` (triage with tools/triage-foreign-residual.ts; if the invalidity is ` +
|
|
730
|
+
`expected, allowlist it via tools/regen-foreign-baseline.ts)`
|
|
731
|
+
);
|
|
732
|
+
failed = true;
|
|
733
|
+
}
|
|
734
|
+
|
|
735
|
+
if (validityStale.length > 0) {
|
|
736
|
+
console.error(
|
|
737
|
+
`\n✗ Stale validity allowlist (R4): ${validityStale.length} allowlisted ` +
|
|
738
|
+
`pair(s) now render VALID — prune with tools/regen-foreign-baseline.ts:`
|
|
739
|
+
);
|
|
740
|
+
for (const p of validityStale) console.error(` ${p}`);
|
|
741
|
+
failed = true;
|
|
742
|
+
}
|
|
743
|
+
|
|
515
744
|
if (failed) {
|
|
516
745
|
exitCode = 1;
|
|
517
746
|
} else {
|
|
518
747
|
console.log(
|
|
519
748
|
`\n✓ No regression vs baseline ` +
|
|
520
|
-
`(parse-rate ${REGRESSION_TOLERANCE_PTS}pts,
|
|
749
|
+
`(parse-rate ${REGRESSION_TOLERANCE_PTS}pts, per-pattern parse (R5) + ` +
|
|
750
|
+
`fidelity + correctness + ` +
|
|
751
|
+
`precision + multiset-recall + role + value + execution ratchets` +
|
|
752
|
+
(validityChecked ? ` + canonical validity (R4)` : '') +
|
|
753
|
+
`).`
|
|
521
754
|
);
|
|
522
755
|
exitCode = 0;
|
|
523
756
|
}
|
|
@@ -2,7 +2,9 @@ import { describe, it, expect } from 'vitest';
|
|
|
2
2
|
import {
|
|
3
3
|
collectActions,
|
|
4
4
|
collectActionsMultiset,
|
|
5
|
+
collectRoleValueSignature,
|
|
5
6
|
computeFidelity,
|
|
7
|
+
computeMultisetRecall,
|
|
6
8
|
computePrecision,
|
|
7
9
|
spuriousActions,
|
|
8
10
|
FIDELITY_THRESHOLD,
|
|
@@ -119,6 +121,196 @@ describe('computePrecision', () => {
|
|
|
119
121
|
});
|
|
120
122
|
});
|
|
121
123
|
|
|
124
|
+
describe('computeMultisetRecall', () => {
|
|
125
|
+
it('catches the dropped duplicate that Set-based recall is blind to', () => {
|
|
126
|
+
// The `bind-two-way` shape. EN `bind $n to #a bind $n to #b` is [bind, bind];
|
|
127
|
+
// a parse that truncates to the first command yields [bind]. Every Set-based
|
|
128
|
+
// signal reads this as perfect — which is why the row recorded fidelity 1.0
|
|
129
|
+
// across all 24 languages while every one of them dropped half its body.
|
|
130
|
+
const en = ['bind', 'bind'];
|
|
131
|
+
const truncated = ['bind'];
|
|
132
|
+
|
|
133
|
+
expect(computeFidelity(collectSet(en), collectSet(truncated))).toBe(1); // recall is fooled
|
|
134
|
+
expect(computePrecision(en, truncated)).toBe(1); // precision is fooled too
|
|
135
|
+
expect(computeMultisetRecall(en, truncated)).toBe(0.5); // this is not
|
|
136
|
+
});
|
|
137
|
+
|
|
138
|
+
it('scores a faithful (reordered) parse 1.0', () => {
|
|
139
|
+
expect(computeMultisetRecall(['bind', 'bind'], ['bind', 'bind'])).toBe(1);
|
|
140
|
+
expect(computeMultisetRecall(['add', 'on', 'remove'], ['on', 'remove', 'add'])).toBe(1);
|
|
141
|
+
});
|
|
142
|
+
|
|
143
|
+
it('is not fooled by an added duplicate (that is precision’s job)', () => {
|
|
144
|
+
// A candidate with a phantom extra still has full recall; precision catches it.
|
|
145
|
+
expect(computeMultisetRecall(['bind'], ['bind', 'bind'])).toBe(1);
|
|
146
|
+
expect(computePrecision(['bind'], ['bind', 'bind'])).toBe(0.5);
|
|
147
|
+
});
|
|
148
|
+
|
|
149
|
+
it('returns undefined when there is no reference to score against', () => {
|
|
150
|
+
expect(computeMultisetRecall([], ['bind'])).toBeUndefined();
|
|
151
|
+
});
|
|
152
|
+
});
|
|
153
|
+
|
|
154
|
+
/** The deduped Set signature `collectActions` produces, from a multiset. */
|
|
155
|
+
function collectSet(actions: readonly string[]): string[] {
|
|
156
|
+
return [...new Set(actions)].sort();
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
describe('collectRoleValueSignature', () => {
|
|
160
|
+
it('emits every invariant-shaped value class, from a roles Map', () => {
|
|
161
|
+
const node = {
|
|
162
|
+
kind: 'event-handler',
|
|
163
|
+
action: 'on',
|
|
164
|
+
roles: new Map<string, unknown>([
|
|
165
|
+
['event', { type: 'expression', raw: 'draggable:start' }], // colon-qualified (the #633 class)
|
|
166
|
+
]),
|
|
167
|
+
body: [
|
|
168
|
+
{
|
|
169
|
+
kind: 'command',
|
|
170
|
+
action: 'toggle',
|
|
171
|
+
roles: new Map<string, unknown>([
|
|
172
|
+
['patient', { type: 'selector', value: '.active', selectorKind: 'class' }],
|
|
173
|
+
]),
|
|
174
|
+
},
|
|
175
|
+
{
|
|
176
|
+
kind: 'command',
|
|
177
|
+
action: 'set',
|
|
178
|
+
roles: new Map<string, unknown>([
|
|
179
|
+
['destination', { type: 'expression', raw: ':x' }], // sigil ref
|
|
180
|
+
]),
|
|
181
|
+
},
|
|
182
|
+
{
|
|
183
|
+
kind: 'command',
|
|
184
|
+
action: 'wait',
|
|
185
|
+
roles: new Map<string, unknown>([
|
|
186
|
+
['duration', { type: 'literal', value: '200ms', dataType: 'duration' }],
|
|
187
|
+
]),
|
|
188
|
+
},
|
|
189
|
+
{
|
|
190
|
+
kind: 'command',
|
|
191
|
+
action: 'fetch',
|
|
192
|
+
roles: new Map<string, unknown>([['source', { type: 'literal', value: '/api/data' }]]),
|
|
193
|
+
},
|
|
194
|
+
],
|
|
195
|
+
};
|
|
196
|
+
expect(collectRoleValueSignature(node)).toEqual([
|
|
197
|
+
'fetch.source=/api/data',
|
|
198
|
+
'on.event=draggable:start',
|
|
199
|
+
'set.destination=:x',
|
|
200
|
+
'toggle.patient=.active',
|
|
201
|
+
'wait.duration=200ms',
|
|
202
|
+
]);
|
|
203
|
+
});
|
|
204
|
+
|
|
205
|
+
it('excludes non-invariant values: references, prose, bare words, mixed raws, flags, property-paths', () => {
|
|
206
|
+
const node = {
|
|
207
|
+
kind: 'command',
|
|
208
|
+
action: 'put',
|
|
209
|
+
roles: new Map<string, unknown>([
|
|
210
|
+
['destination', { type: 'reference', value: 'me' }], // fillSchemaDefaults noise
|
|
211
|
+
['patient', { type: 'literal', value: 'Hello world', dataType: 'string' }], // legitimately translated
|
|
212
|
+
['source', { type: 'expression', raw: 'startX' }], // bare identifier — v1 exclusion
|
|
213
|
+
['modifier', { type: 'expression', raw: '次 .item' }], // native words + code mixed
|
|
214
|
+
['condition', { type: 'expression', raw: '#modal exists' }], // selector-prefixed prose (if-exists class)
|
|
215
|
+
['url', { type: 'literal', value: '/api/search?q=${my' }], // truncated template interpolation
|
|
216
|
+
['flagRole', { type: 'flag', name: 'async', enabled: true }],
|
|
217
|
+
[
|
|
218
|
+
'pathRole',
|
|
219
|
+
{ type: 'property-path', object: { type: 'reference', value: 'me' }, property: 'value' },
|
|
220
|
+
],
|
|
221
|
+
]),
|
|
222
|
+
};
|
|
223
|
+
expect(collectRoleValueSignature(node)).toEqual([]);
|
|
224
|
+
});
|
|
225
|
+
|
|
226
|
+
it('is a multiset — a dropped duplicate value is visible via computeMultisetRecall', () => {
|
|
227
|
+
// The bind-two-way shape: two `bind`s to two different sigil refs; a
|
|
228
|
+
// truncating parse keeps only the first. Both entries share NO key with a
|
|
229
|
+
// Set — but if both bound the SAME ref, dedup would hide the drop:
|
|
230
|
+
const twoBinds = {
|
|
231
|
+
action: 'compound',
|
|
232
|
+
statements: [
|
|
233
|
+
{
|
|
234
|
+
action: 'bind',
|
|
235
|
+
roles: new Map<string, unknown>([['source', { type: 'expression', raw: '$name' }]]),
|
|
236
|
+
},
|
|
237
|
+
{
|
|
238
|
+
action: 'bind',
|
|
239
|
+
roles: new Map<string, unknown>([['source', { type: 'expression', raw: '$name' }]]),
|
|
240
|
+
},
|
|
241
|
+
],
|
|
242
|
+
};
|
|
243
|
+
const oneBind = {
|
|
244
|
+
action: 'bind',
|
|
245
|
+
roles: new Map<string, unknown>([['source', { type: 'expression', raw: '$name' }]]),
|
|
246
|
+
};
|
|
247
|
+
const ref = collectRoleValueSignature(twoBinds);
|
|
248
|
+
const cand = collectRoleValueSignature(oneBind);
|
|
249
|
+
expect(ref).toEqual(['bind.source=$name', 'bind.source=$name']);
|
|
250
|
+
expect(cand).toEqual(['bind.source=$name']);
|
|
251
|
+
expect(computeMultisetRecall(ref, cand)).toBe(0.5);
|
|
252
|
+
});
|
|
253
|
+
|
|
254
|
+
it('recurses into behavior-shaped nodes (eventHandlers + initBlock)', () => {
|
|
255
|
+
// No other walker test exercises these two CHILD_FIELDS; ad-hoc walkers
|
|
256
|
+
// that omit them see behavior bodies as empty.
|
|
257
|
+
const behavior = {
|
|
258
|
+
kind: 'behavior',
|
|
259
|
+
action: 'behavior',
|
|
260
|
+
roles: new Map<string, unknown>([['name', { type: 'expression', raw: 'Draggable' }]]),
|
|
261
|
+
eventHandlers: [
|
|
262
|
+
{
|
|
263
|
+
kind: 'event-handler',
|
|
264
|
+
action: 'on',
|
|
265
|
+
roles: new Map<string, unknown>([
|
|
266
|
+
['event', { type: 'literal', value: 'draggable:start' }],
|
|
267
|
+
]),
|
|
268
|
+
body: [
|
|
269
|
+
{
|
|
270
|
+
kind: 'command',
|
|
271
|
+
action: 'add',
|
|
272
|
+
roles: new Map<string, unknown>([
|
|
273
|
+
['patient', { type: 'selector', value: '.dragging' }],
|
|
274
|
+
]),
|
|
275
|
+
},
|
|
276
|
+
],
|
|
277
|
+
},
|
|
278
|
+
],
|
|
279
|
+
initBlock: [
|
|
280
|
+
{
|
|
281
|
+
kind: 'command',
|
|
282
|
+
action: 'set',
|
|
283
|
+
roles: new Map<string, unknown>([['destination', { type: 'expression', raw: '*width' }]]),
|
|
284
|
+
},
|
|
285
|
+
],
|
|
286
|
+
};
|
|
287
|
+
expect(collectRoleValueSignature(behavior)).toEqual([
|
|
288
|
+
'add.patient=.dragging',
|
|
289
|
+
'on.event=draggable:start',
|
|
290
|
+
'set.destination=*width',
|
|
291
|
+
]);
|
|
292
|
+
});
|
|
293
|
+
|
|
294
|
+
it('reads plain-object roles (synthetic/JSON-shaped nodes) too', () => {
|
|
295
|
+
const node = {
|
|
296
|
+
action: 'toggle',
|
|
297
|
+
roles: { patient: { type: 'selector', value: '#count' } },
|
|
298
|
+
};
|
|
299
|
+
expect(collectRoleValueSignature(node)).toEqual(['toggle.patient=#count']);
|
|
300
|
+
});
|
|
301
|
+
|
|
302
|
+
it('skips the structural compound wrapper and handles non-object input', () => {
|
|
303
|
+
const node = {
|
|
304
|
+
action: 'compound',
|
|
305
|
+
roles: new Map<string, unknown>([['patient', { type: 'selector', value: '.x' }]]),
|
|
306
|
+
statements: [{ action: 'toggle', roles: { patient: { type: 'selector', value: '.x' } } }],
|
|
307
|
+
};
|
|
308
|
+
expect(collectRoleValueSignature(node)).toEqual(['toggle.patient=.x']);
|
|
309
|
+
expect(collectRoleValueSignature(null)).toEqual([]);
|
|
310
|
+
expect(collectRoleValueSignature(undefined)).toEqual([]);
|
|
311
|
+
});
|
|
312
|
+
});
|
|
313
|
+
|
|
122
314
|
describe('spuriousActions', () => {
|
|
123
315
|
it('lists the hallucinated commands a render/parse introduced', () => {
|
|
124
316
|
expect(spuriousActions(['add', 'on', 'remove'], ['add', 'on', 'remove', 'toggle'])).toEqual([
|