@hyperfixi/testing-framework 2.7.2 → 2.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +393 -0
- package/dist/assertions.d.mts +26 -1
- package/dist/assertions.d.ts +26 -1
- package/dist/index.d.mts +5 -112
- package/dist/index.d.ts +5 -112
- package/dist/runner.d.mts +112 -0
- package/dist/runner.d.ts +112 -0
- package/dist/runner.js +1102 -0
- package/dist/runner.js.map +1 -0
- package/dist/runner.mjs +1097 -0
- package/dist/runner.mjs.map +1 -0
- package/dist/{assertions-CsGP61iW.d.mts → types-D-rCVkf3.d.mts} +1 -24
- package/dist/{assertions-CsGP61iW.d.ts → types-D-rCVkf3.d.ts} +1 -24
- package/package.json +13 -27
- package/src/multilingual/canonical-validity.test.ts +69 -0
- package/src/multilingual/canonical-validity.ts +132 -0
- package/src/multilingual/cli.ts +247 -14
- package/src/multilingual/fidelity.test.ts +192 -0
- package/src/multilingual/fidelity.ts +153 -0
- package/src/multilingual/foreign-canonical-validity.test.ts +0 -0
- package/src/multilingual/foreign-canonical-validity.ts +158 -0
- package/src/multilingual/orchestrator.ts +47 -1
- package/src/multilingual/reporters/console-reporter.ts +51 -0
- package/src/multilingual/reporters/regression-reporter.test.ts +78 -0
- package/src/multilingual/reporters/regression-reporter.ts +28 -1
- package/src/multilingual/tools/diagnose-coverage.ts +118 -0
- package/src/multilingual/tools/triage-r1.ts +149 -0
- package/src/multilingual/types.ts +74 -0
- package/src/multilingual/validators/parse-validator.ts +9 -1
- package/src/runner.test.ts +7 -2
- package/src/vocab/batch3-roundtrip.test.ts +184 -0
- package/src/vocab/checks.test.ts +362 -0
- package/src/vocab/checks.ts +311 -0
- package/src/vocab/cli.ts +196 -0
- package/src/vocab/dump.ts +80 -0
- package/src/vocab/model.ts +110 -0
- package/src/vocab/report.ts +120 -0
- package/src/vocab/types.ts +100 -0
|
@@ -121,6 +121,39 @@ export function computeFidelity(
|
|
|
121
121
|
return hits / reference.length;
|
|
122
122
|
}
|
|
123
123
|
|
|
124
|
+
/**
|
|
125
|
+
* R0-recall on the **multiset** in [0, 1]: the fraction of the reference's
|
|
126
|
+
* actions — counting duplicates — also present in the candidate.
|
|
127
|
+
*
|
|
128
|
+
* {@link computeFidelity} scores the deduped Set signature, so a candidate that
|
|
129
|
+
* drops a REPEATED command scores 1.0: reference `[bind, bind]` collapses to
|
|
130
|
+
* `{bind}`, which `[bind]` satisfies in full. That is how `bind-two-way` sat at
|
|
131
|
+
* fidelity 1.0 across all 24 languages while every one of them parsed only the
|
|
132
|
+
* first of its two `bind`s. R1 (role signatures) is a Set too, and is equally
|
|
133
|
+
* blind. {@link computePrecision} catches the mirror case — a candidate that ADDS
|
|
134
|
+
* a duplicate — so before this signal existed the ratchet saw spurious commands
|
|
135
|
+
* but never dropped ones.
|
|
136
|
+
*
|
|
137
|
+
* Pass multisets (see {@link collectActionsMultiset}) on both sides.
|
|
138
|
+
* Returns `undefined` when the reference has no actions to compare against.
|
|
139
|
+
*/
|
|
140
|
+
export function computeMultisetRecall(
|
|
141
|
+
reference: readonly string[],
|
|
142
|
+
candidate: readonly string[]
|
|
143
|
+
): number | undefined {
|
|
144
|
+
if (reference.length === 0) return undefined;
|
|
145
|
+
const cand = actionCounts(candidate);
|
|
146
|
+
let matched = 0;
|
|
147
|
+
for (const a of reference) {
|
|
148
|
+
const remaining = cand.get(a) ?? 0;
|
|
149
|
+
if (remaining > 0) {
|
|
150
|
+
matched++;
|
|
151
|
+
cand.set(a, remaining - 1);
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
return matched / reference.length;
|
|
155
|
+
}
|
|
156
|
+
|
|
124
157
|
/**
|
|
125
158
|
* Structural **precision** in [0, 1]: the fraction of the *candidate's* actions
|
|
126
159
|
* that are justified by the reference (multiset-aware). The complement of
|
|
@@ -222,3 +255,123 @@ function walkRoles(node: unknown, acc: Set<string>, depth: number): void {
|
|
|
222
255
|
}
|
|
223
256
|
}
|
|
224
257
|
}
|
|
258
|
+
|
|
259
|
+
/**
|
|
260
|
+
* Role-value kinds never emitted by the R3 walker. `reference` values (`me`,
|
|
261
|
+
* `it`, …) are mostly `fillSchemaDefaults` injections present identically on
|
|
262
|
+
* both sides — noise. `flag` names and `property-path` properties are bare
|
|
263
|
+
* identifiers, excluded by v1 (see {@link collectRoleValueSignature}).
|
|
264
|
+
*/
|
|
265
|
+
const VALUE_KIND_EXCLUSIONS = new Set(['reference', 'flag', 'property-path']);
|
|
266
|
+
|
|
267
|
+
/**
|
|
268
|
+
* A role value is compared cross-language only when its WHOLE surface form is
|
|
269
|
+
* code-shaped — language-invariant by construction, never legitimately
|
|
270
|
+
* translated. Conservative v1 whitelist; when a sweep firing turns out to be a
|
|
271
|
+
* legit translation difference, tighten here and document the exclusion.
|
|
272
|
+
*/
|
|
273
|
+
const INVARIANT_VALUE_PATTERNS: readonly RegExp[] = [
|
|
274
|
+
/^[#.[<@*]/, // selectors: #id .class [attr] <tag/> @attr *style
|
|
275
|
+
/^[:$^][A-Za-z_]\w*$/, // sigil refs: :local $global ^element
|
|
276
|
+
/^\d+(\.\d+)?(ms|s|m|h)?$/, // numbers and time literals: 2 1.5 200ms
|
|
277
|
+
/^[A-Za-z_][\w-]*:[\w-]+$/, // colon-qualified event names: draggable:start
|
|
278
|
+
/^(\.{0,2}\/|https?:)/, // URLs / paths: /api/data ./x ../y https://…
|
|
279
|
+
];
|
|
280
|
+
|
|
281
|
+
function isInvariantSurface(surface: string): boolean {
|
|
282
|
+
// Whole-surface rule, sweep-validated exclusions:
|
|
283
|
+
// - whitespace ⇒ mixed content. `if #modal exists` captures its condition as
|
|
284
|
+
// `#modal exists` — starts selector-shaped, but `exists` is prose that every
|
|
285
|
+
// language legitimately translates (16-language false firing without this).
|
|
286
|
+
// - `${` ⇒ template interpolation. `/api/search?q=${my value}` is an
|
|
287
|
+
// expression, and tokenizers split it at different points per language, so
|
|
288
|
+
// the captured surface isn't comparable verbatim.
|
|
289
|
+
if (/\s/.test(surface) || surface.includes('${')) return false;
|
|
290
|
+
return INVARIANT_VALUE_PATTERNS.some(re => re.test(surface));
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
/**
|
|
294
|
+
* The comparable surface form of a role value, or `undefined` when the value
|
|
295
|
+
* carries none: `.value` for `literal`/`selector` (string/number/boolean
|
|
296
|
+
* coerced), `.raw` for `expression`. Kinds in {@link VALUE_KIND_EXCLUSIONS}
|
|
297
|
+
* and values without a string `type` discriminator yield `undefined`.
|
|
298
|
+
*/
|
|
299
|
+
function roleValueSurface(value: unknown): string | undefined {
|
|
300
|
+
if (value === null || typeof value !== 'object') return undefined;
|
|
301
|
+
const rec = value as { type?: unknown; value?: unknown; raw?: unknown };
|
|
302
|
+
if (typeof rec.type !== 'string' || VALUE_KIND_EXCLUSIONS.has(rec.type)) return undefined;
|
|
303
|
+
if (rec.type === 'expression') {
|
|
304
|
+
return typeof rec.raw === 'string' ? rec.raw : undefined;
|
|
305
|
+
}
|
|
306
|
+
const v = rec.value;
|
|
307
|
+
return typeof v === 'string' || typeof v === 'number' || typeof v === 'boolean'
|
|
308
|
+
? String(v)
|
|
309
|
+
: undefined;
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
/**
|
|
313
|
+
* R3 — role-VALUE signature (invariant values only, multiset).
|
|
314
|
+
*
|
|
315
|
+
* R0/R1 compare actions and role *types*; values are never compared because
|
|
316
|
+
* they are legitimately translated. That leaves a live defect class with zero
|
|
317
|
+
* signal: right action counts, right role types, wrong role VALUE — the #633
|
|
318
|
+
* class, where ms captured `trigger` events named `draggable` instead of
|
|
319
|
+
* `draggable:start` (correct multiset, correct `trigger.event:literal`
|
|
320
|
+
* signature, silently wrong runtime behavior), and 18 other languages carried
|
|
321
|
+
* the same corruption with no side-effect at all.
|
|
322
|
+
*
|
|
323
|
+
* The subset of values compared here is language-invariant by construction —
|
|
324
|
+
* code, not prose (see {@link INVARIANT_VALUE_PATTERNS}). Emits a **multiset**
|
|
325
|
+
* of `` `action.role=value` `` entries (duplicates preserved, sorted); score
|
|
326
|
+
* with {@link computeMultisetRecall} so a dropped duplicate value is visible.
|
|
327
|
+
*
|
|
328
|
+
* Deliberately EXCLUDED in v1: bare-word identifiers (`startX` — usually
|
|
329
|
+
* invariant, but property/variable names occasionally get localized in seeds),
|
|
330
|
+
* string literals (message strings are legitimately translated), expression
|
|
331
|
+
* raws mixing native words + code (`次 .item`), and `reference` values
|
|
332
|
+
* (`me`/`it` — mostly `fillSchemaDefaults` injections on both sides).
|
|
333
|
+
*
|
|
334
|
+
* Blind spot: recall fires when a *translation* loses/corrupts an invariant
|
|
335
|
+
* value. If the **en reference itself** corrupts a value, every language flags
|
|
336
|
+
* at once — a 24-language R3 firestorm on one pattern means "suspect the en
|
|
337
|
+
* parse first" (unlike R0, where en corruption moves nothing).
|
|
338
|
+
*
|
|
339
|
+
* The roles container is a ReadonlyMap on live nodes (serializes to {} in
|
|
340
|
+
* results.json), so collect at validation time, never from the JSON.
|
|
341
|
+
*/
|
|
342
|
+
export function collectRoleValueSignature(node: unknown): string[] {
|
|
343
|
+
const acc: string[] = [];
|
|
344
|
+
walkRoleValues(node, acc, 0);
|
|
345
|
+
return acc.sort();
|
|
346
|
+
}
|
|
347
|
+
|
|
348
|
+
function walkRoleValues(node: unknown, acc: string[], depth: number): void {
|
|
349
|
+
if (depth > 64 || node === null || typeof node !== 'object') return;
|
|
350
|
+
|
|
351
|
+
const rec = node as Record<string, unknown>;
|
|
352
|
+
const action = rec.action;
|
|
353
|
+
if (typeof action === 'string' && !STRUCTURAL_ACTIONS.has(action)) {
|
|
354
|
+
const roles = rec.roles;
|
|
355
|
+
const entries: Array<[unknown, unknown]> =
|
|
356
|
+
roles instanceof Map
|
|
357
|
+
? [...roles.entries()]
|
|
358
|
+
: roles && typeof roles === 'object'
|
|
359
|
+
? Object.entries(roles)
|
|
360
|
+
: [];
|
|
361
|
+
for (const [role, value] of entries) {
|
|
362
|
+
if (value === undefined || value === null) continue;
|
|
363
|
+
const surface = roleValueSurface(value);
|
|
364
|
+
if (surface === undefined || !isInvariantSurface(surface)) continue;
|
|
365
|
+
acc.push(`${action}.${String(role)}=${surface}`);
|
|
366
|
+
}
|
|
367
|
+
}
|
|
368
|
+
|
|
369
|
+
for (const field of CHILD_FIELDS) {
|
|
370
|
+
const child = rec[field];
|
|
371
|
+
if (Array.isArray(child)) {
|
|
372
|
+
for (const c of child) walkRoleValues(c, acc, depth + 1);
|
|
373
|
+
} else if (child && typeof child === 'object') {
|
|
374
|
+
walkRoleValues(child, acc, depth + 1);
|
|
375
|
+
}
|
|
376
|
+
}
|
|
377
|
+
}
|
|
Binary file
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Foreign→English canonical-validity gate
|
|
3
|
+
* ---------------------------------------
|
|
4
|
+
* The sibling en-render gate (canonical-validity.ts) renders every corpus English
|
|
5
|
+
* reference and parses it on the real `hyperscript.org` engine. But the PRODUCTION
|
|
6
|
+
* path is foreign→English: an authored non-English source is parsed and rendered to
|
|
7
|
+
* English (`preprocessToEnglish` → `render`). A parse can be role-faithful (the
|
|
8
|
+
* fidelity ratchet scores ~1.0) yet still render English the canonical parser rejects.
|
|
9
|
+
* (The build-time `@hyperscript-tools/i18n` transpiler now parse-gates its ENGLISH
|
|
10
|
+
* side with the same loader recipe; faithful foreign-output gating there still awaits
|
|
11
|
+
* the v2 semantic-engine transpiler, since its GrammarTransformer is lossy in reverse.)
|
|
12
|
+
*
|
|
13
|
+
* This gate closes that blind spot for the multilingual path: for every language, it
|
|
14
|
+
* renders each authored `pattern_translation` to English and parses the result on
|
|
15
|
+
* the canonical engine, failing on any invalid (pattern, language) pair that is not
|
|
16
|
+
* in the committed allowlist. The allowlist is keyed by pattern id → the languages
|
|
17
|
+
* that currently fail, so a fix that clears a family across languages shrinks (or
|
|
18
|
+
* removes) its entry — the list only ever ratchets down.
|
|
19
|
+
*
|
|
20
|
+
* Denominator: only (pattern, language) pairs whose EN reference the canonical parser
|
|
21
|
+
* already accepts are scored, so a handful of inherently non-canonical corpus rows
|
|
22
|
+
* never distort the signal (same fairness rule as the en gate).
|
|
23
|
+
*
|
|
24
|
+
* DB dependency: reads authored translations from `pattern_translations`, which only
|
|
25
|
+
* exist after `npm run populate`. Generate the baseline and run the gate against a
|
|
26
|
+
* freshly populated DB (CI's multilingual-validation job populates; see the
|
|
27
|
+
* provenance-stamp discipline in packages/patterns-reference/CLAUDE.md).
|
|
28
|
+
*/
|
|
29
|
+
|
|
30
|
+
import { getAllPatterns, getTranslationsByLanguage } from '@hyperfixi/patterns-reference';
|
|
31
|
+
import { parseSemantic, render } from '@lokascript/semantic';
|
|
32
|
+
import { loadCanonicalParser, type CanonicalValidate } from './canonical-validity';
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* The 23 non-English priority languages (the `browser-priority` corpus set). English
|
|
36
|
+
* is the reference, scored by the sibling en gate.
|
|
37
|
+
*/
|
|
38
|
+
export const FOREIGN_LANGUAGES = [
|
|
39
|
+
'es',
|
|
40
|
+
'fr',
|
|
41
|
+
'pt',
|
|
42
|
+
'it',
|
|
43
|
+
'id',
|
|
44
|
+
'ms',
|
|
45
|
+
'sw',
|
|
46
|
+
'zh',
|
|
47
|
+
'vi',
|
|
48
|
+
'tl',
|
|
49
|
+
'ja',
|
|
50
|
+
'ko',
|
|
51
|
+
'tr',
|
|
52
|
+
'qu',
|
|
53
|
+
'hi',
|
|
54
|
+
'bn',
|
|
55
|
+
'ar',
|
|
56
|
+
'de',
|
|
57
|
+
'ru',
|
|
58
|
+
'uk',
|
|
59
|
+
'pl',
|
|
60
|
+
'th',
|
|
61
|
+
'he',
|
|
62
|
+
] as const;
|
|
63
|
+
|
|
64
|
+
export interface ForeignValidityFailure {
|
|
65
|
+
/** Corpus pattern id (`code_example` id the translation belongs to). */
|
|
66
|
+
id: string;
|
|
67
|
+
language: string;
|
|
68
|
+
/** The authored foreign source that was rendered. */
|
|
69
|
+
foreign: string;
|
|
70
|
+
/** The English the renderer produced from the foreign parse. */
|
|
71
|
+
rendered: string;
|
|
72
|
+
error: string;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
export interface ForeignValidityResult {
|
|
76
|
+
/** (pattern, language) pairs whose EN reference the canonical parser accepts. */
|
|
77
|
+
checked: number;
|
|
78
|
+
valid: number;
|
|
79
|
+
failures: ForeignValidityFailure[];
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* Render every authored foreign `pattern_translation` to English and parse it on the
|
|
84
|
+
* canonical engine. Only pairs whose EN reference is itself canonical-valid are scored.
|
|
85
|
+
*/
|
|
86
|
+
export async function checkForeignRenderValidity(opts?: {
|
|
87
|
+
validate?: CanonicalValidate;
|
|
88
|
+
languages?: readonly string[];
|
|
89
|
+
/** Translations fetched per language (defaults comfortably above the corpus size). */
|
|
90
|
+
perLanguageLimit?: number;
|
|
91
|
+
}): Promise<ForeignValidityResult> {
|
|
92
|
+
const validate = opts?.validate ?? (await loadCanonicalParser());
|
|
93
|
+
const languages = opts?.languages ?? FOREIGN_LANGUAGES;
|
|
94
|
+
const limit = opts?.perLanguageLimit ?? 500;
|
|
95
|
+
|
|
96
|
+
const patterns = await getAllPatterns();
|
|
97
|
+
const enAccepts = new Map(patterns.map(p => [p.id, validate(p.rawCode).length === 0]));
|
|
98
|
+
|
|
99
|
+
const failures: ForeignValidityFailure[] = [];
|
|
100
|
+
let checked = 0;
|
|
101
|
+
let valid = 0;
|
|
102
|
+
|
|
103
|
+
for (const language of languages) {
|
|
104
|
+
const translations = await getTranslationsByLanguage(language, limit);
|
|
105
|
+
for (const t of translations) {
|
|
106
|
+
if (!enAccepts.get(t.codeExampleId)) continue; // fair denominator: EN accepts the reference
|
|
107
|
+
checked++;
|
|
108
|
+
|
|
109
|
+
let rendered: string;
|
|
110
|
+
let errors: string[];
|
|
111
|
+
try {
|
|
112
|
+
const node = parseSemantic(t.hyperscript, language).node;
|
|
113
|
+
rendered = node ? render(node, 'en') : '(no node)';
|
|
114
|
+
errors = validate(rendered);
|
|
115
|
+
} catch (e) {
|
|
116
|
+
// parseSemantic/render only — validate never throws (its tokenizer-level
|
|
117
|
+
// throws fold into the returned array), so an invalid render is reported
|
|
118
|
+
// WITH the render that caused it, not discarded as '(threw)'.
|
|
119
|
+
rendered = '(threw)';
|
|
120
|
+
errors = ['threw: ' + (e as Error).message.split('\n')[0]];
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
if (errors.length === 0) {
|
|
124
|
+
valid++;
|
|
125
|
+
} else {
|
|
126
|
+
failures.push({
|
|
127
|
+
id: t.codeExampleId,
|
|
128
|
+
language,
|
|
129
|
+
foreign: t.hyperscript,
|
|
130
|
+
rendered,
|
|
131
|
+
error: errors[0] ?? 'unknown error',
|
|
132
|
+
});
|
|
133
|
+
}
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
return { checked, valid, failures };
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
/**
|
|
141
|
+
* Group failures into the committed allowlist shape: `{ patternId: [langs…] }`
|
|
142
|
+
* (languages sorted for a stable diff). Used by the baseline generator and by the
|
|
143
|
+
* gate's stale-entry check.
|
|
144
|
+
*/
|
|
145
|
+
export function groupFailuresByPattern(
|
|
146
|
+
failures: readonly ForeignValidityFailure[]
|
|
147
|
+
): Record<string, string[]> {
|
|
148
|
+
const byPattern = new Map<string, Set<string>>();
|
|
149
|
+
for (const f of failures) {
|
|
150
|
+
if (!byPattern.has(f.id)) byPattern.set(f.id, new Set());
|
|
151
|
+
byPattern.get(f.id)!.add(f.language);
|
|
152
|
+
}
|
|
153
|
+
const out: Record<string, string[]> = {};
|
|
154
|
+
for (const id of [...byPattern.keys()].sort()) {
|
|
155
|
+
out[id] = [...byPattern.get(id)!].sort();
|
|
156
|
+
}
|
|
157
|
+
return out;
|
|
158
|
+
}
|
|
@@ -20,7 +20,13 @@ import type {
|
|
|
20
20
|
Reporter,
|
|
21
21
|
BundleInfo,
|
|
22
22
|
} from './types';
|
|
23
|
-
import {
|
|
23
|
+
import {
|
|
24
|
+
computeFidelity,
|
|
25
|
+
computePrecision,
|
|
26
|
+
computeMultisetRecall,
|
|
27
|
+
spuriousActions,
|
|
28
|
+
FIDELITY_THRESHOLD,
|
|
29
|
+
} from './fidelity';
|
|
24
30
|
|
|
25
31
|
const execAsync = promisify(exec);
|
|
26
32
|
|
|
@@ -258,6 +264,8 @@ export class TestOrchestrator {
|
|
|
258
264
|
const multisetReference = new Map<string, string[]>();
|
|
259
265
|
// R1: codeExampleId -> English role signature (action.role:valueType set).
|
|
260
266
|
const roleReference = new Map<string, string[]>();
|
|
267
|
+
// R3: codeExampleId -> English role-VALUE signature (invariant values, multiset).
|
|
268
|
+
const roleValueReference = new Map<string, string[]>();
|
|
261
269
|
for (const r of en.parseResults) {
|
|
262
270
|
if (r.success && r.actionSignature && r.actionSignature.length > 0) {
|
|
263
271
|
reference.set(r.pattern.codeExampleId, r.actionSignature);
|
|
@@ -268,6 +276,9 @@ export class TestOrchestrator {
|
|
|
268
276
|
if (r.success && r.roleSignature && r.roleSignature.length > 0) {
|
|
269
277
|
roleReference.set(r.pattern.codeExampleId, r.roleSignature);
|
|
270
278
|
}
|
|
279
|
+
if (r.success && r.roleValueSignature && r.roleValueSignature.length > 0) {
|
|
280
|
+
roleValueReference.set(r.pattern.codeExampleId, r.roleValueSignature);
|
|
281
|
+
}
|
|
271
282
|
}
|
|
272
283
|
|
|
273
284
|
for (const lang of languageResults) {
|
|
@@ -277,7 +288,9 @@ export class TestOrchestrator {
|
|
|
277
288
|
const lossy: string[] = [];
|
|
278
289
|
const scores: number[] = [];
|
|
279
290
|
const precisionScores: number[] = [];
|
|
291
|
+
const multisetRecallScores: number[] = [];
|
|
280
292
|
const roleScores: number[] = [];
|
|
293
|
+
const valueRecallScores: number[] = [];
|
|
281
294
|
|
|
282
295
|
for (const result of lang.parseResults) {
|
|
283
296
|
if (!result.success || !result.actionSignature) continue;
|
|
@@ -294,6 +307,9 @@ export class TestOrchestrator {
|
|
|
294
307
|
|
|
295
308
|
// R0-precision — fraction of THIS parse's actions justified by the en
|
|
296
309
|
// multiset reference (catches phantom/spurious commands recall misses).
|
|
310
|
+
// R0-recall-multiset — the mirror: fraction of the en reference's actions,
|
|
311
|
+
// counting duplicates, present here (catches a DROPPED repeated command,
|
|
312
|
+
// which the Set-based fidelity/roleFidelity above cannot see).
|
|
297
313
|
const multisetRef = multisetReference.get(result.pattern.codeExampleId);
|
|
298
314
|
if (multisetRef && result.actionMultisetSignature) {
|
|
299
315
|
const precision = computePrecision(multisetRef, result.actionMultisetSignature);
|
|
@@ -301,6 +317,11 @@ export class TestOrchestrator {
|
|
|
301
317
|
result.precision = precision;
|
|
302
318
|
precisionScores.push(precision);
|
|
303
319
|
}
|
|
320
|
+
const multisetRecall = computeMultisetRecall(multisetRef, result.actionMultisetSignature);
|
|
321
|
+
if (multisetRecall !== undefined) {
|
|
322
|
+
result.multisetRecall = multisetRecall;
|
|
323
|
+
multisetRecallScores.push(multisetRecall);
|
|
324
|
+
}
|
|
304
325
|
}
|
|
305
326
|
|
|
306
327
|
// R1 — role recall vs the en role signature (role name + value type).
|
|
@@ -312,6 +333,23 @@ export class TestOrchestrator {
|
|
|
312
333
|
roleScores.push(roleFidelity);
|
|
313
334
|
}
|
|
314
335
|
}
|
|
336
|
+
|
|
337
|
+
// R3 — invariant role-VALUE recall vs the en value signature (multiset).
|
|
338
|
+
// Values are compared verbatim: the filtered subset (selectors, sigil
|
|
339
|
+
// refs, time literals, colon-qualified events, URLs) is code, not prose.
|
|
340
|
+
const valueRef = roleValueReference.get(result.pattern.codeExampleId);
|
|
341
|
+
if (valueRef && result.roleValueSignature) {
|
|
342
|
+
const valueRecall = computeMultisetRecall(valueRef, result.roleValueSignature);
|
|
343
|
+
if (valueRecall !== undefined) {
|
|
344
|
+
result.valueRecall = valueRecall;
|
|
345
|
+
valueRecallScores.push(valueRecall);
|
|
346
|
+
if (valueRecall < 1) {
|
|
347
|
+
// Missing = reference entries not covered here (multiset diff;
|
|
348
|
+
// spuriousActions with the arguments swapped).
|
|
349
|
+
result.valueRecallMissing = spuriousActions(result.roleValueSignature, valueRef);
|
|
350
|
+
}
|
|
351
|
+
}
|
|
352
|
+
}
|
|
315
353
|
}
|
|
316
354
|
|
|
317
355
|
lang.avgFidelity =
|
|
@@ -320,10 +358,18 @@ export class TestOrchestrator {
|
|
|
320
358
|
precisionScores.length > 0
|
|
321
359
|
? precisionScores.reduce((a, b) => a + b, 0) / precisionScores.length
|
|
322
360
|
: undefined;
|
|
361
|
+
lang.avgMultisetRecall =
|
|
362
|
+
multisetRecallScores.length > 0
|
|
363
|
+
? multisetRecallScores.reduce((a, b) => a + b, 0) / multisetRecallScores.length
|
|
364
|
+
: undefined;
|
|
323
365
|
lang.avgRoleFidelity =
|
|
324
366
|
roleScores.length > 0
|
|
325
367
|
? roleScores.reduce((a, b) => a + b, 0) / roleScores.length
|
|
326
368
|
: undefined;
|
|
369
|
+
lang.avgValueRecall =
|
|
370
|
+
valueRecallScores.length > 0
|
|
371
|
+
? valueRecallScores.reduce((a, b) => a + b, 0) / valueRecallScores.length
|
|
372
|
+
: undefined;
|
|
327
373
|
lang.degeneratePasses = degenerate.sort();
|
|
328
374
|
lang.lossyPasses = lossy.sort();
|
|
329
375
|
}
|
|
@@ -158,11 +158,62 @@ export class ConsoleReporter implements Reporter {
|
|
|
158
158
|
}
|
|
159
159
|
|
|
160
160
|
this.reportDegeneratePasses(results);
|
|
161
|
+
this.reportValueRecall(results);
|
|
161
162
|
this.reportExecutionFailures(results);
|
|
162
163
|
|
|
163
164
|
this.log('');
|
|
164
165
|
}
|
|
165
166
|
|
|
167
|
+
/**
|
|
168
|
+
* R3 — surface parses that lost/corrupted a language-invariant role VALUE
|
|
169
|
+
* (selector, sigil ref, time literal, colon-qualified event name, URL) vs
|
|
170
|
+
* the en reference. These score 1.0 on every action/type-based signal yet
|
|
171
|
+
* behave differently at runtime, so they're reported separately. Report-only:
|
|
172
|
+
* the ratchet gate lives in the --regression path.
|
|
173
|
+
*/
|
|
174
|
+
private reportValueRecall(results: TestResults): void {
|
|
175
|
+
const scored = results.languageResults.filter(l => l.avgValueRecall !== undefined);
|
|
176
|
+
if (scored.length === 0) return;
|
|
177
|
+
|
|
178
|
+
const avg = scored.reduce((sum, l) => sum + (l.avgValueRecall ?? 0), 0) / scored.length;
|
|
179
|
+
this.log('');
|
|
180
|
+
this.log(
|
|
181
|
+
this.bright(`Role values (R3): avgValueRecall ${avg.toFixed(4)} over invariant values`)
|
|
182
|
+
);
|
|
183
|
+
|
|
184
|
+
// pattern id -> "lang: missing entries" rows
|
|
185
|
+
const byPattern = new Map<string, string[]>();
|
|
186
|
+
let instances = 0;
|
|
187
|
+
for (const lang of scored) {
|
|
188
|
+
for (const r of lang.parseResults) {
|
|
189
|
+
if (r.valueRecall === undefined || r.valueRecall >= 1) continue;
|
|
190
|
+
instances++;
|
|
191
|
+
let rows = byPattern.get(r.pattern.codeExampleId);
|
|
192
|
+
if (!rows) {
|
|
193
|
+
rows = [];
|
|
194
|
+
byPattern.set(r.pattern.codeExampleId, rows);
|
|
195
|
+
}
|
|
196
|
+
rows.push(`${lang.language}: missing ${(r.valueRecallMissing ?? []).join(', ')}`);
|
|
197
|
+
}
|
|
198
|
+
}
|
|
199
|
+
if (byPattern.size === 0) return;
|
|
200
|
+
|
|
201
|
+
this.log(
|
|
202
|
+
this.yellow(
|
|
203
|
+
`⚠ Invariant value loss: ${instances} instance(s) across ${byPattern.size} pattern(s)`
|
|
204
|
+
)
|
|
205
|
+
);
|
|
206
|
+
this.log(
|
|
207
|
+
this.dim(
|
|
208
|
+
' (an action.role=value present in the en reference is absent — not parse failures)'
|
|
209
|
+
)
|
|
210
|
+
);
|
|
211
|
+
for (const [id, rows] of [...byPattern.entries()].sort((a, b) => b[1].length - a[1].length)) {
|
|
212
|
+
this.log(` ${this.dim('-')} ${id}`);
|
|
213
|
+
for (const row of rows.sort()) this.log(` ${this.dim(row)}`);
|
|
214
|
+
}
|
|
215
|
+
}
|
|
216
|
+
|
|
166
217
|
/**
|
|
167
218
|
* R2 — surface curated-subset patterns whose jsdom execution diverged from
|
|
168
219
|
* the en reference's DOM effects. These can score 1.0 on parse fidelity yet
|
|
@@ -404,3 +404,81 @@ describe('RegressionReporter precision ratchet — R0-precision (trust floor)',
|
|
|
404
404
|
expect(saved.languages.ja!.avgPrecision).toBeCloseTo(0.95, 5);
|
|
405
405
|
});
|
|
406
406
|
});
|
|
407
|
+
|
|
408
|
+
describe('RegressionReporter per-pattern parse ratchet — R5', () => {
|
|
409
|
+
let dir: string;
|
|
410
|
+
afterEach(() => {
|
|
411
|
+
if (dir) rmSync(dir, { recursive: true, force: true });
|
|
412
|
+
});
|
|
413
|
+
|
|
414
|
+
function reporterWith(baseline: Baseline): RegressionReporter {
|
|
415
|
+
dir = mkdtempSync(join(tmpdir(), 'parse-ratchet-'));
|
|
416
|
+
const path = join(dir, 'baseline.json');
|
|
417
|
+
writeFileSync(path, JSON.stringify(baseline));
|
|
418
|
+
return new RegressionReporter(path);
|
|
419
|
+
}
|
|
420
|
+
|
|
421
|
+
/** Minimal ParseResult for a pattern that no longer parses. */
|
|
422
|
+
function fail(id: string): ParseResult {
|
|
423
|
+
return { ...pass(id), success: false };
|
|
424
|
+
}
|
|
425
|
+
|
|
426
|
+
function baselineOf(ids: string[], failing: string[] = []): Baseline {
|
|
427
|
+
const patterns: Record<string, { success: boolean; confidence: number | undefined }> = {};
|
|
428
|
+
for (const id of ids) {
|
|
429
|
+
patterns[id] = { success: !failing.includes(id), confidence: 1 };
|
|
430
|
+
}
|
|
431
|
+
return {
|
|
432
|
+
timestamp: '',
|
|
433
|
+
commit: 'base',
|
|
434
|
+
languages: {
|
|
435
|
+
ja: {
|
|
436
|
+
parseSuccess: ids.length - failing.length,
|
|
437
|
+
parseFailure: failing.length,
|
|
438
|
+
parseRate: (ids.length - failing.length) / ids.length,
|
|
439
|
+
avgConfidence: 1,
|
|
440
|
+
avgFidelity: 1,
|
|
441
|
+
degeneratePasses: [],
|
|
442
|
+
lossyPasses: [],
|
|
443
|
+
bundleSize: undefined,
|
|
444
|
+
patterns,
|
|
445
|
+
},
|
|
446
|
+
},
|
|
447
|
+
bundles: {},
|
|
448
|
+
};
|
|
449
|
+
}
|
|
450
|
+
|
|
451
|
+
it('flags a baseline pass that no longer parses', () => {
|
|
452
|
+
const reporter = reporterWith(baselineOf(['a', 'b', 'c']));
|
|
453
|
+
reporter.reportComplete(results(lang([pass('a'), fail('b'), pass('c')], [])));
|
|
454
|
+
const r = reporter.getRegressionResults().find(x => x.language === 'ja')!;
|
|
455
|
+
expect(r.newFailures).toEqual(['b']);
|
|
456
|
+
});
|
|
457
|
+
|
|
458
|
+
it('does not flag a pattern that was already failing in the baseline', () => {
|
|
459
|
+
const reporter = reporterWith(baselineOf(['a', 'b'], ['b']));
|
|
460
|
+
reporter.reportComplete(results(lang([pass('a'), fail('b')], [])));
|
|
461
|
+
const r = reporter.getRegressionResults().find(x => x.language === 'ja')!;
|
|
462
|
+
expect(r.newFailures).toEqual([]);
|
|
463
|
+
});
|
|
464
|
+
|
|
465
|
+
it('never retro-flags when the baseline has no per-pattern data', () => {
|
|
466
|
+
const noPatterns = baselineOf(['a', 'b']);
|
|
467
|
+
delete (noPatterns.languages.ja as { patterns?: unknown }).patterns;
|
|
468
|
+
const reporter = reporterWith(noPatterns);
|
|
469
|
+
reporter.reportComplete(results(lang([fail('a'), fail('b')], [])));
|
|
470
|
+
const r = reporter.getRegressionResults().find(x => x.language === 'ja')!;
|
|
471
|
+
expect(r.newFailures).toEqual([]);
|
|
472
|
+
});
|
|
473
|
+
|
|
474
|
+
// The list is the R5 gate's evidence, not just a reporting nicety: it used to
|
|
475
|
+
// be `.slice(0, 10)`, which would have under-reported how much broke — and, if
|
|
476
|
+
// the gate ever counted it, capped the count at 10.
|
|
477
|
+
it('returns every new failure, uncapped', () => {
|
|
478
|
+
const ids = Array.from({ length: 25 }, (_, i) => `p${i}`);
|
|
479
|
+
const reporter = reporterWith(baselineOf(ids));
|
|
480
|
+
reporter.reportComplete(results(lang(ids.map(fail), [])));
|
|
481
|
+
const r = reporter.getRegressionResults().find(x => x.language === 'ja')!;
|
|
482
|
+
expect(r.newFailures).toHaveLength(25);
|
|
483
|
+
});
|
|
484
|
+
});
|
|
@@ -118,12 +118,25 @@ export class RegressionReporter implements Reporter {
|
|
|
118
118
|
langResult.avgPrecision !== undefined && baselineLang.avgPrecision !== undefined
|
|
119
119
|
? langResult.avgPrecision - baselineLang.avgPrecision
|
|
120
120
|
: 0;
|
|
121
|
+
// R0-recall-multiset — same both-sides guard. Negative = a repeated command
|
|
122
|
+
// is now being dropped, which avgFidelity (a Set) cannot see.
|
|
123
|
+
const avgMultisetRecallDelta =
|
|
124
|
+
langResult.avgMultisetRecall !== undefined && baselineLang.avgMultisetRecall !== undefined
|
|
125
|
+
? langResult.avgMultisetRecall - baselineLang.avgMultisetRecall
|
|
126
|
+
: 0;
|
|
121
127
|
// R1 — only meaningful when BOTH sides carry role data; an un-regenerated
|
|
122
128
|
// baseline (no avgRoleFidelity yet) must never retro-flag.
|
|
123
129
|
const avgRoleFidelityDelta =
|
|
124
130
|
langResult.avgRoleFidelity !== undefined && baselineLang.avgRoleFidelity !== undefined
|
|
125
131
|
? langResult.avgRoleFidelity - baselineLang.avgRoleFidelity
|
|
126
132
|
: 0;
|
|
133
|
+
// R3 — same both-sides guard. Negative = a language-invariant role VALUE
|
|
134
|
+
// (selector/sigil/time/colon-event/URL) is now lost or corrupted, which
|
|
135
|
+
// every action/type-based signal scores as perfect.
|
|
136
|
+
const avgValueRecallDelta =
|
|
137
|
+
langResult.avgValueRecall !== undefined && baselineLang.avgValueRecall !== undefined
|
|
138
|
+
? langResult.avgValueRecall - baselineLang.avgValueRecall
|
|
139
|
+
: 0;
|
|
127
140
|
// R2 — same both-sides guard: an un-regenerated baseline (no execution
|
|
128
141
|
// data yet) must never retro-flag.
|
|
129
142
|
const avgExecutionFidelityDelta =
|
|
@@ -163,7 +176,9 @@ export class RegressionReporter implements Reporter {
|
|
|
163
176
|
avgConfidenceDelta,
|
|
164
177
|
avgFidelityDelta,
|
|
165
178
|
avgPrecisionDelta,
|
|
179
|
+
avgMultisetRecallDelta,
|
|
166
180
|
avgRoleFidelityDelta,
|
|
181
|
+
avgValueRecallDelta,
|
|
167
182
|
avgExecutionFidelityDelta,
|
|
168
183
|
bundleSizeDelta: bundleSizeDelta !== undefined ? bundleSizeDelta : undefined,
|
|
169
184
|
newFailures,
|
|
@@ -211,7 +226,11 @@ export class RegressionReporter implements Reporter {
|
|
|
211
226
|
}
|
|
212
227
|
}
|
|
213
228
|
|
|
214
|
-
|
|
229
|
+
// Returned UNCAPPED: this list is the R5 parse ratchet's evidence (cli.ts),
|
|
230
|
+
// not just a reporting nicety, and a gate that silently truncated itself
|
|
231
|
+
// would under-report how much broke. Display sites do their own capping
|
|
232
|
+
// (console-reporter shows 5; the CLI shows 20).
|
|
233
|
+
return newFailures;
|
|
215
234
|
}
|
|
216
235
|
|
|
217
236
|
/**
|
|
@@ -364,9 +383,17 @@ export class RegressionReporter implements Reporter {
|
|
|
364
383
|
// R0-precision — fraction of each parse's actions justified by the en
|
|
365
384
|
// reference (phantom-command signal recall can't see). Recorded + ratcheted.
|
|
366
385
|
avgPrecision: langResult.avgPrecision ?? undefined,
|
|
386
|
+
// R0-recall-multiset — the mirror of precision: a DROPPED repeated command
|
|
387
|
+
// (`[bind, bind]` parsed as `[bind]`) is invisible to the Set-based
|
|
388
|
+
// avgFidelity and avgRoleFidelity. Recorded + ratcheted.
|
|
389
|
+
avgMultisetRecall: langResult.avgMultisetRecall ?? undefined,
|
|
367
390
|
// R1 — role fidelity (role name + value type vs the en reference).
|
|
368
391
|
// Recorded + ratcheted; burn-down is NOT part of the parsing-track goal.
|
|
369
392
|
avgRoleFidelity: langResult.avgRoleFidelity ?? undefined,
|
|
393
|
+
// R3 — invariant role-VALUE recall (verbatim value comparison over the
|
|
394
|
+
// code-shaped subset: selectors, sigil refs, time literals,
|
|
395
|
+
// colon-qualified event names, URLs). Recorded + ratcheted.
|
|
396
|
+
avgValueRecall: langResult.avgValueRecall ?? undefined,
|
|
370
397
|
// R2 — execution fidelity over the curated subset (DOM effects vs the
|
|
371
398
|
// en reference in jsdom). Recorded + ratcheted; burn-down deferred.
|
|
372
399
|
avgExecutionFidelity: langResult.avgExecutionFidelity ?? undefined,
|