@hyperfixi/testing-framework 2.7.1 → 2.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assertions.d.mts +26 -1
- package/dist/assertions.d.ts +26 -1
- package/dist/index.d.mts +5 -112
- package/dist/index.d.ts +5 -112
- package/dist/runner.d.mts +112 -0
- package/dist/runner.d.ts +112 -0
- package/dist/runner.js +1102 -0
- package/dist/runner.js.map +1 -0
- package/dist/runner.mjs +1097 -0
- package/dist/runner.mjs.map +1 -0
- package/dist/{assertions-CsGP61iW.d.mts → types-D-rCVkf3.d.mts} +1 -24
- package/dist/{assertions-CsGP61iW.d.ts → types-D-rCVkf3.d.ts} +1 -24
- package/package.json +11 -26
- package/src/multilingual/canonical-validity.test.ts +69 -0
- package/src/multilingual/canonical-validity.ts +132 -0
- package/src/multilingual/cli.ts +185 -1
- package/src/multilingual/fidelity.test.ts +192 -0
- package/src/multilingual/fidelity.ts +153 -0
- package/src/multilingual/foreign-canonical-validity.test.ts +0 -0
- package/src/multilingual/foreign-canonical-validity.ts +158 -0
- package/src/multilingual/orchestrator.ts +47 -1
- package/src/multilingual/reporters/console-reporter.ts +51 -0
- package/src/multilingual/reporters/regression-reporter.ts +23 -0
- package/src/multilingual/tools/diagnose-coverage.ts +118 -0
- package/src/multilingual/tools/triage-r1.ts +149 -0
- package/src/multilingual/types.ts +74 -0
- package/src/multilingual/validators/parse-validator.ts +9 -1
- package/src/runner.test.ts +7 -2
- package/src/vocab/batch3-roundtrip.test.ts +184 -0
- package/src/vocab/checks.test.ts +362 -0
- package/src/vocab/checks.ts +311 -0
- package/src/vocab/cli.ts +196 -0
- package/src/vocab/dump.ts +80 -0
- package/src/vocab/model.ts +110 -0
- package/src/vocab/report.ts +120 -0
- package/src/vocab/types.ts +100 -0
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Foreign→English canonical-validity gate
|
|
3
|
+
* ---------------------------------------
|
|
4
|
+
* The sibling en-render gate (canonical-validity.ts) renders every corpus English
|
|
5
|
+
* reference and parses it on the real `hyperscript.org` engine. But the PRODUCTION
|
|
6
|
+
* path is foreign→English: an authored non-English source is parsed and rendered to
|
|
7
|
+
* English (`preprocessToEnglish` → `render`). A parse can be role-faithful (the
|
|
8
|
+
* fidelity ratchet scores ~1.0) yet still render English the canonical parser rejects.
|
|
9
|
+
* (The build-time `@hyperscript-tools/i18n` transpiler now parse-gates its ENGLISH
|
|
10
|
+
* side with the same loader recipe; faithful foreign-output gating there still awaits
|
|
11
|
+
* the v2 semantic-engine transpiler, since its GrammarTransformer is lossy in reverse.)
|
|
12
|
+
*
|
|
13
|
+
* This gate closes that blind spot for the multilingual path: for every language, it
|
|
14
|
+
* renders each authored `pattern_translation` to English and parses the result on
|
|
15
|
+
* the canonical engine, failing on any invalid (pattern, language) pair that is not
|
|
16
|
+
* in the committed allowlist. The allowlist is keyed by pattern id → the languages
|
|
17
|
+
* that currently fail, so a fix that clears a family across languages shrinks (or
|
|
18
|
+
* removes) its entry — the list only ever ratchets down.
|
|
19
|
+
*
|
|
20
|
+
* Denominator: only (pattern, language) pairs whose EN reference the canonical parser
|
|
21
|
+
* already accepts are scored, so a handful of inherently non-canonical corpus rows
|
|
22
|
+
* never distort the signal (same fairness rule as the en gate).
|
|
23
|
+
*
|
|
24
|
+
* DB dependency: reads authored translations from `pattern_translations`, which only
|
|
25
|
+
* exist after `npm run populate`. Generate the baseline and run the gate against a
|
|
26
|
+
* freshly populated DB (CI's multilingual-validation job populates; see the
|
|
27
|
+
* provenance-stamp discipline in packages/patterns-reference/CLAUDE.md).
|
|
28
|
+
*/
|
|
29
|
+
|
|
30
|
+
import { getAllPatterns, getTranslationsByLanguage } from '@hyperfixi/patterns-reference';
|
|
31
|
+
import { parseSemantic, render } from '@lokascript/semantic';
|
|
32
|
+
import { loadCanonicalParser, type CanonicalValidate } from './canonical-validity';
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* The 23 non-English priority languages (the `browser-priority` corpus set). English
|
|
36
|
+
* is the reference, scored by the sibling en gate.
|
|
37
|
+
*/
|
|
38
|
+
export const FOREIGN_LANGUAGES = [
|
|
39
|
+
'es',
|
|
40
|
+
'fr',
|
|
41
|
+
'pt',
|
|
42
|
+
'it',
|
|
43
|
+
'id',
|
|
44
|
+
'ms',
|
|
45
|
+
'sw',
|
|
46
|
+
'zh',
|
|
47
|
+
'vi',
|
|
48
|
+
'tl',
|
|
49
|
+
'ja',
|
|
50
|
+
'ko',
|
|
51
|
+
'tr',
|
|
52
|
+
'qu',
|
|
53
|
+
'hi',
|
|
54
|
+
'bn',
|
|
55
|
+
'ar',
|
|
56
|
+
'de',
|
|
57
|
+
'ru',
|
|
58
|
+
'uk',
|
|
59
|
+
'pl',
|
|
60
|
+
'th',
|
|
61
|
+
'he',
|
|
62
|
+
] as const;
|
|
63
|
+
|
|
64
|
+
export interface ForeignValidityFailure {
|
|
65
|
+
/** Corpus pattern id (`code_example` id the translation belongs to). */
|
|
66
|
+
id: string;
|
|
67
|
+
language: string;
|
|
68
|
+
/** The authored foreign source that was rendered. */
|
|
69
|
+
foreign: string;
|
|
70
|
+
/** The English the renderer produced from the foreign parse. */
|
|
71
|
+
rendered: string;
|
|
72
|
+
error: string;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
export interface ForeignValidityResult {
|
|
76
|
+
/** (pattern, language) pairs whose EN reference the canonical parser accepts. */
|
|
77
|
+
checked: number;
|
|
78
|
+
valid: number;
|
|
79
|
+
failures: ForeignValidityFailure[];
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* Render every authored foreign `pattern_translation` to English and parse it on the
|
|
84
|
+
* canonical engine. Only pairs whose EN reference is itself canonical-valid are scored.
|
|
85
|
+
*/
|
|
86
|
+
export async function checkForeignRenderValidity(opts?: {
|
|
87
|
+
validate?: CanonicalValidate;
|
|
88
|
+
languages?: readonly string[];
|
|
89
|
+
/** Translations fetched per language (defaults comfortably above the corpus size). */
|
|
90
|
+
perLanguageLimit?: number;
|
|
91
|
+
}): Promise<ForeignValidityResult> {
|
|
92
|
+
const validate = opts?.validate ?? (await loadCanonicalParser());
|
|
93
|
+
const languages = opts?.languages ?? FOREIGN_LANGUAGES;
|
|
94
|
+
const limit = opts?.perLanguageLimit ?? 500;
|
|
95
|
+
|
|
96
|
+
const patterns = await getAllPatterns();
|
|
97
|
+
const enAccepts = new Map(patterns.map(p => [p.id, validate(p.rawCode).length === 0]));
|
|
98
|
+
|
|
99
|
+
const failures: ForeignValidityFailure[] = [];
|
|
100
|
+
let checked = 0;
|
|
101
|
+
let valid = 0;
|
|
102
|
+
|
|
103
|
+
for (const language of languages) {
|
|
104
|
+
const translations = await getTranslationsByLanguage(language, limit);
|
|
105
|
+
for (const t of translations) {
|
|
106
|
+
if (!enAccepts.get(t.codeExampleId)) continue; // fair denominator: EN accepts the reference
|
|
107
|
+
checked++;
|
|
108
|
+
|
|
109
|
+
let rendered: string;
|
|
110
|
+
let errors: string[];
|
|
111
|
+
try {
|
|
112
|
+
const node = parseSemantic(t.hyperscript, language).node;
|
|
113
|
+
rendered = node ? render(node, 'en') : '(no node)';
|
|
114
|
+
errors = validate(rendered);
|
|
115
|
+
} catch (e) {
|
|
116
|
+
// parseSemantic/render only — validate never throws (its tokenizer-level
|
|
117
|
+
// throws fold into the returned array), so an invalid render is reported
|
|
118
|
+
// WITH the render that caused it, not discarded as '(threw)'.
|
|
119
|
+
rendered = '(threw)';
|
|
120
|
+
errors = ['threw: ' + (e as Error).message.split('\n')[0]];
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
if (errors.length === 0) {
|
|
124
|
+
valid++;
|
|
125
|
+
} else {
|
|
126
|
+
failures.push({
|
|
127
|
+
id: t.codeExampleId,
|
|
128
|
+
language,
|
|
129
|
+
foreign: t.hyperscript,
|
|
130
|
+
rendered,
|
|
131
|
+
error: errors[0] ?? 'unknown error',
|
|
132
|
+
});
|
|
133
|
+
}
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
return { checked, valid, failures };
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
/**
|
|
141
|
+
* Group failures into the committed allowlist shape: `{ patternId: [langs…] }`
|
|
142
|
+
* (languages sorted for a stable diff). Used by the baseline generator and by the
|
|
143
|
+
* gate's stale-entry check.
|
|
144
|
+
*/
|
|
145
|
+
export function groupFailuresByPattern(
|
|
146
|
+
failures: readonly ForeignValidityFailure[]
|
|
147
|
+
): Record<string, string[]> {
|
|
148
|
+
const byPattern = new Map<string, Set<string>>();
|
|
149
|
+
for (const f of failures) {
|
|
150
|
+
if (!byPattern.has(f.id)) byPattern.set(f.id, new Set());
|
|
151
|
+
byPattern.get(f.id)!.add(f.language);
|
|
152
|
+
}
|
|
153
|
+
const out: Record<string, string[]> = {};
|
|
154
|
+
for (const id of [...byPattern.keys()].sort()) {
|
|
155
|
+
out[id] = [...byPattern.get(id)!].sort();
|
|
156
|
+
}
|
|
157
|
+
return out;
|
|
158
|
+
}
|
|
@@ -20,7 +20,13 @@ import type {
|
|
|
20
20
|
Reporter,
|
|
21
21
|
BundleInfo,
|
|
22
22
|
} from './types';
|
|
23
|
-
import {
|
|
23
|
+
import {
|
|
24
|
+
computeFidelity,
|
|
25
|
+
computePrecision,
|
|
26
|
+
computeMultisetRecall,
|
|
27
|
+
spuriousActions,
|
|
28
|
+
FIDELITY_THRESHOLD,
|
|
29
|
+
} from './fidelity';
|
|
24
30
|
|
|
25
31
|
const execAsync = promisify(exec);
|
|
26
32
|
|
|
@@ -258,6 +264,8 @@ export class TestOrchestrator {
|
|
|
258
264
|
const multisetReference = new Map<string, string[]>();
|
|
259
265
|
// R1: codeExampleId -> English role signature (action.role:valueType set).
|
|
260
266
|
const roleReference = new Map<string, string[]>();
|
|
267
|
+
// R3: codeExampleId -> English role-VALUE signature (invariant values, multiset).
|
|
268
|
+
const roleValueReference = new Map<string, string[]>();
|
|
261
269
|
for (const r of en.parseResults) {
|
|
262
270
|
if (r.success && r.actionSignature && r.actionSignature.length > 0) {
|
|
263
271
|
reference.set(r.pattern.codeExampleId, r.actionSignature);
|
|
@@ -268,6 +276,9 @@ export class TestOrchestrator {
|
|
|
268
276
|
if (r.success && r.roleSignature && r.roleSignature.length > 0) {
|
|
269
277
|
roleReference.set(r.pattern.codeExampleId, r.roleSignature);
|
|
270
278
|
}
|
|
279
|
+
if (r.success && r.roleValueSignature && r.roleValueSignature.length > 0) {
|
|
280
|
+
roleValueReference.set(r.pattern.codeExampleId, r.roleValueSignature);
|
|
281
|
+
}
|
|
271
282
|
}
|
|
272
283
|
|
|
273
284
|
for (const lang of languageResults) {
|
|
@@ -277,7 +288,9 @@ export class TestOrchestrator {
|
|
|
277
288
|
const lossy: string[] = [];
|
|
278
289
|
const scores: number[] = [];
|
|
279
290
|
const precisionScores: number[] = [];
|
|
291
|
+
const multisetRecallScores: number[] = [];
|
|
280
292
|
const roleScores: number[] = [];
|
|
293
|
+
const valueRecallScores: number[] = [];
|
|
281
294
|
|
|
282
295
|
for (const result of lang.parseResults) {
|
|
283
296
|
if (!result.success || !result.actionSignature) continue;
|
|
@@ -294,6 +307,9 @@ export class TestOrchestrator {
|
|
|
294
307
|
|
|
295
308
|
// R0-precision — fraction of THIS parse's actions justified by the en
|
|
296
309
|
// multiset reference (catches phantom/spurious commands recall misses).
|
|
310
|
+
// R0-recall-multiset — the mirror: fraction of the en reference's actions,
|
|
311
|
+
// counting duplicates, present here (catches a DROPPED repeated command,
|
|
312
|
+
// which the Set-based fidelity/roleFidelity above cannot see).
|
|
297
313
|
const multisetRef = multisetReference.get(result.pattern.codeExampleId);
|
|
298
314
|
if (multisetRef && result.actionMultisetSignature) {
|
|
299
315
|
const precision = computePrecision(multisetRef, result.actionMultisetSignature);
|
|
@@ -301,6 +317,11 @@ export class TestOrchestrator {
|
|
|
301
317
|
result.precision = precision;
|
|
302
318
|
precisionScores.push(precision);
|
|
303
319
|
}
|
|
320
|
+
const multisetRecall = computeMultisetRecall(multisetRef, result.actionMultisetSignature);
|
|
321
|
+
if (multisetRecall !== undefined) {
|
|
322
|
+
result.multisetRecall = multisetRecall;
|
|
323
|
+
multisetRecallScores.push(multisetRecall);
|
|
324
|
+
}
|
|
304
325
|
}
|
|
305
326
|
|
|
306
327
|
// R1 — role recall vs the en role signature (role name + value type).
|
|
@@ -312,6 +333,23 @@ export class TestOrchestrator {
|
|
|
312
333
|
roleScores.push(roleFidelity);
|
|
313
334
|
}
|
|
314
335
|
}
|
|
336
|
+
|
|
337
|
+
// R3 — invariant role-VALUE recall vs the en value signature (multiset).
|
|
338
|
+
// Values are compared verbatim: the filtered subset (selectors, sigil
|
|
339
|
+
// refs, time literals, colon-qualified events, URLs) is code, not prose.
|
|
340
|
+
const valueRef = roleValueReference.get(result.pattern.codeExampleId);
|
|
341
|
+
if (valueRef && result.roleValueSignature) {
|
|
342
|
+
const valueRecall = computeMultisetRecall(valueRef, result.roleValueSignature);
|
|
343
|
+
if (valueRecall !== undefined) {
|
|
344
|
+
result.valueRecall = valueRecall;
|
|
345
|
+
valueRecallScores.push(valueRecall);
|
|
346
|
+
if (valueRecall < 1) {
|
|
347
|
+
// Missing = reference entries not covered here (multiset diff;
|
|
348
|
+
// spuriousActions with the arguments swapped).
|
|
349
|
+
result.valueRecallMissing = spuriousActions(result.roleValueSignature, valueRef);
|
|
350
|
+
}
|
|
351
|
+
}
|
|
352
|
+
}
|
|
315
353
|
}
|
|
316
354
|
|
|
317
355
|
lang.avgFidelity =
|
|
@@ -320,10 +358,18 @@ export class TestOrchestrator {
|
|
|
320
358
|
precisionScores.length > 0
|
|
321
359
|
? precisionScores.reduce((a, b) => a + b, 0) / precisionScores.length
|
|
322
360
|
: undefined;
|
|
361
|
+
lang.avgMultisetRecall =
|
|
362
|
+
multisetRecallScores.length > 0
|
|
363
|
+
? multisetRecallScores.reduce((a, b) => a + b, 0) / multisetRecallScores.length
|
|
364
|
+
: undefined;
|
|
323
365
|
lang.avgRoleFidelity =
|
|
324
366
|
roleScores.length > 0
|
|
325
367
|
? roleScores.reduce((a, b) => a + b, 0) / roleScores.length
|
|
326
368
|
: undefined;
|
|
369
|
+
lang.avgValueRecall =
|
|
370
|
+
valueRecallScores.length > 0
|
|
371
|
+
? valueRecallScores.reduce((a, b) => a + b, 0) / valueRecallScores.length
|
|
372
|
+
: undefined;
|
|
327
373
|
lang.degeneratePasses = degenerate.sort();
|
|
328
374
|
lang.lossyPasses = lossy.sort();
|
|
329
375
|
}
|
|
@@ -158,11 +158,62 @@ export class ConsoleReporter implements Reporter {
|
|
|
158
158
|
}
|
|
159
159
|
|
|
160
160
|
this.reportDegeneratePasses(results);
|
|
161
|
+
this.reportValueRecall(results);
|
|
161
162
|
this.reportExecutionFailures(results);
|
|
162
163
|
|
|
163
164
|
this.log('');
|
|
164
165
|
}
|
|
165
166
|
|
|
167
|
+
/**
|
|
168
|
+
* R3 — surface parses that lost/corrupted a language-invariant role VALUE
|
|
169
|
+
* (selector, sigil ref, time literal, colon-qualified event name, URL) vs
|
|
170
|
+
* the en reference. These score 1.0 on every action/type-based signal yet
|
|
171
|
+
* behave differently at runtime, so they're reported separately. Report-only:
|
|
172
|
+
* the ratchet gate lives in the --regression path.
|
|
173
|
+
*/
|
|
174
|
+
private reportValueRecall(results: TestResults): void {
|
|
175
|
+
const scored = results.languageResults.filter(l => l.avgValueRecall !== undefined);
|
|
176
|
+
if (scored.length === 0) return;
|
|
177
|
+
|
|
178
|
+
const avg = scored.reduce((sum, l) => sum + (l.avgValueRecall ?? 0), 0) / scored.length;
|
|
179
|
+
this.log('');
|
|
180
|
+
this.log(
|
|
181
|
+
this.bright(`Role values (R3): avgValueRecall ${avg.toFixed(4)} over invariant values`)
|
|
182
|
+
);
|
|
183
|
+
|
|
184
|
+
// pattern id -> "lang: missing entries" rows
|
|
185
|
+
const byPattern = new Map<string, string[]>();
|
|
186
|
+
let instances = 0;
|
|
187
|
+
for (const lang of scored) {
|
|
188
|
+
for (const r of lang.parseResults) {
|
|
189
|
+
if (r.valueRecall === undefined || r.valueRecall >= 1) continue;
|
|
190
|
+
instances++;
|
|
191
|
+
let rows = byPattern.get(r.pattern.codeExampleId);
|
|
192
|
+
if (!rows) {
|
|
193
|
+
rows = [];
|
|
194
|
+
byPattern.set(r.pattern.codeExampleId, rows);
|
|
195
|
+
}
|
|
196
|
+
rows.push(`${lang.language}: missing ${(r.valueRecallMissing ?? []).join(', ')}`);
|
|
197
|
+
}
|
|
198
|
+
}
|
|
199
|
+
if (byPattern.size === 0) return;
|
|
200
|
+
|
|
201
|
+
this.log(
|
|
202
|
+
this.yellow(
|
|
203
|
+
`⚠ Invariant value loss: ${instances} instance(s) across ${byPattern.size} pattern(s)`
|
|
204
|
+
)
|
|
205
|
+
);
|
|
206
|
+
this.log(
|
|
207
|
+
this.dim(
|
|
208
|
+
' (an action.role=value present in the en reference is absent — not parse failures)'
|
|
209
|
+
)
|
|
210
|
+
);
|
|
211
|
+
for (const [id, rows] of [...byPattern.entries()].sort((a, b) => b[1].length - a[1].length)) {
|
|
212
|
+
this.log(` ${this.dim('-')} ${id}`);
|
|
213
|
+
for (const row of rows.sort()) this.log(` ${this.dim(row)}`);
|
|
214
|
+
}
|
|
215
|
+
}
|
|
216
|
+
|
|
166
217
|
/**
|
|
167
218
|
* R2 — surface curated-subset patterns whose jsdom execution diverged from
|
|
168
219
|
* the en reference's DOM effects. These can score 1.0 on parse fidelity yet
|
|
@@ -118,12 +118,25 @@ export class RegressionReporter implements Reporter {
|
|
|
118
118
|
langResult.avgPrecision !== undefined && baselineLang.avgPrecision !== undefined
|
|
119
119
|
? langResult.avgPrecision - baselineLang.avgPrecision
|
|
120
120
|
: 0;
|
|
121
|
+
// R0-recall-multiset — same both-sides guard. Negative = a repeated command
|
|
122
|
+
// is now being dropped, which avgFidelity (a Set) cannot see.
|
|
123
|
+
const avgMultisetRecallDelta =
|
|
124
|
+
langResult.avgMultisetRecall !== undefined && baselineLang.avgMultisetRecall !== undefined
|
|
125
|
+
? langResult.avgMultisetRecall - baselineLang.avgMultisetRecall
|
|
126
|
+
: 0;
|
|
121
127
|
// R1 — only meaningful when BOTH sides carry role data; an un-regenerated
|
|
122
128
|
// baseline (no avgRoleFidelity yet) must never retro-flag.
|
|
123
129
|
const avgRoleFidelityDelta =
|
|
124
130
|
langResult.avgRoleFidelity !== undefined && baselineLang.avgRoleFidelity !== undefined
|
|
125
131
|
? langResult.avgRoleFidelity - baselineLang.avgRoleFidelity
|
|
126
132
|
: 0;
|
|
133
|
+
// R3 — same both-sides guard. Negative = a language-invariant role VALUE
|
|
134
|
+
// (selector/sigil/time/colon-event/URL) is now lost or corrupted, which
|
|
135
|
+
// every action/type-based signal scores as perfect.
|
|
136
|
+
const avgValueRecallDelta =
|
|
137
|
+
langResult.avgValueRecall !== undefined && baselineLang.avgValueRecall !== undefined
|
|
138
|
+
? langResult.avgValueRecall - baselineLang.avgValueRecall
|
|
139
|
+
: 0;
|
|
127
140
|
// R2 — same both-sides guard: an un-regenerated baseline (no execution
|
|
128
141
|
// data yet) must never retro-flag.
|
|
129
142
|
const avgExecutionFidelityDelta =
|
|
@@ -163,7 +176,9 @@ export class RegressionReporter implements Reporter {
|
|
|
163
176
|
avgConfidenceDelta,
|
|
164
177
|
avgFidelityDelta,
|
|
165
178
|
avgPrecisionDelta,
|
|
179
|
+
avgMultisetRecallDelta,
|
|
166
180
|
avgRoleFidelityDelta,
|
|
181
|
+
avgValueRecallDelta,
|
|
167
182
|
avgExecutionFidelityDelta,
|
|
168
183
|
bundleSizeDelta: bundleSizeDelta !== undefined ? bundleSizeDelta : undefined,
|
|
169
184
|
newFailures,
|
|
@@ -364,9 +379,17 @@ export class RegressionReporter implements Reporter {
|
|
|
364
379
|
// R0-precision — fraction of each parse's actions justified by the en
|
|
365
380
|
// reference (phantom-command signal recall can't see). Recorded + ratcheted.
|
|
366
381
|
avgPrecision: langResult.avgPrecision ?? undefined,
|
|
382
|
+
// R0-recall-multiset — the mirror of precision: a DROPPED repeated command
|
|
383
|
+
// (`[bind, bind]` parsed as `[bind]`) is invisible to the Set-based
|
|
384
|
+
// avgFidelity and avgRoleFidelity. Recorded + ratcheted.
|
|
385
|
+
avgMultisetRecall: langResult.avgMultisetRecall ?? undefined,
|
|
367
386
|
// R1 — role fidelity (role name + value type vs the en reference).
|
|
368
387
|
// Recorded + ratcheted; burn-down is NOT part of the parsing-track goal.
|
|
369
388
|
avgRoleFidelity: langResult.avgRoleFidelity ?? undefined,
|
|
389
|
+
// R3 — invariant role-VALUE recall (verbatim value comparison over the
|
|
390
|
+
// code-shaped subset: selectors, sigil refs, time literals,
|
|
391
|
+
// colon-qualified event names, URLs). Recorded + ratcheted.
|
|
392
|
+
avgValueRecall: langResult.avgValueRecall ?? undefined,
|
|
370
393
|
// R2 — execution fidelity over the curated subset (DOM effects vs the
|
|
371
394
|
// en reference in jsdom). Recorded + ratcheted; burn-down deferred.
|
|
372
395
|
avgExecutionFidelity: langResult.avgExecutionFidelity ?? undefined,
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* diagnose-coverage.ts
|
|
3
|
+
*
|
|
4
|
+
* Read-only sweep that counts how often the semantic parser matches a pattern
|
|
5
|
+
* covering only PART of its input — the `unconsumed-input` diagnostic.
|
|
6
|
+
*
|
|
7
|
+
* Why this exists: confidence is computed from role coverage (how many of the
|
|
8
|
+
* pattern's own roles were filled), never from input coverage. A short pattern
|
|
9
|
+
* can fill every role, score 1.0, and silently drop a trailing clause. Turning
|
|
10
|
+
* that into a scoring penalty would move the multilingual fidelity baseline
|
|
11
|
+
* across all 24 languages, so the firing rate has to be MEASURED first — on the
|
|
12
|
+
* real corpus, per language, with the dropped spans visible.
|
|
13
|
+
*
|
|
14
|
+
* This tool never writes a baseline and never gates. It only reads.
|
|
15
|
+
*/
|
|
16
|
+
import { parseSemantic } from '@lokascript/semantic';
|
|
17
|
+
|
|
18
|
+
import { loadPatterns } from '../pattern-loader';
|
|
19
|
+
import type { TestConfig } from '../types';
|
|
20
|
+
|
|
21
|
+
interface LanguageTally {
|
|
22
|
+
language: string;
|
|
23
|
+
total: number;
|
|
24
|
+
fired: number;
|
|
25
|
+
/** A few representative dropped spans, for eyeballing false positives. */
|
|
26
|
+
samples: Array<{ source: string; message: string; confidence: number }>;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
const MAX_SAMPLES_PER_LANGUAGE = 3;
|
|
30
|
+
const UNCONSUMED_CODE = 'unconsumed-input';
|
|
31
|
+
|
|
32
|
+
function unconsumedMessages(node: unknown): string[] {
|
|
33
|
+
const diagnostics = (node as { diagnostics?: Array<{ code?: string; message: string }> } | null)
|
|
34
|
+
?.diagnostics;
|
|
35
|
+
if (!diagnostics?.length) return [];
|
|
36
|
+
return diagnostics.filter(d => d.code === UNCONSUMED_CODE).map(d => d.message);
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* Sweep the corpus and report the `unconsumed-input` firing rate per language.
|
|
41
|
+
*
|
|
42
|
+
* @returns the total number of firings (0 when the signal never trips)
|
|
43
|
+
*/
|
|
44
|
+
export async function diagnoseCoverage(config: TestConfig): Promise<number> {
|
|
45
|
+
const patterns = await loadPatterns(config);
|
|
46
|
+
const byLanguage = new Map<string, LanguageTally>();
|
|
47
|
+
|
|
48
|
+
for (const pattern of patterns) {
|
|
49
|
+
const tally = byLanguage.get(pattern.language) ?? {
|
|
50
|
+
language: pattern.language,
|
|
51
|
+
total: 0,
|
|
52
|
+
fired: 0,
|
|
53
|
+
samples: [],
|
|
54
|
+
};
|
|
55
|
+
tally.total++;
|
|
56
|
+
|
|
57
|
+
let messages: string[] = [];
|
|
58
|
+
let confidence = 0;
|
|
59
|
+
try {
|
|
60
|
+
const result = parseSemantic(pattern.hyperscript, pattern.language);
|
|
61
|
+
confidence = result.confidence;
|
|
62
|
+
messages = unconsumedMessages(result.node);
|
|
63
|
+
} catch {
|
|
64
|
+
// A parse failure is not an input-coverage problem — it is already visible
|
|
65
|
+
// to the parse-rate gate. Skip it rather than double-counting.
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
const firstMessage = messages[0];
|
|
69
|
+
if (firstMessage !== undefined) {
|
|
70
|
+
tally.fired++;
|
|
71
|
+
if (tally.samples.length < MAX_SAMPLES_PER_LANGUAGE) {
|
|
72
|
+
tally.samples.push({ source: pattern.hyperscript, message: firstMessage, confidence });
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
byLanguage.set(pattern.language, tally);
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
report([...byLanguage.values()].sort((a, b) => b.fired / b.total - a.fired / a.total));
|
|
79
|
+
return [...byLanguage.values()].reduce((sum, t) => sum + t.fired, 0);
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
function report(tallies: LanguageTally[]): void {
|
|
83
|
+
const total = tallies.reduce((s, t) => s + t.total, 0);
|
|
84
|
+
const fired = tallies.reduce((s, t) => s + t.fired, 0);
|
|
85
|
+
|
|
86
|
+
console.log('\nInput-coverage diagnostic — `unconsumed-input` firing rate');
|
|
87
|
+
console.log('(a fired row parsed successfully but ignored part of its source)\n');
|
|
88
|
+
console.log(' lang fired / total rate');
|
|
89
|
+
console.log(' ─────────────────────────────────');
|
|
90
|
+
for (const t of tallies) {
|
|
91
|
+
const rate = t.total ? (t.fired / t.total) * 100 : 0;
|
|
92
|
+
console.log(
|
|
93
|
+
` ${t.language.padEnd(6)} ${String(t.fired).padStart(5)} / ${String(t.total).padStart(6)} ${rate.toFixed(1).padStart(6)}%`
|
|
94
|
+
);
|
|
95
|
+
}
|
|
96
|
+
console.log(' ─────────────────────────────────');
|
|
97
|
+
const overall = total ? (fired / total) * 100 : 0;
|
|
98
|
+
console.log(
|
|
99
|
+
` ALL ${String(fired).padStart(5)} / ${String(total).padStart(6)} ${overall.toFixed(1).padStart(6)}%\n`
|
|
100
|
+
);
|
|
101
|
+
|
|
102
|
+
const withSamples = tallies.filter(t => t.samples.length > 0);
|
|
103
|
+
if (withSamples.length === 0) {
|
|
104
|
+
console.log('No unconsumed input anywhere in the corpus.\n');
|
|
105
|
+
return;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
console.log('Sample dropped spans (check these for false positives — trailing');
|
|
109
|
+
console.log('particles and legitimately-optional tokens are expected residue):\n');
|
|
110
|
+
for (const t of withSamples) {
|
|
111
|
+
console.log(` [${t.language}]`);
|
|
112
|
+
for (const s of t.samples) {
|
|
113
|
+
console.log(` source (conf ${s.confidence.toFixed(2)}): ${s.source}`);
|
|
114
|
+
console.log(` ${s.message}`);
|
|
115
|
+
}
|
|
116
|
+
console.log('');
|
|
117
|
+
}
|
|
118
|
+
}
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* triage-r1.ts
|
|
3
|
+
*
|
|
4
|
+
* Read-only sweep that itemizes R1 role-fidelity misses per language: for every
|
|
5
|
+
* corpus pattern whose roleFidelity < 1, the `action.role:type` entries present
|
|
6
|
+
* in the en reference parse but absent from the translation's parse — clustered
|
|
7
|
+
* by entry so the dominant failure families are visible, with the entries the
|
|
8
|
+
* translation captured INSTEAD for the same action (the mistype pairing).
|
|
9
|
+
*
|
|
10
|
+
* Why this exists: avgRoleFidelity summarizes the SOV-six gap (qu ~0.95, the
|
|
11
|
+
* rest ~0.97) to one number per language, and the regression gate only reports
|
|
12
|
+
* DROPS. Planning an improvement arc needs the misses themselves — which roles,
|
|
13
|
+
* on which actions, in which patterns — measured with exactly the pipeline the
|
|
14
|
+
* gate scores (ParseValidator → fillSchemaDefaults → collectRoleSignature →
|
|
15
|
+
* Set recall vs the en reference), so a "fix" that moves the probe also moves
|
|
16
|
+
* the signal.
|
|
17
|
+
*
|
|
18
|
+
* This tool never writes a baseline and never gates. It only reads.
|
|
19
|
+
*/
|
|
20
|
+
import { computeFidelity } from '../fidelity';
|
|
21
|
+
import { loadPatterns } from '../pattern-loader';
|
|
22
|
+
import { ParseValidator } from '../validators/parse-validator';
|
|
23
|
+
import type { ParseResult, TestConfig } from '../types';
|
|
24
|
+
|
|
25
|
+
interface MissCluster {
|
|
26
|
+
/** The en-reference `action.role:type` entry the language failed to recall. */
|
|
27
|
+
entry: string;
|
|
28
|
+
/** Pattern ids where this entry was missed. */
|
|
29
|
+
patterns: string[];
|
|
30
|
+
/**
|
|
31
|
+
* What the language's parse captured for the SAME action in those patterns
|
|
32
|
+
* (entries absent from the en reference) — usually the mistype the miss
|
|
33
|
+
* paired with (`trigger.patient:literal` opposite a missing
|
|
34
|
+
* `trigger.event:literal`). Empty when the whole command dropped.
|
|
35
|
+
*/
|
|
36
|
+
haveInstead: Set<string>;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
const MAX_PATTERN_IDS_SHOWN = 6;
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* Sweep the corpus and itemize R1 misses for the requested languages
|
|
43
|
+
* (default: every non-en language in the run).
|
|
44
|
+
*/
|
|
45
|
+
export async function triageR1(config: TestConfig): Promise<void> {
|
|
46
|
+
// The en rows ARE the reference — force-load them even when the caller
|
|
47
|
+
// filtered --languages to the triage targets.
|
|
48
|
+
const requested = config.languages?.filter(l => l !== 'en');
|
|
49
|
+
const loadConfig: TestConfig =
|
|
50
|
+
requested && requested.length > 0
|
|
51
|
+
? { ...config, languages: [...requested, 'en' as const] }
|
|
52
|
+
: config;
|
|
53
|
+
|
|
54
|
+
const patterns = await loadPatterns(loadConfig);
|
|
55
|
+
const validator = new ParseValidator();
|
|
56
|
+
const results = await validator.validate(patterns);
|
|
57
|
+
|
|
58
|
+
const reference = new Map<string, string[]>();
|
|
59
|
+
for (const r of results) {
|
|
60
|
+
if (r.pattern.language !== 'en') continue;
|
|
61
|
+
if (r.success && r.roleSignature && r.roleSignature.length > 0) {
|
|
62
|
+
reference.set(r.pattern.codeExampleId, r.roleSignature);
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
const byLanguage = new Map<string, ParseResult[]>();
|
|
67
|
+
for (const r of results) {
|
|
68
|
+
if (r.pattern.language === 'en') continue;
|
|
69
|
+
if (requested && requested.length > 0 && !requested.includes(r.pattern.language)) continue;
|
|
70
|
+
const list = byLanguage.get(r.pattern.language) ?? [];
|
|
71
|
+
list.push(r);
|
|
72
|
+
byLanguage.set(r.pattern.language, list);
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
console.log('\nR1 role-fidelity triage (action.role:type recall vs the en reference)');
|
|
76
|
+
console.log('━'.repeat(72));
|
|
77
|
+
|
|
78
|
+
for (const [language, langResults] of [...byLanguage.entries()].sort()) {
|
|
79
|
+
const clusters = new Map<string, MissCluster>();
|
|
80
|
+
const scores: number[] = [];
|
|
81
|
+
let patternsWithMisses = 0;
|
|
82
|
+
|
|
83
|
+
for (const result of langResults) {
|
|
84
|
+
if (!result.success || !result.roleSignature) continue;
|
|
85
|
+
const ref = reference.get(result.pattern.codeExampleId);
|
|
86
|
+
if (!ref) continue;
|
|
87
|
+
|
|
88
|
+
const fidelity = computeFidelity(ref, result.roleSignature);
|
|
89
|
+
if (fidelity === undefined) continue;
|
|
90
|
+
scores.push(fidelity);
|
|
91
|
+
if (fidelity >= 1) continue;
|
|
92
|
+
|
|
93
|
+
patternsWithMisses++;
|
|
94
|
+
const candidate = new Set(result.roleSignature);
|
|
95
|
+
const refSet = new Set(ref);
|
|
96
|
+
const missing = ref.filter(e => !candidate.has(e));
|
|
97
|
+
// Entries this parse has that the reference doesn't — candidates for the
|
|
98
|
+
// "captured instead" pairing, keyed by action.
|
|
99
|
+
const extrasByAction = new Map<string, string[]>();
|
|
100
|
+
for (const e of result.roleSignature) {
|
|
101
|
+
if (refSet.has(e)) continue;
|
|
102
|
+
const action = e.slice(0, e.indexOf('.'));
|
|
103
|
+
const list = extrasByAction.get(action) ?? [];
|
|
104
|
+
list.push(e);
|
|
105
|
+
extrasByAction.set(action, list);
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
for (const entry of missing) {
|
|
109
|
+
const cluster = clusters.get(entry) ?? {
|
|
110
|
+
entry,
|
|
111
|
+
patterns: [],
|
|
112
|
+
haveInstead: new Set<string>(),
|
|
113
|
+
};
|
|
114
|
+
cluster.patterns.push(result.pattern.codeExampleId);
|
|
115
|
+
const action = entry.slice(0, entry.indexOf('.'));
|
|
116
|
+
for (const e of extrasByAction.get(action) ?? []) cluster.haveInstead.add(e);
|
|
117
|
+
clusters.set(entry, cluster);
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
const avg = scores.length > 0 ? scores.reduce((a, b) => a + b, 0) / scores.length : undefined;
|
|
122
|
+
const totalMisses = [...clusters.values()].reduce((a, c) => a + c.patterns.length, 0);
|
|
123
|
+
|
|
124
|
+
console.log(
|
|
125
|
+
`\n[${language}] avgRoleFidelity ${avg?.toFixed(4) ?? 'n/a'} — ` +
|
|
126
|
+
`${totalMisses} miss(es) across ${patternsWithMisses} pattern(s), ` +
|
|
127
|
+
`${clusters.size} distinct entry(ies)`
|
|
128
|
+
);
|
|
129
|
+
|
|
130
|
+
const sorted = [...clusters.values()].sort(
|
|
131
|
+
(a, b) => b.patterns.length - a.patterns.length || a.entry.localeCompare(b.entry)
|
|
132
|
+
);
|
|
133
|
+
for (const cluster of sorted) {
|
|
134
|
+
const ids =
|
|
135
|
+
cluster.patterns.length > MAX_PATTERN_IDS_SHOWN
|
|
136
|
+
? `${cluster.patterns.slice(0, MAX_PATTERN_IDS_SHOWN).join(', ')}, +${
|
|
137
|
+
cluster.patterns.length - MAX_PATTERN_IDS_SHOWN
|
|
138
|
+
} more`
|
|
139
|
+
: cluster.patterns.join(', ');
|
|
140
|
+
console.log(` ×${cluster.patterns.length} missing ${cluster.entry}`);
|
|
141
|
+
console.log(` in: ${ids}`);
|
|
142
|
+
if (cluster.haveInstead.size > 0) {
|
|
143
|
+
console.log(` captured instead: ${[...cluster.haveInstead].sort().join(', ')}`);
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
console.log();
|
|
149
|
+
}
|