@hyperfixi/testing-framework 2.7.2 → 2.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/dist/assertions.d.mts +26 -1
  2. package/dist/assertions.d.ts +26 -1
  3. package/dist/index.d.mts +5 -112
  4. package/dist/index.d.ts +5 -112
  5. package/dist/runner.d.mts +112 -0
  6. package/dist/runner.d.ts +112 -0
  7. package/dist/runner.js +1102 -0
  8. package/dist/runner.js.map +1 -0
  9. package/dist/runner.mjs +1097 -0
  10. package/dist/runner.mjs.map +1 -0
  11. package/dist/{assertions-CsGP61iW.d.mts → types-D-rCVkf3.d.mts} +1 -24
  12. package/dist/{assertions-CsGP61iW.d.ts → types-D-rCVkf3.d.ts} +1 -24
  13. package/package.json +11 -26
  14. package/src/multilingual/canonical-validity.test.ts +69 -0
  15. package/src/multilingual/canonical-validity.ts +132 -0
  16. package/src/multilingual/cli.ts +185 -1
  17. package/src/multilingual/fidelity.test.ts +192 -0
  18. package/src/multilingual/fidelity.ts +153 -0
  19. package/src/multilingual/foreign-canonical-validity.test.ts +0 -0
  20. package/src/multilingual/foreign-canonical-validity.ts +158 -0
  21. package/src/multilingual/orchestrator.ts +47 -1
  22. package/src/multilingual/reporters/console-reporter.ts +51 -0
  23. package/src/multilingual/reporters/regression-reporter.ts +23 -0
  24. package/src/multilingual/tools/diagnose-coverage.ts +118 -0
  25. package/src/multilingual/tools/triage-r1.ts +149 -0
  26. package/src/multilingual/types.ts +74 -0
  27. package/src/multilingual/validators/parse-validator.ts +9 -1
  28. package/src/runner.test.ts +7 -2
  29. package/src/vocab/batch3-roundtrip.test.ts +184 -0
  30. package/src/vocab/checks.test.ts +362 -0
  31. package/src/vocab/checks.ts +311 -0
  32. package/src/vocab/cli.ts +196 -0
  33. package/src/vocab/dump.ts +80 -0
  34. package/src/vocab/model.ts +110 -0
  35. package/src/vocab/report.ts +120 -0
  36. package/src/vocab/types.ts +100 -0
@@ -0,0 +1,158 @@
1
+ /**
2
+ * Foreign→English canonical-validity gate
3
+ * ---------------------------------------
4
+ * The sibling en-render gate (canonical-validity.ts) renders every corpus English
5
+ * reference and parses it on the real `hyperscript.org` engine. But the PRODUCTION
6
+ * path is foreign→English: an authored non-English source is parsed and rendered to
7
+ * English (`preprocessToEnglish` → `render`). A parse can be role-faithful (the
8
+ * fidelity ratchet scores ~1.0) yet still render English the canonical parser rejects.
9
+ * (The build-time `@hyperscript-tools/i18n` transpiler now parse-gates its ENGLISH
10
+ * side with the same loader recipe; faithful foreign-output gating there still awaits
11
+ * the v2 semantic-engine transpiler, since its GrammarTransformer is lossy in reverse.)
12
+ *
13
+ * This gate closes that blind spot for the multilingual path: for every language, it
14
+ * renders each authored `pattern_translation` to English and parses the result on
15
+ * the canonical engine, failing on any invalid (pattern, language) pair that is not
16
+ * in the committed allowlist. The allowlist is keyed by pattern id → the languages
17
+ * that currently fail, so a fix that clears a family across languages shrinks (or
18
+ * removes) its entry — the list only ever ratchets down.
19
+ *
20
+ * Denominator: only (pattern, language) pairs whose EN reference the canonical parser
21
+ * already accepts are scored, so a handful of inherently non-canonical corpus rows
22
+ * never distort the signal (same fairness rule as the en gate).
23
+ *
24
+ * DB dependency: reads authored translations from `pattern_translations`, which only
25
+ * exist after `npm run populate`. Generate the baseline and run the gate against a
26
+ * freshly populated DB (CI's multilingual-validation job populates; see the
27
+ * provenance-stamp discipline in packages/patterns-reference/CLAUDE.md).
28
+ */
29
+
30
+ import { getAllPatterns, getTranslationsByLanguage } from '@hyperfixi/patterns-reference';
31
+ import { parseSemantic, render } from '@lokascript/semantic';
32
+ import { loadCanonicalParser, type CanonicalValidate } from './canonical-validity';
33
+
34
+ /**
35
+ * The 23 non-English priority languages (the `browser-priority` corpus set). English
36
+ * is the reference, scored by the sibling en gate.
37
+ */
38
+ export const FOREIGN_LANGUAGES = [
39
+ 'es',
40
+ 'fr',
41
+ 'pt',
42
+ 'it',
43
+ 'id',
44
+ 'ms',
45
+ 'sw',
46
+ 'zh',
47
+ 'vi',
48
+ 'tl',
49
+ 'ja',
50
+ 'ko',
51
+ 'tr',
52
+ 'qu',
53
+ 'hi',
54
+ 'bn',
55
+ 'ar',
56
+ 'de',
57
+ 'ru',
58
+ 'uk',
59
+ 'pl',
60
+ 'th',
61
+ 'he',
62
+ ] as const;
63
+
64
+ export interface ForeignValidityFailure {
65
+ /** Corpus pattern id (`code_example` id the translation belongs to). */
66
+ id: string;
67
+ language: string;
68
+ /** The authored foreign source that was rendered. */
69
+ foreign: string;
70
+ /** The English the renderer produced from the foreign parse. */
71
+ rendered: string;
72
+ error: string;
73
+ }
74
+
75
+ export interface ForeignValidityResult {
76
+ /** (pattern, language) pairs whose EN reference the canonical parser accepts. */
77
+ checked: number;
78
+ valid: number;
79
+ failures: ForeignValidityFailure[];
80
+ }
81
+
82
+ /**
83
+ * Render every authored foreign `pattern_translation` to English and parse it on the
84
+ * canonical engine. Only pairs whose EN reference is itself canonical-valid are scored.
85
+ */
86
+ export async function checkForeignRenderValidity(opts?: {
87
+ validate?: CanonicalValidate;
88
+ languages?: readonly string[];
89
+ /** Translations fetched per language (defaults comfortably above the corpus size). */
90
+ perLanguageLimit?: number;
91
+ }): Promise<ForeignValidityResult> {
92
+ const validate = opts?.validate ?? (await loadCanonicalParser());
93
+ const languages = opts?.languages ?? FOREIGN_LANGUAGES;
94
+ const limit = opts?.perLanguageLimit ?? 500;
95
+
96
+ const patterns = await getAllPatterns();
97
+ const enAccepts = new Map(patterns.map(p => [p.id, validate(p.rawCode).length === 0]));
98
+
99
+ const failures: ForeignValidityFailure[] = [];
100
+ let checked = 0;
101
+ let valid = 0;
102
+
103
+ for (const language of languages) {
104
+ const translations = await getTranslationsByLanguage(language, limit);
105
+ for (const t of translations) {
106
+ if (!enAccepts.get(t.codeExampleId)) continue; // fair denominator: EN accepts the reference
107
+ checked++;
108
+
109
+ let rendered: string;
110
+ let errors: string[];
111
+ try {
112
+ const node = parseSemantic(t.hyperscript, language).node;
113
+ rendered = node ? render(node, 'en') : '(no node)';
114
+ errors = validate(rendered);
115
+ } catch (e) {
116
+ // parseSemantic/render only — validate never throws (its tokenizer-level
117
+ // throws fold into the returned array), so an invalid render is reported
118
+ // WITH the render that caused it, not discarded as '(threw)'.
119
+ rendered = '(threw)';
120
+ errors = ['threw: ' + (e as Error).message.split('\n')[0]];
121
+ }
122
+
123
+ if (errors.length === 0) {
124
+ valid++;
125
+ } else {
126
+ failures.push({
127
+ id: t.codeExampleId,
128
+ language,
129
+ foreign: t.hyperscript,
130
+ rendered,
131
+ error: errors[0] ?? 'unknown error',
132
+ });
133
+ }
134
+ }
135
+ }
136
+
137
+ return { checked, valid, failures };
138
+ }
139
+
140
+ /**
141
+ * Group failures into the committed allowlist shape: `{ patternId: [langs…] }`
142
+ * (languages sorted for a stable diff). Used by the baseline generator and by the
143
+ * gate's stale-entry check.
144
+ */
145
+ export function groupFailuresByPattern(
146
+ failures: readonly ForeignValidityFailure[]
147
+ ): Record<string, string[]> {
148
+ const byPattern = new Map<string, Set<string>>();
149
+ for (const f of failures) {
150
+ if (!byPattern.has(f.id)) byPattern.set(f.id, new Set());
151
+ byPattern.get(f.id)!.add(f.language);
152
+ }
153
+ const out: Record<string, string[]> = {};
154
+ for (const id of [...byPattern.keys()].sort()) {
155
+ out[id] = [...byPattern.get(id)!].sort();
156
+ }
157
+ return out;
158
+ }
@@ -20,7 +20,13 @@ import type {
20
20
  Reporter,
21
21
  BundleInfo,
22
22
  } from './types';
23
- import { computeFidelity, computePrecision, FIDELITY_THRESHOLD } from './fidelity';
23
+ import {
24
+ computeFidelity,
25
+ computePrecision,
26
+ computeMultisetRecall,
27
+ spuriousActions,
28
+ FIDELITY_THRESHOLD,
29
+ } from './fidelity';
24
30
 
25
31
  const execAsync = promisify(exec);
26
32
 
@@ -258,6 +264,8 @@ export class TestOrchestrator {
258
264
  const multisetReference = new Map<string, string[]>();
259
265
  // R1: codeExampleId -> English role signature (action.role:valueType set).
260
266
  const roleReference = new Map<string, string[]>();
267
+ // R3: codeExampleId -> English role-VALUE signature (invariant values, multiset).
268
+ const roleValueReference = new Map<string, string[]>();
261
269
  for (const r of en.parseResults) {
262
270
  if (r.success && r.actionSignature && r.actionSignature.length > 0) {
263
271
  reference.set(r.pattern.codeExampleId, r.actionSignature);
@@ -268,6 +276,9 @@ export class TestOrchestrator {
268
276
  if (r.success && r.roleSignature && r.roleSignature.length > 0) {
269
277
  roleReference.set(r.pattern.codeExampleId, r.roleSignature);
270
278
  }
279
+ if (r.success && r.roleValueSignature && r.roleValueSignature.length > 0) {
280
+ roleValueReference.set(r.pattern.codeExampleId, r.roleValueSignature);
281
+ }
271
282
  }
272
283
 
273
284
  for (const lang of languageResults) {
@@ -277,7 +288,9 @@ export class TestOrchestrator {
277
288
  const lossy: string[] = [];
278
289
  const scores: number[] = [];
279
290
  const precisionScores: number[] = [];
291
+ const multisetRecallScores: number[] = [];
280
292
  const roleScores: number[] = [];
293
+ const valueRecallScores: number[] = [];
281
294
 
282
295
  for (const result of lang.parseResults) {
283
296
  if (!result.success || !result.actionSignature) continue;
@@ -294,6 +307,9 @@ export class TestOrchestrator {
294
307
 
295
308
  // R0-precision — fraction of THIS parse's actions justified by the en
296
309
  // multiset reference (catches phantom/spurious commands recall misses).
310
+ // R0-recall-multiset — the mirror: fraction of the en reference's actions,
311
+ // counting duplicates, present here (catches a DROPPED repeated command,
312
+ // which the Set-based fidelity/roleFidelity above cannot see).
297
313
  const multisetRef = multisetReference.get(result.pattern.codeExampleId);
298
314
  if (multisetRef && result.actionMultisetSignature) {
299
315
  const precision = computePrecision(multisetRef, result.actionMultisetSignature);
@@ -301,6 +317,11 @@ export class TestOrchestrator {
301
317
  result.precision = precision;
302
318
  precisionScores.push(precision);
303
319
  }
320
+ const multisetRecall = computeMultisetRecall(multisetRef, result.actionMultisetSignature);
321
+ if (multisetRecall !== undefined) {
322
+ result.multisetRecall = multisetRecall;
323
+ multisetRecallScores.push(multisetRecall);
324
+ }
304
325
  }
305
326
 
306
327
  // R1 — role recall vs the en role signature (role name + value type).
@@ -312,6 +333,23 @@ export class TestOrchestrator {
312
333
  roleScores.push(roleFidelity);
313
334
  }
314
335
  }
336
+
337
+ // R3 — invariant role-VALUE recall vs the en value signature (multiset).
338
+ // Values are compared verbatim: the filtered subset (selectors, sigil
339
+ // refs, time literals, colon-qualified events, URLs) is code, not prose.
340
+ const valueRef = roleValueReference.get(result.pattern.codeExampleId);
341
+ if (valueRef && result.roleValueSignature) {
342
+ const valueRecall = computeMultisetRecall(valueRef, result.roleValueSignature);
343
+ if (valueRecall !== undefined) {
344
+ result.valueRecall = valueRecall;
345
+ valueRecallScores.push(valueRecall);
346
+ if (valueRecall < 1) {
347
+ // Missing = reference entries not covered here (multiset diff;
348
+ // spuriousActions with the arguments swapped).
349
+ result.valueRecallMissing = spuriousActions(result.roleValueSignature, valueRef);
350
+ }
351
+ }
352
+ }
315
353
  }
316
354
 
317
355
  lang.avgFidelity =
@@ -320,10 +358,18 @@ export class TestOrchestrator {
320
358
  precisionScores.length > 0
321
359
  ? precisionScores.reduce((a, b) => a + b, 0) / precisionScores.length
322
360
  : undefined;
361
+ lang.avgMultisetRecall =
362
+ multisetRecallScores.length > 0
363
+ ? multisetRecallScores.reduce((a, b) => a + b, 0) / multisetRecallScores.length
364
+ : undefined;
323
365
  lang.avgRoleFidelity =
324
366
  roleScores.length > 0
325
367
  ? roleScores.reduce((a, b) => a + b, 0) / roleScores.length
326
368
  : undefined;
369
+ lang.avgValueRecall =
370
+ valueRecallScores.length > 0
371
+ ? valueRecallScores.reduce((a, b) => a + b, 0) / valueRecallScores.length
372
+ : undefined;
327
373
  lang.degeneratePasses = degenerate.sort();
328
374
  lang.lossyPasses = lossy.sort();
329
375
  }
@@ -158,11 +158,62 @@ export class ConsoleReporter implements Reporter {
158
158
  }
159
159
 
160
160
  this.reportDegeneratePasses(results);
161
+ this.reportValueRecall(results);
161
162
  this.reportExecutionFailures(results);
162
163
 
163
164
  this.log('');
164
165
  }
165
166
 
167
+ /**
168
+ * R3 — surface parses that lost/corrupted a language-invariant role VALUE
169
+ * (selector, sigil ref, time literal, colon-qualified event name, URL) vs
170
+ * the en reference. These score 1.0 on every action/type-based signal yet
171
+ * behave differently at runtime, so they're reported separately. Report-only:
172
+ * the ratchet gate lives in the --regression path.
173
+ */
174
+ private reportValueRecall(results: TestResults): void {
175
+ const scored = results.languageResults.filter(l => l.avgValueRecall !== undefined);
176
+ if (scored.length === 0) return;
177
+
178
+ const avg = scored.reduce((sum, l) => sum + (l.avgValueRecall ?? 0), 0) / scored.length;
179
+ this.log('');
180
+ this.log(
181
+ this.bright(`Role values (R3): avgValueRecall ${avg.toFixed(4)} over invariant values`)
182
+ );
183
+
184
+ // pattern id -> "lang: missing entries" rows
185
+ const byPattern = new Map<string, string[]>();
186
+ let instances = 0;
187
+ for (const lang of scored) {
188
+ for (const r of lang.parseResults) {
189
+ if (r.valueRecall === undefined || r.valueRecall >= 1) continue;
190
+ instances++;
191
+ let rows = byPattern.get(r.pattern.codeExampleId);
192
+ if (!rows) {
193
+ rows = [];
194
+ byPattern.set(r.pattern.codeExampleId, rows);
195
+ }
196
+ rows.push(`${lang.language}: missing ${(r.valueRecallMissing ?? []).join(', ')}`);
197
+ }
198
+ }
199
+ if (byPattern.size === 0) return;
200
+
201
+ this.log(
202
+ this.yellow(
203
+ `⚠ Invariant value loss: ${instances} instance(s) across ${byPattern.size} pattern(s)`
204
+ )
205
+ );
206
+ this.log(
207
+ this.dim(
208
+ ' (an action.role=value present in the en reference is absent — not parse failures)'
209
+ )
210
+ );
211
+ for (const [id, rows] of [...byPattern.entries()].sort((a, b) => b[1].length - a[1].length)) {
212
+ this.log(` ${this.dim('-')} ${id}`);
213
+ for (const row of rows.sort()) this.log(` ${this.dim(row)}`);
214
+ }
215
+ }
216
+
166
217
  /**
167
218
  * R2 — surface curated-subset patterns whose jsdom execution diverged from
168
219
  * the en reference's DOM effects. These can score 1.0 on parse fidelity yet
@@ -118,12 +118,25 @@ export class RegressionReporter implements Reporter {
118
118
  langResult.avgPrecision !== undefined && baselineLang.avgPrecision !== undefined
119
119
  ? langResult.avgPrecision - baselineLang.avgPrecision
120
120
  : 0;
121
+ // R0-recall-multiset — same both-sides guard. Negative = a repeated command
122
+ // is now being dropped, which avgFidelity (a Set) cannot see.
123
+ const avgMultisetRecallDelta =
124
+ langResult.avgMultisetRecall !== undefined && baselineLang.avgMultisetRecall !== undefined
125
+ ? langResult.avgMultisetRecall - baselineLang.avgMultisetRecall
126
+ : 0;
121
127
  // R1 — only meaningful when BOTH sides carry role data; an un-regenerated
122
128
  // baseline (no avgRoleFidelity yet) must never retro-flag.
123
129
  const avgRoleFidelityDelta =
124
130
  langResult.avgRoleFidelity !== undefined && baselineLang.avgRoleFidelity !== undefined
125
131
  ? langResult.avgRoleFidelity - baselineLang.avgRoleFidelity
126
132
  : 0;
133
+ // R3 — same both-sides guard. Negative = a language-invariant role VALUE
134
+ // (selector/sigil/time/colon-event/URL) is now lost or corrupted, which
135
+ // every action/type-based signal scores as perfect.
136
+ const avgValueRecallDelta =
137
+ langResult.avgValueRecall !== undefined && baselineLang.avgValueRecall !== undefined
138
+ ? langResult.avgValueRecall - baselineLang.avgValueRecall
139
+ : 0;
127
140
  // R2 — same both-sides guard: an un-regenerated baseline (no execution
128
141
  // data yet) must never retro-flag.
129
142
  const avgExecutionFidelityDelta =
@@ -163,7 +176,9 @@ export class RegressionReporter implements Reporter {
163
176
  avgConfidenceDelta,
164
177
  avgFidelityDelta,
165
178
  avgPrecisionDelta,
179
+ avgMultisetRecallDelta,
166
180
  avgRoleFidelityDelta,
181
+ avgValueRecallDelta,
167
182
  avgExecutionFidelityDelta,
168
183
  bundleSizeDelta: bundleSizeDelta !== undefined ? bundleSizeDelta : undefined,
169
184
  newFailures,
@@ -364,9 +379,17 @@ export class RegressionReporter implements Reporter {
364
379
  // R0-precision — fraction of each parse's actions justified by the en
365
380
  // reference (phantom-command signal recall can't see). Recorded + ratcheted.
366
381
  avgPrecision: langResult.avgPrecision ?? undefined,
382
+ // R0-recall-multiset — the mirror of precision: a DROPPED repeated command
383
+ // (`[bind, bind]` parsed as `[bind]`) is invisible to the Set-based
384
+ // avgFidelity and avgRoleFidelity. Recorded + ratcheted.
385
+ avgMultisetRecall: langResult.avgMultisetRecall ?? undefined,
367
386
  // R1 — role fidelity (role name + value type vs the en reference).
368
387
  // Recorded + ratcheted; burn-down is NOT part of the parsing-track goal.
369
388
  avgRoleFidelity: langResult.avgRoleFidelity ?? undefined,
389
+ // R3 — invariant role-VALUE recall (verbatim value comparison over the
390
+ // code-shaped subset: selectors, sigil refs, time literals,
391
+ // colon-qualified event names, URLs). Recorded + ratcheted.
392
+ avgValueRecall: langResult.avgValueRecall ?? undefined,
370
393
  // R2 — execution fidelity over the curated subset (DOM effects vs the
371
394
  // en reference in jsdom). Recorded + ratcheted; burn-down deferred.
372
395
  avgExecutionFidelity: langResult.avgExecutionFidelity ?? undefined,
@@ -0,0 +1,118 @@
1
+ /**
2
+ * diagnose-coverage.ts
3
+ *
4
+ * Read-only sweep that counts how often the semantic parser matches a pattern
5
+ * covering only PART of its input — the `unconsumed-input` diagnostic.
6
+ *
7
+ * Why this exists: confidence is computed from role coverage (how many of the
8
+ * pattern's own roles were filled), never from input coverage. A short pattern
9
+ * can fill every role, score 1.0, and silently drop a trailing clause. Turning
10
+ * that into a scoring penalty would move the multilingual fidelity baseline
11
+ * across all 24 languages, so the firing rate has to be MEASURED first — on the
12
+ * real corpus, per language, with the dropped spans visible.
13
+ *
14
+ * This tool never writes a baseline and never gates. It only reads.
15
+ */
16
+ import { parseSemantic } from '@lokascript/semantic';
17
+
18
+ import { loadPatterns } from '../pattern-loader';
19
+ import type { TestConfig } from '../types';
20
+
21
+ interface LanguageTally {
22
+ language: string;
23
+ total: number;
24
+ fired: number;
25
+ /** A few representative dropped spans, for eyeballing false positives. */
26
+ samples: Array<{ source: string; message: string; confidence: number }>;
27
+ }
28
+
29
+ const MAX_SAMPLES_PER_LANGUAGE = 3;
30
+ const UNCONSUMED_CODE = 'unconsumed-input';
31
+
32
+ function unconsumedMessages(node: unknown): string[] {
33
+ const diagnostics = (node as { diagnostics?: Array<{ code?: string; message: string }> } | null)
34
+ ?.diagnostics;
35
+ if (!diagnostics?.length) return [];
36
+ return diagnostics.filter(d => d.code === UNCONSUMED_CODE).map(d => d.message);
37
+ }
38
+
39
+ /**
40
+ * Sweep the corpus and report the `unconsumed-input` firing rate per language.
41
+ *
42
+ * @returns the total number of firings (0 when the signal never trips)
43
+ */
44
+ export async function diagnoseCoverage(config: TestConfig): Promise<number> {
45
+ const patterns = await loadPatterns(config);
46
+ const byLanguage = new Map<string, LanguageTally>();
47
+
48
+ for (const pattern of patterns) {
49
+ const tally = byLanguage.get(pattern.language) ?? {
50
+ language: pattern.language,
51
+ total: 0,
52
+ fired: 0,
53
+ samples: [],
54
+ };
55
+ tally.total++;
56
+
57
+ let messages: string[] = [];
58
+ let confidence = 0;
59
+ try {
60
+ const result = parseSemantic(pattern.hyperscript, pattern.language);
61
+ confidence = result.confidence;
62
+ messages = unconsumedMessages(result.node);
63
+ } catch {
64
+ // A parse failure is not an input-coverage problem — it is already visible
65
+ // to the parse-rate gate. Skip it rather than double-counting.
66
+ }
67
+
68
+ const firstMessage = messages[0];
69
+ if (firstMessage !== undefined) {
70
+ tally.fired++;
71
+ if (tally.samples.length < MAX_SAMPLES_PER_LANGUAGE) {
72
+ tally.samples.push({ source: pattern.hyperscript, message: firstMessage, confidence });
73
+ }
74
+ }
75
+ byLanguage.set(pattern.language, tally);
76
+ }
77
+
78
+ report([...byLanguage.values()].sort((a, b) => b.fired / b.total - a.fired / a.total));
79
+ return [...byLanguage.values()].reduce((sum, t) => sum + t.fired, 0);
80
+ }
81
+
82
+ function report(tallies: LanguageTally[]): void {
83
+ const total = tallies.reduce((s, t) => s + t.total, 0);
84
+ const fired = tallies.reduce((s, t) => s + t.fired, 0);
85
+
86
+ console.log('\nInput-coverage diagnostic — `unconsumed-input` firing rate');
87
+ console.log('(a fired row parsed successfully but ignored part of its source)\n');
88
+ console.log(' lang fired / total rate');
89
+ console.log(' ─────────────────────────────────');
90
+ for (const t of tallies) {
91
+ const rate = t.total ? (t.fired / t.total) * 100 : 0;
92
+ console.log(
93
+ ` ${t.language.padEnd(6)} ${String(t.fired).padStart(5)} / ${String(t.total).padStart(6)} ${rate.toFixed(1).padStart(6)}%`
94
+ );
95
+ }
96
+ console.log(' ─────────────────────────────────');
97
+ const overall = total ? (fired / total) * 100 : 0;
98
+ console.log(
99
+ ` ALL ${String(fired).padStart(5)} / ${String(total).padStart(6)} ${overall.toFixed(1).padStart(6)}%\n`
100
+ );
101
+
102
+ const withSamples = tallies.filter(t => t.samples.length > 0);
103
+ if (withSamples.length === 0) {
104
+ console.log('No unconsumed input anywhere in the corpus.\n');
105
+ return;
106
+ }
107
+
108
+ console.log('Sample dropped spans (check these for false positives — trailing');
109
+ console.log('particles and legitimately-optional tokens are expected residue):\n');
110
+ for (const t of withSamples) {
111
+ console.log(` [${t.language}]`);
112
+ for (const s of t.samples) {
113
+ console.log(` source (conf ${s.confidence.toFixed(2)}): ${s.source}`);
114
+ console.log(` ${s.message}`);
115
+ }
116
+ console.log('');
117
+ }
118
+ }
@@ -0,0 +1,149 @@
1
+ /**
2
+ * triage-r1.ts
3
+ *
4
+ * Read-only sweep that itemizes R1 role-fidelity misses per language: for every
5
+ * corpus pattern whose roleFidelity < 1, the `action.role:type` entries present
6
+ * in the en reference parse but absent from the translation's parse — clustered
7
+ * by entry so the dominant failure families are visible, with the entries the
8
+ * translation captured INSTEAD for the same action (the mistype pairing).
9
+ *
10
+ * Why this exists: avgRoleFidelity summarizes the SOV-six gap (qu ~0.95, the
11
+ * rest ~0.97) to one number per language, and the regression gate only reports
12
+ * DROPS. Planning an improvement arc needs the misses themselves — which roles,
13
+ * on which actions, in which patterns — measured with exactly the pipeline the
14
+ * gate scores (ParseValidator → fillSchemaDefaults → collectRoleSignature →
15
+ * Set recall vs the en reference), so a "fix" that moves the probe also moves
16
+ * the signal.
17
+ *
18
+ * This tool never writes a baseline and never gates. It only reads.
19
+ */
20
+ import { computeFidelity } from '../fidelity';
21
+ import { loadPatterns } from '../pattern-loader';
22
+ import { ParseValidator } from '../validators/parse-validator';
23
+ import type { ParseResult, TestConfig } from '../types';
24
+
25
+ interface MissCluster {
26
+ /** The en-reference `action.role:type` entry the language failed to recall. */
27
+ entry: string;
28
+ /** Pattern ids where this entry was missed. */
29
+ patterns: string[];
30
+ /**
31
+ * What the language's parse captured for the SAME action in those patterns
32
+ * (entries absent from the en reference) — usually the mistype the miss
33
+ * paired with (`trigger.patient:literal` opposite a missing
34
+ * `trigger.event:literal`). Empty when the whole command dropped.
35
+ */
36
+ haveInstead: Set<string>;
37
+ }
38
+
39
+ const MAX_PATTERN_IDS_SHOWN = 6;
40
+
41
+ /**
42
+ * Sweep the corpus and itemize R1 misses for the requested languages
43
+ * (default: every non-en language in the run).
44
+ */
45
+ export async function triageR1(config: TestConfig): Promise<void> {
46
+ // The en rows ARE the reference — force-load them even when the caller
47
+ // filtered --languages to the triage targets.
48
+ const requested = config.languages?.filter(l => l !== 'en');
49
+ const loadConfig: TestConfig =
50
+ requested && requested.length > 0
51
+ ? { ...config, languages: [...requested, 'en' as const] }
52
+ : config;
53
+
54
+ const patterns = await loadPatterns(loadConfig);
55
+ const validator = new ParseValidator();
56
+ const results = await validator.validate(patterns);
57
+
58
+ const reference = new Map<string, string[]>();
59
+ for (const r of results) {
60
+ if (r.pattern.language !== 'en') continue;
61
+ if (r.success && r.roleSignature && r.roleSignature.length > 0) {
62
+ reference.set(r.pattern.codeExampleId, r.roleSignature);
63
+ }
64
+ }
65
+
66
+ const byLanguage = new Map<string, ParseResult[]>();
67
+ for (const r of results) {
68
+ if (r.pattern.language === 'en') continue;
69
+ if (requested && requested.length > 0 && !requested.includes(r.pattern.language)) continue;
70
+ const list = byLanguage.get(r.pattern.language) ?? [];
71
+ list.push(r);
72
+ byLanguage.set(r.pattern.language, list);
73
+ }
74
+
75
+ console.log('\nR1 role-fidelity triage (action.role:type recall vs the en reference)');
76
+ console.log('━'.repeat(72));
77
+
78
+ for (const [language, langResults] of [...byLanguage.entries()].sort()) {
79
+ const clusters = new Map<string, MissCluster>();
80
+ const scores: number[] = [];
81
+ let patternsWithMisses = 0;
82
+
83
+ for (const result of langResults) {
84
+ if (!result.success || !result.roleSignature) continue;
85
+ const ref = reference.get(result.pattern.codeExampleId);
86
+ if (!ref) continue;
87
+
88
+ const fidelity = computeFidelity(ref, result.roleSignature);
89
+ if (fidelity === undefined) continue;
90
+ scores.push(fidelity);
91
+ if (fidelity >= 1) continue;
92
+
93
+ patternsWithMisses++;
94
+ const candidate = new Set(result.roleSignature);
95
+ const refSet = new Set(ref);
96
+ const missing = ref.filter(e => !candidate.has(e));
97
+ // Entries this parse has that the reference doesn't — candidates for the
98
+ // "captured instead" pairing, keyed by action.
99
+ const extrasByAction = new Map<string, string[]>();
100
+ for (const e of result.roleSignature) {
101
+ if (refSet.has(e)) continue;
102
+ const action = e.slice(0, e.indexOf('.'));
103
+ const list = extrasByAction.get(action) ?? [];
104
+ list.push(e);
105
+ extrasByAction.set(action, list);
106
+ }
107
+
108
+ for (const entry of missing) {
109
+ const cluster = clusters.get(entry) ?? {
110
+ entry,
111
+ patterns: [],
112
+ haveInstead: new Set<string>(),
113
+ };
114
+ cluster.patterns.push(result.pattern.codeExampleId);
115
+ const action = entry.slice(0, entry.indexOf('.'));
116
+ for (const e of extrasByAction.get(action) ?? []) cluster.haveInstead.add(e);
117
+ clusters.set(entry, cluster);
118
+ }
119
+ }
120
+
121
+ const avg = scores.length > 0 ? scores.reduce((a, b) => a + b, 0) / scores.length : undefined;
122
+ const totalMisses = [...clusters.values()].reduce((a, c) => a + c.patterns.length, 0);
123
+
124
+ console.log(
125
+ `\n[${language}] avgRoleFidelity ${avg?.toFixed(4) ?? 'n/a'} — ` +
126
+ `${totalMisses} miss(es) across ${patternsWithMisses} pattern(s), ` +
127
+ `${clusters.size} distinct entry(ies)`
128
+ );
129
+
130
+ const sorted = [...clusters.values()].sort(
131
+ (a, b) => b.patterns.length - a.patterns.length || a.entry.localeCompare(b.entry)
132
+ );
133
+ for (const cluster of sorted) {
134
+ const ids =
135
+ cluster.patterns.length > MAX_PATTERN_IDS_SHOWN
136
+ ? `${cluster.patterns.slice(0, MAX_PATTERN_IDS_SHOWN).join(', ')}, +${
137
+ cluster.patterns.length - MAX_PATTERN_IDS_SHOWN
138
+ } more`
139
+ : cluster.patterns.join(', ');
140
+ console.log(` ×${cluster.patterns.length} missing ${cluster.entry}`);
141
+ console.log(` in: ${ids}`);
142
+ if (cluster.haveInstead.size > 0) {
143
+ console.log(` captured instead: ${[...cluster.haveInstead].sort().join(', ')}`);
144
+ }
145
+ }
146
+ }
147
+
148
+ console.log();
149
+ }