@hyperfixi/testing-framework 2.7.2 → 2.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +393 -0
- package/dist/assertions.d.mts +26 -1
- package/dist/assertions.d.ts +26 -1
- package/dist/index.d.mts +5 -112
- package/dist/index.d.ts +5 -112
- package/dist/runner.d.mts +112 -0
- package/dist/runner.d.ts +112 -0
- package/dist/runner.js +1102 -0
- package/dist/runner.js.map +1 -0
- package/dist/runner.mjs +1097 -0
- package/dist/runner.mjs.map +1 -0
- package/dist/{assertions-CsGP61iW.d.mts → types-D-rCVkf3.d.mts} +1 -24
- package/dist/{assertions-CsGP61iW.d.ts → types-D-rCVkf3.d.ts} +1 -24
- package/package.json +13 -27
- package/src/multilingual/canonical-validity.test.ts +69 -0
- package/src/multilingual/canonical-validity.ts +132 -0
- package/src/multilingual/cli.ts +247 -14
- package/src/multilingual/fidelity.test.ts +192 -0
- package/src/multilingual/fidelity.ts +153 -0
- package/src/multilingual/foreign-canonical-validity.test.ts +0 -0
- package/src/multilingual/foreign-canonical-validity.ts +158 -0
- package/src/multilingual/orchestrator.ts +47 -1
- package/src/multilingual/reporters/console-reporter.ts +51 -0
- package/src/multilingual/reporters/regression-reporter.test.ts +78 -0
- package/src/multilingual/reporters/regression-reporter.ts +28 -1
- package/src/multilingual/tools/diagnose-coverage.ts +118 -0
- package/src/multilingual/tools/triage-r1.ts +149 -0
- package/src/multilingual/types.ts +74 -0
- package/src/multilingual/validators/parse-validator.ts +9 -1
- package/src/runner.test.ts +7 -2
- package/src/vocab/batch3-roundtrip.test.ts +184 -0
- package/src/vocab/checks.test.ts +362 -0
- package/src/vocab/checks.ts +311 -0
- package/src/vocab/cli.ts +196 -0
- package/src/vocab/dump.ts +80 -0
- package/src/vocab/model.ts +110 -0
- package/src/vocab/report.ts +120 -0
- package/src/vocab/types.ts +100 -0
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* diagnose-coverage.ts
|
|
3
|
+
*
|
|
4
|
+
* Read-only sweep that counts how often the semantic parser matches a pattern
|
|
5
|
+
* covering only PART of its input — the `unconsumed-input` diagnostic.
|
|
6
|
+
*
|
|
7
|
+
* Why this exists: confidence is computed from role coverage (how many of the
|
|
8
|
+
* pattern's own roles were filled), never from input coverage. A short pattern
|
|
9
|
+
* can fill every role, score 1.0, and silently drop a trailing clause. Turning
|
|
10
|
+
* that into a scoring penalty would move the multilingual fidelity baseline
|
|
11
|
+
* across all 24 languages, so the firing rate has to be MEASURED first — on the
|
|
12
|
+
* real corpus, per language, with the dropped spans visible.
|
|
13
|
+
*
|
|
14
|
+
* This tool never writes a baseline and never gates. It only reads.
|
|
15
|
+
*/
|
|
16
|
+
import { parseSemantic } from '@lokascript/semantic';
|
|
17
|
+
|
|
18
|
+
import { loadPatterns } from '../pattern-loader';
|
|
19
|
+
import type { TestConfig } from '../types';
|
|
20
|
+
|
|
21
|
+
interface LanguageTally {
|
|
22
|
+
language: string;
|
|
23
|
+
total: number;
|
|
24
|
+
fired: number;
|
|
25
|
+
/** A few representative dropped spans, for eyeballing false positives. */
|
|
26
|
+
samples: Array<{ source: string; message: string; confidence: number }>;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
const MAX_SAMPLES_PER_LANGUAGE = 3;
|
|
30
|
+
const UNCONSUMED_CODE = 'unconsumed-input';
|
|
31
|
+
|
|
32
|
+
function unconsumedMessages(node: unknown): string[] {
|
|
33
|
+
const diagnostics = (node as { diagnostics?: Array<{ code?: string; message: string }> } | null)
|
|
34
|
+
?.diagnostics;
|
|
35
|
+
if (!diagnostics?.length) return [];
|
|
36
|
+
return diagnostics.filter(d => d.code === UNCONSUMED_CODE).map(d => d.message);
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* Sweep the corpus and report the `unconsumed-input` firing rate per language.
|
|
41
|
+
*
|
|
42
|
+
* @returns the total number of firings (0 when the signal never trips)
|
|
43
|
+
*/
|
|
44
|
+
export async function diagnoseCoverage(config: TestConfig): Promise<number> {
|
|
45
|
+
const patterns = await loadPatterns(config);
|
|
46
|
+
const byLanguage = new Map<string, LanguageTally>();
|
|
47
|
+
|
|
48
|
+
for (const pattern of patterns) {
|
|
49
|
+
const tally = byLanguage.get(pattern.language) ?? {
|
|
50
|
+
language: pattern.language,
|
|
51
|
+
total: 0,
|
|
52
|
+
fired: 0,
|
|
53
|
+
samples: [],
|
|
54
|
+
};
|
|
55
|
+
tally.total++;
|
|
56
|
+
|
|
57
|
+
let messages: string[] = [];
|
|
58
|
+
let confidence = 0;
|
|
59
|
+
try {
|
|
60
|
+
const result = parseSemantic(pattern.hyperscript, pattern.language);
|
|
61
|
+
confidence = result.confidence;
|
|
62
|
+
messages = unconsumedMessages(result.node);
|
|
63
|
+
} catch {
|
|
64
|
+
// A parse failure is not an input-coverage problem — it is already visible
|
|
65
|
+
// to the parse-rate gate. Skip it rather than double-counting.
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
const firstMessage = messages[0];
|
|
69
|
+
if (firstMessage !== undefined) {
|
|
70
|
+
tally.fired++;
|
|
71
|
+
if (tally.samples.length < MAX_SAMPLES_PER_LANGUAGE) {
|
|
72
|
+
tally.samples.push({ source: pattern.hyperscript, message: firstMessage, confidence });
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
byLanguage.set(pattern.language, tally);
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
report([...byLanguage.values()].sort((a, b) => b.fired / b.total - a.fired / a.total));
|
|
79
|
+
return [...byLanguage.values()].reduce((sum, t) => sum + t.fired, 0);
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
function report(tallies: LanguageTally[]): void {
|
|
83
|
+
const total = tallies.reduce((s, t) => s + t.total, 0);
|
|
84
|
+
const fired = tallies.reduce((s, t) => s + t.fired, 0);
|
|
85
|
+
|
|
86
|
+
console.log('\nInput-coverage diagnostic — `unconsumed-input` firing rate');
|
|
87
|
+
console.log('(a fired row parsed successfully but ignored part of its source)\n');
|
|
88
|
+
console.log(' lang fired / total rate');
|
|
89
|
+
console.log(' ─────────────────────────────────');
|
|
90
|
+
for (const t of tallies) {
|
|
91
|
+
const rate = t.total ? (t.fired / t.total) * 100 : 0;
|
|
92
|
+
console.log(
|
|
93
|
+
` ${t.language.padEnd(6)} ${String(t.fired).padStart(5)} / ${String(t.total).padStart(6)} ${rate.toFixed(1).padStart(6)}%`
|
|
94
|
+
);
|
|
95
|
+
}
|
|
96
|
+
console.log(' ─────────────────────────────────');
|
|
97
|
+
const overall = total ? (fired / total) * 100 : 0;
|
|
98
|
+
console.log(
|
|
99
|
+
` ALL ${String(fired).padStart(5)} / ${String(total).padStart(6)} ${overall.toFixed(1).padStart(6)}%\n`
|
|
100
|
+
);
|
|
101
|
+
|
|
102
|
+
const withSamples = tallies.filter(t => t.samples.length > 0);
|
|
103
|
+
if (withSamples.length === 0) {
|
|
104
|
+
console.log('No unconsumed input anywhere in the corpus.\n');
|
|
105
|
+
return;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
console.log('Sample dropped spans (check these for false positives — trailing');
|
|
109
|
+
console.log('particles and legitimately-optional tokens are expected residue):\n');
|
|
110
|
+
for (const t of withSamples) {
|
|
111
|
+
console.log(` [${t.language}]`);
|
|
112
|
+
for (const s of t.samples) {
|
|
113
|
+
console.log(` source (conf ${s.confidence.toFixed(2)}): ${s.source}`);
|
|
114
|
+
console.log(` ${s.message}`);
|
|
115
|
+
}
|
|
116
|
+
console.log('');
|
|
117
|
+
}
|
|
118
|
+
}
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* triage-r1.ts
|
|
3
|
+
*
|
|
4
|
+
* Read-only sweep that itemizes R1 role-fidelity misses per language: for every
|
|
5
|
+
* corpus pattern whose roleFidelity < 1, the `action.role:type` entries present
|
|
6
|
+
* in the en reference parse but absent from the translation's parse — clustered
|
|
7
|
+
* by entry so the dominant failure families are visible, with the entries the
|
|
8
|
+
* translation captured INSTEAD for the same action (the mistype pairing).
|
|
9
|
+
*
|
|
10
|
+
* Why this exists: avgRoleFidelity summarizes the SOV-six gap (qu ~0.95, the
|
|
11
|
+
* rest ~0.97) to one number per language, and the regression gate only reports
|
|
12
|
+
* DROPS. Planning an improvement arc needs the misses themselves — which roles,
|
|
13
|
+
* on which actions, in which patterns — measured with exactly the pipeline the
|
|
14
|
+
* gate scores (ParseValidator → fillSchemaDefaults → collectRoleSignature →
|
|
15
|
+
* Set recall vs the en reference), so a "fix" that moves the probe also moves
|
|
16
|
+
* the signal.
|
|
17
|
+
*
|
|
18
|
+
* This tool never writes a baseline and never gates. It only reads.
|
|
19
|
+
*/
|
|
20
|
+
import { computeFidelity } from '../fidelity';
|
|
21
|
+
import { loadPatterns } from '../pattern-loader';
|
|
22
|
+
import { ParseValidator } from '../validators/parse-validator';
|
|
23
|
+
import type { ParseResult, TestConfig } from '../types';
|
|
24
|
+
|
|
25
|
+
interface MissCluster {
|
|
26
|
+
/** The en-reference `action.role:type` entry the language failed to recall. */
|
|
27
|
+
entry: string;
|
|
28
|
+
/** Pattern ids where this entry was missed. */
|
|
29
|
+
patterns: string[];
|
|
30
|
+
/**
|
|
31
|
+
* What the language's parse captured for the SAME action in those patterns
|
|
32
|
+
* (entries absent from the en reference) — usually the mistype the miss
|
|
33
|
+
* paired with (`trigger.patient:literal` opposite a missing
|
|
34
|
+
* `trigger.event:literal`). Empty when the whole command dropped.
|
|
35
|
+
*/
|
|
36
|
+
haveInstead: Set<string>;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
const MAX_PATTERN_IDS_SHOWN = 6;
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* Sweep the corpus and itemize R1 misses for the requested languages
|
|
43
|
+
* (default: every non-en language in the run).
|
|
44
|
+
*/
|
|
45
|
+
export async function triageR1(config: TestConfig): Promise<void> {
|
|
46
|
+
// The en rows ARE the reference — force-load them even when the caller
|
|
47
|
+
// filtered --languages to the triage targets.
|
|
48
|
+
const requested = config.languages?.filter(l => l !== 'en');
|
|
49
|
+
const loadConfig: TestConfig =
|
|
50
|
+
requested && requested.length > 0
|
|
51
|
+
? { ...config, languages: [...requested, 'en' as const] }
|
|
52
|
+
: config;
|
|
53
|
+
|
|
54
|
+
const patterns = await loadPatterns(loadConfig);
|
|
55
|
+
const validator = new ParseValidator();
|
|
56
|
+
const results = await validator.validate(patterns);
|
|
57
|
+
|
|
58
|
+
const reference = new Map<string, string[]>();
|
|
59
|
+
for (const r of results) {
|
|
60
|
+
if (r.pattern.language !== 'en') continue;
|
|
61
|
+
if (r.success && r.roleSignature && r.roleSignature.length > 0) {
|
|
62
|
+
reference.set(r.pattern.codeExampleId, r.roleSignature);
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
const byLanguage = new Map<string, ParseResult[]>();
|
|
67
|
+
for (const r of results) {
|
|
68
|
+
if (r.pattern.language === 'en') continue;
|
|
69
|
+
if (requested && requested.length > 0 && !requested.includes(r.pattern.language)) continue;
|
|
70
|
+
const list = byLanguage.get(r.pattern.language) ?? [];
|
|
71
|
+
list.push(r);
|
|
72
|
+
byLanguage.set(r.pattern.language, list);
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
console.log('\nR1 role-fidelity triage (action.role:type recall vs the en reference)');
|
|
76
|
+
console.log('━'.repeat(72));
|
|
77
|
+
|
|
78
|
+
for (const [language, langResults] of [...byLanguage.entries()].sort()) {
|
|
79
|
+
const clusters = new Map<string, MissCluster>();
|
|
80
|
+
const scores: number[] = [];
|
|
81
|
+
let patternsWithMisses = 0;
|
|
82
|
+
|
|
83
|
+
for (const result of langResults) {
|
|
84
|
+
if (!result.success || !result.roleSignature) continue;
|
|
85
|
+
const ref = reference.get(result.pattern.codeExampleId);
|
|
86
|
+
if (!ref) continue;
|
|
87
|
+
|
|
88
|
+
const fidelity = computeFidelity(ref, result.roleSignature);
|
|
89
|
+
if (fidelity === undefined) continue;
|
|
90
|
+
scores.push(fidelity);
|
|
91
|
+
if (fidelity >= 1) continue;
|
|
92
|
+
|
|
93
|
+
patternsWithMisses++;
|
|
94
|
+
const candidate = new Set(result.roleSignature);
|
|
95
|
+
const refSet = new Set(ref);
|
|
96
|
+
const missing = ref.filter(e => !candidate.has(e));
|
|
97
|
+
// Entries this parse has that the reference doesn't — candidates for the
|
|
98
|
+
// "captured instead" pairing, keyed by action.
|
|
99
|
+
const extrasByAction = new Map<string, string[]>();
|
|
100
|
+
for (const e of result.roleSignature) {
|
|
101
|
+
if (refSet.has(e)) continue;
|
|
102
|
+
const action = e.slice(0, e.indexOf('.'));
|
|
103
|
+
const list = extrasByAction.get(action) ?? [];
|
|
104
|
+
list.push(e);
|
|
105
|
+
extrasByAction.set(action, list);
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
for (const entry of missing) {
|
|
109
|
+
const cluster = clusters.get(entry) ?? {
|
|
110
|
+
entry,
|
|
111
|
+
patterns: [],
|
|
112
|
+
haveInstead: new Set<string>(),
|
|
113
|
+
};
|
|
114
|
+
cluster.patterns.push(result.pattern.codeExampleId);
|
|
115
|
+
const action = entry.slice(0, entry.indexOf('.'));
|
|
116
|
+
for (const e of extrasByAction.get(action) ?? []) cluster.haveInstead.add(e);
|
|
117
|
+
clusters.set(entry, cluster);
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
const avg = scores.length > 0 ? scores.reduce((a, b) => a + b, 0) / scores.length : undefined;
|
|
122
|
+
const totalMisses = [...clusters.values()].reduce((a, c) => a + c.patterns.length, 0);
|
|
123
|
+
|
|
124
|
+
console.log(
|
|
125
|
+
`\n[${language}] avgRoleFidelity ${avg?.toFixed(4) ?? 'n/a'} — ` +
|
|
126
|
+
`${totalMisses} miss(es) across ${patternsWithMisses} pattern(s), ` +
|
|
127
|
+
`${clusters.size} distinct entry(ies)`
|
|
128
|
+
);
|
|
129
|
+
|
|
130
|
+
const sorted = [...clusters.values()].sort(
|
|
131
|
+
(a, b) => b.patterns.length - a.patterns.length || a.entry.localeCompare(b.entry)
|
|
132
|
+
);
|
|
133
|
+
for (const cluster of sorted) {
|
|
134
|
+
const ids =
|
|
135
|
+
cluster.patterns.length > MAX_PATTERN_IDS_SHOWN
|
|
136
|
+
? `${cluster.patterns.slice(0, MAX_PATTERN_IDS_SHOWN).join(', ')}, +${
|
|
137
|
+
cluster.patterns.length - MAX_PATTERN_IDS_SHOWN
|
|
138
|
+
} more`
|
|
139
|
+
: cluster.patterns.join(', ');
|
|
140
|
+
console.log(` ×${cluster.patterns.length} missing ${cluster.entry}`);
|
|
141
|
+
console.log(` in: ${ids}`);
|
|
142
|
+
if (cluster.haveInstead.size > 0) {
|
|
143
|
+
console.log(` captured instead: ${[...cluster.haveInstead].sort().join(', ')}`);
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
console.log();
|
|
149
|
+
}
|
|
@@ -100,6 +100,19 @@ export interface TestConfig {
|
|
|
100
100
|
|
|
101
101
|
/** Number of patterns per language in quick mode */
|
|
102
102
|
quickModeLimit?: number;
|
|
103
|
+
|
|
104
|
+
/**
|
|
105
|
+
* Report the semantic parser's `unconsumed-input` firing rate per language,
|
|
106
|
+
* then exit. Read-only: never gates, never writes a baseline.
|
|
107
|
+
*/
|
|
108
|
+
diagnoseCoverage?: boolean;
|
|
109
|
+
|
|
110
|
+
/**
|
|
111
|
+
* Itemize R1 role-fidelity misses (missing `action.role:type` entries vs the
|
|
112
|
+
* en reference, clustered) per language, then exit. Read-only: never gates,
|
|
113
|
+
* never writes a baseline. Scope with `languages`.
|
|
114
|
+
*/
|
|
115
|
+
triageR1?: boolean;
|
|
103
116
|
}
|
|
104
117
|
|
|
105
118
|
/**
|
|
@@ -138,6 +151,17 @@ export interface ParseResult {
|
|
|
138
151
|
*/
|
|
139
152
|
roleSignature?: string[];
|
|
140
153
|
|
|
154
|
+
/**
|
|
155
|
+
* R3 — role-VALUE signature: `action.role=value` multiset, filtered to
|
|
156
|
+
* language-invariant surface forms (selectors, sigil refs, numbers/time
|
|
157
|
+
* literals, colon-qualified event names, URLs — code, not prose). The one
|
|
158
|
+
* value class that CAN be compared verbatim across languages; catches
|
|
159
|
+
* right-action/right-type/wrong-VALUE corruption (`draggable` captured for
|
|
160
|
+
* `draggable:start`) that every type-based signal misses. Set by the
|
|
161
|
+
* validator (roles are a ReadonlyMap — serialize to {} in results.json).
|
|
162
|
+
*/
|
|
163
|
+
roleValueSignature?: string[];
|
|
164
|
+
|
|
141
165
|
/**
|
|
142
166
|
* Structural fidelity vs the English reference parse, in [0, 1]: the fraction
|
|
143
167
|
* of the English parse's distinct actions also present in this language's
|
|
@@ -157,6 +181,15 @@ export interface ParseResult {
|
|
|
157
181
|
*/
|
|
158
182
|
precision?: number;
|
|
159
183
|
|
|
184
|
+
/**
|
|
185
|
+
* R0-recall on the MULTISET vs the English reference, in [0, 1]. `fidelity`
|
|
186
|
+
* above scores the deduped Set, so a parse that drops a REPEATED command
|
|
187
|
+
* (`[bind, bind]` → `[bind]`) still scores 1.0. This counts duplicates, and is
|
|
188
|
+
* the mirror of `precision`: together they see both a dropped and an added
|
|
189
|
+
* copy of a command. `undefined` when there is no usable English reference.
|
|
190
|
+
*/
|
|
191
|
+
multisetRecall?: number;
|
|
192
|
+
|
|
160
193
|
/**
|
|
161
194
|
* R1 — role fidelity vs the English reference, in [0, 1]: the fraction of the
|
|
162
195
|
* English parse's role-signature entries also present here. Catches a parse
|
|
@@ -164,6 +197,23 @@ export interface ParseResult {
|
|
|
164
197
|
* patient/destination executes wrongly while action-fidelity scores 1.0).
|
|
165
198
|
*/
|
|
166
199
|
roleFidelity?: number;
|
|
200
|
+
|
|
201
|
+
/**
|
|
202
|
+
* R3 — invariant role-VALUE recall vs the English reference, in [0, 1]: the
|
|
203
|
+
* fraction of the English parse's `roleValueSignature` entries (multiset —
|
|
204
|
+
* duplicates counted) also present here. Falls below 1.0 when a translation
|
|
205
|
+
* loses or corrupts a language-invariant value (selector, sigil ref, time
|
|
206
|
+
* literal, colon-qualified event name, URL) while actions and role types
|
|
207
|
+
* still match. `undefined` when the en reference has no invariant values.
|
|
208
|
+
*/
|
|
209
|
+
valueRecall?: number;
|
|
210
|
+
|
|
211
|
+
/**
|
|
212
|
+
* R3 — the en reference's `action.role=value` entries missing from this
|
|
213
|
+
* parse (multiset difference). Populated only when `valueRecall` < 1, for
|
|
214
|
+
* diagnostics ("which value was lost/corrupted?").
|
|
215
|
+
*/
|
|
216
|
+
valueRecallMissing?: string[];
|
|
167
217
|
}
|
|
168
218
|
|
|
169
219
|
/**
|
|
@@ -205,6 +255,14 @@ export interface LanguageResults {
|
|
|
205
255
|
*/
|
|
206
256
|
avgPrecision?: number | undefined;
|
|
207
257
|
|
|
258
|
+
/**
|
|
259
|
+
* R0-recall-multiset — mean `multisetRecall` over successful parses with an
|
|
260
|
+
* English reference. Recorded + ratcheted alongside avgFidelity: a drop means a
|
|
261
|
+
* parser change started dropping a REPEATED command, which the Set-based
|
|
262
|
+
* avgFidelity and avgRoleFidelity cannot see.
|
|
263
|
+
*/
|
|
264
|
+
avgMultisetRecall?: number | undefined;
|
|
265
|
+
|
|
208
266
|
/**
|
|
209
267
|
* R1 — mean `roleFidelity` over successful parses with an English reference.
|
|
210
268
|
* Recorded + ratcheted in the baseline; the burn-down is deliberately NOT part
|
|
@@ -212,6 +270,14 @@ export interface LanguageResults {
|
|
|
212
270
|
*/
|
|
213
271
|
avgRoleFidelity?: number | undefined;
|
|
214
272
|
|
|
273
|
+
/**
|
|
274
|
+
* R3 — mean `valueRecall` over successful parses with an English reference.
|
|
275
|
+
* Recorded + ratcheted alongside avgRoleFidelity: a drop means a translation
|
|
276
|
+
* started losing/corrupting a language-invariant role VALUE (the #633 class),
|
|
277
|
+
* which every action/type-based signal above scores as perfect.
|
|
278
|
+
*/
|
|
279
|
+
avgValueRecall?: number | undefined;
|
|
280
|
+
|
|
215
281
|
/**
|
|
216
282
|
* R2 — mean executionFidelity over the curated execution subset (see
|
|
217
283
|
* validators/execution-validator.ts): 1 per pattern whose jsdom DOM-effect
|
|
@@ -298,8 +364,12 @@ export interface Baseline {
|
|
|
298
364
|
avgFidelity?: number | undefined;
|
|
299
365
|
/** R0-precision — mean precision vs the English reference (see ParseResult.precision). */
|
|
300
366
|
avgPrecision?: number | undefined;
|
|
367
|
+
/** R0-recall-multiset — mean multisetRecall vs the English reference (see ParseResult.multisetRecall). */
|
|
368
|
+
avgMultisetRecall?: number | undefined;
|
|
301
369
|
/** R1 — mean role fidelity vs the English reference (see ParseResult.roleFidelity). */
|
|
302
370
|
avgRoleFidelity?: number | undefined;
|
|
371
|
+
/** R3 — mean invariant role-VALUE recall vs the English reference (see ParseResult.valueRecall). */
|
|
372
|
+
avgValueRecall?: number | undefined;
|
|
303
373
|
/** R2 — mean execution fidelity over the curated execution subset (see LanguageResults.avgExecutionFidelity). */
|
|
304
374
|
avgExecutionFidelity?: number | undefined;
|
|
305
375
|
/** R2 — curated-subset pattern IDs whose execution diverged from the en reference. */
|
|
@@ -335,8 +405,12 @@ export interface RegressionResult {
|
|
|
335
405
|
avgFidelityDelta: number;
|
|
336
406
|
/** R0-precision — absolute change in avgPrecision (current − baseline). 0 when either side lacks data. Negative = phantom commands introduced. */
|
|
337
407
|
avgPrecisionDelta: number;
|
|
408
|
+
/** R0-recall-multiset — absolute change in avgMultisetRecall (current − baseline). 0 when either side lacks data. Negative = a repeated command is being dropped. */
|
|
409
|
+
avgMultisetRecallDelta: number;
|
|
338
410
|
/** R1 — absolute change in avgRoleFidelity (current − baseline). 0 when either side lacks data. */
|
|
339
411
|
avgRoleFidelityDelta: number;
|
|
412
|
+
/** R3 — absolute change in avgValueRecall (current − baseline). 0 when either side lacks data. Negative = an invariant role VALUE is being lost/corrupted. */
|
|
413
|
+
avgValueRecallDelta: number;
|
|
340
414
|
/** R2 — absolute change in avgExecutionFidelity (current − baseline). 0 when either side lacks data. */
|
|
341
415
|
avgExecutionFidelityDelta: number;
|
|
342
416
|
/**
|
|
@@ -6,7 +6,12 @@ import { MultilingualHyperscript } from '@hyperfixi/core/multilingual';
|
|
|
6
6
|
import type { SemanticNode } from '@lokascript/semantic';
|
|
7
7
|
import { fillSchemaDefaults } from '@lokascript/semantic';
|
|
8
8
|
import type { PatternTranslation, ParseResult, Validator } from '../types';
|
|
9
|
-
import {
|
|
9
|
+
import {
|
|
10
|
+
collectActions,
|
|
11
|
+
collectActionsMultiset,
|
|
12
|
+
collectRoleSignature,
|
|
13
|
+
collectRoleValueSignature,
|
|
14
|
+
} from '../fidelity';
|
|
10
15
|
|
|
11
16
|
/**
|
|
12
17
|
* Parse Validator
|
|
@@ -106,6 +111,9 @@ export class ParseValidator implements Validator<ParseResult[]> {
|
|
|
106
111
|
// R1 role signature (role name + value type per command) — collected here
|
|
107
112
|
// because live nodes carry roles as a ReadonlyMap that serializes to {}.
|
|
108
113
|
roleSignature: collectRoleSignature(semanticNode),
|
|
114
|
+
// R3 role-VALUE signature (invariant-shaped values only, multiset) —
|
|
115
|
+
// catches right-action/right-type/wrong-VALUE corruption (the #633 class).
|
|
116
|
+
roleValueSignature: collectRoleValueSignature(semanticNode),
|
|
109
117
|
duration: performance.now() - startTime,
|
|
110
118
|
};
|
|
111
119
|
} catch (error) {
|
package/src/runner.test.ts
CHANGED
|
@@ -438,7 +438,11 @@ describe('CoreTestRunner', () => {
|
|
|
438
438
|
|
|
439
439
|
const result = await runner.runTest(test, context);
|
|
440
440
|
|
|
441
|
-
|
|
441
|
+
// Tolerance below the 50ms sleep: setTimeout clamping can fire the
|
|
442
|
+
// timer ~1-2ms early on CI runners (observed: 49ms in pre-publish run
|
|
443
|
+
// 29800757714), and the assertion is about duration RECORDING, not
|
|
444
|
+
// timer precision.
|
|
445
|
+
expect(result.duration).toBeGreaterThanOrEqual(45);
|
|
442
446
|
});
|
|
443
447
|
});
|
|
444
448
|
});
|
|
@@ -686,7 +690,8 @@ describe('measurePerformance', () => {
|
|
|
686
690
|
|
|
687
691
|
const metrics = await measurePerformance(mockPage, testFn);
|
|
688
692
|
|
|
689
|
-
|
|
693
|
+
// Same timer-clamping tolerance as the duration test above.
|
|
694
|
+
expect(metrics.loadTime).toBeGreaterThanOrEqual(45);
|
|
690
695
|
});
|
|
691
696
|
});
|
|
692
697
|
|
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Batch 3 red→green proofs — one representative render→parse round-trip per
|
|
3
|
+
* V1 dict-fix class (docs-internal/MULTILINGUAL_NEXT_STEPS.md § "V1 probe
|
|
4
|
+
* conclusion (Batch 3)"). Each of these was a live render→parse break before
|
|
5
|
+
* the Batch 3 dictionary fixes:
|
|
6
|
+
*
|
|
7
|
+
* - select class: the dict word doubled as the pick keyword — the bare select
|
|
8
|
+
* render parsed null (de `auswählen #note`).
|
|
9
|
+
* - wrong-verb class: the dict word was another command's verb — the render
|
|
10
|
+
* parsed as THAT action (bn clone → copy, id close → hide, sw copy → clone,
|
|
11
|
+
* vi prepend → add, qu change-event → toggle).
|
|
12
|
+
* - reset class: the dict event word captured on.event as an expression (a
|
|
13
|
+
* broken listener) or a wrong event; the profile verb round-trips.
|
|
14
|
+
* - submit class: the dict event word doubled as the send verb — corpus
|
|
15
|
+
* on-submit rows captured event "send" (es/pl/tr/vi live).
|
|
16
|
+
* - blur class (it): the noun form dropped blur.patient in command position.
|
|
17
|
+
* - empty class (ko/qu profile-alternatives): the dict adjective renders the
|
|
18
|
+
* empty COMMAND but only the profile verb parsed. (ja is waived: bare 空
|
|
19
|
+
* phantoms the hot `is empty` rows if registered.)
|
|
20
|
+
*
|
|
21
|
+
* Assertions are on captured ACTION and role VALUES, never "it parses".
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
import { describe, expect, it } from 'vitest';
|
|
25
|
+
import { parseSemantic } from '@lokascript/semantic';
|
|
26
|
+
import { GrammarTransformer } from '@lokascript/i18n';
|
|
27
|
+
|
|
28
|
+
/** Verbatim (action, role, value) triples from a parse tree. */
|
|
29
|
+
function triples(node: unknown, acc: string[] = [], depth = 0): string[] {
|
|
30
|
+
if (depth > 64 || node === null || typeof node !== 'object') return acc;
|
|
31
|
+
const rec = node as Record<string, unknown>;
|
|
32
|
+
if (typeof rec.action === 'string') {
|
|
33
|
+
const roles = rec.roles;
|
|
34
|
+
const entries: Array<[unknown, unknown]> =
|
|
35
|
+
roles instanceof Map
|
|
36
|
+
? [...roles.entries()]
|
|
37
|
+
: roles && typeof roles === 'object'
|
|
38
|
+
? Object.entries(roles)
|
|
39
|
+
: [];
|
|
40
|
+
for (const [role, v] of entries) {
|
|
41
|
+
if (v === null || v === undefined) continue;
|
|
42
|
+
const val = (v as { value?: unknown }).value ?? (v as { raw?: unknown }).raw;
|
|
43
|
+
acc.push(`${rec.action}.${String(role)}=${String(val)}`);
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
for (const f of ['body', 'commands', 'children', 'thenBranch', 'elseBranch']) {
|
|
47
|
+
const c = rec[f];
|
|
48
|
+
if (Array.isArray(c)) for (const x of c) triples(x, acc, depth + 1);
|
|
49
|
+
else if (c && typeof c === 'object') triples(c, acc, depth + 1);
|
|
50
|
+
}
|
|
51
|
+
return acc;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
function renderAndParse(en: string, lang: string): { render: string; triples: string[] } {
|
|
55
|
+
const render = new GrammarTransformer('en', lang).transform(en);
|
|
56
|
+
const result = parseSemantic(render, lang);
|
|
57
|
+
return { render, triples: result.node ? triples(result.node) : [] };
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
describe('Batch 3 — select class (dict word was the pick keyword)', () => {
|
|
61
|
+
it('de: `select #note` renders markieren and parses back as select', () => {
|
|
62
|
+
const { render, triples: t } = renderAndParse('select #note', 'de');
|
|
63
|
+
expect(render).toContain('markieren');
|
|
64
|
+
expect(t).toContain('select.patient=#note');
|
|
65
|
+
});
|
|
66
|
+
|
|
67
|
+
it('ar: `select #note` renders ظلل and parses back as select', () => {
|
|
68
|
+
const { render, triples: t } = renderAndParse('select #note', 'ar');
|
|
69
|
+
expect(render).toContain('ظلل');
|
|
70
|
+
expect(t).toContain('select.patient=#note');
|
|
71
|
+
});
|
|
72
|
+
});
|
|
73
|
+
|
|
74
|
+
describe('Batch 3 — wrong-verb class (dict word was another command)', () => {
|
|
75
|
+
it('bn: `clone #card` no longer parses as copy', () => {
|
|
76
|
+
const { render, triples: t } = renderAndParse('clone #card', 'bn');
|
|
77
|
+
expect(render).toContain('ক্লোন');
|
|
78
|
+
expect(t).toContain('clone.patient=#card');
|
|
79
|
+
});
|
|
80
|
+
|
|
81
|
+
it('id: `close #modal` no longer parses as hide', () => {
|
|
82
|
+
const { render, triples: t } = renderAndParse('close #modal', 'id');
|
|
83
|
+
expect(render).toContain('tutupkan');
|
|
84
|
+
expect(t).toContain('close.patient=#modal');
|
|
85
|
+
});
|
|
86
|
+
|
|
87
|
+
it('sw: `copy #text` no longer parses as clone', () => {
|
|
88
|
+
const { render, triples: t } = renderAndParse('copy #text', 'sw');
|
|
89
|
+
expect(render).toContain('nakala');
|
|
90
|
+
expect(t).toContain('copy.patient=#text');
|
|
91
|
+
});
|
|
92
|
+
|
|
93
|
+
it('vi: `prepend "x" to #list` no longer parses as add', () => {
|
|
94
|
+
const { render, triples: t } = renderAndParse('prepend "x" to #list', 'vi');
|
|
95
|
+
expect(render).toContain('thêm vào đầu');
|
|
96
|
+
expect(t.some(x => x.startsWith('prepend.'))).toBe(true);
|
|
97
|
+
});
|
|
98
|
+
});
|
|
99
|
+
|
|
100
|
+
describe('Batch 3 — reset class (broken/wrong-event listener)', () => {
|
|
101
|
+
it.each([
|
|
102
|
+
['it', 'reimpostare'],
|
|
103
|
+
['ko', '재설정'],
|
|
104
|
+
['pl', 'zresetuj'],
|
|
105
|
+
['pt', 'redefinir'],
|
|
106
|
+
['ru', 'сбросить'],
|
|
107
|
+
['uk', 'скинути'],
|
|
108
|
+
['qu', 'musuqchay'],
|
|
109
|
+
])('%s: on-reset render captures on.event="reset" (canonical)', (langCode, word) => {
|
|
110
|
+
const { render, triples: t } = renderAndParse('on reset log "done"', langCode);
|
|
111
|
+
expect(render).toContain(word);
|
|
112
|
+
expect(t).toContain('on.event=reset');
|
|
113
|
+
});
|
|
114
|
+
});
|
|
115
|
+
|
|
116
|
+
describe('Batch 3 — submit class (dict word was the send verb)', () => {
|
|
117
|
+
it.each([
|
|
118
|
+
['es', 'envío'],
|
|
119
|
+
['pl', 'wysłaniu'],
|
|
120
|
+
['tr', 'gönderme'],
|
|
121
|
+
['vi', 'nộp'],
|
|
122
|
+
['qu', 'apaykachay'],
|
|
123
|
+
])(
|
|
124
|
+
'%s: corpus-shaped on-submit render captures on.event="submit", not "send"',
|
|
125
|
+
(langCode, word) => {
|
|
126
|
+
const { render, triples: t } = renderAndParse(
|
|
127
|
+
'on submit add @disabled to <button/> in me put "Submitting..." into <button/> in me',
|
|
128
|
+
langCode
|
|
129
|
+
);
|
|
130
|
+
expect(render).toContain(word);
|
|
131
|
+
expect(t).toContain('on.event=submit');
|
|
132
|
+
expect(t).not.toContain('on.event=send');
|
|
133
|
+
}
|
|
134
|
+
);
|
|
135
|
+
});
|
|
136
|
+
|
|
137
|
+
describe('Batch 3 — qu change (dict word was the toggle verb)', () => {
|
|
138
|
+
it('qu: on-change render captures on.event="change", not "toggle"', () => {
|
|
139
|
+
const { render, triples: t } = renderAndParse('on change log "x"', 'qu');
|
|
140
|
+
expect(render).toContain('kambiay');
|
|
141
|
+
expect(t).toContain('on.event=change');
|
|
142
|
+
});
|
|
143
|
+
});
|
|
144
|
+
|
|
145
|
+
describe('Batch 3 — it blur (noun form dropped the command patient)', () => {
|
|
146
|
+
it('it: blur command render captures blur.patient', () => {
|
|
147
|
+
const { render, triples: t } = renderAndParse('on keydown[key=="Escape"] blur me', 'it');
|
|
148
|
+
expect(render).toContain('sfuocare');
|
|
149
|
+
expect(t).toContain('blur.patient=me');
|
|
150
|
+
});
|
|
151
|
+
|
|
152
|
+
it('it: blur event position still captures on.event="blur"', () => {
|
|
153
|
+
const { triples: t } = renderAndParse('on blur log "x"', 'it');
|
|
154
|
+
expect(t).toContain('on.event=blur');
|
|
155
|
+
});
|
|
156
|
+
});
|
|
157
|
+
|
|
158
|
+
describe('Batch 3 — empty class (ko/qu profile alternatives)', () => {
|
|
159
|
+
it('ko: the dict adjective now parses the empty command', () => {
|
|
160
|
+
const t = triples(parseSemantic('#list 를 비어있는', 'ko').node);
|
|
161
|
+
expect(t).toContain('empty.patient=#list');
|
|
162
|
+
});
|
|
163
|
+
|
|
164
|
+
it('qu: apostrophe-less chusaq now parses the empty command', () => {
|
|
165
|
+
const t = triples(parseSemantic('#list ta chusaq', 'qu').node);
|
|
166
|
+
expect(t).toContain('empty.patient=#list');
|
|
167
|
+
});
|
|
168
|
+
|
|
169
|
+
it('ko/qu: the hot `is empty` expression rows keep parsing without a phantom empty action', () => {
|
|
170
|
+
for (const [text, langCode] of [
|
|
171
|
+
[
|
|
172
|
+
'블러 할 때 만약 내 값 이다 비어있는 .error 를 추가 나 에 아니면 .error 를 제거 나 에서 끝',
|
|
173
|
+
'ko',
|
|
174
|
+
],
|
|
175
|
+
[
|
|
176
|
+
'paqariy pi sichus noqaq chanin kanqa chusaq .error ta noqa man yapay manachus .error ta noqa manta qichuy tukuy',
|
|
177
|
+
'qu',
|
|
178
|
+
],
|
|
179
|
+
] as const) {
|
|
180
|
+
const t = triples(parseSemantic(text, langCode).node);
|
|
181
|
+
expect(t.some(x => x.startsWith('empty.'))).toBe(false);
|
|
182
|
+
}
|
|
183
|
+
});
|
|
184
|
+
});
|