@hyperfixi/testing-framework 2.7.2 → 2.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assertions.d.mts +26 -1
- package/dist/assertions.d.ts +26 -1
- package/dist/index.d.mts +5 -112
- package/dist/index.d.ts +5 -112
- package/dist/runner.d.mts +112 -0
- package/dist/runner.d.ts +112 -0
- package/dist/runner.js +1102 -0
- package/dist/runner.js.map +1 -0
- package/dist/runner.mjs +1097 -0
- package/dist/runner.mjs.map +1 -0
- package/dist/{assertions-CsGP61iW.d.mts → types-D-rCVkf3.d.mts} +1 -24
- package/dist/{assertions-CsGP61iW.d.ts → types-D-rCVkf3.d.ts} +1 -24
- package/package.json +11 -26
- package/src/multilingual/canonical-validity.test.ts +69 -0
- package/src/multilingual/canonical-validity.ts +132 -0
- package/src/multilingual/cli.ts +185 -1
- package/src/multilingual/fidelity.test.ts +192 -0
- package/src/multilingual/fidelity.ts +153 -0
- package/src/multilingual/foreign-canonical-validity.test.ts +0 -0
- package/src/multilingual/foreign-canonical-validity.ts +158 -0
- package/src/multilingual/orchestrator.ts +47 -1
- package/src/multilingual/reporters/console-reporter.ts +51 -0
- package/src/multilingual/reporters/regression-reporter.ts +23 -0
- package/src/multilingual/tools/diagnose-coverage.ts +118 -0
- package/src/multilingual/tools/triage-r1.ts +149 -0
- package/src/multilingual/types.ts +74 -0
- package/src/multilingual/validators/parse-validator.ts +9 -1
- package/src/runner.test.ts +7 -2
- package/src/vocab/batch3-roundtrip.test.ts +184 -0
- package/src/vocab/checks.test.ts +362 -0
- package/src/vocab/checks.ts +311 -0
- package/src/vocab/cli.ts +196 -0
- package/src/vocab/dump.ts +80 -0
- package/src/vocab/model.ts +110 -0
- package/src/vocab/report.ts +120 -0
- package/src/vocab/types.ts +100 -0
|
@@ -0,0 +1,362 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Each check is proven against a seeded disagreement (fires) and a seeded
|
|
3
|
+
* agreement (stays silent) — never "does it run".
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
import { describe, expect, it } from 'vitest';
|
|
7
|
+
import { runChecks } from './checks';
|
|
8
|
+
import { buildLedger } from './report';
|
|
9
|
+
import type { Finding, LangVocab, VocabModel } from './types';
|
|
10
|
+
import { findingKey } from './types';
|
|
11
|
+
|
|
12
|
+
function lang(partial: Partial<LangVocab>): LangVocab {
|
|
13
|
+
return {
|
|
14
|
+
language: 'xx',
|
|
15
|
+
keywords: {},
|
|
16
|
+
roleMarkers: {},
|
|
17
|
+
schemaMarkers: [],
|
|
18
|
+
grammarMarkers: [],
|
|
19
|
+
...partial,
|
|
20
|
+
};
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
const model = (l: Partial<LangVocab>): VocabModel => ({ languages: [lang(l)] });
|
|
24
|
+
|
|
25
|
+
const of = (findings: Finding[], check: string, tier?: string) =>
|
|
26
|
+
findings.filter(f => f.check === check && (!tier || f.tier === tier));
|
|
27
|
+
|
|
28
|
+
describe('V1 — profile keywords vs i18n dictionary', () => {
|
|
29
|
+
it('fires an error when the dictionary translates a concept differently', () => {
|
|
30
|
+
const findings = runChecks(
|
|
31
|
+
model({
|
|
32
|
+
keywords: { toggle: { primary: 'alternar', alternatives: ['conmutar'] } },
|
|
33
|
+
dictionary: { commands: { toggle: 'cambiar' } },
|
|
34
|
+
})
|
|
35
|
+
);
|
|
36
|
+
const errors = of(findings, 'V1', 'error');
|
|
37
|
+
expect(errors).toHaveLength(1);
|
|
38
|
+
expect(errors[0].key).toBe('toggle');
|
|
39
|
+
expect(errors[0].message).toContain('cambiar');
|
|
40
|
+
});
|
|
41
|
+
|
|
42
|
+
it('accepts a dictionary value matching an alternative form (set containment, not 1:1)', () => {
|
|
43
|
+
const findings = runChecks(
|
|
44
|
+
model({
|
|
45
|
+
keywords: { toggle: { primary: 'alternar', alternatives: ['conmutar'] } },
|
|
46
|
+
dictionary: { commands: { toggle: 'Conmutar' } }, // case difference must normalize away
|
|
47
|
+
})
|
|
48
|
+
);
|
|
49
|
+
expect(of(findings, 'V1', 'error')).toHaveLength(0);
|
|
50
|
+
});
|
|
51
|
+
|
|
52
|
+
it('V1b: warns on a profile concept missing from the dictionary, infos on dictionary-only keys', () => {
|
|
53
|
+
const findings = runChecks(
|
|
54
|
+
model({
|
|
55
|
+
keywords: { toggle: { primary: 'alternar' } },
|
|
56
|
+
dictionary: { commands: { add: 'añadir' } },
|
|
57
|
+
})
|
|
58
|
+
);
|
|
59
|
+
expect(of(findings, 'V1b', 'warn').map(f => f.key)).toEqual(['toggle']);
|
|
60
|
+
expect(of(findings, 'V1b', 'info').map(f => f.key)).toEqual(['add']);
|
|
61
|
+
});
|
|
62
|
+
|
|
63
|
+
it('stays silent without a dictionary (no S3 surface for the language)', () => {
|
|
64
|
+
const findings = runChecks(model({ keywords: { toggle: { primary: 'x' } } }));
|
|
65
|
+
expect(of(findings, 'V1')).toHaveLength(0);
|
|
66
|
+
expect(of(findings, 'V1b')).toHaveLength(0);
|
|
67
|
+
});
|
|
68
|
+
});
|
|
69
|
+
|
|
70
|
+
describe('V2 — role markers three-way', () => {
|
|
71
|
+
it('fires an error when the grammar renders a marker the parse side does not know', () => {
|
|
72
|
+
const findings = runChecks(
|
|
73
|
+
model({
|
|
74
|
+
roleMarkers: { destination: { primary: 'e' } },
|
|
75
|
+
schemaMarkers: [{ action: 'set', role: 'destination', marker: 'de', kind: 'override' }],
|
|
76
|
+
grammarMarkers: [{ form: 'ya', role: 'destination' }], // the tr allomorph class
|
|
77
|
+
})
|
|
78
|
+
);
|
|
79
|
+
const errors = of(findings, 'V2', 'error');
|
|
80
|
+
expect(errors).toHaveLength(1);
|
|
81
|
+
expect(errors[0].key).toBe('destination:ya');
|
|
82
|
+
});
|
|
83
|
+
|
|
84
|
+
it('accepts a grammar form known via a schema variant (union across S1+S2)', () => {
|
|
85
|
+
const findings = runChecks(
|
|
86
|
+
model({
|
|
87
|
+
roleMarkers: { destination: { primary: 'e' } },
|
|
88
|
+
schemaMarkers: [{ action: 'set', role: 'destination', marker: 'ya', kind: 'variant' }],
|
|
89
|
+
grammarMarkers: [
|
|
90
|
+
{ form: 'ya', role: 'destination' },
|
|
91
|
+
{ form: 'e', role: 'destination' },
|
|
92
|
+
],
|
|
93
|
+
})
|
|
94
|
+
);
|
|
95
|
+
expect(of(findings, 'V2', 'error')).toHaveLength(0);
|
|
96
|
+
});
|
|
97
|
+
|
|
98
|
+
it('downgrades to info when the role has no parse-side markers at all', () => {
|
|
99
|
+
const findings = runChecks(model({ grammarMarkers: [{ form: 'на', role: 'locative' }] }));
|
|
100
|
+
expect(of(findings, 'V2', 'error')).toHaveLength(0);
|
|
101
|
+
expect(of(findings, 'V2', 'info')).toHaveLength(1);
|
|
102
|
+
});
|
|
103
|
+
|
|
104
|
+
it('accepts an event marker known only via the hardcoded SOV table (surface #6)', () => {
|
|
105
|
+
const withoutSurface6 = runChecks(
|
|
106
|
+
model({
|
|
107
|
+
roleMarkers: { event: { primary: 'に' } },
|
|
108
|
+
grammarMarkers: [{ form: 'で', role: 'event' }], // the ja class from the first ledger
|
|
109
|
+
})
|
|
110
|
+
);
|
|
111
|
+
expect(of(withoutSurface6, 'V2', 'error')).toHaveLength(1);
|
|
112
|
+
|
|
113
|
+
const withSurface6 = runChecks(
|
|
114
|
+
model({
|
|
115
|
+
roleMarkers: { event: { primary: 'に' } },
|
|
116
|
+
grammarMarkers: [{ form: 'で', role: 'event' }],
|
|
117
|
+
sovEventMarkers: ['で'],
|
|
118
|
+
})
|
|
119
|
+
);
|
|
120
|
+
expect(of(withSurface6, 'V2', 'error')).toHaveLength(0);
|
|
121
|
+
});
|
|
122
|
+
|
|
123
|
+
it('warns when the profile canonical marker never appears among the grammar forms', () => {
|
|
124
|
+
const findings = runChecks(
|
|
125
|
+
model({
|
|
126
|
+
roleMarkers: { destination: { primary: 'zu', alternatives: ['nach'] } },
|
|
127
|
+
grammarMarkers: [{ form: 'nach', role: 'destination' }],
|
|
128
|
+
})
|
|
129
|
+
);
|
|
130
|
+
const warns = of(findings, 'V2', 'warn');
|
|
131
|
+
expect(warns).toHaveLength(1);
|
|
132
|
+
expect(warns[0].key).toBe('destination:primary');
|
|
133
|
+
});
|
|
134
|
+
});
|
|
135
|
+
|
|
136
|
+
describe('V3 — event names', () => {
|
|
137
|
+
it('fires an error when the dictionary and eventNameTranslations disagree on an event', () => {
|
|
138
|
+
const findings = runChecks(
|
|
139
|
+
model({
|
|
140
|
+
eventTranslations: { クリック: 'click' },
|
|
141
|
+
dictionary: { events: { click: '押す' } },
|
|
142
|
+
})
|
|
143
|
+
);
|
|
144
|
+
const errors = of(findings, 'V3', 'error');
|
|
145
|
+
expect(errors).toHaveLength(1);
|
|
146
|
+
expect(errors[0].key).toBe('click');
|
|
147
|
+
});
|
|
148
|
+
|
|
149
|
+
it('accepts agreement (any native form recognized for the event)', () => {
|
|
150
|
+
const findings = runChecks(
|
|
151
|
+
model({
|
|
152
|
+
eventTranslations: { クリック: 'click', おす: 'click' },
|
|
153
|
+
dictionary: { events: { click: 'クリック' } },
|
|
154
|
+
})
|
|
155
|
+
);
|
|
156
|
+
expect(of(findings, 'V3', 'error')).toHaveLength(0);
|
|
157
|
+
});
|
|
158
|
+
|
|
159
|
+
it('V3b: infos when a language has no eventNameTranslations table', () => {
|
|
160
|
+
const findings = runChecks(model({ dictionary: { events: { click: 'klik' } } }));
|
|
161
|
+
expect(of(findings, 'V3b', 'info')).toHaveLength(1);
|
|
162
|
+
expect(of(findings, 'V3', 'error')).toHaveLength(0);
|
|
163
|
+
});
|
|
164
|
+
});
|
|
165
|
+
|
|
166
|
+
describe('V3c — dictionary events NOT covered by eventNameTranslations', () => {
|
|
167
|
+
// The gap this closes: V3's `continue` branch skips any dict event whose
|
|
168
|
+
// English key is absent from S5b, so a broken form there was invisible
|
|
169
|
+
// (id tekan_mouse/lepas_mouse, zh 鼠标进入, fr sourisappuyée).
|
|
170
|
+
const base = { eventTranslations: { クリック: 'click' } }; // covered lang, but no mousedown key
|
|
171
|
+
|
|
172
|
+
it('warns (broken-listener) when the uncovered form resolves to nothing', () => {
|
|
173
|
+
const findings = runChecks(
|
|
174
|
+
model({ ...base, dictionary: { events: { mousedown: 'tekan_mouse' } } })
|
|
175
|
+
);
|
|
176
|
+
const warns = of(findings, 'V3c', 'warn');
|
|
177
|
+
expect(warns).toHaveLength(1);
|
|
178
|
+
expect(warns[0].key).toBe('mousedown');
|
|
179
|
+
expect(warns[0].message).toContain('broken-listener');
|
|
180
|
+
});
|
|
181
|
+
|
|
182
|
+
it('warns (wrong-event) when S5b maps the form to a DIFFERENT event', () => {
|
|
183
|
+
const findings = runChecks(
|
|
184
|
+
model({
|
|
185
|
+
eventTranslations: { 鼠标进入: 'mouseover' },
|
|
186
|
+
dictionary: { events: { mouseenter: '鼠标进入' } },
|
|
187
|
+
})
|
|
188
|
+
);
|
|
189
|
+
const warns = of(findings, 'V3c', 'warn');
|
|
190
|
+
expect(warns).toHaveLength(1);
|
|
191
|
+
expect(warns[0].key).toBe('mouseenter');
|
|
192
|
+
expect(warns[0].message).toContain('"mouseover"');
|
|
193
|
+
expect(warns[0].message).toContain('wrong-event');
|
|
194
|
+
});
|
|
195
|
+
|
|
196
|
+
it('stays silent when the tokenizer keyword table resolves the form (parse authority)', () => {
|
|
197
|
+
// sw panya_juu: S5b says mouseover, but the tokenizer deliberately maps
|
|
198
|
+
// it to mouseup for the corpus — tokenizer wins, no warn.
|
|
199
|
+
const findings = runChecks(
|
|
200
|
+
model({
|
|
201
|
+
eventTranslations: { 'panya juu': 'mouseover' },
|
|
202
|
+
dictionary: { events: { mouseup: 'panya_juu' } },
|
|
203
|
+
normalizeWord: w => (w === 'panya_juu' ? 'mouseup' : undefined),
|
|
204
|
+
})
|
|
205
|
+
);
|
|
206
|
+
expect(of(findings, 'V3c')).toHaveLength(0);
|
|
207
|
+
});
|
|
208
|
+
|
|
209
|
+
it('stays silent for denylisted (intentionally-English) pairs', () => {
|
|
210
|
+
const findings = runChecks(
|
|
211
|
+
model({
|
|
212
|
+
...base,
|
|
213
|
+
dictionary: { events: { mousedown: 'kaput_form' } },
|
|
214
|
+
eventDenylist: new Set(['mousedown']),
|
|
215
|
+
})
|
|
216
|
+
);
|
|
217
|
+
expect(of(findings, 'V3c')).toHaveLength(0);
|
|
218
|
+
});
|
|
219
|
+
|
|
220
|
+
it('stays silent for English passthrough forms', () => {
|
|
221
|
+
const findings = runChecks(model({ ...base, dictionary: { events: { keyup: 'keyup' } } }));
|
|
222
|
+
expect(of(findings, 'V3c')).toHaveLength(0);
|
|
223
|
+
});
|
|
224
|
+
|
|
225
|
+
it('stays silent when S5b covers the form under the SAME event via underscore↔space', () => {
|
|
226
|
+
const findings = runChecks(
|
|
227
|
+
model({
|
|
228
|
+
eventTranslations: { クリック: 'click', 'tekan tombol': 'keydown' },
|
|
229
|
+
dictionary: { events: { keydown: 'tekan_tombol' } },
|
|
230
|
+
})
|
|
231
|
+
);
|
|
232
|
+
expect(of(findings, 'V3c')).toHaveLength(0);
|
|
233
|
+
});
|
|
234
|
+
|
|
235
|
+
it('never fires for a language without an eventNameTranslations table (V3b owns that)', () => {
|
|
236
|
+
const findings = runChecks(model({ dictionary: { events: { mousedown: 'tekan_mouse' } } }));
|
|
237
|
+
expect(of(findings, 'V3c')).toHaveLength(0);
|
|
238
|
+
});
|
|
239
|
+
});
|
|
240
|
+
|
|
241
|
+
describe('V4 — tokenizer classification', () => {
|
|
242
|
+
const classify = (word: string) => (word === 'kaydır' ? 'identifier' : 'keyword');
|
|
243
|
+
|
|
244
|
+
it('fires an error when a vocab word does not classify as keyword/particle', () => {
|
|
245
|
+
const findings = runChecks(
|
|
246
|
+
model({
|
|
247
|
+
keywords: { scroll: { primary: 'kaydır' } },
|
|
248
|
+
classify,
|
|
249
|
+
})
|
|
250
|
+
);
|
|
251
|
+
const errors = of(findings, 'V4', 'error');
|
|
252
|
+
expect(errors).toHaveLength(1);
|
|
253
|
+
expect(errors[0].key).toBe('kaydır');
|
|
254
|
+
expect(errors[0].message).toContain("'identifier'");
|
|
255
|
+
});
|
|
256
|
+
|
|
257
|
+
it('collects from all four sources and dedupes the word into one finding', () => {
|
|
258
|
+
const findings = runChecks(
|
|
259
|
+
model({
|
|
260
|
+
keywords: { scroll: { primary: 'kaydır' } },
|
|
261
|
+
roleMarkers: { patient: { primary: 'kaydır' } },
|
|
262
|
+
schemaMarkers: [{ action: 'go', role: 'patient', marker: 'kaydır', kind: 'override' }],
|
|
263
|
+
grammarMarkers: [{ form: 'kaydır', role: 'patient' }],
|
|
264
|
+
classify,
|
|
265
|
+
})
|
|
266
|
+
);
|
|
267
|
+
const errors = of(findings, 'V4', 'error');
|
|
268
|
+
expect(errors).toHaveLength(1);
|
|
269
|
+
expect(errors[0].source).toContain('profile.keywords.scroll');
|
|
270
|
+
});
|
|
271
|
+
|
|
272
|
+
it('accepts particles and skips multi-word forms', () => {
|
|
273
|
+
const findings = runChecks(
|
|
274
|
+
model({
|
|
275
|
+
roleMarkers: { patient: { primary: 'を' }, event: { primary: 'do not' } },
|
|
276
|
+
classify: () => 'particle',
|
|
277
|
+
})
|
|
278
|
+
);
|
|
279
|
+
expect(of(findings, 'V4')).toHaveLength(0);
|
|
280
|
+
});
|
|
281
|
+
|
|
282
|
+
it('stays silent without a tokenizer', () => {
|
|
283
|
+
const findings = runChecks(model({ keywords: { scroll: { primary: 'kaydır' } } }));
|
|
284
|
+
expect(of(findings, 'V4')).toHaveLength(0);
|
|
285
|
+
});
|
|
286
|
+
});
|
|
287
|
+
|
|
288
|
+
describe('check filter + ledger/waivers', () => {
|
|
289
|
+
const disagreeing = (): VocabModel =>
|
|
290
|
+
model({
|
|
291
|
+
keywords: { toggle: { primary: 'a' } },
|
|
292
|
+
dictionary: { commands: { toggle: 'b' } },
|
|
293
|
+
});
|
|
294
|
+
|
|
295
|
+
it('--check filters to the requested checks only', () => {
|
|
296
|
+
const findings = runChecks(disagreeing(), ['V4']);
|
|
297
|
+
expect(findings).toHaveLength(0);
|
|
298
|
+
});
|
|
299
|
+
|
|
300
|
+
it('waivers suppress exactly the keyed error and flag stale keys', () => {
|
|
301
|
+
const findings = runChecks(disagreeing());
|
|
302
|
+
const key = findingKey(of(findings, 'V1', 'error')[0]);
|
|
303
|
+
const ledger = buildLedger(findings, [
|
|
304
|
+
{ key, reason: 'legit morphology divergence' },
|
|
305
|
+
{ key: 'V1|zz|ghost', reason: 'stale' },
|
|
306
|
+
]);
|
|
307
|
+
expect(ledger.unwaivedErrors).toBe(0);
|
|
308
|
+
expect(ledger.findings.find(f => findingKey(f) === key)?.waived).toBe(
|
|
309
|
+
'legit morphology divergence'
|
|
310
|
+
);
|
|
311
|
+
expect(ledger.staleWaivers).toEqual(['V1|zz|ghost']);
|
|
312
|
+
});
|
|
313
|
+
|
|
314
|
+
it('class waivers (wildcard segments) cover every finding in the class', () => {
|
|
315
|
+
const twoLangs: VocabModel = {
|
|
316
|
+
languages: [
|
|
317
|
+
lang({
|
|
318
|
+
language: 'es',
|
|
319
|
+
keywords: { toggle: { primary: 'a' } },
|
|
320
|
+
dictionary: { commands: { toggle: 'b' } },
|
|
321
|
+
}),
|
|
322
|
+
lang({
|
|
323
|
+
language: 'fr',
|
|
324
|
+
keywords: { toggle: { primary: 'c' } },
|
|
325
|
+
dictionary: { commands: { toggle: 'd' } },
|
|
326
|
+
}),
|
|
327
|
+
],
|
|
328
|
+
};
|
|
329
|
+
const findings = runChecks(twoLangs);
|
|
330
|
+
expect(of(findings, 'V1', 'error')).toHaveLength(2);
|
|
331
|
+
|
|
332
|
+
const ledger = buildLedger(findings, [
|
|
333
|
+
{ key: 'V1|*|*', reason: 'pending Arc B reconciliation' },
|
|
334
|
+
]);
|
|
335
|
+
expect(ledger.unwaivedErrors).toBe(0);
|
|
336
|
+
expect(
|
|
337
|
+
ledger.findings
|
|
338
|
+
.filter(f => f.waived)
|
|
339
|
+
.map(f => f.language)
|
|
340
|
+
.sort()
|
|
341
|
+
).toEqual(['es', 'fr']);
|
|
342
|
+
expect(ledger.staleWaivers).toHaveLength(0);
|
|
343
|
+
});
|
|
344
|
+
|
|
345
|
+
it('a wildcard waiver matching nothing is stale', () => {
|
|
346
|
+
const findings = runChecks(disagreeing());
|
|
347
|
+
const ledger = buildLedger(findings, [{ key: 'V3|*|*', reason: 'nothing here' }]);
|
|
348
|
+
expect(ledger.staleWaivers).toEqual(['V3|*|*']);
|
|
349
|
+
});
|
|
350
|
+
|
|
351
|
+
it('waivers never apply to warn/info tiers', () => {
|
|
352
|
+
const findings = runChecks(
|
|
353
|
+
model({ keywords: { toggle: { primary: 'a' } }, dictionary: { commands: {} } })
|
|
354
|
+
);
|
|
355
|
+
const warn = of(findings, 'V1b', 'warn')[0];
|
|
356
|
+
const ledger = buildLedger(findings, [{ key: findingKey(warn), reason: 'nope' }]);
|
|
357
|
+
expect(
|
|
358
|
+
ledger.findings.find(f => f.check === 'V1b' && f.tier === 'warn')?.waived
|
|
359
|
+
).toBeUndefined();
|
|
360
|
+
expect(ledger.staleWaivers).toHaveLength(1);
|
|
361
|
+
});
|
|
362
|
+
});
|
|
@@ -0,0 +1,311 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The V1–V4 cross-surface consistency checks
|
|
3
|
+
* (check matrix: docs-internal/HANDOFF_vocab-consistency.md).
|
|
4
|
+
*
|
|
5
|
+
* Tiering rule of thumb: a CONFLICT between two surfaces is an error; a
|
|
6
|
+
* MISSING entry is warn/info depending on the pair's direction of authority.
|
|
7
|
+
* Comparison is always over normalized form SETS with containment — never
|
|
8
|
+
* naive 1:1 string equality (markers have allomorphs and alternatives).
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
import type { CheckId, Finding, LangVocab, VocabModel } from './types';
|
|
12
|
+
import { formSet, norm } from './types';
|
|
13
|
+
|
|
14
|
+
const ACCEPTED_TOKEN_KINDS = new Set(['keyword', 'particle']);
|
|
15
|
+
|
|
16
|
+
/** Look an English concept key up across all dictionary categories. */
|
|
17
|
+
function findInDictionary(
|
|
18
|
+
dictionary: NonNullable<LangVocab['dictionary']>,
|
|
19
|
+
concept: string
|
|
20
|
+
): { category: string; value: string } | undefined {
|
|
21
|
+
for (const [category, entries] of Object.entries(dictionary)) {
|
|
22
|
+
const value = entries[concept];
|
|
23
|
+
if (value !== undefined) return { category, value };
|
|
24
|
+
}
|
|
25
|
+
return undefined;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/** V1/V1b — S1 profile keywords ↔ S3 i18n dictionary. */
|
|
29
|
+
function checkKeywords(lang: LangVocab, findings: Finding[]): void {
|
|
30
|
+
if (!lang.dictionary) return;
|
|
31
|
+
|
|
32
|
+
const dictKeys = new Set<string>();
|
|
33
|
+
for (const entries of Object.values(lang.dictionary)) {
|
|
34
|
+
for (const key of Object.keys(entries)) dictKeys.add(key);
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
for (const [concept, entry] of Object.entries(lang.keywords)) {
|
|
38
|
+
const hit = findInDictionary(lang.dictionary, concept);
|
|
39
|
+
if (!hit) {
|
|
40
|
+
findings.push({
|
|
41
|
+
check: 'V1b',
|
|
42
|
+
tier: 'warn',
|
|
43
|
+
language: lang.language,
|
|
44
|
+
key: concept,
|
|
45
|
+
message: `profile concept "${concept}" (${entry.primary}) has no i18n dictionary entry`,
|
|
46
|
+
source: `profile.keywords.${concept}`,
|
|
47
|
+
});
|
|
48
|
+
continue;
|
|
49
|
+
}
|
|
50
|
+
const forms = formSet(entry);
|
|
51
|
+
if (forms.size > 0 && !forms.has(norm(hit.value))) {
|
|
52
|
+
findings.push({
|
|
53
|
+
check: 'V1',
|
|
54
|
+
tier: 'error',
|
|
55
|
+
language: lang.language,
|
|
56
|
+
key: concept,
|
|
57
|
+
message: `"${concept}": profile says "${entry.primary}"${
|
|
58
|
+
entry.alternatives?.length ? ` (alts: ${entry.alternatives.join(', ')})` : ''
|
|
59
|
+
} but dictionary ${hit.category} says "${hit.value}"`,
|
|
60
|
+
source: `dictionary.${hit.category}.${concept}`,
|
|
61
|
+
});
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
// Dictionary-only keys: S3 legitimately holds render vocabulary S1 never
|
|
66
|
+
// needs — coverage note, not a defect.
|
|
67
|
+
for (const key of dictKeys) {
|
|
68
|
+
if (!(key in lang.keywords)) {
|
|
69
|
+
findings.push({
|
|
70
|
+
check: 'V1b',
|
|
71
|
+
tier: 'info',
|
|
72
|
+
language: lang.language,
|
|
73
|
+
key,
|
|
74
|
+
message: `dictionary key "${key}" has no profile concept (dictionary-only vocabulary)`,
|
|
75
|
+
});
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/** V2 — role markers three-way: S1 profile ↔ S2 schemas ↔ S4 grammar profile. */
|
|
81
|
+
function checkRoleMarkers(lang: LangVocab, findings: Finding[]): void {
|
|
82
|
+
// Parse-side union per role: profile markers + every schema override/variant.
|
|
83
|
+
const parseSet = new Map<string, Set<string>>();
|
|
84
|
+
const add = (role: string, word: string) => {
|
|
85
|
+
if (!word) return;
|
|
86
|
+
let set = parseSet.get(role);
|
|
87
|
+
if (!set) parseSet.set(role, (set = new Set()));
|
|
88
|
+
set.add(norm(word));
|
|
89
|
+
};
|
|
90
|
+
for (const [role, entry] of Object.entries(lang.roleMarkers)) {
|
|
91
|
+
for (const f of formSet(entry)) add(role, f);
|
|
92
|
+
}
|
|
93
|
+
for (const sm of lang.schemaMarkers) add(sm.role, sm.marker);
|
|
94
|
+
// Surface #6: hardcoded SOV event markers count as parse-side knowledge for
|
|
95
|
+
// the event role (they live in semantic-parser.ts, not in any vocab file).
|
|
96
|
+
for (const marker of lang.sovEventMarkers ?? []) add('event', marker);
|
|
97
|
+
|
|
98
|
+
// Render-side forms per role (S4).
|
|
99
|
+
const renderSet = new Map<string, Set<string>>();
|
|
100
|
+
for (const gm of lang.grammarMarkers) {
|
|
101
|
+
let set = renderSet.get(gm.role);
|
|
102
|
+
if (!set) renderSet.set(gm.role, (set = new Set()));
|
|
103
|
+
set.add(norm(gm.form));
|
|
104
|
+
for (const a of gm.alternatives ?? []) if (a) set.add(norm(a));
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
// Error: the transformer renders a marker the parse side does not know for
|
|
108
|
+
// that role (the tr `ya`-vs-`e` class). Only when the parse side HAS
|
|
109
|
+
// markers for the role — an entirely unmarked role is a coverage note.
|
|
110
|
+
for (const [role, forms] of renderSet) {
|
|
111
|
+
const parse = parseSet.get(role);
|
|
112
|
+
for (const form of forms) {
|
|
113
|
+
if (!parse || parse.size === 0) {
|
|
114
|
+
findings.push({
|
|
115
|
+
check: 'V2',
|
|
116
|
+
tier: 'info',
|
|
117
|
+
language: lang.language,
|
|
118
|
+
key: `${role}:${form}`,
|
|
119
|
+
message: `grammar renders "${form}" for role ${role}, which has no parse-side markers at all`,
|
|
120
|
+
source: `grammar.${role}`,
|
|
121
|
+
});
|
|
122
|
+
} else if (!parse.has(form)) {
|
|
123
|
+
findings.push({
|
|
124
|
+
check: 'V2',
|
|
125
|
+
tier: 'error',
|
|
126
|
+
language: lang.language,
|
|
127
|
+
key: `${role}:${form}`,
|
|
128
|
+
message: `grammar renders "${form}" for role ${role} but neither the profile role marker nor any schema override/variant knows it (parse side: ${[...parse].join(', ')})`,
|
|
129
|
+
source: `grammar.${role}`,
|
|
130
|
+
});
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
// Warn: the profile's canonical marker never appears among the grammar
|
|
136
|
+
// forms for a role the grammar does render — the two canonical copies
|
|
137
|
+
// disagree even if parsing still succeeds via alternatives.
|
|
138
|
+
for (const [role, entry] of Object.entries(lang.roleMarkers)) {
|
|
139
|
+
const render = renderSet.get(role);
|
|
140
|
+
if (!render || !entry.primary) continue;
|
|
141
|
+
if (!render.has(norm(entry.primary))) {
|
|
142
|
+
findings.push({
|
|
143
|
+
check: 'V2',
|
|
144
|
+
tier: 'warn',
|
|
145
|
+
language: lang.language,
|
|
146
|
+
key: `${role}:primary`,
|
|
147
|
+
message: `profile canonical marker "${entry.primary}" for role ${role} is not among the grammar forms (${[...render].join(', ')})`,
|
|
148
|
+
source: `profile.roleMarkers.${role}`,
|
|
149
|
+
});
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/** V3/V3b — event names: S5b eventNameTranslations ↔ S3 dictionary `events`. */
|
|
155
|
+
function checkEventNames(lang: LangVocab, findings: Finding[]): void {
|
|
156
|
+
if (!lang.eventTranslations) {
|
|
157
|
+
findings.push({
|
|
158
|
+
check: 'V3b',
|
|
159
|
+
tier: 'info',
|
|
160
|
+
language: lang.language,
|
|
161
|
+
key: 'coverage',
|
|
162
|
+
message: `language has no eventNameTranslations table (native event words fall back to English)`,
|
|
163
|
+
});
|
|
164
|
+
return;
|
|
165
|
+
}
|
|
166
|
+
const events = lang.dictionary?.events;
|
|
167
|
+
if (!events) return;
|
|
168
|
+
|
|
169
|
+
// Invert S5b: English event → set of native forms.
|
|
170
|
+
const englishToNatives = new Map<string, Set<string>>();
|
|
171
|
+
for (const [native, english] of Object.entries(lang.eventTranslations)) {
|
|
172
|
+
let set = englishToNatives.get(english);
|
|
173
|
+
if (!set) englishToNatives.set(english, (set = new Set()));
|
|
174
|
+
set.add(norm(native));
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
for (const [english, native] of Object.entries(events)) {
|
|
178
|
+
const natives = englishToNatives.get(english);
|
|
179
|
+
if (!natives) {
|
|
180
|
+
// V3c — S5b has NO entry for this event, so V3 above is blind here: a
|
|
181
|
+
// broken dictionary event word hides exactly in this branch (id
|
|
182
|
+
// tekan_mouse/lepas_mouse — corpus-live, zh 鼠标进入, fr sourisappuyée).
|
|
183
|
+
// Verify the dict form still resolves to the canonical event on the
|
|
184
|
+
// parse side. Warn-tier: never gates, no waivers.
|
|
185
|
+
checkUncoveredEvent(lang, english, native, findings);
|
|
186
|
+
continue;
|
|
187
|
+
}
|
|
188
|
+
if (!natives.has(norm(native))) {
|
|
189
|
+
findings.push({
|
|
190
|
+
check: 'V3',
|
|
191
|
+
tier: 'error',
|
|
192
|
+
language: lang.language,
|
|
193
|
+
key: english,
|
|
194
|
+
message: `event "${english}": dictionary renders "${native}" but eventNameTranslations only recognizes ${[...natives].map(n => `"${n}"`).join(', ')}`,
|
|
195
|
+
source: `dictionary.events.${english}`,
|
|
196
|
+
});
|
|
197
|
+
}
|
|
198
|
+
}
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
/**
|
|
202
|
+
* V3c — coverage companion to V3: a dictionary event whose English key is
|
|
203
|
+
* ABSENT from eventNameTranslations. V3 compares only covered events, so a
|
|
204
|
+
* dict form here can be arbitrarily broken without any signal (the S5b
|
|
205
|
+
* coverage-gap family: V3's own `continue` branch). Resolution order follows
|
|
206
|
+
* the Batch 2 verdict — the TOKENIZER keyword table is the parse authority
|
|
207
|
+
* (clears deliberate mappings like sw panya_juu→mouseup); S5b under another
|
|
208
|
+
* key is the wrong-event signal (zh 鼠标进入→mouseover). Denylisted pairs are
|
|
209
|
+
* intentionally English — never flagged.
|
|
210
|
+
*/
|
|
211
|
+
function checkUncoveredEvent(
|
|
212
|
+
lang: LangVocab,
|
|
213
|
+
english: string,
|
|
214
|
+
native: string,
|
|
215
|
+
findings: Finding[]
|
|
216
|
+
): void {
|
|
217
|
+
if (lang.eventDenylist?.has(english)) return;
|
|
218
|
+
const dictNorm = norm(native);
|
|
219
|
+
const englishNorm = norm(english);
|
|
220
|
+
if (dictNorm === englishNorm) return; // English passthrough always round-trips
|
|
221
|
+
|
|
222
|
+
// 1) Parse authority: single-token keyword-table normalization.
|
|
223
|
+
let resolved: string | undefined;
|
|
224
|
+
if (!/\s/.test(native) && lang.normalizeWord) {
|
|
225
|
+
const n = lang.normalizeWord(native);
|
|
226
|
+
if (n && norm(n) !== dictNorm) resolved = norm(n);
|
|
227
|
+
}
|
|
228
|
+
// 2) S5b under a DIFFERENT English key (underscore↔space tolerant).
|
|
229
|
+
if (!resolved && lang.eventTranslations) {
|
|
230
|
+
const spaced = dictNorm.replace(/_/g, ' ');
|
|
231
|
+
for (const [nat, eng] of Object.entries(lang.eventTranslations)) {
|
|
232
|
+
const natNorm = norm(nat);
|
|
233
|
+
if (natNorm === dictNorm || natNorm === spaced) {
|
|
234
|
+
resolved = norm(eng);
|
|
235
|
+
break;
|
|
236
|
+
}
|
|
237
|
+
}
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
if (resolved === englishNorm) return; // round-trips; S5b just lacks the row (alias candidate)
|
|
241
|
+
|
|
242
|
+
findings.push({
|
|
243
|
+
check: 'V3c',
|
|
244
|
+
tier: 'warn',
|
|
245
|
+
language: lang.language,
|
|
246
|
+
key: english,
|
|
247
|
+
message: resolved
|
|
248
|
+
? `event "${english}": dictionary renders "${native}" but the parse side resolves it to "${resolved}" (S5b has no ${english} entry — V3-invisible) — wrong-event listener risk`
|
|
249
|
+
: `event "${english}": dictionary renders "${native}" which the parse side does not resolve to any event (S5b has no ${english} entry — V3-invisible) — broken-listener risk`,
|
|
250
|
+
source: `dictionary.events.${english}`,
|
|
251
|
+
});
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
/** V4 — every parse/render vocab word must classify as keyword/particle in that language's tokenizer. */
|
|
255
|
+
function checkTokenizerClassification(lang: LangVocab, findings: Finding[]): void {
|
|
256
|
+
if (!lang.classify) return;
|
|
257
|
+
|
|
258
|
+
// (word → sources) — dedupe so one bad word yields one finding.
|
|
259
|
+
const words = new Map<string, string[]>();
|
|
260
|
+
const collect = (word: string | undefined, source: string) => {
|
|
261
|
+
if (!word) return;
|
|
262
|
+
if (/\s/.test(word)) return; // classifyToken is word-level; multi-word forms are out of its scope
|
|
263
|
+
const list = words.get(word);
|
|
264
|
+
if (list) list.push(source);
|
|
265
|
+
else words.set(word, [source]);
|
|
266
|
+
};
|
|
267
|
+
|
|
268
|
+
for (const [concept, entry] of Object.entries(lang.keywords)) {
|
|
269
|
+
collect(entry.primary, `profile.keywords.${concept}`);
|
|
270
|
+
for (const a of entry.alternatives ?? []) collect(a, `profile.keywords.${concept}`);
|
|
271
|
+
}
|
|
272
|
+
for (const [role, entry] of Object.entries(lang.roleMarkers)) {
|
|
273
|
+
collect(entry.primary, `profile.roleMarkers.${role}`);
|
|
274
|
+
for (const a of entry.alternatives ?? []) collect(a, `profile.roleMarkers.${role}`);
|
|
275
|
+
}
|
|
276
|
+
for (const sm of lang.schemaMarkers) {
|
|
277
|
+
collect(sm.marker, `schema.${sm.action}.${sm.role}`);
|
|
278
|
+
}
|
|
279
|
+
for (const gm of lang.grammarMarkers) {
|
|
280
|
+
collect(gm.form, `grammar.${gm.role}`);
|
|
281
|
+
for (const a of gm.alternatives ?? []) collect(a, `grammar.${gm.role}`);
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
for (const [word, sources] of words) {
|
|
285
|
+
const kind = lang.classify(word);
|
|
286
|
+
if (!ACCEPTED_TOKEN_KINDS.has(kind)) {
|
|
287
|
+
findings.push({
|
|
288
|
+
check: 'V4',
|
|
289
|
+
tier: 'error',
|
|
290
|
+
language: lang.language,
|
|
291
|
+
key: word,
|
|
292
|
+
message: `"${word}" classifies as '${kind}' (not keyword/particle) — it cannot tokenize as vocabulary`,
|
|
293
|
+
source:
|
|
294
|
+
sources.slice(0, 3).join(', ') +
|
|
295
|
+
(sources.length > 3 ? ` (+${sources.length - 3} more)` : ''),
|
|
296
|
+
});
|
|
297
|
+
}
|
|
298
|
+
}
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
export function runChecks(model: VocabModel, checks?: readonly CheckId[]): Finding[] {
|
|
302
|
+
const enabled = (id: CheckId) => !checks || checks.includes(id);
|
|
303
|
+
const findings: Finding[] = [];
|
|
304
|
+
for (const lang of model.languages) {
|
|
305
|
+
if (enabled('V1') || enabled('V1b')) checkKeywords(lang, findings);
|
|
306
|
+
if (enabled('V2')) checkRoleMarkers(lang, findings);
|
|
307
|
+
if (enabled('V3') || enabled('V3b') || enabled('V3c')) checkEventNames(lang, findings);
|
|
308
|
+
if (enabled('V4')) checkTokenizerClassification(lang, findings);
|
|
309
|
+
}
|
|
310
|
+
return checks ? findings.filter(f => checks.includes(f.check)) : findings;
|
|
311
|
+
}
|