@hyperfixi/testing-framework 2.7.2 → 2.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +393 -0
- package/dist/assertions.d.mts +26 -1
- package/dist/assertions.d.ts +26 -1
- package/dist/index.d.mts +5 -112
- package/dist/index.d.ts +5 -112
- package/dist/runner.d.mts +112 -0
- package/dist/runner.d.ts +112 -0
- package/dist/runner.js +1102 -0
- package/dist/runner.js.map +1 -0
- package/dist/runner.mjs +1097 -0
- package/dist/runner.mjs.map +1 -0
- package/dist/{assertions-CsGP61iW.d.mts → types-D-rCVkf3.d.mts} +1 -24
- package/dist/{assertions-CsGP61iW.d.ts → types-D-rCVkf3.d.ts} +1 -24
- package/package.json +13 -27
- package/src/multilingual/canonical-validity.test.ts +69 -0
- package/src/multilingual/canonical-validity.ts +132 -0
- package/src/multilingual/cli.ts +247 -14
- package/src/multilingual/fidelity.test.ts +192 -0
- package/src/multilingual/fidelity.ts +153 -0
- package/src/multilingual/foreign-canonical-validity.test.ts +0 -0
- package/src/multilingual/foreign-canonical-validity.ts +158 -0
- package/src/multilingual/orchestrator.ts +47 -1
- package/src/multilingual/reporters/console-reporter.ts +51 -0
- package/src/multilingual/reporters/regression-reporter.test.ts +78 -0
- package/src/multilingual/reporters/regression-reporter.ts +28 -1
- package/src/multilingual/tools/diagnose-coverage.ts +118 -0
- package/src/multilingual/tools/triage-r1.ts +149 -0
- package/src/multilingual/types.ts +74 -0
- package/src/multilingual/validators/parse-validator.ts +9 -1
- package/src/runner.test.ts +7 -2
- package/src/vocab/batch3-roundtrip.test.ts +184 -0
- package/src/vocab/checks.test.ts +362 -0
- package/src/vocab/checks.ts +311 -0
- package/src/vocab/cli.ts +196 -0
- package/src/vocab/dump.ts +80 -0
- package/src/vocab/model.ts +110 -0
- package/src/vocab/report.ts +120 -0
- package/src/vocab/types.ts +100 -0
package/src/vocab/cli.ts
ADDED
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Vocab-consistency CLI (Arc A — docs-internal/HANDOFF_vocab-consistency.md).
|
|
3
|
+
*
|
|
4
|
+
* npx tsx src/vocab/cli.ts validate [--language xx[,yy]] [--check V1,V4]
|
|
5
|
+
* [--json [path]] [--waivers path] [--warn-only]
|
|
6
|
+
* npx tsx src/vocab/cli.ts dump [keywords|markers|events] [--concept toggle]
|
|
7
|
+
* [--language xx[,yy]] [--format md|tsv|json]
|
|
8
|
+
*
|
|
9
|
+
* Exit codes (validate): 0 = no unwaived error-tier findings (or --warn-only);
|
|
10
|
+
* 1 = unwaived errors; 2 = refused (stale dist).
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import * as fs from 'node:fs';
|
|
14
|
+
import * as path from 'node:path';
|
|
15
|
+
import { fileURLToPath } from 'node:url';
|
|
16
|
+
import type { CheckId } from './types';
|
|
17
|
+
import { loadWaivers, buildLedger, printLedger } from './report';
|
|
18
|
+
import type { DumpFormat, DumpTarget } from './dump';
|
|
19
|
+
|
|
20
|
+
// --- stale-dist guard (same pattern as the multilingual CLI) -----------------
|
|
21
|
+
// The model loads @lokascript/semantic and @lokascript/i18n through their
|
|
22
|
+
// package entries, i.e. their dist/. A stale dist scores code that differs
|
|
23
|
+
// from the checkout, so refuse to run rather than report vacuously.
|
|
24
|
+
|
|
25
|
+
const DIST_GUARD_PACKAGES = ['intent', 'framework', 'semantic', 'i18n'];
|
|
26
|
+
|
|
27
|
+
function hasNewerTs(dir: string, builtAt: number): boolean {
|
|
28
|
+
if (!fs.existsSync(dir)) return false;
|
|
29
|
+
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
|
|
30
|
+
const p = path.join(dir, entry.name);
|
|
31
|
+
if (entry.isDirectory()) {
|
|
32
|
+
if (hasNewerTs(p, builtAt)) return true;
|
|
33
|
+
} else if (entry.name.endsWith('.ts') && fs.statSync(p).mtimeMs > builtAt) {
|
|
34
|
+
return true;
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
return false;
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
function findStaleDists(): string[] {
|
|
41
|
+
const here = path.dirname(fileURLToPath(import.meta.url));
|
|
42
|
+
const packagesRoot = path.resolve(here, '../../..');
|
|
43
|
+
const stale: string[] = [];
|
|
44
|
+
for (const name of DIST_GUARD_PACKAGES) {
|
|
45
|
+
const srcDir = path.join(packagesRoot, name, 'src');
|
|
46
|
+
const marker = path.join(packagesRoot, name, 'dist', 'index.js');
|
|
47
|
+
if (!fs.existsSync(marker) || hasNewerTs(srcDir, fs.statSync(marker).mtimeMs)) {
|
|
48
|
+
stale.push(name);
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
return stale;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
// --- arg parsing --------------------------------------------------------------
|
|
55
|
+
|
|
56
|
+
interface Args {
|
|
57
|
+
command: 'validate' | 'dump';
|
|
58
|
+
languages?: string[] | undefined;
|
|
59
|
+
checks?: CheckId[] | undefined;
|
|
60
|
+
json?: string | true | undefined;
|
|
61
|
+
waivers: string;
|
|
62
|
+
warnOnly: boolean;
|
|
63
|
+
target: DumpTarget;
|
|
64
|
+
concept?: string | undefined;
|
|
65
|
+
format: DumpFormat;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
const VALID_CHECKS: readonly CheckId[] = ['V1', 'V1b', 'V2', 'V3', 'V3b', 'V3c', 'V4'];
|
|
69
|
+
|
|
70
|
+
function parseArgs(argv: string[]): Args {
|
|
71
|
+
const here = path.dirname(fileURLToPath(import.meta.url));
|
|
72
|
+
const args: Args = {
|
|
73
|
+
command: 'validate',
|
|
74
|
+
waivers: path.resolve(here, '../../vocab-waivers.json'),
|
|
75
|
+
warnOnly: false,
|
|
76
|
+
target: 'keywords',
|
|
77
|
+
format: 'md',
|
|
78
|
+
};
|
|
79
|
+
const rest = [...argv];
|
|
80
|
+
while (rest.length > 0) {
|
|
81
|
+
const a = rest.shift()!;
|
|
82
|
+
switch (a) {
|
|
83
|
+
case 'validate':
|
|
84
|
+
args.command = 'validate';
|
|
85
|
+
break;
|
|
86
|
+
case 'dump':
|
|
87
|
+
args.command = 'dump';
|
|
88
|
+
break;
|
|
89
|
+
case 'keywords':
|
|
90
|
+
case 'markers':
|
|
91
|
+
case 'events':
|
|
92
|
+
args.target = a;
|
|
93
|
+
break;
|
|
94
|
+
case '--language':
|
|
95
|
+
case '-l':
|
|
96
|
+
case '--languages':
|
|
97
|
+
args.languages = (rest.shift() ?? '').split(',').filter(Boolean);
|
|
98
|
+
break;
|
|
99
|
+
case '--check':
|
|
100
|
+
case '--checks': {
|
|
101
|
+
const ids = (rest.shift() ?? '').split(',').filter(Boolean) as CheckId[];
|
|
102
|
+
for (const id of ids) {
|
|
103
|
+
if (!VALID_CHECKS.includes(id))
|
|
104
|
+
throw new Error(`unknown check "${id}" (valid: ${VALID_CHECKS.join(', ')})`);
|
|
105
|
+
}
|
|
106
|
+
args.checks = ids;
|
|
107
|
+
break;
|
|
108
|
+
}
|
|
109
|
+
case '--json':
|
|
110
|
+
args.json = rest[0] && !rest[0].startsWith('-') ? rest.shift()! : true;
|
|
111
|
+
break;
|
|
112
|
+
case '--waivers':
|
|
113
|
+
args.waivers = path.resolve(rest.shift() ?? args.waivers);
|
|
114
|
+
break;
|
|
115
|
+
case '--warn-only':
|
|
116
|
+
args.warnOnly = true;
|
|
117
|
+
break;
|
|
118
|
+
case '--concept':
|
|
119
|
+
args.concept = rest.shift();
|
|
120
|
+
break;
|
|
121
|
+
case '--format': {
|
|
122
|
+
const f = rest.shift() as DumpFormat;
|
|
123
|
+
if (!['md', 'tsv', 'json'].includes(f)) throw new Error(`unknown format "${f}"`);
|
|
124
|
+
args.format = f;
|
|
125
|
+
break;
|
|
126
|
+
}
|
|
127
|
+
case '--help':
|
|
128
|
+
case '-h':
|
|
129
|
+
console.log(
|
|
130
|
+
'vocab cli — validate (V1–V4 cross-surface consistency) | dump (concept × language table)\n' +
|
|
131
|
+
'see file header for flags'
|
|
132
|
+
);
|
|
133
|
+
process.exit(0);
|
|
134
|
+
break;
|
|
135
|
+
default:
|
|
136
|
+
throw new Error(`unknown argument "${a}"`);
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
return args;
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
// --- main ----------------------------------------------------------------------
|
|
143
|
+
|
|
144
|
+
async function main(): Promise<void> {
|
|
145
|
+
const args = parseArgs(process.argv.slice(2));
|
|
146
|
+
|
|
147
|
+
const stale = findStaleDists();
|
|
148
|
+
if (stale.length > 0) {
|
|
149
|
+
console.error(
|
|
150
|
+
`REFUSING to run: stale dist for ${stale.join(', ')} (src newer than dist/index.js).\n` +
|
|
151
|
+
`Rebuild first: npm run check:fresh (or npm run build --prefix packages/<pkg>)`
|
|
152
|
+
);
|
|
153
|
+
process.exit(2);
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
// Deferred so the guard runs before the package entries are loaded.
|
|
157
|
+
const { loadVocabModel } = await import('./model');
|
|
158
|
+
const model = loadVocabModel(args.languages);
|
|
159
|
+
|
|
160
|
+
if (args.command === 'dump') {
|
|
161
|
+
const { renderDump } = await import('./dump');
|
|
162
|
+
console.log(renderDump(model, args.target, args.format, args.concept));
|
|
163
|
+
return;
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
const { runChecks } = await import('./checks');
|
|
167
|
+
const findings = runChecks(model, args.checks);
|
|
168
|
+
const waivers = loadWaivers(args.waivers);
|
|
169
|
+
const ledger = buildLedger(findings, waivers);
|
|
170
|
+
|
|
171
|
+
printLedger(ledger);
|
|
172
|
+
|
|
173
|
+
if (args.json) {
|
|
174
|
+
const target = args.json === true ? undefined : args.json;
|
|
175
|
+
const payload = JSON.stringify(ledger, null, 2);
|
|
176
|
+
if (target) {
|
|
177
|
+
fs.writeFileSync(target, payload);
|
|
178
|
+
console.log(`ledger written: ${target}`);
|
|
179
|
+
} else {
|
|
180
|
+
console.log(payload);
|
|
181
|
+
}
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
if (ledger.unwaivedErrors > 0 && !args.warnOnly) {
|
|
185
|
+
console.error(`\n✗ ${ledger.unwaivedErrors} unwaived error(s)`);
|
|
186
|
+
process.exit(1);
|
|
187
|
+
}
|
|
188
|
+
console.log(
|
|
189
|
+
'\n✓ vocab consistency: no unwaived errors' + (args.warnOnly ? ' enforced (--warn-only)' : '')
|
|
190
|
+
);
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
main().catch(err => {
|
|
194
|
+
console.error(err instanceof Error ? err.message : err);
|
|
195
|
+
process.exit(2);
|
|
196
|
+
});
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `dump` — one concept × language table over the loaded vocab model, so a
|
|
3
|
+
* translation can be reviewed in a single view instead of five files.
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
import type { VocabModel } from './types';
|
|
7
|
+
|
|
8
|
+
export type DumpTarget = 'keywords' | 'markers' | 'events';
|
|
9
|
+
export type DumpFormat = 'md' | 'tsv' | 'json';
|
|
10
|
+
|
|
11
|
+
interface Table {
|
|
12
|
+
header: string[];
|
|
13
|
+
rows: string[][];
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
function buildTable(model: VocabModel, target: DumpTarget, conceptFilter?: string): Table {
|
|
17
|
+
const langs = model.languages.map(l => l.language);
|
|
18
|
+
const header = [
|
|
19
|
+
target === 'markers' ? 'role' : target === 'events' ? 'event' : 'concept',
|
|
20
|
+
...langs,
|
|
21
|
+
];
|
|
22
|
+
|
|
23
|
+
const keys = new Set<string>();
|
|
24
|
+
for (const lang of model.languages) {
|
|
25
|
+
if (target === 'keywords') Object.keys(lang.keywords).forEach(k => keys.add(k));
|
|
26
|
+
else if (target === 'markers') Object.keys(lang.roleMarkers).forEach(k => keys.add(k));
|
|
27
|
+
else for (const english of Object.values(lang.eventTranslations ?? {})) keys.add(english);
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
const rows: string[][] = [];
|
|
31
|
+
for (const key of [...keys].sort()) {
|
|
32
|
+
if (conceptFilter && key !== conceptFilter) continue;
|
|
33
|
+
const row = [key];
|
|
34
|
+
for (const lang of model.languages) {
|
|
35
|
+
if (target === 'keywords') {
|
|
36
|
+
const e = lang.keywords[key];
|
|
37
|
+
row.push(
|
|
38
|
+
e ? e.primary + (e.alternatives?.length ? ` (${e.alternatives.join('|')})` : '') : ''
|
|
39
|
+
);
|
|
40
|
+
} else if (target === 'markers') {
|
|
41
|
+
const e = lang.roleMarkers[key];
|
|
42
|
+
row.push(
|
|
43
|
+
e ? e.primary + (e.alternatives?.length ? ` (${e.alternatives.join('|')})` : '') : ''
|
|
44
|
+
);
|
|
45
|
+
} else {
|
|
46
|
+
const natives = Object.entries(lang.eventTranslations ?? {})
|
|
47
|
+
.filter(([, english]) => english === key)
|
|
48
|
+
.map(([native]) => native);
|
|
49
|
+
row.push(natives.join('|'));
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
rows.push(row);
|
|
53
|
+
}
|
|
54
|
+
return { header, rows };
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
export function renderDump(
|
|
58
|
+
model: VocabModel,
|
|
59
|
+
target: DumpTarget,
|
|
60
|
+
format: DumpFormat,
|
|
61
|
+
conceptFilter?: string
|
|
62
|
+
): string {
|
|
63
|
+
const { header, rows } = buildTable(model, target, conceptFilter);
|
|
64
|
+
if (format === 'json') {
|
|
65
|
+
return JSON.stringify(
|
|
66
|
+
rows.map(r => Object.fromEntries(header.map((h, i) => [h, r[i]]))),
|
|
67
|
+
null,
|
|
68
|
+
2
|
|
69
|
+
);
|
|
70
|
+
}
|
|
71
|
+
if (format === 'tsv') {
|
|
72
|
+
return [header.join('\t'), ...rows.map(r => r.join('\t'))].join('\n');
|
|
73
|
+
}
|
|
74
|
+
const md = [
|
|
75
|
+
`| ${header.join(' | ')} |`,
|
|
76
|
+
`| ${header.map(() => '---').join(' | ')} |`,
|
|
77
|
+
...rows.map(r => `| ${r.map(c => c.replace(/\|/g, '\\|')).join(' | ')} |`),
|
|
78
|
+
];
|
|
79
|
+
return md.join('\n');
|
|
80
|
+
}
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Loads the real five-surface vocab model.
|
|
3
|
+
*
|
|
4
|
+
* S1/S2/S5 come from `@lokascript/semantic`, S3/S4 from `@lokascript/i18n` —
|
|
5
|
+
* all via the package entries, so a stale `dist/` scores code that differs
|
|
6
|
+
* from the checkout (the CLI guards this before loading).
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import {
|
|
10
|
+
KNOWN_PROFILES,
|
|
11
|
+
commandSchemas,
|
|
12
|
+
getTokenizer,
|
|
13
|
+
eventNameTranslations,
|
|
14
|
+
getSOVEventMarkers,
|
|
15
|
+
getEventLocalizationDenylist,
|
|
16
|
+
} from '@lokascript/semantic';
|
|
17
|
+
import { dictionaries, profiles as grammarProfiles } from '@lokascript/i18n';
|
|
18
|
+
import type { LangVocab, VocabModel } from './types';
|
|
19
|
+
|
|
20
|
+
export function loadVocabModel(languageFilter?: readonly string[]): VocabModel {
|
|
21
|
+
const codes = Object.keys(KNOWN_PROFILES).filter(
|
|
22
|
+
code => !languageFilter || languageFilter.includes(code)
|
|
23
|
+
);
|
|
24
|
+
const sovEventMarkers = getSOVEventMarkers();
|
|
25
|
+
const eventDenylists = getEventLocalizationDenylist();
|
|
26
|
+
|
|
27
|
+
const languages: LangVocab[] = [];
|
|
28
|
+
for (const code of codes) {
|
|
29
|
+
const profile = KNOWN_PROFILES[code];
|
|
30
|
+
if (!profile) continue;
|
|
31
|
+
|
|
32
|
+
const keywords: LangVocab['keywords'] = {};
|
|
33
|
+
for (const [concept, t] of Object.entries(profile.keywords ?? {})) {
|
|
34
|
+
keywords[concept] = { primary: t.primary, alternatives: t.alternatives };
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
const roleMarkers: LangVocab['roleMarkers'] = {};
|
|
38
|
+
for (const [role, m] of Object.entries(profile.roleMarkers ?? {})) {
|
|
39
|
+
if (m) roleMarkers[role] = { primary: m.primary, alternatives: m.alternatives };
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
const schemaMarkers: LangVocab['schemaMarkers'] = [];
|
|
43
|
+
for (const [action, schema] of Object.entries(commandSchemas)) {
|
|
44
|
+
for (const spec of schema.roles ?? []) {
|
|
45
|
+
const override = spec.markerOverride?.[code];
|
|
46
|
+
if (override) {
|
|
47
|
+
schemaMarkers.push({ action, role: spec.role, marker: override, kind: 'override' });
|
|
48
|
+
}
|
|
49
|
+
for (const variant of spec.markerVariants?.[code] ?? []) {
|
|
50
|
+
if (variant)
|
|
51
|
+
schemaMarkers.push({ action, role: spec.role, marker: variant, kind: 'variant' });
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
const grammar = (
|
|
57
|
+
grammarProfiles as Record<string, { markers?: readonly unknown[] } | undefined>
|
|
58
|
+
)[code];
|
|
59
|
+
const grammarMarkers: LangVocab['grammarMarkers'] = [];
|
|
60
|
+
for (const raw of grammar?.markers ?? []) {
|
|
61
|
+
const m = raw as {
|
|
62
|
+
form: string;
|
|
63
|
+
role: string;
|
|
64
|
+
position?: string;
|
|
65
|
+
alternatives?: readonly string[];
|
|
66
|
+
};
|
|
67
|
+
if (m.form)
|
|
68
|
+
grammarMarkers.push({
|
|
69
|
+
form: m.form,
|
|
70
|
+
role: m.role,
|
|
71
|
+
position: m.position,
|
|
72
|
+
alternatives: m.alternatives,
|
|
73
|
+
});
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
const dictionary = (
|
|
77
|
+
dictionaries as unknown as Record<string, Record<string, Record<string, string>> | undefined>
|
|
78
|
+
)[code];
|
|
79
|
+
|
|
80
|
+
const tokenizer = getTokenizer(code);
|
|
81
|
+
|
|
82
|
+
languages.push({
|
|
83
|
+
language: code,
|
|
84
|
+
keywords,
|
|
85
|
+
roleMarkers,
|
|
86
|
+
schemaMarkers,
|
|
87
|
+
dictionary,
|
|
88
|
+
grammarMarkers,
|
|
89
|
+
eventTranslations: eventNameTranslations[code],
|
|
90
|
+
sovEventMarkers: sovEventMarkers[code],
|
|
91
|
+
classify: tokenizer ? (word: string) => tokenizer.classifyToken(word) : undefined,
|
|
92
|
+
normalizeWord: tokenizer
|
|
93
|
+
? (word: string) => {
|
|
94
|
+
// Parse-authority resolution: what the keyword table turns this
|
|
95
|
+
// word into. Only trustworthy when the word survives as ONE token
|
|
96
|
+
// (a shattered compound resolves to its first fragment — that is
|
|
97
|
+
// the broken-listener class itself, not a resolution).
|
|
98
|
+
const stream = tokenizer.tokenize(word) as {
|
|
99
|
+
tokens?: readonly { normalized?: string }[];
|
|
100
|
+
};
|
|
101
|
+
const toks = stream.tokens ?? [];
|
|
102
|
+
return toks.length === 1 ? toks[0]?.normalized : undefined;
|
|
103
|
+
}
|
|
104
|
+
: undefined,
|
|
105
|
+
eventDenylist: eventDenylists[code],
|
|
106
|
+
});
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
return { languages };
|
|
110
|
+
}
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Waiver application + console/JSON reporting for the vocab checks.
|
|
3
|
+
*
|
|
4
|
+
* Waivers apply to error-tier findings only (warn/info never gate), keyed by
|
|
5
|
+
* `check|language|key` with a mandatory reason. A waiver that matches nothing
|
|
6
|
+
* is stale and reported so the file cannot rot silently.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import * as fs from 'node:fs';
|
|
10
|
+
import type { Finding, Tier, Waiver } from './types';
|
|
11
|
+
|
|
12
|
+
export interface Ledger {
|
|
13
|
+
generatedAt: string;
|
|
14
|
+
totals: Record<Tier, number>;
|
|
15
|
+
unwaivedErrors: number;
|
|
16
|
+
staleWaivers: string[];
|
|
17
|
+
findings: Array<Finding & { waived?: string | undefined }>;
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
export function loadWaivers(path: string): Waiver[] {
|
|
21
|
+
if (!fs.existsSync(path)) return [];
|
|
22
|
+
const raw = JSON.parse(fs.readFileSync(path, 'utf8')) as Waiver[];
|
|
23
|
+
for (const w of raw) {
|
|
24
|
+
if (!w.key || !w.reason) throw new Error(`waiver missing key or reason: ${JSON.stringify(w)}`);
|
|
25
|
+
}
|
|
26
|
+
return raw;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* A waiver key is `check|language|key` where any segment may be `*` — a class
|
|
31
|
+
* waiver covering every finding in that class (e.g. `V1|*|*` = "all V1s,
|
|
32
|
+
* pending Arc B"). Exact keys still work; first matching waiver wins.
|
|
33
|
+
*/
|
|
34
|
+
function waiverMatches(waiverKey: string, f: Finding): boolean {
|
|
35
|
+
const parts = waiverKey.split('|');
|
|
36
|
+
if (parts.length !== 3) return false;
|
|
37
|
+
const [c, l, k] = parts;
|
|
38
|
+
return (
|
|
39
|
+
(c === '*' || c === f.check) && (l === '*' || l === f.language) && (k === '*' || k === f.key)
|
|
40
|
+
);
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
export function buildLedger(findings: Finding[], waivers: Waiver[]): Ledger {
|
|
44
|
+
const used = new Set<string>();
|
|
45
|
+
|
|
46
|
+
const annotated: Array<Finding & { waived?: string | undefined }> = findings.map(f => {
|
|
47
|
+
const w = f.tier === 'error' ? waivers.find(w => waiverMatches(w.key, f)) : undefined;
|
|
48
|
+
if (w) used.add(w.key);
|
|
49
|
+
return w ? { ...f, waived: w.reason } : { ...f };
|
|
50
|
+
});
|
|
51
|
+
|
|
52
|
+
const totals: Record<Tier, number> = { error: 0, warn: 0, info: 0 };
|
|
53
|
+
for (const f of annotated) totals[f.tier]++;
|
|
54
|
+
|
|
55
|
+
return {
|
|
56
|
+
generatedAt: new Date().toISOString(),
|
|
57
|
+
totals,
|
|
58
|
+
unwaivedErrors: annotated.filter(f => f.tier === 'error' && !f.waived).length,
|
|
59
|
+
staleWaivers: waivers.map(w => w.key).filter(k => !used.has(k)),
|
|
60
|
+
findings: annotated,
|
|
61
|
+
};
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
const PRINT_CAPS: Record<Tier, number> = { error: 50, warn: 20, info: 0 };
|
|
65
|
+
|
|
66
|
+
export function printLedger(ledger: Ledger): void {
|
|
67
|
+
const { totals } = ledger;
|
|
68
|
+
console.log('Vocab consistency — cross-surface check (V1–V4)');
|
|
69
|
+
console.log(
|
|
70
|
+
` errors: ${totals.error} (${ledger.unwaivedErrors} unwaived) · warns: ${totals.warn} · infos: ${totals.info}\n`
|
|
71
|
+
);
|
|
72
|
+
|
|
73
|
+
// Per-check × per-language matrix of error/warn counts.
|
|
74
|
+
const cells = new Map<string, { error: number; warn: number; info: number }>();
|
|
75
|
+
for (const f of ledger.findings) {
|
|
76
|
+
const key = `${f.check}|${f.language}`;
|
|
77
|
+
const cell = cells.get(key) ?? { error: 0, warn: 0, info: 0 };
|
|
78
|
+
cell[f.tier]++;
|
|
79
|
+
cells.set(key, cell);
|
|
80
|
+
}
|
|
81
|
+
const checks = [...new Set(ledger.findings.map(f => f.check))].sort();
|
|
82
|
+
const langs = [...new Set(ledger.findings.map(f => f.language))].sort();
|
|
83
|
+
if (checks.length > 0) {
|
|
84
|
+
console.log(` ${'lang'.padEnd(6)}${checks.map(c => c.padEnd(14)).join('')}`);
|
|
85
|
+
for (const lang of langs) {
|
|
86
|
+
const row = checks
|
|
87
|
+
.map(c => {
|
|
88
|
+
const cell = cells.get(`${c}|${lang}`);
|
|
89
|
+
if (!cell) return '—'.padEnd(14);
|
|
90
|
+
return `${cell.error}e/${cell.warn}w/${cell.info}i`.padEnd(14);
|
|
91
|
+
})
|
|
92
|
+
.join('');
|
|
93
|
+
console.log(` ${lang.padEnd(6)}${row}`);
|
|
94
|
+
}
|
|
95
|
+
console.log('');
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
for (const tier of ['error', 'warn'] as const) {
|
|
99
|
+
const cap = PRINT_CAPS[tier];
|
|
100
|
+
const list = ledger.findings.filter(f => f.tier === tier && !f.waived);
|
|
101
|
+
if (list.length === 0) continue;
|
|
102
|
+
console.log(
|
|
103
|
+
` ${tier.toUpperCase()}S${list.length > cap ? ` (first ${cap} of ${list.length} — full list via --json)` : ''}:`
|
|
104
|
+
);
|
|
105
|
+
for (const f of list.slice(0, cap)) {
|
|
106
|
+
console.log(
|
|
107
|
+
` [${f.check}|${f.language}|${f.key}] ${f.message}${f.source ? ` (${f.source})` : ''}`
|
|
108
|
+
);
|
|
109
|
+
}
|
|
110
|
+
console.log('');
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
const waived = ledger.findings.filter(f => f.waived);
|
|
114
|
+
if (waived.length > 0) console.log(` waived errors: ${waived.length}`);
|
|
115
|
+
if (ledger.staleWaivers.length > 0) {
|
|
116
|
+
console.log(
|
|
117
|
+
` STALE waivers (match no finding — remove them): ${ledger.staleWaivers.join(', ')}`
|
|
118
|
+
);
|
|
119
|
+
}
|
|
120
|
+
}
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Vocab-consistency model + finding types (Arc A — HANDOFF_vocab-consistency.md).
|
|
3
|
+
*
|
|
4
|
+
* Checks operate on a `VocabModel`, not on the packages directly, so tests can
|
|
5
|
+
* seed synthetic disagreements without importing the real surfaces.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
export type Tier = 'error' | 'warn' | 'info';
|
|
9
|
+
|
|
10
|
+
export type CheckId = 'V1' | 'V1b' | 'V2' | 'V3' | 'V3b' | 'V3c' | 'V4';
|
|
11
|
+
|
|
12
|
+
export interface Finding {
|
|
13
|
+
check: CheckId;
|
|
14
|
+
tier: Tier;
|
|
15
|
+
language: string;
|
|
16
|
+
/** Stable identifier within (check, language) — concept, role, event or word. */
|
|
17
|
+
key: string;
|
|
18
|
+
message: string;
|
|
19
|
+
/** Where the offending entry lives (e.g. `schema.bind.source`, `grammar.destination`). */
|
|
20
|
+
source?: string | undefined;
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
/** Stable waiver key: check|language|key. */
|
|
24
|
+
export function findingKey(f: Pick<Finding, 'check' | 'language' | 'key'>): string {
|
|
25
|
+
return `${f.check}|${f.language}|${f.key}`;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
export interface Waiver {
|
|
29
|
+
key: string;
|
|
30
|
+
reason: string;
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/** One language's view over the five authoring surfaces. */
|
|
34
|
+
export interface LangVocab {
|
|
35
|
+
language: string;
|
|
36
|
+
/** S1 — semantic profile keywords: concept → forms. */
|
|
37
|
+
keywords: Record<string, { primary: string; alternatives?: readonly string[] | undefined }>;
|
|
38
|
+
/** S1 — semantic profile role markers: role → forms. */
|
|
39
|
+
roleMarkers: Record<string, { primary: string; alternatives?: readonly string[] | undefined }>;
|
|
40
|
+
/** S2 — command-schema per-language markers (flattened over all schemas/roles). */
|
|
41
|
+
schemaMarkers: Array<{
|
|
42
|
+
action: string;
|
|
43
|
+
role: string;
|
|
44
|
+
marker: string;
|
|
45
|
+
kind: 'override' | 'variant';
|
|
46
|
+
}>;
|
|
47
|
+
/** S3 — i18n dictionary: category → English key → native value. Absent if the language has no dictionary. */
|
|
48
|
+
dictionary?: Record<string, Record<string, string>> | undefined;
|
|
49
|
+
/** S4 — i18n grammar-profile markers (render side). */
|
|
50
|
+
grammarMarkers: Array<{
|
|
51
|
+
form: string;
|
|
52
|
+
role: string;
|
|
53
|
+
position?: string | undefined;
|
|
54
|
+
alternatives?: readonly string[] | undefined;
|
|
55
|
+
}>;
|
|
56
|
+
/** S5b — native event word → English event name. Absent for the 11 uncovered languages. */
|
|
57
|
+
eventTranslations?: Record<string, string> | undefined;
|
|
58
|
+
/**
|
|
59
|
+
* Surface #6 — hardcoded SOV event markers from `semantic-parser.ts`
|
|
60
|
+
* (`getSOVEventMarkers()`): parse-side event-marker knowledge that lives in
|
|
61
|
+
* neither profiles, schemas, nor grammar profiles. Feeds V2's parse-side
|
|
62
|
+
* union for the `event` role only.
|
|
63
|
+
*/
|
|
64
|
+
sovEventMarkers?: readonly string[] | undefined;
|
|
65
|
+
/** S5a — word-level tokenizer classification. Absent if no tokenizer registered. */
|
|
66
|
+
classify?: ((word: string) => string) | undefined;
|
|
67
|
+
/**
|
|
68
|
+
* S5a — tokenizer keyword-table normalization for a SINGLE word: what the
|
|
69
|
+
* parse side actually resolves the word to (undefined when the word doesn't
|
|
70
|
+
* tokenize as exactly one token, or has no normalization). The parse
|
|
71
|
+
* authority for V3c — S5b can be aspirational (Batch 2).
|
|
72
|
+
*/
|
|
73
|
+
normalizeWord?: ((word: string) => string | undefined) | undefined;
|
|
74
|
+
/**
|
|
75
|
+
* English events INTENTIONALLY kept English for this language (the
|
|
76
|
+
* test-locked `eventLocalizationDenylist` from `@lokascript/semantic`):
|
|
77
|
+
* a dict event word for a denylisted pair is expected not to round-trip —
|
|
78
|
+
* V3c must not flag it.
|
|
79
|
+
*/
|
|
80
|
+
eventDenylist?: ReadonlySet<string> | undefined;
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
export interface VocabModel {
|
|
84
|
+
languages: LangVocab[];
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/** NFC + lowercase + trim — the comparison normal form for every check. */
|
|
88
|
+
export function norm(s: string): string {
|
|
89
|
+
return s.normalize('NFC').toLowerCase().trim();
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
export function formSet(entry: {
|
|
93
|
+
primary: string;
|
|
94
|
+
alternatives?: readonly string[] | undefined;
|
|
95
|
+
}): Set<string> {
|
|
96
|
+
const out = new Set<string>();
|
|
97
|
+
if (entry.primary) out.add(norm(entry.primary));
|
|
98
|
+
for (const a of entry.alternatives ?? []) if (a) out.add(norm(a));
|
|
99
|
+
return out;
|
|
100
|
+
}
|