@hyperfixi/testing-framework 3.1.1 → 3.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +186 -1
- package/dist/index.js +5 -1
- package/dist/index.js.map +1 -1
- package/dist/index.mjs +5 -1
- package/dist/index.mjs.map +1 -1
- package/dist/runner.js +5 -1
- package/dist/runner.js.map +1 -1
- package/dist/runner.mjs +5 -1
- package/dist/runner.mjs.map +1 -1
- package/package.json +21 -19
- package/src/multilingual/README.md +39 -0
- package/src/multilingual/en-reference-equivalences.test.ts +321 -0
- package/src/multilingual/en-reference-preservation.test.ts +124 -0
- package/src/multilingual/en-reference-preservation.ts +495 -0
- package/src/multilingual/engine-parser-parity.test.ts +67 -0
- package/src/multilingual/pattern-loader.test.ts +43 -0
- package/src/multilingual/pattern-loader.ts +7 -2
- package/src/multilingual/shipped-examples-execution.test.ts +76 -1
- package/src/multilingual/shipped-examples-execution.ts +73 -3
- package/src/multilingual/shipped-sources-engine.test.ts +106 -0
- package/src/multilingual/shipped-sources-validity.ts +65 -0
- package/src/multilingual/validators/execution-validator.test.ts +32 -3
- package/src/multilingual/validators/execution-validator.ts +22 -5
- package/src/multilingual/value-matrix-gate.ts +118 -0
- package/src/multilingual/value-matrix.accepted.test.ts +65 -0
- package/src/multilingual/value-matrix.assign.test.ts +13 -0
- package/src/multilingual/value-matrix.chain-phrases.test.ts +13 -0
- package/src/multilingual/value-matrix.chain.test.ts +13 -0
- package/src/multilingual/value-matrix.count.test.ts +13 -0
- package/src/multilingual/value-matrix.get-phrases.test.ts +13 -0
- package/src/multilingual/value-matrix.get.test.ts +13 -0
- package/src/multilingual/value-matrix.if-phrases.test.ts +13 -0
- package/src/multilingual/value-matrix.if.test.ts +13 -0
- package/src/multilingual/value-matrix.increment.test.ts +13 -0
- package/src/multilingual/value-matrix.isolation.test.ts +57 -0
- package/src/multilingual/value-matrix.names.test.ts +62 -0
- package/src/multilingual/value-matrix.put-phrases.test.ts +13 -0
- package/src/multilingual/value-matrix.put.test.ts +13 -0
- package/src/multilingual/value-matrix.set-phrases.test.ts +13 -0
- package/src/multilingual/value-matrix.set.test.ts +13 -0
- package/src/multilingual/value-matrix.times.test.ts +13 -0
- package/src/multilingual/value-matrix.ts +1208 -0
- package/src/multilingual/value-matrix.while.test.ts +13 -0
- package/src/runner.ts +7 -1
|
@@ -0,0 +1,495 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* En-reference preservation gate
|
|
3
|
+
* ------------------------------
|
|
4
|
+
* WHY THIS EXISTS
|
|
5
|
+
* Every stored translation is `render(parse_en(src), L)`, and every other
|
|
6
|
+
* multilingual signal — the eleven `--regression` ratchets, render-fidelity, and
|
|
7
|
+
* foreign-canonical-validity (R4) — scores a language against, or renders from,
|
|
8
|
+
* that same English parse. So a construct the ENGLISH parse drops is dropped in
|
|
9
|
+
* all 23 languages at once while every signal stays green: en defines the
|
|
10
|
+
* reference. `canonical-validity` does render en→en, but only asks whether the
|
|
11
|
+
* output PARSES on the real engine, and only for rows that engine accepts.
|
|
12
|
+
* Measured 2026-09-25: 24 corpus rows lose content or change meaning this way
|
|
13
|
+
* (repeat-while lost `< 10`, so its loop no longer terminates; `go back` rendered
|
|
14
|
+
* `go url back`, which navigates to a page called "back") — and one of them,
|
|
15
|
+
* morph-form-update, was introduced by #1167 with CI fully green.
|
|
16
|
+
*
|
|
17
|
+
* WHAT IT ASSERTS
|
|
18
|
+
* For every translatable corpus row — each plain row, and each `_="…"` body of a
|
|
19
|
+
* translatable markup row (exactly the bodies the corpus writer renders) — the
|
|
20
|
+
* English re-render `render(parse_en(src), 'en')` carries the source's content:
|
|
21
|
+
* it equals the source under the NAMED EQUIVALENCES below, ignoring whitespace.
|
|
22
|
+
* Failures are recorded against a committed allowlist that only shrinks (the R4
|
|
23
|
+
* discipline): a new offender fails, an allowlisted unit whose render changed
|
|
24
|
+
* fails (re-triage it), and an entry that now passes must be pruned.
|
|
25
|
+
*
|
|
26
|
+
* NAMED EQUIVALENCES
|
|
27
|
+
* The renderer legitimately respells some constructs. Each respelling allowed
|
|
28
|
+
* here is listed in `EQUIVALENCES`, and each one is pinned in
|
|
29
|
+
* `en-reference-equivalences.test.ts` by showing that both spellings are the
|
|
30
|
+
* same program on the real `hyperscript.org` engine (identical parse trees, or
|
|
31
|
+
* identical effects where the node types differ). A respelling that is not in
|
|
32
|
+
* the list is a difference. The pins are why the list is narrow: dropping `the`
|
|
33
|
+
* is NOT allowed in general, because `halt the event` is valid and
|
|
34
|
+
* `halt event` is not.
|
|
35
|
+
*
|
|
36
|
+
* Whitespace is ignored everywhere, including inside string literals — the same
|
|
37
|
+
* known limitation as the writer's own guard (`reRenderPreservesContent`): it can
|
|
38
|
+
* hide a re-spaced string, never a dropped token.
|
|
39
|
+
*/
|
|
40
|
+
import {
|
|
41
|
+
findHyperscriptAttributes,
|
|
42
|
+
getAllPatterns,
|
|
43
|
+
isMarkupRow,
|
|
44
|
+
type Pattern,
|
|
45
|
+
} from '@hyperfixi/patterns-reference';
|
|
46
|
+
import { parseSemantic, render } from '@lokascript/semantic';
|
|
47
|
+
|
|
48
|
+
// =============================================================================
|
|
49
|
+
// Named equivalences
|
|
50
|
+
// =============================================================================
|
|
51
|
+
|
|
52
|
+
export interface Equivalence {
|
|
53
|
+
/** Stable id; `en-reference-equivalences.test.ts` pins each one. */
|
|
54
|
+
readonly id: string;
|
|
55
|
+
/** What the renderer respells, and why the two spellings are one program. */
|
|
56
|
+
readonly description: string;
|
|
57
|
+
/** A source spelling and the rendered spelling it must compare equal to. */
|
|
58
|
+
readonly example: readonly [source: string, rendered: string];
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
export const EQUIVALENCES: readonly Equivalence[] = [
|
|
62
|
+
{
|
|
63
|
+
id: 'then-separator',
|
|
64
|
+
description: '`then` between commands is an optional separator',
|
|
65
|
+
example: ['on click toggle .a add .b to me', 'on click toggle .a then add .b to me'],
|
|
66
|
+
},
|
|
67
|
+
{
|
|
68
|
+
id: 'quote-style',
|
|
69
|
+
description: 'a single-quoted string literal is the same literal double-quoted',
|
|
70
|
+
example: ["on click put 'Saved!' into me", 'on click put "Saved!" into me'],
|
|
71
|
+
},
|
|
72
|
+
{
|
|
73
|
+
id: 'quoted-url',
|
|
74
|
+
description:
|
|
75
|
+
'a naked URL is the same value as the quoted URL (only for URL-shaped text with no ' +
|
|
76
|
+
'whitespace and no `${…}` — interpolation is exactly where quoting could matter)',
|
|
77
|
+
example: ['on click fetch /api/data', 'on click fetch "/api/data"'],
|
|
78
|
+
},
|
|
79
|
+
{
|
|
80
|
+
id: 'article-before-query',
|
|
81
|
+
description: '`a`/`an` directly before a query literal (`make a <div/>`) is an article',
|
|
82
|
+
example: ['on click make a <div.card/>', 'on click make <div.card/>'],
|
|
83
|
+
},
|
|
84
|
+
{
|
|
85
|
+
id: 'the-before-positional',
|
|
86
|
+
description: '`the` directly before `next`/`previous` is an article',
|
|
87
|
+
example: ['on click show the next <div/>', 'on click show next <div/>'],
|
|
88
|
+
},
|
|
89
|
+
{
|
|
90
|
+
id: 'the-before-target',
|
|
91
|
+
description: '`the` directly before `target` is an article',
|
|
92
|
+
example: ['on click log the target', 'on click log target'],
|
|
93
|
+
},
|
|
94
|
+
{
|
|
95
|
+
id: 'dotted-possessive',
|
|
96
|
+
description: '`my.x` is `my x`; `it.x` and `its.x` are `its x`',
|
|
97
|
+
example: ['on click put it.name into #o', 'on click put its name into #o'],
|
|
98
|
+
},
|
|
99
|
+
{
|
|
100
|
+
id: 'of-possessive',
|
|
101
|
+
description: "`the X of #id` is `#id's X`",
|
|
102
|
+
example: ['on click set the *color of #t to "red"', 'on click set #t\'s *color to "red"'],
|
|
103
|
+
},
|
|
104
|
+
{
|
|
105
|
+
id: 'go-to-url',
|
|
106
|
+
description: '`to` in `go to url` is optional',
|
|
107
|
+
example: ['on click go to url "/page"', 'on click go url "/page"'],
|
|
108
|
+
},
|
|
109
|
+
{
|
|
110
|
+
id: 'with-object-braces',
|
|
111
|
+
description: 'naked named arguments after `with` are the braced object literal',
|
|
112
|
+
example: [
|
|
113
|
+
'on click fetch /x with method:"POST", body:form',
|
|
114
|
+
'on click fetch /x with {method:"POST", body:form}',
|
|
115
|
+
],
|
|
116
|
+
},
|
|
117
|
+
{
|
|
118
|
+
id: 'trailing-end',
|
|
119
|
+
description: 'an `end` at end of input is optional (end of input closes every open block)',
|
|
120
|
+
example: ['on click repeat 3 times log 1', 'on click repeat 3 times log 1 end'],
|
|
121
|
+
},
|
|
122
|
+
{
|
|
123
|
+
id: 'settle-me',
|
|
124
|
+
description: "`settle`'s target defaults to `me`",
|
|
125
|
+
example: ['on click settle', 'on click settle me'],
|
|
126
|
+
},
|
|
127
|
+
{
|
|
128
|
+
id: 'pseudo-command-me',
|
|
129
|
+
description: 'the pseudo-command `m() me` is `call me.m()`',
|
|
130
|
+
example: ['on load click() me', 'on load call me.click()'],
|
|
131
|
+
},
|
|
132
|
+
{
|
|
133
|
+
id: 'quoted-event-name',
|
|
134
|
+
description:
|
|
135
|
+
"`send`/`trigger`'s event name may be quoted: a string whose text is a plain name is " +
|
|
136
|
+
'that name (upstream reads either as its eventName; only a plain name, since text ' +
|
|
137
|
+
'that is not one cannot be written bare)',
|
|
138
|
+
example: ['on click send "hello" to ChatSocket', 'on click send hello to ChatSocket'],
|
|
139
|
+
},
|
|
140
|
+
{
|
|
141
|
+
id: 'handler-from-me',
|
|
142
|
+
description:
|
|
143
|
+
"a handler head's `from me` is its default source: the handler listens on `me` " +
|
|
144
|
+
'either way (only in a head — `on <event>[(params)][filter] from me` — never a ' +
|
|
145
|
+
"command's `from me`, where `take`/`remove` give it meaning)",
|
|
146
|
+
example: ['on pointerdown(y) from me log y', 'on pointerdown(y) log y'],
|
|
147
|
+
},
|
|
148
|
+
];
|
|
149
|
+
|
|
150
|
+
// =============================================================================
|
|
151
|
+
// Normalization
|
|
152
|
+
// =============================================================================
|
|
153
|
+
|
|
154
|
+
type Token =
|
|
155
|
+
| { readonly kind: 'string'; readonly quote: '"' | "'" | '`'; readonly body: string }
|
|
156
|
+
| { readonly kind: 'word'; readonly text: string };
|
|
157
|
+
|
|
158
|
+
/**
|
|
159
|
+
* Characters that always stand alone, so `with {` and `click()` split cleanly
|
|
160
|
+
* (and a re-spaced `left:` / `left :` is one token sequence, which keeps
|
|
161
|
+
* `describeDifference` quiet about pure re-spacing).
|
|
162
|
+
*/
|
|
163
|
+
const PUNCTUATION = new Set(['(', ')', '{', '}', '[', ']', ',', ':', ';']);
|
|
164
|
+
|
|
165
|
+
/**
|
|
166
|
+
* A `'` right after a word character or a closing bracket is a possessive
|
|
167
|
+
* (`#price's value`, `<form/>'s`), not the start of a string literal.
|
|
168
|
+
*/
|
|
169
|
+
function isApostrophe(src: string, index: number): boolean {
|
|
170
|
+
const previous = src[index - 1];
|
|
171
|
+
return previous !== undefined && /[\w)\]>}]/.test(previous);
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
/** Split hyperscript into string literals and whitespace/punctuation-delimited words. */
|
|
175
|
+
export function tokenize(src: string): Token[] {
|
|
176
|
+
const out: Token[] = [];
|
|
177
|
+
let word = '';
|
|
178
|
+
const flush = () => {
|
|
179
|
+
if (word) out.push({ kind: 'word', text: word });
|
|
180
|
+
word = '';
|
|
181
|
+
};
|
|
182
|
+
for (let i = 0; i < src.length; i++) {
|
|
183
|
+
const c = src[i]!;
|
|
184
|
+
if (/\s/.test(c)) {
|
|
185
|
+
flush();
|
|
186
|
+
} else if (c === '"' || c === '`' || (c === "'" && !isApostrophe(src, i))) {
|
|
187
|
+
flush();
|
|
188
|
+
let body = '';
|
|
189
|
+
let j = i + 1;
|
|
190
|
+
while (j < src.length && src[j] !== c) {
|
|
191
|
+
if (src[j] === '\\' && j + 1 < src.length) {
|
|
192
|
+
body += src[j]! + src[j + 1]!;
|
|
193
|
+
j += 2;
|
|
194
|
+
} else {
|
|
195
|
+
body += src[j]!;
|
|
196
|
+
j++;
|
|
197
|
+
}
|
|
198
|
+
}
|
|
199
|
+
out.push({ kind: 'string', quote: c, body });
|
|
200
|
+
i = j;
|
|
201
|
+
} else if (PUNCTUATION.has(c)) {
|
|
202
|
+
flush();
|
|
203
|
+
out.push({ kind: 'word', text: c });
|
|
204
|
+
} else {
|
|
205
|
+
word += c;
|
|
206
|
+
}
|
|
207
|
+
}
|
|
208
|
+
flush();
|
|
209
|
+
return out;
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
type WordToken = Extract<Token, { kind: 'word' }>;
|
|
213
|
+
|
|
214
|
+
/** Type guard only. (Folding a text check into a predicate would make its FALSE branch
|
|
215
|
+
* claim "not a word at all" and mis-narrow every later check.) */
|
|
216
|
+
const isWord = (t: Token | undefined): t is WordToken => t?.kind === 'word';
|
|
217
|
+
|
|
218
|
+
/** Whether `t` is the word `text`. Deliberately not a type predicate — see isWord. */
|
|
219
|
+
const wordIs = (t: Token | undefined, text: string): boolean => isWord(t) && t.text === text;
|
|
220
|
+
|
|
221
|
+
const URL_SHAPED = /^(?:\/|https?:\/\/|wss?:\/\/)[^\s"'`${}]*$/;
|
|
222
|
+
|
|
223
|
+
/** Index of the `close` matching the `open` at `start`, or -1. */
|
|
224
|
+
function matchingIndex(tokens: Token[], start: number, open: string, close: string): number {
|
|
225
|
+
let depth = 0;
|
|
226
|
+
for (let i = start; i < tokens.length; i++) {
|
|
227
|
+
if (wordIs(tokens[i], open)) depth++;
|
|
228
|
+
else if (wordIs(tokens[i], close) && --depth === 0) return i;
|
|
229
|
+
}
|
|
230
|
+
return -1;
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
/**
|
|
234
|
+
* Apply every named equivalence, in a fixed order, and return the canonical
|
|
235
|
+
* token list. Order matters only where two rules read the same word: the
|
|
236
|
+
* of-possessive needs its `the` before `the-before-positional` could drop it.
|
|
237
|
+
*/
|
|
238
|
+
export function canonicalTokens(src: string): Token[] {
|
|
239
|
+
let t = tokenize(src);
|
|
240
|
+
|
|
241
|
+
// of-possessive: `the X of #id` → `#id's X`
|
|
242
|
+
for (let i = 0; i + 3 < t.length; i++) {
|
|
243
|
+
const [a, x, of, owner] = [t[i], t[i + 1], t[i + 2], t[i + 3]];
|
|
244
|
+
if (
|
|
245
|
+
wordIs(a, 'the') &&
|
|
246
|
+
isWord(x) &&
|
|
247
|
+
/^\*?[\w-]+$/.test(x.text) &&
|
|
248
|
+
wordIs(of, 'of') &&
|
|
249
|
+
isWord(owner) &&
|
|
250
|
+
/^#[\w-]+$/.test(owner.text)
|
|
251
|
+
) {
|
|
252
|
+
t.splice(i, 4, { kind: 'word', text: `${owner.text}'s` }, x);
|
|
253
|
+
}
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
const out: Token[] = [];
|
|
257
|
+
for (let i = 0; i < t.length; i++) {
|
|
258
|
+
const tok = t[i]!;
|
|
259
|
+
const next = t[i + 1];
|
|
260
|
+
// then-separator
|
|
261
|
+
if (wordIs(tok, 'then')) continue;
|
|
262
|
+
// article-before-query
|
|
263
|
+
if ((wordIs(tok, 'a') || wordIs(tok, 'an')) && isWord(next) && next.text.startsWith('<')) {
|
|
264
|
+
continue;
|
|
265
|
+
}
|
|
266
|
+
// the-before-positional
|
|
267
|
+
if (wordIs(tok, 'the') && (wordIs(next, 'next') || wordIs(next, 'previous'))) continue;
|
|
268
|
+
// the-before-target (`the target.closest("li")` is `target.closest("li")`)
|
|
269
|
+
if (wordIs(tok, 'the') && isWord(next) && /^target(?:\.|$)/.test(next.text)) continue;
|
|
270
|
+
// go-to-url
|
|
271
|
+
if (wordIs(tok, 'to') && wordIs(out[out.length - 1], 'go') && wordIs(next, 'url')) continue;
|
|
272
|
+
// dotted-possessive
|
|
273
|
+
if (isWord(tok)) {
|
|
274
|
+
const m = /^(my|its|it)\.(.+)$/.exec(tok.text);
|
|
275
|
+
if (m) {
|
|
276
|
+
out.push(
|
|
277
|
+
{ kind: 'word', text: m[1] === 'my' ? 'my' : 'its' },
|
|
278
|
+
{ kind: 'word', text: m[2]! }
|
|
279
|
+
);
|
|
280
|
+
continue;
|
|
281
|
+
}
|
|
282
|
+
}
|
|
283
|
+
// settle-me
|
|
284
|
+
if (wordIs(tok, 'me') && wordIs(out[out.length - 1], 'settle')) continue;
|
|
285
|
+
// quoted-event-name
|
|
286
|
+
if (
|
|
287
|
+
tok.kind === 'string' &&
|
|
288
|
+
tok.quote !== '`' &&
|
|
289
|
+
(wordIs(out[out.length - 1], 'send') || wordIs(out[out.length - 1], 'trigger')) &&
|
|
290
|
+
/^[A-Za-z_$][\w$]*$/.test(tok.body)
|
|
291
|
+
) {
|
|
292
|
+
out.push({ kind: 'word', text: tok.body });
|
|
293
|
+
continue;
|
|
294
|
+
}
|
|
295
|
+
out.push(tok);
|
|
296
|
+
}
|
|
297
|
+
t = out;
|
|
298
|
+
|
|
299
|
+
// handler-from-me: `on <event> [( … )] [[ … ]] from me` → the same head without
|
|
300
|
+
// it. Only at a feature position: the start, or after an `end` or a header's `)`.
|
|
301
|
+
for (let i = 0; i + 1 < t.length; i++) {
|
|
302
|
+
const previous = t[i - 1];
|
|
303
|
+
const atFeature = !previous || wordIs(previous, 'end') || wordIs(previous, ')');
|
|
304
|
+
if (!atFeature || !wordIs(t[i], 'on') || !isWord(t[i + 1])) continue;
|
|
305
|
+
let k = i + 2;
|
|
306
|
+
if (wordIs(t[k], '(')) k = matchingIndex(t, k, '(', ')') + 1;
|
|
307
|
+
if (k > 0 && wordIs(t[k], '[')) k = matchingIndex(t, k, '[', ']') + 1;
|
|
308
|
+
if (k > 0 && wordIs(t[k], 'from') && wordIs(t[k + 1], 'me')) t.splice(k, 2);
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
// with-object-braces: `with { … }` → `with …`
|
|
312
|
+
for (let i = 0; i + 1 < t.length; i++) {
|
|
313
|
+
if (wordIs(t[i], 'with') && wordIs(t[i + 1], '{')) {
|
|
314
|
+
const close = matchingIndex(t, i + 1, '{', '}');
|
|
315
|
+
if (close > 0) {
|
|
316
|
+
t.splice(close, 1);
|
|
317
|
+
t.splice(i + 1, 1);
|
|
318
|
+
}
|
|
319
|
+
}
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
// pseudo-command-me: `m ( ) me` → `call me.m ( )`
|
|
323
|
+
for (let i = 0; i + 3 < t.length; i++) {
|
|
324
|
+
const [name, open, close, target] = [t[i], t[i + 1], t[i + 2], t[i + 3]];
|
|
325
|
+
if (
|
|
326
|
+
isWord(name) &&
|
|
327
|
+
/^[A-Za-z_]\w*$/.test(name.text) &&
|
|
328
|
+
wordIs(open, '(') &&
|
|
329
|
+
wordIs(close, ')') &&
|
|
330
|
+
wordIs(target, 'me')
|
|
331
|
+
) {
|
|
332
|
+
t.splice(
|
|
333
|
+
i,
|
|
334
|
+
4,
|
|
335
|
+
{ kind: 'word', text: 'call' },
|
|
336
|
+
{ kind: 'word', text: `me.${name.text}` },
|
|
337
|
+
{ kind: 'word', text: '(' },
|
|
338
|
+
{ kind: 'word', text: ')' }
|
|
339
|
+
);
|
|
340
|
+
}
|
|
341
|
+
}
|
|
342
|
+
|
|
343
|
+
// trailing-end
|
|
344
|
+
while (wordIs(t[t.length - 1], 'end')) t.pop();
|
|
345
|
+
|
|
346
|
+
return t;
|
|
347
|
+
}
|
|
348
|
+
|
|
349
|
+
/** One token's canonical spelling (quote-style and quoted-url apply here). */
|
|
350
|
+
function spell(tok: Token): string {
|
|
351
|
+
if (tok.kind === 'word') return tok.text;
|
|
352
|
+
if (tok.quote === '`') return `\`${tok.body}\``;
|
|
353
|
+
if (URL_SHAPED.test(tok.body)) return tok.body;
|
|
354
|
+
return `"${tok.body}"`;
|
|
355
|
+
}
|
|
356
|
+
|
|
357
|
+
/** The comparison form: canonical tokens, all whitespace removed. */
|
|
358
|
+
export function normalizeForComparison(src: string): string {
|
|
359
|
+
return canonicalTokens(src).map(spell).join('').replace(/\s+/g, '');
|
|
360
|
+
}
|
|
361
|
+
|
|
362
|
+
/** Whether `rendered` carries all of `source`'s content, under the named equivalences. */
|
|
363
|
+
export function preservesContent(source: string, rendered: string): boolean {
|
|
364
|
+
return normalizeForComparison(source) === normalizeForComparison(rendered);
|
|
365
|
+
}
|
|
366
|
+
|
|
367
|
+
/**
|
|
368
|
+
* What changed, as runs of canonical tokens: `lost` are in the source and not
|
|
369
|
+
* the render, `added` the reverse. For messages and the allowlist only — the
|
|
370
|
+
* verdict is `preservesContent`.
|
|
371
|
+
*/
|
|
372
|
+
export function describeDifference(
|
|
373
|
+
source: string,
|
|
374
|
+
rendered: string
|
|
375
|
+
): { lost: string[]; added: string[] } {
|
|
376
|
+
const a = canonicalTokens(source).map(spell);
|
|
377
|
+
const b = canonicalTokens(rendered).map(spell);
|
|
378
|
+
// Longest-common-subsequence table (corpus rows are well under 300 tokens).
|
|
379
|
+
const lcs: number[][] = Array.from({ length: a.length + 1 }, () =>
|
|
380
|
+
new Array<number>(b.length + 1).fill(0)
|
|
381
|
+
);
|
|
382
|
+
for (let i = a.length - 1; i >= 0; i--) {
|
|
383
|
+
for (let j = b.length - 1; j >= 0; j--) {
|
|
384
|
+
lcs[i]![j] =
|
|
385
|
+
a[i] === b[j] ? lcs[i + 1]![j + 1]! + 1 : Math.max(lcs[i + 1]![j]!, lcs[i]![j + 1]!);
|
|
386
|
+
}
|
|
387
|
+
}
|
|
388
|
+
const lost: string[] = [];
|
|
389
|
+
const added: string[] = [];
|
|
390
|
+
let runLost: string[] = [];
|
|
391
|
+
let runAdded: string[] = [];
|
|
392
|
+
const flush = () => {
|
|
393
|
+
if (runLost.length) lost.push(runLost.join(' '));
|
|
394
|
+
if (runAdded.length) added.push(runAdded.join(' '));
|
|
395
|
+
runLost = [];
|
|
396
|
+
runAdded = [];
|
|
397
|
+
};
|
|
398
|
+
let i = 0;
|
|
399
|
+
let j = 0;
|
|
400
|
+
while (i < a.length || j < b.length) {
|
|
401
|
+
if (i < a.length && j < b.length && a[i] === b[j]) {
|
|
402
|
+
flush();
|
|
403
|
+
i++;
|
|
404
|
+
j++;
|
|
405
|
+
} else if (j >= b.length || (i < a.length && lcs[i + 1]![j]! >= lcs[i]![j + 1]!)) {
|
|
406
|
+
runLost.push(a[i++]!);
|
|
407
|
+
} else {
|
|
408
|
+
runAdded.push(b[j++]!);
|
|
409
|
+
}
|
|
410
|
+
}
|
|
411
|
+
flush();
|
|
412
|
+
return { lost, added };
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
// =============================================================================
|
|
416
|
+
// The corpus check
|
|
417
|
+
// =============================================================================
|
|
418
|
+
|
|
419
|
+
/** One piece of hyperscript the corpus writer renders. */
|
|
420
|
+
export interface PreservationUnit {
|
|
421
|
+
/** `<pattern id>` for a plain row; `<pattern id>#<n>` for a markup row's n-th `_` body. */
|
|
422
|
+
readonly key: string;
|
|
423
|
+
readonly id: string;
|
|
424
|
+
readonly source: string;
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
export interface PreservationFailure extends PreservationUnit {
|
|
428
|
+
/** The English re-render, or null when the parse or render failed outright. */
|
|
429
|
+
readonly rendered: string | null;
|
|
430
|
+
readonly lost: string[];
|
|
431
|
+
readonly added: string[];
|
|
432
|
+
}
|
|
433
|
+
|
|
434
|
+
export interface PreservationResult {
|
|
435
|
+
/** Units checked (translatable plain rows + translatable markup bodies). */
|
|
436
|
+
checked: number;
|
|
437
|
+
preserved: number;
|
|
438
|
+
failures: PreservationFailure[];
|
|
439
|
+
}
|
|
440
|
+
|
|
441
|
+
/**
|
|
442
|
+
* The units the corpus writer renders: every translatable plain row, and each
|
|
443
|
+
* `_` body of a translatable markup row. Non-translatable rows are copied
|
|
444
|
+
* verbatim into every language, so nothing they contain can be lost.
|
|
445
|
+
*/
|
|
446
|
+
export function collectUnits(
|
|
447
|
+
patterns: ReadonlyArray<Pick<Pattern, 'id' | 'rawCode' | 'translatable'>>
|
|
448
|
+
): PreservationUnit[] {
|
|
449
|
+
const units: PreservationUnit[] = [];
|
|
450
|
+
for (const p of patterns) {
|
|
451
|
+
if (!p.translatable) continue;
|
|
452
|
+
if (isMarkupRow(p.rawCode)) {
|
|
453
|
+
findHyperscriptAttributes(p.rawCode).forEach((span, n) => {
|
|
454
|
+
if (span.body.trim()) units.push({ key: `${p.id}#${n}`, id: p.id, source: span.body });
|
|
455
|
+
});
|
|
456
|
+
} else {
|
|
457
|
+
units.push({ key: p.id, id: p.id, source: p.rawCode });
|
|
458
|
+
}
|
|
459
|
+
}
|
|
460
|
+
return units;
|
|
461
|
+
}
|
|
462
|
+
|
|
463
|
+
/** Render `source` to English through its own English parse; null on failure. */
|
|
464
|
+
export function renderEnglish(source: string): string | null {
|
|
465
|
+
try {
|
|
466
|
+
const node = parseSemantic(source, 'en').node;
|
|
467
|
+
return node ? render(node, 'en') : null;
|
|
468
|
+
} catch {
|
|
469
|
+
return null;
|
|
470
|
+
}
|
|
471
|
+
}
|
|
472
|
+
|
|
473
|
+
export async function checkEnReferencePreservation(opts?: {
|
|
474
|
+
patterns?: ReadonlyArray<Pick<Pattern, 'id' | 'rawCode' | 'translatable'>>;
|
|
475
|
+
}): Promise<PreservationResult> {
|
|
476
|
+
const patterns = opts?.patterns ?? (await getAllPatterns({ limit: 1000 }));
|
|
477
|
+
const units = collectUnits(patterns);
|
|
478
|
+
const failures: PreservationFailure[] = [];
|
|
479
|
+
let preserved = 0;
|
|
480
|
+
|
|
481
|
+
for (const unit of units) {
|
|
482
|
+
const rendered = renderEnglish(unit.source);
|
|
483
|
+
if (rendered !== null && preservesContent(unit.source, rendered)) {
|
|
484
|
+
preserved++;
|
|
485
|
+
continue;
|
|
486
|
+
}
|
|
487
|
+
const diff =
|
|
488
|
+
rendered === null
|
|
489
|
+
? { lost: [unit.source], added: [] }
|
|
490
|
+
: describeDifference(unit.source, rendered);
|
|
491
|
+
failures.push({ ...unit, rendered, ...diff });
|
|
492
|
+
}
|
|
493
|
+
|
|
494
|
+
return { checked: units.length, preserved, failures };
|
|
495
|
+
}
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Engine parser parity: the canonical-validity gates' strings on `@hyperfixi/engine`.
|
|
3
|
+
*
|
|
4
|
+
* Every English string those gates put to upstream `hyperscript.org`'s parser — each corpus
|
|
5
|
+
* row, its English re-render, and the English render of every authored translation — is put
|
|
6
|
+
* to the new engine's parser as well. The two must agree on whether it parses. The engine is
|
|
7
|
+
* meant to replace core's, and the multilingual product reaches an engine as English text, so
|
|
8
|
+
* a string upstream reads and the engine rejects (or the reverse) is a translation that works
|
|
9
|
+
* on one host and not the other.
|
|
10
|
+
*
|
|
11
|
+
* Unlike the foreign canonical-validity gate, this needs no fresh `populate`: it compares two
|
|
12
|
+
* parsers on the SAME strings, so a stale patterns.db changes which strings are asked, not
|
|
13
|
+
* whether the parsers agree. It always runs.
|
|
14
|
+
*
|
|
15
|
+
* On a failure, run one string on both engines:
|
|
16
|
+
* `npx tsx packages/engine/tools/probe.mts '<source>'`.
|
|
17
|
+
*/
|
|
18
|
+
import { describe, it, expect, beforeAll } from 'vitest';
|
|
19
|
+
import {
|
|
20
|
+
checkCorpusRenderValidity,
|
|
21
|
+
loadCanonicalParser,
|
|
22
|
+
type CanonicalValidate,
|
|
23
|
+
} from './canonical-validity';
|
|
24
|
+
import { checkForeignRenderValidity } from './foreign-canonical-validity';
|
|
25
|
+
|
|
26
|
+
describe('engine parser parity (R4 strings on @hyperfixi/engine)', () => {
|
|
27
|
+
const asked = new Set<string>();
|
|
28
|
+
const disagreements: string[] = [];
|
|
29
|
+
|
|
30
|
+
beforeAll(async () => {
|
|
31
|
+
const { api, everything, register } = await import('@hyperfixi/engine');
|
|
32
|
+
register(...everything);
|
|
33
|
+
const upstream = await loadCanonicalParser();
|
|
34
|
+
const engine: CanonicalValidate = source => {
|
|
35
|
+
try {
|
|
36
|
+
return api.parse(source).errors.map(error => error.message);
|
|
37
|
+
} catch (e) {
|
|
38
|
+
return ['threw: ' + (e instanceof Error ? e.message.split('\n')[0] : String(e))];
|
|
39
|
+
}
|
|
40
|
+
};
|
|
41
|
+
// Upstream's verdict is the one handed back, so the gates' own logic is unchanged.
|
|
42
|
+
const both: CanonicalValidate = source => {
|
|
43
|
+
const up = upstream(source);
|
|
44
|
+
if (!asked.has(source)) {
|
|
45
|
+
asked.add(source);
|
|
46
|
+
const mine = engine(source);
|
|
47
|
+
if ((up.length === 0) !== (mine.length === 0)) {
|
|
48
|
+
disagreements.push(
|
|
49
|
+
`${JSON.stringify(source)}\n upstream: ${up[0]?.split('\n')[0] ?? 'accepts'}` +
|
|
50
|
+
`\n engine: ${mine[0]?.split('\n')[0] ?? 'accepts'}`
|
|
51
|
+
);
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
return up;
|
|
55
|
+
};
|
|
56
|
+
await checkCorpusRenderValidity({ validate: both });
|
|
57
|
+
await checkForeignRenderValidity({ validate: both });
|
|
58
|
+
}, 300_000);
|
|
59
|
+
|
|
60
|
+
it('asks about a real corpus (sanity: patterns.db has rows)', () => {
|
|
61
|
+
expect(asked.size).toBeGreaterThan(100);
|
|
62
|
+
});
|
|
63
|
+
|
|
64
|
+
it('the two parsers agree on every string', () => {
|
|
65
|
+
expect(disagreements).toEqual([]);
|
|
66
|
+
});
|
|
67
|
+
});
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* What the multilingual sweep grades: translations only.
|
|
3
|
+
*
|
|
4
|
+
* A non-translatable row is stored as written in every language. For a markup
|
|
5
|
+
* row the shape check dropped it; a PLAIN non-translatable row
|
|
6
|
+
* (intercept-cache-strategies, from 2026-09-25) would otherwise be graded as
|
|
7
|
+
* the English source under every other language's parser.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
import { describe, expect, it, vi } from 'vitest';
|
|
11
|
+
import type { TestConfig } from './types';
|
|
12
|
+
|
|
13
|
+
const row = (codeExampleId: string, language: string, hyperscript: string, method: string) => ({
|
|
14
|
+
codeExampleId,
|
|
15
|
+
language,
|
|
16
|
+
hyperscript,
|
|
17
|
+
translationMethod: method,
|
|
18
|
+
wordOrder: 'SOV',
|
|
19
|
+
confidence: 1,
|
|
20
|
+
verifiedParses: true,
|
|
21
|
+
});
|
|
22
|
+
|
|
23
|
+
vi.mock('@hyperfixi/patterns-reference', () => ({
|
|
24
|
+
getTranslationsByLanguage: async () => [
|
|
25
|
+
row('toggle', 'ja', '.active を 切り替え', 'semantic-render'),
|
|
26
|
+
row('intercept', 'ja', 'intercept /\nend', 'non-translatable-identity'),
|
|
27
|
+
row('wiring', 'ja', '<div sse-connect="/e"></div>', 'non-translatable-identity'),
|
|
28
|
+
],
|
|
29
|
+
getVerifiedTranslations: async () => [],
|
|
30
|
+
getHighConfidenceTranslations: async () => [],
|
|
31
|
+
getAllPatterns: async () => [],
|
|
32
|
+
getPatternStats: async () => ({ byLanguage: { ja: {} } }),
|
|
33
|
+
}));
|
|
34
|
+
|
|
35
|
+
const { loadPatterns } = await import('./pattern-loader');
|
|
36
|
+
|
|
37
|
+
describe('loadPatterns', () => {
|
|
38
|
+
it('grades translations, not rows copied as written', async () => {
|
|
39
|
+
const config: TestConfig = { languages: ['ja'], mode: 'full' };
|
|
40
|
+
const loaded = await loadPatterns(config);
|
|
41
|
+
expect(loaded.map(p => p.codeExampleId)).toEqual(['toggle']);
|
|
42
|
+
});
|
|
43
|
+
});
|
|
@@ -63,8 +63,13 @@ async function loadTranslationsForLanguage(
|
|
|
63
63
|
translations = await getTranslationsByLanguage(language, 1000);
|
|
64
64
|
}
|
|
65
65
|
|
|
66
|
-
//
|
|
67
|
-
|
|
66
|
+
// A non-translatable row is stored as written in every language: a copy, not
|
|
67
|
+
// a translation. Grading it would score the English source under another
|
|
68
|
+
// language's parser. (Markup rows are dropped by shape in loadPatterns; this
|
|
69
|
+
// also covers plain rows, such as intercept-cache-strategies.)
|
|
70
|
+
return translations
|
|
71
|
+
.filter(t => t.translationMethod !== 'non-translatable-identity')
|
|
72
|
+
.map(t => mapToPatternTranslation(t));
|
|
68
73
|
}
|
|
69
74
|
|
|
70
75
|
/**
|