@lokascript/i18n 2.11.1 → 3.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +13 -0
- package/dist/browser.cjs +5 -1690
- package/dist/browser.cjs.map +1 -1
- package/dist/browser.d.cts +2 -2
- package/dist/browser.d.ts +2 -2
- package/dist/browser.js +6 -1685
- package/dist/browser.js.map +1 -1
- package/dist/dictionaries/index.cjs +5 -1
- package/dist/dictionaries/index.cjs.map +1 -1
- package/dist/dictionaries/index.js +5 -1
- package/dist/dictionaries/index.js.map +1 -1
- package/dist/{transformer-CsOeqayN.d.cts → index-BykxjYST.d.cts} +1 -232
- package/dist/{transformer-DWCTG1DQ.d.ts → index-DuIef8O7.d.ts} +1 -232
- package/dist/index.cjs +16 -1878
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +1 -1
- package/dist/index.d.ts +1 -1
- package/dist/index.js +17 -1873
- package/dist/index.js.map +1 -1
- package/dist/lokascript-i18n.min.js +1 -1
- package/dist/lokascript-i18n.min.js.map +1 -1
- package/dist/lokascript-i18n.mjs +49 -2680
- package/dist/lokascript-i18n.mjs.map +1 -1
- package/dist/plugins/vite.cjs +5 -1
- package/dist/plugins/vite.cjs.map +1 -1
- package/dist/plugins/vite.js +5 -1
- package/dist/plugins/vite.js.map +1 -1
- package/dist/plugins/webpack.cjs +5 -1
- package/dist/plugins/webpack.cjs.map +1 -1
- package/dist/plugins/webpack.js +5 -1
- package/dist/plugins/webpack.js.map +1 -1
- package/package.json +4 -4
- package/src/browser.ts +0 -7
- package/src/compatibility/browser-tests/grammar-demo.spec.ts +22 -8
- package/src/constants.ts +1 -0
- package/src/dictionaries/bn.ts +5 -1
- package/src/grammar/index.ts +15 -9
- package/src/grammar/profiles.test.ts +440 -0
- package/src/index.ts +0 -7
- package/src/lexicon-parity.test.ts +77 -0
- package/src/grammar/grammar.test.ts +0 -2751
- package/src/grammar/transformer.ts +0 -2737
|
@@ -1,2737 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Grammar-Aware Transformer
|
|
3
|
-
*
|
|
4
|
-
* Transforms hyperscript statements between languages using the
|
|
5
|
-
* generalized grammar system. The key insight is that semantic
|
|
6
|
-
* roles are universal - only their surface realization differs.
|
|
7
|
-
*/
|
|
8
|
-
|
|
9
|
-
import type {
|
|
10
|
-
LanguageProfile,
|
|
11
|
-
ParsedStatement,
|
|
12
|
-
ParsedElement,
|
|
13
|
-
SemanticRole,
|
|
14
|
-
GrammarRule,
|
|
15
|
-
LineMetadata,
|
|
16
|
-
} from './types';
|
|
17
|
-
import { reorderRoles, insertMarkers, joinTokens } from './types';
|
|
18
|
-
import { getProfile, profiles } from './profiles';
|
|
19
|
-
import { hasDirectMapping, getDirectMapping } from './direct-mappings';
|
|
20
|
-
import { dictionaries } from '../dictionaries';
|
|
21
|
-
import { findInDictionary, translateFromEnglish } from '../types';
|
|
22
|
-
import {
|
|
23
|
-
ENGLISH_MODIFIER_ROLES,
|
|
24
|
-
ENGLISH_COMMANDS,
|
|
25
|
-
COMMAND_PRIMARY_ROLES,
|
|
26
|
-
CONDITIONAL_KEYWORDS,
|
|
27
|
-
THEN_KEYWORDS,
|
|
28
|
-
} from '../constants';
|
|
29
|
-
|
|
30
|
-
// =============================================================================
|
|
31
|
-
// Compound Statement Handling
|
|
32
|
-
// =============================================================================
|
|
33
|
-
|
|
34
|
-
/**
|
|
35
|
-
* Get all command keywords including translated ones for a locale.
|
|
36
|
-
*/
|
|
37
|
-
function getCommandKeywordsForLocale(locale: string): Set<string> {
|
|
38
|
-
const keywords = new Set(ENGLISH_COMMANDS);
|
|
39
|
-
|
|
40
|
-
// Add translated command keywords from dictionaries
|
|
41
|
-
const dict = dictionaries[locale];
|
|
42
|
-
if (dict?.commands) {
|
|
43
|
-
Object.values(dict.commands).forEach(cmd => {
|
|
44
|
-
if (typeof cmd === 'string') {
|
|
45
|
-
keywords.add(cmd.toLowerCase());
|
|
46
|
-
}
|
|
47
|
-
});
|
|
48
|
-
}
|
|
49
|
-
|
|
50
|
-
return keywords;
|
|
51
|
-
}
|
|
52
|
-
|
|
53
|
-
/**
|
|
54
|
-
* English copula forms. A command keyword IMMEDIATELY after one of these is a
|
|
55
|
-
* predicate adjective, not a command verb: in `if my value is empty add .error
|
|
56
|
-
* to me` the `empty` belongs to the condition (`empty` is also a hyperscript
|
|
57
|
-
* command, v0.9.90). Without this guard the condition/body scans cut the
|
|
58
|
-
* condition at `is` and displace the adjective into the body's argument zone,
|
|
59
|
-
* where it anchors a spurious `empty-{lang}-generated` parse AND steals a
|
|
60
|
-
* neighboring role (the empty ×8 bn/hi/tr family). Source-locale copulas are
|
|
61
|
-
* added via the dictionary in `isPredicateAdjectivePosition`.
|
|
62
|
-
*/
|
|
63
|
-
const EN_COPULAS = ['is', 'are', 'was', 'were', 'am', 'be'];
|
|
64
|
-
|
|
65
|
-
function getCopulasForLocale(locale: string): Set<string> {
|
|
66
|
-
const copulas = new Set(EN_COPULAS);
|
|
67
|
-
if (locale !== 'en') {
|
|
68
|
-
for (const form of EN_COPULAS) {
|
|
69
|
-
copulas.add(translateWord(form, 'en', locale).toLowerCase());
|
|
70
|
-
}
|
|
71
|
-
}
|
|
72
|
-
return copulas;
|
|
73
|
-
}
|
|
74
|
-
|
|
75
|
-
/**
|
|
76
|
-
* True when tokens[i] sits right after a copula — a predicate-adjective
|
|
77
|
-
* position that must never be read as the start of a body command.
|
|
78
|
-
*/
|
|
79
|
-
function isPredicateAdjectivePosition(tokens: string[], i: number, copulas: Set<string>): boolean {
|
|
80
|
-
const prev = tokens[i - 1]?.toLowerCase();
|
|
81
|
-
return !!prev && copulas.has(prev);
|
|
82
|
-
}
|
|
83
|
-
|
|
84
|
-
/**
|
|
85
|
-
* Source-locale surface forms of the `for` command keyword and the loop's
|
|
86
|
-
* `in` preposition, for the loop-head test below.
|
|
87
|
-
*/
|
|
88
|
-
function getForLoopWordsForLocale(locale: string): { forWords: Set<string>; inWords: Set<string> } {
|
|
89
|
-
const forWords = new Set(['for']);
|
|
90
|
-
const inWords = new Set(['in']);
|
|
91
|
-
if (locale !== 'en') {
|
|
92
|
-
forWords.add(translateWord('for', 'en', locale).toLowerCase());
|
|
93
|
-
inWords.add(translateWord('in', 'en', locale).toLowerCase());
|
|
94
|
-
}
|
|
95
|
-
return { forWords, inWords };
|
|
96
|
-
}
|
|
97
|
-
|
|
98
|
-
/**
|
|
99
|
-
* True when a `for` at tokens[i] heads a real loop. Hyperscript's only
|
|
100
|
-
* statement-head `for` is `for <var> in <iterable>`, so a `for` with no `in`
|
|
101
|
-
* among the tokens before the next command keyword is a ROLE PHRASE — take's
|
|
102
|
-
* target (`take .active from .tab-button for me`) or a duration (`for 2s`) —
|
|
103
|
-
* and must never split the statement: the split shattered the phrase into a
|
|
104
|
-
* dangling `then for me` clause, which six SOV languages then parsed as a
|
|
105
|
-
* spurious `for` loop with patient "me" (the take-class ×6 family).
|
|
106
|
-
*/
|
|
107
|
-
function isLoopHeadFor(
|
|
108
|
-
tokens: string[],
|
|
109
|
-
i: number,
|
|
110
|
-
inWords: Set<string>,
|
|
111
|
-
commandKeywords: Set<string>
|
|
112
|
-
): boolean {
|
|
113
|
-
for (let j = i + 1; j < tokens.length; j++) {
|
|
114
|
-
const lt = tokens[j].toLowerCase();
|
|
115
|
-
if (inWords.has(lt)) return true;
|
|
116
|
-
if (commandKeywords.has(lt)) return false;
|
|
117
|
-
}
|
|
118
|
-
return false;
|
|
119
|
-
}
|
|
120
|
-
|
|
121
|
-
/**
|
|
122
|
-
* Repair a FRONTED Hebrew accusative marker in transformed output.
|
|
123
|
-
*
|
|
124
|
-
* When an event-handler body leads with a command-modifier (`on click once add …`,
|
|
125
|
-
* via {@link GrammarTransformer.tryTransformEventWithModifierBody}) or is a control
|
|
126
|
-
* block (`on blur if … add … end`, via `tryTransformEventWithBlockBody`), the body
|
|
127
|
-
* command's accusative marker את can be emitted AHEAD of its verb — `add .error to me`
|
|
128
|
-
* renders `… את הוסף .error …` instead of the canonical `… הוסף את .error …`. את before
|
|
129
|
-
* a verb is always ungrammatical Hebrew (it only ever marks a FOLLOWING definite
|
|
130
|
-
* object), and the semantic parser drops the command in every parse path (fused-event,
|
|
131
|
-
* multi-clause, conditional-body) when the marker is fronted but parses it when the
|
|
132
|
-
* marker follows the verb. So an `<accusative-marker> <command-verb>` adjacency is a
|
|
133
|
-
* pure transformer artifact: swap it back. Idempotent and safe — only touches `את
|
|
134
|
-
* <verb>`, never the ~40 generated `<verb> את {patient}` patterns that embed את legitimately.
|
|
135
|
-
*/
|
|
136
|
-
function repairHebrewFrontedAccusative(text: string): string {
|
|
137
|
-
const ACC = 'את'; // hebrewProfile.markers patient marker
|
|
138
|
-
const verbs = getCommandKeywordsForLocale('he');
|
|
139
|
-
const tokens = text.split(/\s+/);
|
|
140
|
-
let changed = false;
|
|
141
|
-
for (let i = 0; i + 1 < tokens.length; i++) {
|
|
142
|
-
if (tokens[i] === ACC && verbs.has(tokens[i + 1].toLowerCase())) {
|
|
143
|
-
[tokens[i], tokens[i + 1]] = [tokens[i + 1], tokens[i]];
|
|
144
|
-
changed = true;
|
|
145
|
-
i++; // skip the marker we just moved
|
|
146
|
-
}
|
|
147
|
-
}
|
|
148
|
-
return changed ? tokens.join(' ') : text;
|
|
149
|
-
}
|
|
150
|
-
|
|
151
|
-
/**
|
|
152
|
-
* Split a compound statement into parts at "then" boundaries, newlines,
|
|
153
|
-
* AND command keyword boundaries.
|
|
154
|
-
*
|
|
155
|
-
* Example: "on click wait 1s then increment #count then toggle .active"
|
|
156
|
-
* Returns: ["on click wait 1s", "increment #count", "toggle .active"]
|
|
157
|
-
*
|
|
158
|
-
* Example: "on click\n increment #count\n toggle .highlight"
|
|
159
|
-
* Returns: ["on click", "increment #count", "toggle .highlight"]
|
|
160
|
-
*
|
|
161
|
-
* Example: "wait 2s toggle .highlight"
|
|
162
|
-
* Returns: ["wait 2s", "toggle .highlight"]
|
|
163
|
-
*/
|
|
164
|
-
/**
|
|
165
|
-
* A reactive-block structure pulled apart by `extractBlockStructure`.
|
|
166
|
-
* Block-syntactic tokens (head, optional condition expression, optional
|
|
167
|
-
* connector, optional `end` tail) live outside the inner body so the
|
|
168
|
-
* body can be transformed by the regular pipeline (SOV reordering,
|
|
169
|
-
* possessives, etc.) without having block delimiters dragged into role
|
|
170
|
-
* values or moved by the canonical-order reorder.
|
|
171
|
-
*/
|
|
172
|
-
interface BlockStructure {
|
|
173
|
-
headKeyword: string;
|
|
174
|
-
/** Condition expression for `when X changes Y` / `unless X Y`. */
|
|
175
|
-
prefixExpr?: string;
|
|
176
|
-
/** `changes` for `when`; undefined for `live`/`unless`. */
|
|
177
|
-
connector?: string;
|
|
178
|
-
body: string;
|
|
179
|
-
/** `end` if the source had one; undefined otherwise. */
|
|
180
|
-
tailKeyword?: string;
|
|
181
|
-
}
|
|
182
|
-
|
|
183
|
-
/**
|
|
184
|
-
* Detect a reactive block (`live ... end`, `when X changes Y [end]`,
|
|
185
|
-
* `unless X Y [end]`) and decompose it. Returns `null` when the input
|
|
186
|
-
* is not a reactive block, when there's content after the matched
|
|
187
|
-
* `end`, or when the heuristic can't locate a body — all of which fall
|
|
188
|
-
* through to the standard `parseStatement` path.
|
|
189
|
-
*/
|
|
190
|
-
function extractBlockStructure(input: string, sourceLocale: string): BlockStructure | null {
|
|
191
|
-
const tokens = input.split(/\s+/);
|
|
192
|
-
const head = tokens[0]?.toLowerCase();
|
|
193
|
-
if (!head || !BLOCK_HEAD_KEYWORDS.has(head)) return null;
|
|
194
|
-
|
|
195
|
-
// Depth-aware match for the closing `end` so nested blocks
|
|
196
|
-
// (`live when X changes Y end end`) slice correctly.
|
|
197
|
-
let depth = 1;
|
|
198
|
-
let endIdx = -1;
|
|
199
|
-
for (let i = 1; i < tokens.length; i++) {
|
|
200
|
-
const t = tokens[i].toLowerCase();
|
|
201
|
-
if (BLOCK_HEAD_KEYWORDS.has(t)) depth++;
|
|
202
|
-
else if (t === 'end') {
|
|
203
|
-
depth--;
|
|
204
|
-
if (depth === 0) {
|
|
205
|
-
endIdx = i;
|
|
206
|
-
break;
|
|
207
|
-
}
|
|
208
|
-
}
|
|
209
|
-
}
|
|
210
|
-
|
|
211
|
-
// If there's trailing content after the matched `end`, bail out and
|
|
212
|
-
// let the existing splitter handle it. (`splitOnThen` normally
|
|
213
|
-
// separates trailing code before we get here.)
|
|
214
|
-
if (endIdx !== -1 && endIdx !== tokens.length - 1) return null;
|
|
215
|
-
|
|
216
|
-
const inner = endIdx !== -1 ? tokens.slice(1, endIdx) : tokens.slice(1);
|
|
217
|
-
const base: BlockStructure = { headKeyword: tokens[0], body: '' };
|
|
218
|
-
if (endIdx !== -1) base.tailKeyword = tokens[endIdx];
|
|
219
|
-
|
|
220
|
-
if (head === 'live') {
|
|
221
|
-
return { ...base, body: inner.join(' ') };
|
|
222
|
-
}
|
|
223
|
-
|
|
224
|
-
if (head === 'when') {
|
|
225
|
-
// Reactive: `when <expr> changes <body>`. Without `changes`, fall
|
|
226
|
-
// through to the standard event-wait path (parseConditional).
|
|
227
|
-
const idx = inner.findIndex(t => t.toLowerCase() === 'changes');
|
|
228
|
-
if (idx >= 0) {
|
|
229
|
-
return {
|
|
230
|
-
...base,
|
|
231
|
-
prefixExpr: inner.slice(0, idx).join(' '),
|
|
232
|
-
connector: inner[idx],
|
|
233
|
-
body: inner.slice(idx + 1).join(' '),
|
|
234
|
-
};
|
|
235
|
-
}
|
|
236
|
-
return null;
|
|
237
|
-
}
|
|
238
|
-
|
|
239
|
-
// `unless <cond> <body>`: condition runs up to the first command
|
|
240
|
-
// keyword in `inner`. Heuristic — works because hyperscript bodies
|
|
241
|
-
// always start with a command verb, and `unless` conditions rarely
|
|
242
|
-
// contain bare command keywords as values. A candidate right after a
|
|
243
|
-
// copula (`… is empty`) is a predicate adjective inside the condition,
|
|
244
|
-
// not a body verb — skip it.
|
|
245
|
-
const commands = getCommandKeywordsForLocale(sourceLocale);
|
|
246
|
-
const copulas = getCopulasForLocale(sourceLocale);
|
|
247
|
-
let bodyStart = -1;
|
|
248
|
-
for (let i = 0; i < inner.length; i++) {
|
|
249
|
-
if (commands.has(inner[i].toLowerCase()) && !isPredicateAdjectivePosition(inner, i, copulas)) {
|
|
250
|
-
bodyStart = i;
|
|
251
|
-
break;
|
|
252
|
-
}
|
|
253
|
-
}
|
|
254
|
-
if (bodyStart <= 0) return null;
|
|
255
|
-
return {
|
|
256
|
-
...base,
|
|
257
|
-
prefixExpr: inner.slice(0, bodyStart).join(' '),
|
|
258
|
-
body: inner.slice(bodyStart).join(' '),
|
|
259
|
-
};
|
|
260
|
-
}
|
|
261
|
-
|
|
262
|
-
function splitCompoundStatement(input: string, sourceLocale: string): string[] {
|
|
263
|
-
// First, split on newlines (preserving non-empty lines)
|
|
264
|
-
const lines = input
|
|
265
|
-
.split(/\n/)
|
|
266
|
-
.map(line => line.trim())
|
|
267
|
-
.filter(line => line.length > 0);
|
|
268
|
-
|
|
269
|
-
// If we have multiple lines, treat each as a separate part
|
|
270
|
-
// (but still need to handle "then" within each line)
|
|
271
|
-
const parts: string[] = [];
|
|
272
|
-
|
|
273
|
-
for (const line of lines) {
|
|
274
|
-
const lineParts = splitOnThen(line, sourceLocale);
|
|
275
|
-
// Further split each part on command boundaries
|
|
276
|
-
for (const part of lineParts) {
|
|
277
|
-
const commandParts = splitOnCommandBoundaries(part, sourceLocale);
|
|
278
|
-
parts.push(...commandParts);
|
|
279
|
-
}
|
|
280
|
-
}
|
|
281
|
-
|
|
282
|
-
return parts;
|
|
283
|
-
}
|
|
284
|
-
|
|
285
|
-
// =============================================================================
|
|
286
|
-
// Line Structure Preservation
|
|
287
|
-
// =============================================================================
|
|
288
|
-
|
|
289
|
-
/**
|
|
290
|
-
* Result of splitting with preserved line metadata.
|
|
291
|
-
*/
|
|
292
|
-
interface SplitWithMetadataResult {
|
|
293
|
-
/** The processed parts (commands/statements) */
|
|
294
|
-
parts: string[];
|
|
295
|
-
/** Metadata for each original line (for reconstruction) */
|
|
296
|
-
lineMetadata: LineMetadata[];
|
|
297
|
-
/** Mapping from parts back to their original line indices */
|
|
298
|
-
partToLineIndex: number[];
|
|
299
|
-
}
|
|
300
|
-
|
|
301
|
-
/**
|
|
302
|
-
* Split a compound statement while preserving line structure metadata.
|
|
303
|
-
* This tracks indentation and blank lines for reconstruction.
|
|
304
|
-
*/
|
|
305
|
-
function splitCompoundStatementWithMetadata(
|
|
306
|
-
input: string,
|
|
307
|
-
sourceLocale: string
|
|
308
|
-
): SplitWithMetadataResult {
|
|
309
|
-
const rawLines = input.split('\n');
|
|
310
|
-
const lineMetadata: LineMetadata[] = [];
|
|
311
|
-
const parts: string[] = [];
|
|
312
|
-
const partToLineIndex: number[] = [];
|
|
313
|
-
|
|
314
|
-
for (let lineIndex = 0; lineIndex < rawLines.length; lineIndex++) {
|
|
315
|
-
const rawLine = rawLines[lineIndex];
|
|
316
|
-
|
|
317
|
-
// Capture leading whitespace
|
|
318
|
-
const indentMatch = rawLine.match(/^(\s*)/);
|
|
319
|
-
const originalIndent = indentMatch ? indentMatch[1] : '';
|
|
320
|
-
const trimmed = rawLine.trim();
|
|
321
|
-
|
|
322
|
-
lineMetadata.push({
|
|
323
|
-
content: trimmed,
|
|
324
|
-
originalIndent,
|
|
325
|
-
isBlank: trimmed.length === 0,
|
|
326
|
-
});
|
|
327
|
-
|
|
328
|
-
if (trimmed.length > 0) {
|
|
329
|
-
// Process non-empty lines for "then" and command boundaries
|
|
330
|
-
const lineParts = splitOnThen(trimmed, sourceLocale);
|
|
331
|
-
for (const part of lineParts) {
|
|
332
|
-
const commandParts = splitOnCommandBoundaries(part, sourceLocale);
|
|
333
|
-
for (const cmdPart of commandParts) {
|
|
334
|
-
parts.push(cmdPart);
|
|
335
|
-
partToLineIndex.push(lineIndex);
|
|
336
|
-
}
|
|
337
|
-
}
|
|
338
|
-
}
|
|
339
|
-
}
|
|
340
|
-
|
|
341
|
-
return { parts, lineMetadata, partToLineIndex };
|
|
342
|
-
}
|
|
343
|
-
|
|
344
|
-
/**
|
|
345
|
-
* Normalize indentation to consistent 4-space levels.
|
|
346
|
-
* Preserves relative indentation structure while standardizing spacing.
|
|
347
|
-
*/
|
|
348
|
-
function normalizeIndentation(lineMetadata: LineMetadata[]): string[] {
|
|
349
|
-
// Find non-blank lines with indentation
|
|
350
|
-
const indentedLines = lineMetadata.filter(m => !m.isBlank && m.originalIndent.length > 0);
|
|
351
|
-
|
|
352
|
-
if (indentedLines.length === 0) {
|
|
353
|
-
// No indented lines, return empty strings
|
|
354
|
-
return lineMetadata.map(() => '');
|
|
355
|
-
}
|
|
356
|
-
|
|
357
|
-
// Find minimum non-zero indent (the base unit)
|
|
358
|
-
const indentLengths = indentedLines.map(m => {
|
|
359
|
-
// Convert tabs to 4 spaces for consistent measurement
|
|
360
|
-
const normalized = m.originalIndent.replace(/\t/g, ' ');
|
|
361
|
-
return normalized.length;
|
|
362
|
-
});
|
|
363
|
-
const minIndent = Math.min(...indentLengths);
|
|
364
|
-
const baseUnit = minIndent > 0 ? minIndent : 4;
|
|
365
|
-
|
|
366
|
-
// Normalize each line's indentation
|
|
367
|
-
return lineMetadata.map(meta => {
|
|
368
|
-
if (meta.isBlank) {
|
|
369
|
-
return ''; // Blank lines get no indentation
|
|
370
|
-
}
|
|
371
|
-
if (meta.originalIndent.length === 0) {
|
|
372
|
-
return ''; // No original indent
|
|
373
|
-
}
|
|
374
|
-
|
|
375
|
-
// Convert tabs and calculate level
|
|
376
|
-
const normalized = meta.originalIndent.replace(/\t/g, ' ');
|
|
377
|
-
const level = Math.round(normalized.length / baseUnit);
|
|
378
|
-
return ' '.repeat(level); // 4 spaces per level
|
|
379
|
-
});
|
|
380
|
-
}
|
|
381
|
-
|
|
382
|
-
/**
|
|
383
|
-
* Reconstruct output with preserved line structure.
|
|
384
|
-
* Maps transformed parts back to their original lines with proper indentation.
|
|
385
|
-
*/
|
|
386
|
-
function reconstructWithLineStructure(
|
|
387
|
-
transformedParts: string[],
|
|
388
|
-
lineMetadata: LineMetadata[],
|
|
389
|
-
partToLineIndex: number[],
|
|
390
|
-
targetThen: string
|
|
391
|
-
): string {
|
|
392
|
-
// If there's only one non-blank line, simple case
|
|
393
|
-
const nonBlankCount = lineMetadata.filter(m => !m.isBlank).length;
|
|
394
|
-
if (nonBlankCount <= 1 && transformedParts.length <= 1) {
|
|
395
|
-
const normalizedIndents = normalizeIndentation(lineMetadata);
|
|
396
|
-
const result: string[] = [];
|
|
397
|
-
|
|
398
|
-
for (let i = 0; i < lineMetadata.length; i++) {
|
|
399
|
-
if (lineMetadata[i].isBlank) {
|
|
400
|
-
result.push('');
|
|
401
|
-
} else if (transformedParts.length > 0) {
|
|
402
|
-
result.push(normalizedIndents[i] + transformedParts[0]);
|
|
403
|
-
}
|
|
404
|
-
}
|
|
405
|
-
return result.join('\n');
|
|
406
|
-
}
|
|
407
|
-
|
|
408
|
-
// Normalize indentation
|
|
409
|
-
const normalizedIndents = normalizeIndentation(lineMetadata);
|
|
410
|
-
|
|
411
|
-
// Group transformed parts by their original line
|
|
412
|
-
const partsPerLine: Map<number, string[]> = new Map();
|
|
413
|
-
for (let i = 0; i < transformedParts.length; i++) {
|
|
414
|
-
const lineIdx = partToLineIndex[i];
|
|
415
|
-
if (!partsPerLine.has(lineIdx)) {
|
|
416
|
-
partsPerLine.set(lineIdx, []);
|
|
417
|
-
}
|
|
418
|
-
partsPerLine.get(lineIdx)!.push(transformedParts[i]);
|
|
419
|
-
}
|
|
420
|
-
|
|
421
|
-
// Build result lines
|
|
422
|
-
const result: string[] = [];
|
|
423
|
-
for (let i = 0; i < lineMetadata.length; i++) {
|
|
424
|
-
const meta = lineMetadata[i];
|
|
425
|
-
const indent = normalizedIndents[i];
|
|
426
|
-
|
|
427
|
-
if (meta.isBlank) {
|
|
428
|
-
result.push('');
|
|
429
|
-
} else {
|
|
430
|
-
const lineParts = partsPerLine.get(i) || [];
|
|
431
|
-
if (lineParts.length > 0) {
|
|
432
|
-
// Join multiple parts on same line with "then"
|
|
433
|
-
const lineContent = lineParts.join(` ${targetThen} `);
|
|
434
|
-
result.push(indent + lineContent);
|
|
435
|
-
}
|
|
436
|
-
}
|
|
437
|
-
}
|
|
438
|
-
|
|
439
|
-
return result.join('\n');
|
|
440
|
-
}
|
|
441
|
-
|
|
442
|
-
/**
|
|
443
|
-
* Split a statement on command keyword boundaries.
|
|
444
|
-
* E.g., "wait 2s toggle .highlight" → ["wait 2s", "toggle .highlight"]
|
|
445
|
-
*
|
|
446
|
-
* Special cases:
|
|
447
|
-
* - "on <event> <command>" stays together (event handler with first command)
|
|
448
|
-
* - Modifiers like "to", "from" don't trigger splits
|
|
449
|
-
*/
|
|
450
|
-
/**
|
|
451
|
-
* English modifier keywords that should not trigger command boundary splits.
|
|
452
|
-
* These are the base set; localized equivalents (e.g. Japanese `に`, Spanish
|
|
453
|
-
* `a`, Arabic `إلى`) are layered on per source locale by
|
|
454
|
-
* `getBoundaryModifiersForLocale`, so a preposition in a non-English source
|
|
455
|
-
* is also recognized as a modifier rather than a spurious command boundary.
|
|
456
|
-
*/
|
|
457
|
-
const BOUNDARY_MODIFIERS = new Set([
|
|
458
|
-
'to',
|
|
459
|
-
'into',
|
|
460
|
-
'from',
|
|
461
|
-
'with',
|
|
462
|
-
'by',
|
|
463
|
-
'as',
|
|
464
|
-
'at',
|
|
465
|
-
'in',
|
|
466
|
-
'on',
|
|
467
|
-
'of',
|
|
468
|
-
'over',
|
|
469
|
-
]);
|
|
470
|
-
|
|
471
|
-
/**
|
|
472
|
-
* Boundary modifiers resolved for a given source locale: the English base set
|
|
473
|
-
* (always kept, since input may mix English keywords) plus every grammatical
|
|
474
|
-
* marker form declared by the locale's profile. Profile markers are the
|
|
475
|
-
* surface realizations of semantic roles (destination, source, style, …) —
|
|
476
|
-
* they always bind to a following value, so none should be treated as a
|
|
477
|
-
* command boundary. Cached per locale because profiles are static.
|
|
478
|
-
*/
|
|
479
|
-
const boundaryModifiersCache = new Map<string, Set<string>>();
|
|
480
|
-
|
|
481
|
-
function getBoundaryModifiersForLocale(locale: string): Set<string> {
|
|
482
|
-
const cached = boundaryModifiersCache.get(locale);
|
|
483
|
-
if (cached) return cached;
|
|
484
|
-
|
|
485
|
-
const modifiers = new Set(BOUNDARY_MODIFIERS);
|
|
486
|
-
|
|
487
|
-
const profile = getProfile(locale);
|
|
488
|
-
profile?.markers.forEach(marker => {
|
|
489
|
-
const form = marker.form.replace(/^-|-$/g, '').toLowerCase();
|
|
490
|
-
if (form) modifiers.add(form);
|
|
491
|
-
|
|
492
|
-
marker.alternatives?.forEach(alt => {
|
|
493
|
-
const altForm = alt.replace(/^-|-$/g, '').toLowerCase();
|
|
494
|
-
if (altForm) modifiers.add(altForm);
|
|
495
|
-
});
|
|
496
|
-
});
|
|
497
|
-
|
|
498
|
-
boundaryModifiersCache.set(locale, modifiers);
|
|
499
|
-
return modifiers;
|
|
500
|
-
}
|
|
501
|
-
|
|
502
|
-
/**
|
|
503
|
-
* Commands for which a trailing `on <element>` is the TARGET the command acts
|
|
504
|
-
* on (a destination), not a new event-handler clause. For these, the locative
|
|
505
|
-
* `on` must neither split the statement (`splitOnCommandBoundaries`) nor be read
|
|
506
|
-
* as a fresh `event` role (`buildArgumentModifierMap`) — see those call sites.
|
|
507
|
-
*
|
|
508
|
-
* Deliberately narrow. `on` is overloaded (event-handler head vs. locative
|
|
509
|
-
* target), and the change has cross-language blast radius, so we only enable
|
|
510
|
-
* the locative reading for the DOM class/attribute mutators where it's the
|
|
511
|
-
* documented idiom (`toggle .open on #menu`, `toggle @hidden on #panel`).
|
|
512
|
-
*
|
|
513
|
-
* `trigger`/`send` — `trigger X on Y` / `send X on Y` fire an event on a target
|
|
514
|
-
* element; the trailing `on Y` is that target (destination), not a new clause.
|
|
515
|
-
* Previously excluded: keeping `on Y` attached produced `trigger X → #y` output
|
|
516
|
-
* that was thought to destabilise the semantic parser's multi-line behaviour
|
|
517
|
-
* fallback (behavior-sortable's `trigger sortable:start on me`). That concern is
|
|
518
|
-
* stale — the recent semantic-parser body increments fold the surrounding block
|
|
519
|
-
* cleanly, and splitting instead injected a spurious `then` (`disparar
|
|
520
|
-
* sortable:start entonces en yo`) that glued the following `repeat until event`
|
|
521
|
-
* loop into a then-chain and dropped it. Keeping `on Y` attached restores
|
|
522
|
-
* behavior-sortable to faithful across the SVO languages.
|
|
523
|
-
*
|
|
524
|
-
* Excluded on purpose:
|
|
525
|
-
* - `set` — `set @attr to V on Y` carries BOTH `to` (value) and `on` (target);
|
|
526
|
-
* English marks both as `destination`, so remapping `on` would collide with
|
|
527
|
-
* and clobber the value. Needs distinct value/target roles first (deferred).
|
|
528
|
-
* - `put` — `put X on Y into Z` has the same dual-destination collision.
|
|
529
|
-
*
|
|
530
|
-
* `add`/`remove` use `to`/`from` for their target in practice (not `on`), so
|
|
531
|
-
* their inclusion is a harmless no-op that documents intent.
|
|
532
|
-
*/
|
|
533
|
-
const ON_TARGET_COMMANDS = new Set(['toggle', 'add', 'remove', 'trigger', 'send']);
|
|
534
|
-
|
|
535
|
-
/**
|
|
536
|
-
* Find the command verb of a partially-collected statement: the first command
|
|
537
|
-
* keyword that is neither the event-handler head (`on`/その他 event markers) nor
|
|
538
|
-
* an argument-introducing preposition. For `on click toggle .open` → `toggle`;
|
|
539
|
-
* for `trigger sortable:start` → `trigger`.
|
|
540
|
-
*/
|
|
541
|
-
function commandVerbOf(tokens: string[], commandKeywords: Set<string>): string | null {
|
|
542
|
-
for (const token of tokens) {
|
|
543
|
-
const lt = token.toLowerCase();
|
|
544
|
-
if (commandKeywords.has(lt) && !BOUNDARY_MODIFIERS.has(lt) && !EVENT_KEYWORDS.has(lt)) {
|
|
545
|
-
return lt;
|
|
546
|
-
}
|
|
547
|
-
}
|
|
548
|
-
return null;
|
|
549
|
-
}
|
|
550
|
-
|
|
551
|
-
/**
|
|
552
|
-
* Block-introducing keywords whose body should not be split at command
|
|
553
|
-
* boundaries by `splitOnCommandBoundaries`. Inputs starting with one of
|
|
554
|
-
* these are also routed around `parseStatement` entirely by
|
|
555
|
-
* `extractBlockStructure` + `transformBlock` so block-syntactic tokens
|
|
556
|
-
* never reach `parseCommand`/`parseConditional` (where they'd be
|
|
557
|
-
* misinterpreted as command verbs or swept into role values).
|
|
558
|
-
*
|
|
559
|
-
* `if` deliberately stays out: `if X then Y end` already works via the
|
|
560
|
-
* `splitOnThen` + `parseConditional` path.
|
|
561
|
-
*/
|
|
562
|
-
const BLOCK_HEAD_KEYWORDS = new Set(['live', 'when', 'unless']);
|
|
563
|
-
|
|
564
|
-
/**
|
|
565
|
-
* Source-language block-introducing command keywords whose body is a clause plus a
|
|
566
|
-
* branch/loop body (harness source is English, so these are English-keyed). Used to
|
|
567
|
-
* locate a block body inside an event handler and to track block depth when finding
|
|
568
|
-
* a top-level `else`.
|
|
569
|
-
*/
|
|
570
|
-
const BLOCK_BODY_KEYWORDS = new Set(['if', 'repeat', 'unless', 'while', 'for']);
|
|
571
|
-
|
|
572
|
-
/**
|
|
573
|
-
* SVO targets that mark the object/patient with a particle (he את, zh 把) and so
|
|
574
|
-
* mangle an inline `on <event> unless <cond> <body>` guard — the unless tail is
|
|
575
|
-
* swept into one patient blob and the marker lands ahead of the condition. These
|
|
576
|
-
* route through `tryTransformEventWithUnlessGuard`. SOV/VSO object-markers
|
|
577
|
-
* (ja/ko/tr/ar) are excluded: their event does not lead, so the SVO event-first
|
|
578
|
-
* emission there would be wrong (and they don't exhibit the artifact today).
|
|
579
|
-
*/
|
|
580
|
-
const UNLESS_GUARD_OBJECT_MARKING_LOCALES = new Set(['he', 'zh']);
|
|
581
|
-
|
|
582
|
-
function splitOnCommandBoundaries(input: string, sourceLocale: string): string[] {
|
|
583
|
-
const commandKeywords = getCommandKeywordsForLocale(sourceLocale);
|
|
584
|
-
const boundaryModifiers = getBoundaryModifiersForLocale(sourceLocale);
|
|
585
|
-
const { forWords, inWords } = getForLoopWordsForLocale(sourceLocale);
|
|
586
|
-
const tokens = input.split(/\s+/);
|
|
587
|
-
|
|
588
|
-
if (tokens.length === 0) return [input];
|
|
589
|
-
|
|
590
|
-
const parts: string[] = [];
|
|
591
|
-
let currentPart: string[] = [];
|
|
592
|
-
|
|
593
|
-
// Check if this starts with an event handler pattern (on/em/en/bei/で + event)
|
|
594
|
-
const firstTokenLower = tokens[0]?.toLowerCase();
|
|
595
|
-
const isEventHandler = EVENT_KEYWORDS.has(firstTokenLower);
|
|
596
|
-
|
|
597
|
-
// If it's an event handler, the first command after the event is part of the handler
|
|
598
|
-
// So we need to track whether we've seen the first command yet
|
|
599
|
-
let seenFirstCommand = !isEventHandler; // If not event handler, we're already past the "first command" phase
|
|
600
|
-
|
|
601
|
-
// Track block-scope depth (live/when/bind/if/unless/for/while/...). While
|
|
602
|
-
// inside a block, do not split on command boundaries — the body belongs
|
|
603
|
-
// to the block head and must transform as one unit. See comments on
|
|
604
|
-
// BLOCK_HEAD_KEYWORDS for the failure mode this prevents.
|
|
605
|
-
let blockDepth = 0;
|
|
606
|
-
|
|
607
|
-
for (let i = 0; i < tokens.length; i++) {
|
|
608
|
-
const token = tokens[i];
|
|
609
|
-
const lowerToken = token.toLowerCase();
|
|
610
|
-
|
|
611
|
-
// Update block-scope depth before any split decision.
|
|
612
|
-
if (BLOCK_HEAD_KEYWORDS.has(lowerToken)) {
|
|
613
|
-
blockDepth++;
|
|
614
|
-
} else if (lowerToken === 'end' && blockDepth > 0) {
|
|
615
|
-
blockDepth--;
|
|
616
|
-
}
|
|
617
|
-
|
|
618
|
-
// If this is a command keyword and we already have tokens in current part
|
|
619
|
-
if (commandKeywords.has(lowerToken) && currentPart.length > 0) {
|
|
620
|
-
// Check if the previous token looks like it could end a command
|
|
621
|
-
const prevToken = currentPart[currentPart.length - 1];
|
|
622
|
-
const prevLower = prevToken.toLowerCase();
|
|
623
|
-
|
|
624
|
-
// For event handlers: don't split before the first command
|
|
625
|
-
// E.g., "on click wait 1s" should stay together
|
|
626
|
-
if (!seenFirstCommand) {
|
|
627
|
-
// Mark that we've now seen the first command
|
|
628
|
-
seenFirstCommand = true;
|
|
629
|
-
currentPart.push(token);
|
|
630
|
-
continue;
|
|
631
|
-
}
|
|
632
|
-
|
|
633
|
-
// Don't split inside a block (live/when/bind/unless body, etc.).
|
|
634
|
-
// The block head and its body must transform as one statement.
|
|
635
|
-
if (blockDepth > 0) {
|
|
636
|
-
currentPart.push(token);
|
|
637
|
-
continue;
|
|
638
|
-
}
|
|
639
|
-
|
|
640
|
-
// A `for` that doesn't head a real loop (`for <var> in <iterable>`) is a
|
|
641
|
-
// role phrase of the current command — see isLoopHeadFor.
|
|
642
|
-
if (forWords.has(lowerToken) && !isLoopHeadFor(tokens, i, inWords, commandKeywords)) {
|
|
643
|
-
currentPart.push(token);
|
|
644
|
-
continue;
|
|
645
|
-
}
|
|
646
|
-
|
|
647
|
-
// Locative `on` for a DOM-target command (`toggle X on Y`) is the target
|
|
648
|
-
// element, NOT a new command. `on` lands in `commandKeywords` only
|
|
649
|
-
// incidentally — the EN dictionary registers `commands.on = 'on'` for the
|
|
650
|
-
// event-handler head — so without this guard the destination `on` split
|
|
651
|
-
// the statement, the join re-inserted a spurious `then` (ثم/pagkatapos/…),
|
|
652
|
-
// and the dangling `on Y` was misread as a second event handler. Keep it
|
|
653
|
-
// attached so the role parser can assign it to `destination`. Restricted
|
|
654
|
-
// to ON_TARGET_COMMANDS so `trigger X on me` etc. keep their prior split.
|
|
655
|
-
if (BOUNDARY_MODIFIERS.has(lowerToken)) {
|
|
656
|
-
const verb = commandVerbOf(currentPart, commandKeywords);
|
|
657
|
-
if (verb && ON_TARGET_COMMANDS.has(verb)) {
|
|
658
|
-
currentPart.push(token);
|
|
659
|
-
continue;
|
|
660
|
-
}
|
|
661
|
-
// `set @attr to V on <scope>` (S1 tabs-aria): the trailing `on <scope>`
|
|
662
|
-
// is the element(s) the attribute is set on, not a new command. `set` is
|
|
663
|
-
// deliberately NOT in ON_TARGET_COMMANDS (its role parser would clobber
|
|
664
|
-
// the value), so this is a dedicated guard — kept attached only when a
|
|
665
|
-
// selector/reference scope follows, then positioned by transformSingle's
|
|
666
|
-
// set-scope handler (transformSetWithScope). A `set` whose `on` is
|
|
667
|
-
// followed by a verb is left to split as before.
|
|
668
|
-
if (lowerToken === 'on' && verb === 'set') {
|
|
669
|
-
const nextTok = tokens[i + 1];
|
|
670
|
-
const nextLower = nextTok?.toLowerCase();
|
|
671
|
-
const scopeLike =
|
|
672
|
-
!!nextTok &&
|
|
673
|
-
(/^[#.<@[]/.test(nextTok) ||
|
|
674
|
-
nextLower === 'me' ||
|
|
675
|
-
nextLower === 'it' ||
|
|
676
|
-
nextLower === 'you');
|
|
677
|
-
if (scopeLike) {
|
|
678
|
-
currentPart.push(token);
|
|
679
|
-
continue;
|
|
680
|
-
}
|
|
681
|
-
}
|
|
682
|
-
}
|
|
683
|
-
|
|
684
|
-
if (!boundaryModifiers.has(prevLower) && !commandKeywords.has(prevLower)) {
|
|
685
|
-
// This looks like a command boundary - save current part and start new one
|
|
686
|
-
parts.push(currentPart.join(' '));
|
|
687
|
-
currentPart = [token];
|
|
688
|
-
continue;
|
|
689
|
-
}
|
|
690
|
-
}
|
|
691
|
-
|
|
692
|
-
currentPart.push(token);
|
|
693
|
-
}
|
|
694
|
-
|
|
695
|
-
// Add the last part
|
|
696
|
-
if (currentPart.length > 0) {
|
|
697
|
-
parts.push(currentPart.join(' '));
|
|
698
|
-
}
|
|
699
|
-
|
|
700
|
-
return parts.filter(p => p.length > 0);
|
|
701
|
-
}
|
|
702
|
-
|
|
703
|
-
/**
|
|
704
|
-
* Split a single line on "then" keywords.
|
|
705
|
-
*/
|
|
706
|
-
function splitOnThen(input: string, sourceLocale: string): string[] {
|
|
707
|
-
// Build regex pattern from all known "then" keywords
|
|
708
|
-
const thenKeywords = Array.from(THEN_KEYWORDS);
|
|
709
|
-
|
|
710
|
-
// Add any dictionary-specific "then" keyword for the source locale
|
|
711
|
-
const sourceDict = sourceLocale === 'en' ? null : dictionaries[sourceLocale];
|
|
712
|
-
if (sourceDict?.modifiers?.then) {
|
|
713
|
-
thenKeywords.push(sourceDict.modifiers.then);
|
|
714
|
-
}
|
|
715
|
-
// Also check logical.then since some dictionaries put it there
|
|
716
|
-
if ((sourceDict?.logical as Record<string, string>)?.then) {
|
|
717
|
-
thenKeywords.push((sourceDict?.logical as Record<string, string>).then);
|
|
718
|
-
}
|
|
719
|
-
|
|
720
|
-
// Create a regex that matches any "then" keyword as a whole word
|
|
721
|
-
// Use word boundaries to avoid matching "then" inside other words
|
|
722
|
-
const escapedKeywords = thenKeywords.map(k => k.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'));
|
|
723
|
-
const pattern = new RegExp(`\\s+(${escapedKeywords.join('|')})\\s+`, 'gi');
|
|
724
|
-
|
|
725
|
-
// Split on "then" keywords
|
|
726
|
-
const parts = input.split(pattern).filter(part => {
|
|
727
|
-
// Filter out the "then" keywords themselves (captured by the group)
|
|
728
|
-
const lowerPart = part.toLowerCase().trim();
|
|
729
|
-
return lowerPart && !thenKeywords.some(k => k.toLowerCase() === lowerPart);
|
|
730
|
-
});
|
|
731
|
-
|
|
732
|
-
return parts.map(p => p.trim()).filter(p => p.length > 0);
|
|
733
|
-
}
|
|
734
|
-
|
|
735
|
-
/**
|
|
736
|
-
* Get the "then" keyword in the target language.
|
|
737
|
-
* Checks both modifiers and logical sections since dictionaries vary.
|
|
738
|
-
*/
|
|
739
|
-
function getTargetThenKeyword(targetLocale: string): string {
|
|
740
|
-
if (targetLocale === 'en') return 'then';
|
|
741
|
-
|
|
742
|
-
const targetDict = dictionaries[targetLocale];
|
|
743
|
-
if (!targetDict) return 'then';
|
|
744
|
-
|
|
745
|
-
// Check modifiers first, then logical (dictionaries vary)
|
|
746
|
-
return (
|
|
747
|
-
targetDict.modifiers?.then || (targetDict.logical as Record<string, string>)?.then || 'then'
|
|
748
|
-
);
|
|
749
|
-
}
|
|
750
|
-
|
|
751
|
-
// =============================================================================
|
|
752
|
-
// Derived Constants from Profiles
|
|
753
|
-
// =============================================================================
|
|
754
|
-
|
|
755
|
-
/**
|
|
756
|
-
* Derive event keywords from all language profiles.
|
|
757
|
-
* This replaces the hardcoded eventKeywords array.
|
|
758
|
-
*/
|
|
759
|
-
function deriveEventKeywordsFromProfiles(): Set<string> {
|
|
760
|
-
const keywords = new Set<string>();
|
|
761
|
-
|
|
762
|
-
// Add 'on' as the English default
|
|
763
|
-
keywords.add('on');
|
|
764
|
-
|
|
765
|
-
// Extract event markers from all profiles
|
|
766
|
-
for (const profile of Object.values(profiles)) {
|
|
767
|
-
for (const marker of profile.markers) {
|
|
768
|
-
if (marker.role === 'event') {
|
|
769
|
-
// Strip hyphen notation and add
|
|
770
|
-
const form = marker.form.replace(/^-|-$/g, '').toLowerCase();
|
|
771
|
-
if (form) keywords.add(form);
|
|
772
|
-
|
|
773
|
-
// Add alternatives
|
|
774
|
-
marker.alternatives?.forEach(alt => {
|
|
775
|
-
const altForm = alt.replace(/^-|-$/g, '').toLowerCase();
|
|
776
|
-
if (altForm) keywords.add(altForm);
|
|
777
|
-
});
|
|
778
|
-
}
|
|
779
|
-
}
|
|
780
|
-
}
|
|
781
|
-
|
|
782
|
-
return keywords;
|
|
783
|
-
}
|
|
784
|
-
|
|
785
|
-
/** Event keywords derived from language profiles */
|
|
786
|
-
const EVENT_KEYWORDS = deriveEventKeywordsFromProfiles();
|
|
787
|
-
|
|
788
|
-
/**
|
|
789
|
-
* Conjunctions that join multiple events in an event handler head
|
|
790
|
-
* (`on click or keypress ...`). Source hyperscript is English, so the
|
|
791
|
-
* canonical form is `or`; localized equivalents are added defensively for
|
|
792
|
-
* non-English sources.
|
|
793
|
-
*/
|
|
794
|
-
const EVENT_CONJUNCTIONS = new Set(['or']);
|
|
795
|
-
|
|
796
|
-
/**
|
|
797
|
-
* Command-modifier keywords that may lead an event handler body
|
|
798
|
-
* (`on click async fetch …`, `on click once add …`). They modify the handler /
|
|
799
|
-
* following command rather than acting as the verb, so the transformer must lift
|
|
800
|
-
* them out before role assignment instead of treating them as the action.
|
|
801
|
-
* Source hyperscript is English, so these are the canonical English forms — the
|
|
802
|
-
* semantic parser recognizes the same literals when stripping them pre-parse.
|
|
803
|
-
*/
|
|
804
|
-
const BODY_MODIFIER_KEYWORDS = new Set([
|
|
805
|
-
'async',
|
|
806
|
-
'once',
|
|
807
|
-
'debounced',
|
|
808
|
-
'debounce',
|
|
809
|
-
'throttled',
|
|
810
|
-
'throttle',
|
|
811
|
-
]);
|
|
812
|
-
|
|
813
|
-
// =============================================================================
|
|
814
|
-
// Helper: Dynamic Modifier Map
|
|
815
|
-
// =============================================================================
|
|
816
|
-
|
|
817
|
-
/**
|
|
818
|
-
* Generates a lookup map for semantic roles based on the language profile.
|
|
819
|
-
* Maps markers (e.g., 'to', 'に', 'into', 'إلى') to their semantic roles.
|
|
820
|
-
* This enables parsing non-English input by using the profile's markers.
|
|
821
|
-
*/
|
|
822
|
-
function generateModifierMap(profile: LanguageProfile): Record<string, SemanticRole> {
|
|
823
|
-
const map: Record<string, SemanticRole> = {};
|
|
824
|
-
|
|
825
|
-
// Map markers to roles from the profile
|
|
826
|
-
profile.markers.forEach(marker => {
|
|
827
|
-
// Strip hyphen notation for suffix/prefix markers
|
|
828
|
-
const form = marker.form.replace(/^-|-$/g, '').toLowerCase();
|
|
829
|
-
if (form) {
|
|
830
|
-
map[form] = marker.role;
|
|
831
|
-
}
|
|
832
|
-
|
|
833
|
-
// Map alternatives if they exist (e.g., Korean vowel harmony variants)
|
|
834
|
-
marker.alternatives?.forEach(alt => {
|
|
835
|
-
const altForm = alt.replace(/^-|-$/g, '').toLowerCase();
|
|
836
|
-
if (altForm) {
|
|
837
|
-
map[altForm] = marker.role;
|
|
838
|
-
}
|
|
839
|
-
});
|
|
840
|
-
});
|
|
841
|
-
|
|
842
|
-
// Add English modifiers as fallback (don't override profile-specific markers)
|
|
843
|
-
for (const [key, role] of Object.entries(ENGLISH_MODIFIER_ROLES)) {
|
|
844
|
-
if (!(key in map)) {
|
|
845
|
-
map[key] = role;
|
|
846
|
-
}
|
|
847
|
-
}
|
|
848
|
-
|
|
849
|
-
return map;
|
|
850
|
-
}
|
|
851
|
-
|
|
852
|
-
/**
|
|
853
|
-
* Modifier map for parsing the ARGUMENTS of a statement (everything after the
|
|
854
|
-
* command verb), as opposed to the statement head.
|
|
855
|
-
*
|
|
856
|
-
* Why a separate map: a few languages reuse one word for both the event-handler
|
|
857
|
-
* head marker and a locative/target preposition. English is the prime case —
|
|
858
|
-
* `on` is the event head (`on click …`) AND the toggle/set target preposition
|
|
859
|
-
* (`toggle .x on #y`). `generateModifierMap` maps `on → event` (from the EN
|
|
860
|
-
* profile's event marker), so when the destination `on` was read in argument
|
|
861
|
-
* position it overwrote the already-captured head event (the `click` got
|
|
862
|
-
* dropped, e.g. `toggle @hidden on #panel` → `… عند #panel` with نقر/click gone).
|
|
863
|
-
*
|
|
864
|
-
* The fix: in argument position, remap the event role to `destination`. This is
|
|
865
|
-
* safe because a trigger event only ever appears at the statement head (which is
|
|
866
|
-
* consumed before argument parsing begins). Any "event marker" reached while
|
|
867
|
-
* scanning arguments is therefore being used as a locative — i.e. the element
|
|
868
|
-
* the command acts ON — which is exactly the `destination` role (the semantic
|
|
869
|
-
* parser also models `toggle .x on #y` as patient `.x` + destination `#y`).
|
|
870
|
-
*
|
|
871
|
-
* Restricted to (a) SVO source profiles and (b) ON_TARGET_COMMANDS:
|
|
872
|
-
* - SVO: in VSO/SOV languages the event marker can legitimately appear
|
|
873
|
-
* mid-statement (Arabic VSO renders an event handler as `بدل X عند نقر`,
|
|
874
|
-
* classified as a command), so remapping there would mis-read a real event
|
|
875
|
-
* as a destination. English — and the en→lang gate path — is SVO.
|
|
876
|
-
* - command: only the DOM class/attr mutators use a locative `on` target.
|
|
877
|
-
* For other commands (`trigger X on me`, `set X to V on Y`) the remap is
|
|
878
|
-
* either destabilising or collides with `to`/`into` — see ON_TARGET_COMMANDS.
|
|
879
|
-
*
|
|
880
|
-
* When neither applies the unmodified map is returned (event stays event).
|
|
881
|
-
*/
|
|
882
|
-
function buildArgumentModifierMap(
|
|
883
|
-
profile: LanguageProfile,
|
|
884
|
-
actionVerb: string | undefined
|
|
885
|
-
): Record<string, SemanticRole> {
|
|
886
|
-
const map = generateModifierMap(profile);
|
|
887
|
-
const verb = actionVerb?.toLowerCase();
|
|
888
|
-
if (profile.wordOrder !== 'SVO' || !verb || !ON_TARGET_COMMANDS.has(verb)) {
|
|
889
|
-
return map;
|
|
890
|
-
}
|
|
891
|
-
|
|
892
|
-
const remapped: Record<string, SemanticRole> = {};
|
|
893
|
-
for (const [form, role] of Object.entries(map)) {
|
|
894
|
-
remapped[form] = role === 'event' ? 'destination' : role;
|
|
895
|
-
}
|
|
896
|
-
return remapped;
|
|
897
|
-
}
|
|
898
|
-
|
|
899
|
-
// =============================================================================
|
|
900
|
-
// Statement Parser
|
|
901
|
-
// =============================================================================
|
|
902
|
-
|
|
903
|
-
/**
|
|
904
|
-
* Parse a hyperscript statement into semantic roles
|
|
905
|
-
* This is the core analysis step that identifies WHAT each part means
|
|
906
|
-
*/
|
|
907
|
-
export function parseStatement(input: string, sourceLocale: string = 'en'): ParsedStatement | null {
|
|
908
|
-
const profile = getProfile(sourceLocale);
|
|
909
|
-
if (!profile) return null;
|
|
910
|
-
|
|
911
|
-
const tokens = tokenize(input, profile);
|
|
912
|
-
|
|
913
|
-
// Identify statement type and extract roles
|
|
914
|
-
const statementType = identifyStatementType(tokens, profile);
|
|
915
|
-
|
|
916
|
-
switch (statementType) {
|
|
917
|
-
case 'event-handler':
|
|
918
|
-
return parseEventHandler(tokens, profile);
|
|
919
|
-
case 'command':
|
|
920
|
-
return parseCommand(tokens, profile);
|
|
921
|
-
case 'conditional':
|
|
922
|
-
return parseConditional(tokens, profile);
|
|
923
|
-
default:
|
|
924
|
-
return null;
|
|
925
|
-
}
|
|
926
|
-
}
|
|
927
|
-
|
|
928
|
-
/**
|
|
929
|
-
* Known suffixes that may attach to words without spaces.
|
|
930
|
-
* These are split off during tokenization for proper parsing.
|
|
931
|
-
*/
|
|
932
|
-
const ATTACHED_SUFFIXES: Record<string, string[]> = {
|
|
933
|
-
// Chinese: 时 (time/when) often attaches to events like 点击时 (when clicking)
|
|
934
|
-
zh: ['时', '的', '地', '得'],
|
|
935
|
-
// Japanese: Some particles may attach in casual writing
|
|
936
|
-
ja: [],
|
|
937
|
-
// Korean: Particles sometimes written without spaces
|
|
938
|
-
ko: [],
|
|
939
|
-
};
|
|
940
|
-
|
|
941
|
-
/**
|
|
942
|
-
* Known prefixes that may attach to words without spaces.
|
|
943
|
-
*/
|
|
944
|
-
const ATTACHED_PREFIXES: Record<string, string[]> = {
|
|
945
|
-
// Chinese: 当 (when) sometimes written attached
|
|
946
|
-
zh: ['当'],
|
|
947
|
-
// Arabic: Prepositions that attach
|
|
948
|
-
ar: ['بـ', 'كـ', 'و'],
|
|
949
|
-
};
|
|
950
|
-
|
|
951
|
-
/**
|
|
952
|
-
* Post-process tokens to split attached suffixes/prefixes.
|
|
953
|
-
* E.g., "点击时" → ["点击", "时"]
|
|
954
|
-
*/
|
|
955
|
-
function splitAttachedAffixes(tokens: string[], locale: string): string[] {
|
|
956
|
-
const suffixes = ATTACHED_SUFFIXES[locale] || [];
|
|
957
|
-
const prefixes = ATTACHED_PREFIXES[locale] || [];
|
|
958
|
-
|
|
959
|
-
if (suffixes.length === 0 && prefixes.length === 0) {
|
|
960
|
-
return tokens;
|
|
961
|
-
}
|
|
962
|
-
|
|
963
|
-
const result: string[] = [];
|
|
964
|
-
|
|
965
|
-
for (const token of tokens) {
|
|
966
|
-
// Skip CSS selectors and numbers
|
|
967
|
-
if (/^[#.<@]/.test(token) || /^\d+/.test(token)) {
|
|
968
|
-
result.push(token);
|
|
969
|
-
continue;
|
|
970
|
-
}
|
|
971
|
-
|
|
972
|
-
let processed = token;
|
|
973
|
-
let prefix = '';
|
|
974
|
-
let suffix = '';
|
|
975
|
-
|
|
976
|
-
// Check for attached prefixes
|
|
977
|
-
for (const p of prefixes) {
|
|
978
|
-
if (processed.startsWith(p) && processed.length > p.length) {
|
|
979
|
-
prefix = p;
|
|
980
|
-
processed = processed.slice(p.length);
|
|
981
|
-
break;
|
|
982
|
-
}
|
|
983
|
-
}
|
|
984
|
-
|
|
985
|
-
// Check for attached suffixes
|
|
986
|
-
for (const s of suffixes) {
|
|
987
|
-
if (processed.endsWith(s) && processed.length > s.length) {
|
|
988
|
-
suffix = s;
|
|
989
|
-
processed = processed.slice(0, -s.length);
|
|
990
|
-
break;
|
|
991
|
-
}
|
|
992
|
-
}
|
|
993
|
-
|
|
994
|
-
// Add tokens in order: prefix, main, suffix
|
|
995
|
-
if (prefix) result.push(prefix);
|
|
996
|
-
if (processed) result.push(processed);
|
|
997
|
-
if (suffix) result.push(suffix);
|
|
998
|
-
}
|
|
999
|
-
|
|
1000
|
-
return result;
|
|
1001
|
-
}
|
|
1002
|
-
|
|
1003
|
-
/**
|
|
1004
|
-
* Simple tokenizer that handles:
|
|
1005
|
-
* - Keywords (from dictionary)
|
|
1006
|
-
* - CSS selectors (#id, .class, <tag/>)
|
|
1007
|
-
* - String literals
|
|
1008
|
-
* - Numbers
|
|
1009
|
-
* - Attached suffixes/prefixes (language-specific)
|
|
1010
|
-
*/
|
|
1011
|
-
function tokenize(input: string, profile: LanguageProfile): string[] {
|
|
1012
|
-
// Split on whitespace, preserving selectors and strings
|
|
1013
|
-
const tokens: string[] = [];
|
|
1014
|
-
let current = '';
|
|
1015
|
-
let inSelector = false;
|
|
1016
|
-
let selectorDepth = 0;
|
|
1017
|
-
let bracketDepth = 0;
|
|
1018
|
-
let parenDepth = 0;
|
|
1019
|
-
|
|
1020
|
-
for (let i = 0; i < input.length; i++) {
|
|
1021
|
-
const char = input[i];
|
|
1022
|
-
|
|
1023
|
-
// Track CSS selector context
|
|
1024
|
-
if (char === '<') {
|
|
1025
|
-
inSelector = true;
|
|
1026
|
-
selectorDepth++;
|
|
1027
|
-
} else if (char === '>' && inSelector) {
|
|
1028
|
-
selectorDepth--;
|
|
1029
|
-
if (selectorDepth === 0) inSelector = false;
|
|
1030
|
-
}
|
|
1031
|
-
|
|
1032
|
-
// Track event-guard / attribute brackets so `[key is 'Escape']` (which has
|
|
1033
|
-
// internal spaces) stays a single token instead of splitting into
|
|
1034
|
-
// `[key` / `is` / `'Escape']` — which mis-assigns `is` as the action verb.
|
|
1035
|
-
if (char === '[') {
|
|
1036
|
-
bracketDepth++;
|
|
1037
|
-
} else if (char === ']' && bracketDepth > 0) {
|
|
1038
|
-
bracketDepth--;
|
|
1039
|
-
}
|
|
1040
|
-
|
|
1041
|
-
// Track EVERY parenthesized group as a depth scope so it stays ONE token:
|
|
1042
|
-
//
|
|
1043
|
-
// - An ATTACHED `(` (a call or event destructure: `pointerdown(clientX,
|
|
1044
|
-
// clientY)`, `Resizable(a, b)`) must not split at the comma-space — the
|
|
1045
|
-
// event-handler-head reorder would separate the halves and drop the
|
|
1046
|
-
// whole handler (tl behavior-resizable degenerate).
|
|
1047
|
-
// - A STANDALONE `(` opening an expression (`to (the value of #price as
|
|
1048
|
-
// Number) * (my value as Number)`) must be OPAQUE to role segmentation:
|
|
1049
|
-
// left loose, its interior `of`/`as`/`from` keywords hit the argument
|
|
1050
|
-
// modifier map, so the parser split the expression across roles — the
|
|
1051
|
-
// R1 cluster E mangle (computed-value: the transformer reordered INSIDE
|
|
1052
|
-
// the parens, embedded the event phrase mid-expression, and dropped the
|
|
1053
|
-
// whole second operand in every language). Interior keywords still
|
|
1054
|
-
// translate IN PLACE: role values are re-split on whitespace by
|
|
1055
|
-
// translateMultiWordValue, and translateWord strips paren punctuation
|
|
1056
|
-
// before its dictionary lookup (`($count or 0)` → `($count o 0)` in tl —
|
|
1057
|
-
// the operator translates without the group being torn apart).
|
|
1058
|
-
if (char === '(') {
|
|
1059
|
-
parenDepth++;
|
|
1060
|
-
} else if (char === ')' && parenDepth > 0) {
|
|
1061
|
-
parenDepth--;
|
|
1062
|
-
}
|
|
1063
|
-
|
|
1064
|
-
// Split on whitespace unless inside a selector, a bracket guard, or an
|
|
1065
|
-
// attached parenthesized argument list
|
|
1066
|
-
if (/\s/.test(char) && !inSelector && bracketDepth === 0 && parenDepth === 0) {
|
|
1067
|
-
if (current) {
|
|
1068
|
-
tokens.push(current);
|
|
1069
|
-
current = '';
|
|
1070
|
-
}
|
|
1071
|
-
} else {
|
|
1072
|
-
current += char;
|
|
1073
|
-
}
|
|
1074
|
-
}
|
|
1075
|
-
|
|
1076
|
-
if (current) {
|
|
1077
|
-
tokens.push(current);
|
|
1078
|
-
}
|
|
1079
|
-
|
|
1080
|
-
// Post-process to split attached affixes for languages that need it
|
|
1081
|
-
return splitAttachedAffixes(tokens, profile.code);
|
|
1082
|
-
}
|
|
1083
|
-
|
|
1084
|
-
/**
|
|
1085
|
-
* Identify what type of statement this is
|
|
1086
|
-
*/
|
|
1087
|
-
function identifyStatementType(
|
|
1088
|
-
tokens: string[],
|
|
1089
|
-
profile: LanguageProfile
|
|
1090
|
-
): 'event-handler' | 'command' | 'conditional' | 'unknown' {
|
|
1091
|
-
if (tokens.length === 0) return 'unknown';
|
|
1092
|
-
|
|
1093
|
-
const firstToken = tokens[0].toLowerCase();
|
|
1094
|
-
|
|
1095
|
-
// Check for event handler
|
|
1096
|
-
const eventMarker = profile.markers.find(m => m.role === 'event' && m.position === 'preposition');
|
|
1097
|
-
if (eventMarker && firstToken === eventMarker.form.toLowerCase()) {
|
|
1098
|
-
return 'event-handler';
|
|
1099
|
-
}
|
|
1100
|
-
|
|
1101
|
-
// Check if first token is a known event keyword (derived from profiles)
|
|
1102
|
-
if (EVENT_KEYWORDS.has(firstToken)) {
|
|
1103
|
-
return 'event-handler';
|
|
1104
|
-
}
|
|
1105
|
-
|
|
1106
|
-
// Check for conditional using shared constants
|
|
1107
|
-
if (CONDITIONAL_KEYWORDS.has(firstToken)) {
|
|
1108
|
-
return 'conditional';
|
|
1109
|
-
}
|
|
1110
|
-
|
|
1111
|
-
return 'command';
|
|
1112
|
-
}
|
|
1113
|
-
|
|
1114
|
-
/**
|
|
1115
|
-
* Parse an event handler statement
|
|
1116
|
-
* Pattern: on {event} {command} {target?} {modifiers?}
|
|
1117
|
-
*
|
|
1118
|
-
* Now handles modifiers like "by 3" in "on click increment #count by 3"
|
|
1119
|
-
*/
|
|
1120
|
-
function parseEventHandler(tokens: string[], profile: LanguageProfile): ParsedStatement {
|
|
1121
|
-
const roles = new Map<SemanticRole, ParsedElement>();
|
|
1122
|
-
|
|
1123
|
-
// Skip the event keyword (e.g., 'on', 'で', '当', etc.) - derived from profiles
|
|
1124
|
-
let startIndex = EVENT_KEYWORDS.has(tokens[0]?.toLowerCase()) ? 1 : 0;
|
|
1125
|
-
|
|
1126
|
-
// Next token is the event
|
|
1127
|
-
if (tokens[startIndex]) {
|
|
1128
|
-
const eventTokens: string[] = [tokens[startIndex]];
|
|
1129
|
-
startIndex++;
|
|
1130
|
-
|
|
1131
|
-
// Fold "or"-conjoined events into the event value (e.g.
|
|
1132
|
-
// "on click or keypress[...] toggle .active"). Without this, "or" would be
|
|
1133
|
-
// read as the action verb and the second event swept into role values,
|
|
1134
|
-
// hoisting "or <event2>" ahead of the command on reorder. Keeping the whole
|
|
1135
|
-
// "<event1> or <event2>" string in the event role lets it translate and
|
|
1136
|
-
// reorder as a single event clause. Only consumes "or" that immediately
|
|
1137
|
-
// follows the event, before any source/action — so a later "or" in a guard
|
|
1138
|
-
// or value is untouched.
|
|
1139
|
-
while (
|
|
1140
|
-
tokens[startIndex] &&
|
|
1141
|
-
EVENT_CONJUNCTIONS.has(tokens[startIndex].toLowerCase()) &&
|
|
1142
|
-
tokens[startIndex + 1]
|
|
1143
|
-
) {
|
|
1144
|
-
eventTokens.push(tokens[startIndex], tokens[startIndex + 1]);
|
|
1145
|
-
startIndex += 2;
|
|
1146
|
-
}
|
|
1147
|
-
|
|
1148
|
-
roles.set('event', {
|
|
1149
|
-
role: 'event',
|
|
1150
|
-
value: eventTokens.join(' '),
|
|
1151
|
-
});
|
|
1152
|
-
}
|
|
1153
|
-
|
|
1154
|
-
// Check for event source modifier before the action (e.g., "from #source" in "on input from #firstName set ...")
|
|
1155
|
-
// Only handle 'from' (event source) — other modifiers like circumfix markers (Chinese 时) should not be consumed.
|
|
1156
|
-
if (tokens[startIndex] && tokens[startIndex].toLowerCase() === 'from' && tokens[startIndex + 1]) {
|
|
1157
|
-
startIndex++; // skip 'from'
|
|
1158
|
-
// Collect source value tokens until a command keyword is found
|
|
1159
|
-
const sourceValue: string[] = [];
|
|
1160
|
-
while (tokens[startIndex]) {
|
|
1161
|
-
if (ENGLISH_COMMANDS.has(tokens[startIndex].toLowerCase())) break;
|
|
1162
|
-
sourceValue.push(tokens[startIndex]);
|
|
1163
|
-
startIndex++;
|
|
1164
|
-
}
|
|
1165
|
-
if (sourceValue.length > 0) {
|
|
1166
|
-
const value = sourceValue.join(' ');
|
|
1167
|
-
roles.set('source', {
|
|
1168
|
-
role: 'source',
|
|
1169
|
-
value,
|
|
1170
|
-
isSelector: /^[#.<@]/.test(value),
|
|
1171
|
-
});
|
|
1172
|
-
}
|
|
1173
|
-
}
|
|
1174
|
-
|
|
1175
|
-
// Next token is the action (command verb)
|
|
1176
|
-
if (tokens[startIndex]) {
|
|
1177
|
-
roles.set('action', {
|
|
1178
|
-
role: 'action',
|
|
1179
|
-
value: tokens[startIndex],
|
|
1180
|
-
});
|
|
1181
|
-
startIndex++;
|
|
1182
|
-
}
|
|
1183
|
-
|
|
1184
|
-
// Parse remaining tokens with modifier awareness (like parseCommand does)
|
|
1185
|
-
// This handles "by 3" in "on click increment #count by 3".
|
|
1186
|
-
// Argument map: for DOM-target commands the head event is already captured, so
|
|
1187
|
-
// a locative `on` (`toggle @hidden on #panel`) maps to `destination` rather
|
|
1188
|
-
// than overwriting the head `event`.
|
|
1189
|
-
if (tokens[startIndex]) {
|
|
1190
|
-
const modifierMap = buildArgumentModifierMap(profile, roles.get('action')?.value);
|
|
1191
|
-
let currentRole: SemanticRole = 'patient';
|
|
1192
|
-
let currentValue: string[] = [];
|
|
1193
|
-
|
|
1194
|
-
for (let i = startIndex; i < tokens.length; i++) {
|
|
1195
|
-
const token = tokens[i];
|
|
1196
|
-
const mappedRole = modifierMap[token.toLowerCase()];
|
|
1197
|
-
|
|
1198
|
-
if (mappedRole) {
|
|
1199
|
-
// Save previous role
|
|
1200
|
-
if (currentValue.length > 0) {
|
|
1201
|
-
const value = currentValue.join(' ');
|
|
1202
|
-
roles.set(currentRole, {
|
|
1203
|
-
role: currentRole,
|
|
1204
|
-
value,
|
|
1205
|
-
isSelector: /^[#.<@]/.test(value),
|
|
1206
|
-
});
|
|
1207
|
-
}
|
|
1208
|
-
currentRole = mappedRole;
|
|
1209
|
-
currentValue = [];
|
|
1210
|
-
} else {
|
|
1211
|
-
currentValue.push(token);
|
|
1212
|
-
}
|
|
1213
|
-
}
|
|
1214
|
-
|
|
1215
|
-
// Save final role
|
|
1216
|
-
if (currentValue.length > 0) {
|
|
1217
|
-
const value = currentValue.join(' ');
|
|
1218
|
-
roles.set(currentRole, {
|
|
1219
|
-
role: currentRole,
|
|
1220
|
-
value,
|
|
1221
|
-
isSelector: /^[#.<@]/.test(value),
|
|
1222
|
-
});
|
|
1223
|
-
}
|
|
1224
|
-
}
|
|
1225
|
-
|
|
1226
|
-
return {
|
|
1227
|
-
type: 'event-handler',
|
|
1228
|
-
roles,
|
|
1229
|
-
original: tokens.join(' '),
|
|
1230
|
-
};
|
|
1231
|
-
}
|
|
1232
|
-
|
|
1233
|
-
/**
|
|
1234
|
-
* Parse a command statement
|
|
1235
|
-
* Pattern: {command} {args...}
|
|
1236
|
-
*/
|
|
1237
|
-
function parseCommand(tokens: string[], profile: LanguageProfile): ParsedStatement {
|
|
1238
|
-
const roles = new Map<SemanticRole, ParsedElement>();
|
|
1239
|
-
|
|
1240
|
-
if (tokens.length === 0) {
|
|
1241
|
-
return { type: 'command', roles, original: '' };
|
|
1242
|
-
}
|
|
1243
|
-
|
|
1244
|
-
// First token is the command
|
|
1245
|
-
roles.set('action', {
|
|
1246
|
-
role: 'action',
|
|
1247
|
-
value: tokens[0],
|
|
1248
|
-
});
|
|
1249
|
-
|
|
1250
|
-
// Generate dynamic modifier map from language profile (argument position).
|
|
1251
|
-
// This enables parsing non-English input (e.g., Japanese に, Korean 에, Arabic إلى).
|
|
1252
|
-
// For a DOM-target command a locative `on` (`toggle .active on me`) maps to
|
|
1253
|
-
// `destination` (SVO sources) rather than spawning a bogus `event` role.
|
|
1254
|
-
const modifierMap = buildArgumentModifierMap(profile, tokens[0]);
|
|
1255
|
-
|
|
1256
|
-
let currentRole: SemanticRole = 'patient';
|
|
1257
|
-
let currentValue: string[] = [];
|
|
1258
|
-
|
|
1259
|
-
for (let i = 1; i < tokens.length; i++) {
|
|
1260
|
-
const token = tokens[i];
|
|
1261
|
-
const mappedRole = modifierMap[token.toLowerCase()];
|
|
1262
|
-
|
|
1263
|
-
if (mappedRole) {
|
|
1264
|
-
// Save previous role
|
|
1265
|
-
if (currentValue.length > 0) {
|
|
1266
|
-
const value = currentValue.join(' ');
|
|
1267
|
-
roles.set(currentRole, {
|
|
1268
|
-
role: currentRole,
|
|
1269
|
-
value,
|
|
1270
|
-
isSelector: /^[#.<@]/.test(value),
|
|
1271
|
-
});
|
|
1272
|
-
}
|
|
1273
|
-
currentRole = mappedRole;
|
|
1274
|
-
currentValue = [];
|
|
1275
|
-
} else {
|
|
1276
|
-
currentValue.push(token);
|
|
1277
|
-
}
|
|
1278
|
-
}
|
|
1279
|
-
|
|
1280
|
-
// Save final role
|
|
1281
|
-
if (currentValue.length > 0) {
|
|
1282
|
-
const value = currentValue.join(' ');
|
|
1283
|
-
roles.set(currentRole, {
|
|
1284
|
-
role: currentRole,
|
|
1285
|
-
value,
|
|
1286
|
-
isSelector: /^[#.<@]/.test(value),
|
|
1287
|
-
});
|
|
1288
|
-
}
|
|
1289
|
-
|
|
1290
|
-
return {
|
|
1291
|
-
type: 'command',
|
|
1292
|
-
roles,
|
|
1293
|
-
original: tokens.join(' '),
|
|
1294
|
-
};
|
|
1295
|
-
}
|
|
1296
|
-
|
|
1297
|
-
/**
|
|
1298
|
-
* Re-assign a command's mis-marked primary argument from the default `patient`
|
|
1299
|
-
* role to its true primary role (e.g. `wait`'s leading argument is a `duration`,
|
|
1300
|
-
* not a fronted object).
|
|
1301
|
-
*
|
|
1302
|
-
* The generic argument parser in `parseCommand` / `parseEventHandler` defaults the
|
|
1303
|
-
* first unmarked argument to `patient`. For most commands that is correct, but for a
|
|
1304
|
-
* command whose primary role is a non-patient *and which the target language does not
|
|
1305
|
-
* give a marker* (a duration, a measure — never a BA/object construction), the
|
|
1306
|
-
* `patient` assignment makes `insertMarkers` emit a spurious object-marker (Chinese
|
|
1307
|
-
* `把`, Japanese `を`, Korean `를`). The marked form is ungrammatical and the semantic
|
|
1308
|
-
* parser fails to match it, dropping the command.
|
|
1309
|
-
*
|
|
1310
|
-
* The fix is deliberately scoped to **literal/measure** primary roles
|
|
1311
|
-
* ({@link LITERAL_PRIMARY_ROLES}: `duration`, `quantity`) — values that are *never*
|
|
1312
|
-
* the object of an object-marking construction in any supported language, so moving
|
|
1313
|
-
* them off `patient` can only ever *remove* a spurious marker. Marker-bearing
|
|
1314
|
-
* primaries are intentionally left alone:
|
|
1315
|
-
* - `set`→destination `到`, `fetch`→source `从` carry a *correct* marker; touching
|
|
1316
|
-
* them risks no benefit.
|
|
1317
|
-
* - `send`/`trigger`→event: in a language without an event marker (e.g. Korean has
|
|
1318
|
-
* no event particle) un-marking the leading argument makes the semantic parser
|
|
1319
|
-
* mis-read it as a bare event handler and emit a phantom `on` action — an
|
|
1320
|
-
* over-generation that recall-based fidelity would not catch. So `event`-primary
|
|
1321
|
-
* commands stay on the `patient` default.
|
|
1322
|
-
*
|
|
1323
|
-
* Roles are still reordered by the safety-net in `reorderRoles`, so no value is lost.
|
|
1324
|
-
* A belt-and-suspenders target-marker guard keeps the change inert should a profile
|
|
1325
|
-
* ever add a marker for one of these literal roles.
|
|
1326
|
-
*
|
|
1327
|
-
* Must run *before* `translateElements`, while the `action` value is still the
|
|
1328
|
-
* source-language (English) keyword, so the schema lookup resolves.
|
|
1329
|
-
*
|
|
1330
|
-
* @see docs-internal/ZH_BLOCK_BODY_SCOPE.md (#1 — transformer role model)
|
|
1331
|
-
*/
|
|
1332
|
-
const LITERAL_PRIMARY_ROLES: ReadonlySet<SemanticRole> = new Set<SemanticRole>([
|
|
1333
|
-
'duration',
|
|
1334
|
-
'quantity',
|
|
1335
|
-
]);
|
|
1336
|
-
|
|
1337
|
-
function applyPrimaryRole(parsed: ParsedStatement, targetProfile: LanguageProfile): void {
|
|
1338
|
-
// Only re-mark standalone command statements. In an event handler (`on click
|
|
1339
|
-
// wait 2s …`) a verb-final SOV language without an event particle (e.g. Korean)
|
|
1340
|
-
// relies on the leading argument's patient marker as the structural cue that
|
|
1341
|
-
// anchors the handler; un-marking it makes the semantic parser lose the event.
|
|
1342
|
-
// The block-body / then-chain `wait` clauses this fix targets are each parsed as
|
|
1343
|
-
// their own command statement, so they are still covered.
|
|
1344
|
-
if (parsed.type !== 'command') return;
|
|
1345
|
-
|
|
1346
|
-
const action = parsed.roles.get('action')?.value;
|
|
1347
|
-
if (!action) return;
|
|
1348
|
-
|
|
1349
|
-
const primaryRole = COMMAND_PRIMARY_ROLES[action.toLowerCase()];
|
|
1350
|
-
if (!primaryRole || !LITERAL_PRIMARY_ROLES.has(primaryRole)) return;
|
|
1351
|
-
|
|
1352
|
-
// Only act on a leading argument the generic parser defaulted to `patient`,
|
|
1353
|
-
// and only when the primary slot isn't already filled by an explicit marker.
|
|
1354
|
-
const patientEl = parsed.roles.get('patient');
|
|
1355
|
-
if (!patientEl || parsed.roles.has(primaryRole)) return;
|
|
1356
|
-
|
|
1357
|
-
// Guard: never introduce a marker that wasn't there. If the target language
|
|
1358
|
-
// marks the primary role, leave the command as-is.
|
|
1359
|
-
if (targetProfile.markers.some(m => m.role === primaryRole)) return;
|
|
1360
|
-
|
|
1361
|
-
parsed.roles.delete('patient');
|
|
1362
|
-
parsed.roles.set(primaryRole, { ...patientEl, role: primaryRole });
|
|
1363
|
-
}
|
|
1364
|
-
|
|
1365
|
-
/**
|
|
1366
|
-
* Parse a conditional statement
|
|
1367
|
-
*/
|
|
1368
|
-
function parseConditional(tokens: string[], _profile: LanguageProfile): ParsedStatement {
|
|
1369
|
-
const roles = new Map<SemanticRole, ParsedElement>();
|
|
1370
|
-
|
|
1371
|
-
// First token is the 'if' keyword
|
|
1372
|
-
roles.set('action', {
|
|
1373
|
-
role: 'action',
|
|
1374
|
-
value: tokens[0],
|
|
1375
|
-
});
|
|
1376
|
-
|
|
1377
|
-
// Find 'then' to split condition from body - using shared constants
|
|
1378
|
-
const thenIndex = tokens.findIndex(t => THEN_KEYWORDS.has(t.toLowerCase()));
|
|
1379
|
-
|
|
1380
|
-
if (thenIndex > 1) {
|
|
1381
|
-
const conditionValue = tokens.slice(1, thenIndex).join(' ');
|
|
1382
|
-
roles.set('condition', {
|
|
1383
|
-
role: 'condition',
|
|
1384
|
-
value: conditionValue,
|
|
1385
|
-
});
|
|
1386
|
-
} else if (thenIndex === -1 && tokens.length > 1) {
|
|
1387
|
-
// Block-style `if <cond>` with the body on following lines (no inline
|
|
1388
|
-
// `then`). Capture everything after `if` as the condition; otherwise the
|
|
1389
|
-
// condition is silently dropped and the rendered block becomes a bare
|
|
1390
|
-
// `if`/`אם`/`如果`, which the semantic block parser then rejects (null
|
|
1391
|
-
// parse). This is the dominant failure for nested control-flow bodies in
|
|
1392
|
-
// non-Latin languages (he, zh).
|
|
1393
|
-
roles.set('condition', {
|
|
1394
|
-
role: 'condition',
|
|
1395
|
-
value: tokens.slice(1).join(' '),
|
|
1396
|
-
});
|
|
1397
|
-
}
|
|
1398
|
-
|
|
1399
|
-
return {
|
|
1400
|
-
type: 'conditional',
|
|
1401
|
-
roles,
|
|
1402
|
-
original: tokens.join(' '),
|
|
1403
|
-
};
|
|
1404
|
-
}
|
|
1405
|
-
|
|
1406
|
-
// =============================================================================
|
|
1407
|
-
// Translation
|
|
1408
|
-
// =============================================================================
|
|
1409
|
-
|
|
1410
|
-
/**
|
|
1411
|
-
* Translate words using dictionary with type-safe access.
|
|
1412
|
-
*/
|
|
1413
|
-
function translateWord(word: string, sourceLocale: string, targetLocale: string): string {
|
|
1414
|
-
// Don't translate CSS selectors
|
|
1415
|
-
if (/^[#.<@]/.test(word)) {
|
|
1416
|
-
return word;
|
|
1417
|
-
}
|
|
1418
|
-
|
|
1419
|
-
// Don't translate numbers
|
|
1420
|
-
if (/^\d+/.test(word)) {
|
|
1421
|
-
return word;
|
|
1422
|
-
}
|
|
1423
|
-
|
|
1424
|
-
// A whole parenthesized group fused by the tokenizer (`(the value of #price
|
|
1425
|
-
// as Number)`) reaching a single-token translate path: translate its
|
|
1426
|
-
// interior word-by-word IN ORDER — never reordered, never re-segmented.
|
|
1427
|
-
if (/\s/.test(word) && word.startsWith('(')) {
|
|
1428
|
-
return word
|
|
1429
|
-
.split(/\s+/)
|
|
1430
|
-
.map(w => translateWord(w, sourceLocale, targetLocale))
|
|
1431
|
-
.join(' ');
|
|
1432
|
-
}
|
|
1433
|
-
|
|
1434
|
-
// A word carrying paren punctuation from a fused group after whitespace
|
|
1435
|
-
// re-splitting (`(my` / `valor)` / `(($x`): strip the parens for the
|
|
1436
|
-
// dictionary lookup and re-attach, so interior keywords still translate.
|
|
1437
|
-
if (word.length > 1 && (word.startsWith('(') || word.endsWith(')'))) {
|
|
1438
|
-
const m = word.match(/^(\(*)([^()]+)(\)*)$/);
|
|
1439
|
-
if (m && (m[1] || m[3])) {
|
|
1440
|
-
return m[1] + translateWord(m[2], sourceLocale, targetLocale) + m[3];
|
|
1441
|
-
}
|
|
1442
|
-
}
|
|
1443
|
-
|
|
1444
|
-
const sourceDict = sourceLocale === 'en' ? null : dictionaries[sourceLocale];
|
|
1445
|
-
const targetDict = dictionaries[targetLocale];
|
|
1446
|
-
|
|
1447
|
-
if (!targetDict) return word;
|
|
1448
|
-
|
|
1449
|
-
// If source is not English, first map to English using type-safe lookup
|
|
1450
|
-
let englishWord = word;
|
|
1451
|
-
if (sourceDict) {
|
|
1452
|
-
const found = findInDictionary(sourceDict, word);
|
|
1453
|
-
if (found) {
|
|
1454
|
-
englishWord = found.englishKey;
|
|
1455
|
-
}
|
|
1456
|
-
}
|
|
1457
|
-
|
|
1458
|
-
// Now map English to target locale using type-safe lookup
|
|
1459
|
-
const translated = translateFromEnglish(targetDict, englishWord);
|
|
1460
|
-
return translated ?? word;
|
|
1461
|
-
}
|
|
1462
|
-
|
|
1463
|
-
/**
|
|
1464
|
-
* Possessive markers for each language.
|
|
1465
|
-
* Used to transform "X's Y" patterns to target language structure.
|
|
1466
|
-
*/
|
|
1467
|
-
const POSSESSIVE_MARKERS: Record<
|
|
1468
|
-
string,
|
|
1469
|
-
{ type: 'prefix' | 'suffix' | 'preposition' | 'particle'; marker: string }
|
|
1470
|
-
> = {
|
|
1471
|
-
en: { type: 'suffix', marker: "'s" },
|
|
1472
|
-
es: { type: 'preposition', marker: 'de' },
|
|
1473
|
-
pt: { type: 'preposition', marker: 'de' },
|
|
1474
|
-
fr: { type: 'preposition', marker: 'de' },
|
|
1475
|
-
de: { type: 'preposition', marker: 'von' },
|
|
1476
|
-
ja: { type: 'suffix', marker: 'の' },
|
|
1477
|
-
ko: { type: 'suffix', marker: '의' },
|
|
1478
|
-
zh: { type: 'suffix', marker: '的' },
|
|
1479
|
-
ar: { type: 'preposition', marker: 'لـ' },
|
|
1480
|
-
// Spaced genitive particle (not the glued `'ın`), so the tokenizer can split
|
|
1481
|
-
// it off the selector — consistent with Turkish's other spaced case markers.
|
|
1482
|
-
tr: { type: 'particle', marker: 'ın' },
|
|
1483
|
-
id: { type: 'preposition', marker: 'dari' },
|
|
1484
|
-
// Latin-script genitive: must be a *spaced* particle (`#picker pa`), since a
|
|
1485
|
-
// glued `#pickerpa` can't be split from the selector by the tokenizer the
|
|
1486
|
-
// way a non-Latin suffix (の/의/র) can.
|
|
1487
|
-
qu: { type: 'particle', marker: 'pa' },
|
|
1488
|
-
// Bengali SOV postposition genitive, like ja/ko — a spaced suffix the
|
|
1489
|
-
// tokenizer splits off as a particle. Previously absent, so it fell back to
|
|
1490
|
-
// the English `'s` marker and its possessive property paths never parsed.
|
|
1491
|
-
// (Hindi `का` is intentionally omitted: its `bind` lacks a verb-final
|
|
1492
|
-
// grammar rule, so fixing its possessive alone yields a wrong `on` parse —
|
|
1493
|
-
// tracked as separate follow-up.)
|
|
1494
|
-
bn: { type: 'suffix', marker: 'র' },
|
|
1495
|
-
sw: { type: 'preposition', marker: 'ya' },
|
|
1496
|
-
};
|
|
1497
|
-
|
|
1498
|
-
/**
|
|
1499
|
-
* Transform possessive 's syntax to target language.
|
|
1500
|
-
*
|
|
1501
|
-
* Examples:
|
|
1502
|
-
* me's value → mi valor (Spanish - pronoun becomes possessive adjective)
|
|
1503
|
-
* #button's textContent → textContent de #button (Spanish - prepositional)
|
|
1504
|
-
* me's value → 私の値 (Japanese - の particle)
|
|
1505
|
-
*/
|
|
1506
|
-
function translatePossessive(token: string, sourceLocale: string, targetLocale: string): string {
|
|
1507
|
-
// Check for 's possessive pattern
|
|
1508
|
-
const possessiveMatch = token.match(/^(.+)'s$/i);
|
|
1509
|
-
if (!possessiveMatch) {
|
|
1510
|
-
return token;
|
|
1511
|
-
}
|
|
1512
|
-
|
|
1513
|
-
const owner = possessiveMatch[1];
|
|
1514
|
-
const targetMarker = POSSESSIVE_MARKERS[targetLocale] || POSSESSIVE_MARKERS.en;
|
|
1515
|
-
|
|
1516
|
-
// Check if owner is a pronoun that has a possessive form
|
|
1517
|
-
const pronounPossessives: Record<string, string> = {
|
|
1518
|
-
me: 'my',
|
|
1519
|
-
it: 'its',
|
|
1520
|
-
you: 'your',
|
|
1521
|
-
};
|
|
1522
|
-
|
|
1523
|
-
const lowerOwner = owner.toLowerCase();
|
|
1524
|
-
if (pronounPossessives[lowerOwner]) {
|
|
1525
|
-
// Convert "me's" to "my" then translate
|
|
1526
|
-
const possessiveForm = pronounPossessives[lowerOwner];
|
|
1527
|
-
return translateWord(possessiveForm, 'en', targetLocale);
|
|
1528
|
-
}
|
|
1529
|
-
|
|
1530
|
-
// For selectors and other owners, translate owner and apply target possessive marker
|
|
1531
|
-
const translatedOwner = translateWord(owner, sourceLocale, targetLocale);
|
|
1532
|
-
|
|
1533
|
-
switch (targetMarker.type) {
|
|
1534
|
-
case 'suffix':
|
|
1535
|
-
// Japanese/Korean/Chinese: owner + marker (e.g., #buttonの, #button의)
|
|
1536
|
-
return `${translatedOwner}${targetMarker.marker}`;
|
|
1537
|
-
case 'particle':
|
|
1538
|
-
// Latin-script spaced genitive (Quechua `pa`): owner + space + marker
|
|
1539
|
-
// so the tokenizer can separate it from the selector.
|
|
1540
|
-
return `${translatedOwner} ${targetMarker.marker}`;
|
|
1541
|
-
case 'preposition':
|
|
1542
|
-
// Will be handled by caller - return marker + owner format
|
|
1543
|
-
// Store as special format to be processed later
|
|
1544
|
-
return `__POSS__${targetMarker.marker}__${translatedOwner}__POSS__`;
|
|
1545
|
-
default:
|
|
1546
|
-
return `${translatedOwner}'s`;
|
|
1547
|
-
}
|
|
1548
|
-
}
|
|
1549
|
-
|
|
1550
|
-
// =============================================================================
|
|
1551
|
-
// Possessive Dot Notation
|
|
1552
|
-
// =============================================================================
|
|
1553
|
-
|
|
1554
|
-
/**
|
|
1555
|
-
* Regex to match possessive dot notation patterns.
|
|
1556
|
-
* Matches: my.prop, its.prop, your.prop, me.prop, it.prop, you.prop
|
|
1557
|
-
* Also matches optional chaining: my?.prop, me?.prop, etc.
|
|
1558
|
-
*/
|
|
1559
|
-
const POSSESSIVE_DOT_REGEX = /^(my|its|your|me|it|you)(\??\..+)$/i;
|
|
1560
|
-
|
|
1561
|
-
/**
|
|
1562
|
-
* Map pronoun forms to possessive adjective forms for dictionary lookup.
|
|
1563
|
-
*/
|
|
1564
|
-
const POSSESSIVE_DOT_PRONOUNS: Record<string, string> = {
|
|
1565
|
-
me: 'my',
|
|
1566
|
-
it: 'its',
|
|
1567
|
-
you: 'your',
|
|
1568
|
-
my: 'my',
|
|
1569
|
-
its: 'its',
|
|
1570
|
-
your: 'your',
|
|
1571
|
-
};
|
|
1572
|
-
|
|
1573
|
-
/**
|
|
1574
|
-
* Translate possessive dot notation like my.textContent → mi.textContent.
|
|
1575
|
-
* Handles both possessive adjective forms (my, its, your) and pronoun forms (me, it, you).
|
|
1576
|
-
* Also handles optional chaining (my?.prop).
|
|
1577
|
-
* Returns null if the value doesn't match or no translation is available.
|
|
1578
|
-
*/
|
|
1579
|
-
function translatePossessiveDotNotation(
|
|
1580
|
-
value: string,
|
|
1581
|
-
sourceLocale: string,
|
|
1582
|
-
targetLocale: string
|
|
1583
|
-
): string | null {
|
|
1584
|
-
const match = value.match(POSSESSIVE_DOT_REGEX);
|
|
1585
|
-
if (!match) return null;
|
|
1586
|
-
|
|
1587
|
-
const possessiveWord = match[1].toLowerCase();
|
|
1588
|
-
const propertySuffix = match[2]; // ".textContent" or "?.textContent"
|
|
1589
|
-
|
|
1590
|
-
// Normalize to possessive adjective form for dictionary lookup
|
|
1591
|
-
const possessiveKey = POSSESSIVE_DOT_PRONOUNS[possessiveWord] || possessiveWord;
|
|
1592
|
-
const translated = translateWord(possessiveKey, sourceLocale, targetLocale);
|
|
1593
|
-
|
|
1594
|
-
// Skip if translation is multi-word (can't prefix dot notation)
|
|
1595
|
-
if (translated.includes(' ')) return null;
|
|
1596
|
-
|
|
1597
|
-
if (translated !== possessiveKey) {
|
|
1598
|
-
return translated + propertySuffix;
|
|
1599
|
-
}
|
|
1600
|
-
|
|
1601
|
-
// Try original pronoun form if different from possessive key
|
|
1602
|
-
if (possessiveWord !== possessiveKey) {
|
|
1603
|
-
const alt = translateWord(possessiveWord, sourceLocale, targetLocale);
|
|
1604
|
-
if (alt !== possessiveWord && !alt.includes(' ')) {
|
|
1605
|
-
return alt + propertySuffix;
|
|
1606
|
-
}
|
|
1607
|
-
}
|
|
1608
|
-
|
|
1609
|
-
return null;
|
|
1610
|
-
}
|
|
1611
|
-
|
|
1612
|
-
// =============================================================================
|
|
1613
|
-
// Multi-Word Value Translation
|
|
1614
|
-
// =============================================================================
|
|
1615
|
-
|
|
1616
|
-
/**
|
|
1617
|
-
* Translate a multi-word value, translating each word individually.
|
|
1618
|
-
* Handles possessives like "my value" → "mi valor" in Spanish.
|
|
1619
|
-
* Also handles 's possessive syntax like "me's value" → "mi valor".
|
|
1620
|
-
* Also handles possessive dot notation like "my.textContent" → "mi.textContent".
|
|
1621
|
-
*/
|
|
1622
|
-
function translateMultiWordValue(
|
|
1623
|
-
value: string,
|
|
1624
|
-
sourceLocale: string,
|
|
1625
|
-
targetLocale: string
|
|
1626
|
-
): string {
|
|
1627
|
-
// Mask event-guard / attribute brackets (`[key is 'Escape']`): their contents
|
|
1628
|
-
// are expression syntax, not translatable keywords — translating `is` -> `ni`
|
|
1629
|
-
// etc. inside them breaks the guard. Restore verbatim after translation.
|
|
1630
|
-
if (value.includes('[')) {
|
|
1631
|
-
const guards: string[] = [];
|
|
1632
|
-
const masked = value.replace(/\[[^\]]*\]/g, match => {
|
|
1633
|
-
guards.push(match);
|
|
1634
|
-
return `${guards.length - 1}`;
|
|
1635
|
-
});
|
|
1636
|
-
if (guards.length > 0) {
|
|
1637
|
-
const translated = translateMultiWordValue(masked, sourceLocale, targetLocale);
|
|
1638
|
-
return translated.replace(/(\d+)/g, (_, n) => guards[Number(n)]);
|
|
1639
|
-
}
|
|
1640
|
-
}
|
|
1641
|
-
|
|
1642
|
-
// If it's a single word, check for possessive then translate
|
|
1643
|
-
if (!value.includes(' ')) {
|
|
1644
|
-
// Check for possessive 's
|
|
1645
|
-
if (value.includes("'s")) {
|
|
1646
|
-
return translatePossessive(value, sourceLocale, targetLocale);
|
|
1647
|
-
}
|
|
1648
|
-
// Check for possessive dot notation (my.prop, its.prop, me.prop, etc.)
|
|
1649
|
-
const dotResult = translatePossessiveDotNotation(value, sourceLocale, targetLocale);
|
|
1650
|
-
if (dotResult !== null) return dotResult;
|
|
1651
|
-
|
|
1652
|
-
return translateWord(value, sourceLocale, targetLocale);
|
|
1653
|
-
}
|
|
1654
|
-
|
|
1655
|
-
// Split into words and translate each
|
|
1656
|
-
const words = value.split(/\s+/);
|
|
1657
|
-
const translated: string[] = [];
|
|
1658
|
-
let i = 0;
|
|
1659
|
-
|
|
1660
|
-
while (i < words.length) {
|
|
1661
|
-
const word = words[i];
|
|
1662
|
-
|
|
1663
|
-
// Check for possessive 's pattern FIRST (e.g., "me's value", "#button's textContent")
|
|
1664
|
-
// This must come before selector check because "#button's" starts with #
|
|
1665
|
-
if (word.includes("'s")) {
|
|
1666
|
-
const possessiveResult = translatePossessive(word, sourceLocale, targetLocale);
|
|
1667
|
-
|
|
1668
|
-
// Check if it's a prepositional possessive that needs reordering
|
|
1669
|
-
const prepMatch = possessiveResult.match(/^__POSS__(.+)__(.+)__POSS__$/);
|
|
1670
|
-
if (prepMatch && i + 1 < words.length) {
|
|
1671
|
-
// Prepositional: "X's Y" → "Y marker X" (e.g., "textContent de #button")
|
|
1672
|
-
const marker = prepMatch[1];
|
|
1673
|
-
const owner = prepMatch[2];
|
|
1674
|
-
const property = words[i + 1];
|
|
1675
|
-
const translatedProperty = translateWord(property, sourceLocale, targetLocale);
|
|
1676
|
-
translated.push(`${translatedProperty} ${marker} ${owner}`);
|
|
1677
|
-
i += 2; // Skip property since we consumed it
|
|
1678
|
-
continue;
|
|
1679
|
-
} else if (prepMatch) {
|
|
1680
|
-
// No property following - just output owner with marker prefix
|
|
1681
|
-
const marker = prepMatch[1];
|
|
1682
|
-
const owner = prepMatch[2];
|
|
1683
|
-
translated.push(`${marker} ${owner}`);
|
|
1684
|
-
i++;
|
|
1685
|
-
continue;
|
|
1686
|
-
}
|
|
1687
|
-
|
|
1688
|
-
// Suffix-style possessive (Japanese, Korean, etc.) or pronoun
|
|
1689
|
-
translated.push(possessiveResult);
|
|
1690
|
-
i++;
|
|
1691
|
-
continue;
|
|
1692
|
-
}
|
|
1693
|
-
|
|
1694
|
-
// Skip pure CSS selectors and numbers (but NOT possessives which were handled above)
|
|
1695
|
-
if (/^[#.<@]/.test(word) || /^\d+/.test(word)) {
|
|
1696
|
-
translated.push(word);
|
|
1697
|
-
i++;
|
|
1698
|
-
continue;
|
|
1699
|
-
}
|
|
1700
|
-
|
|
1701
|
-
// Skip quoted strings
|
|
1702
|
-
if (/^["'].*["']$/.test(word)) {
|
|
1703
|
-
translated.push(word);
|
|
1704
|
-
i++;
|
|
1705
|
-
continue;
|
|
1706
|
-
}
|
|
1707
|
-
|
|
1708
|
-
// Check for possessive dot notation (my.prop, its.prop, me.prop, etc.)
|
|
1709
|
-
const dotResult = translatePossessiveDotNotation(word, sourceLocale, targetLocale);
|
|
1710
|
-
if (dotResult !== null) {
|
|
1711
|
-
translated.push(dotResult);
|
|
1712
|
-
i++;
|
|
1713
|
-
continue;
|
|
1714
|
-
}
|
|
1715
|
-
|
|
1716
|
-
translated.push(translateWord(word, sourceLocale, targetLocale));
|
|
1717
|
-
i++;
|
|
1718
|
-
}
|
|
1719
|
-
|
|
1720
|
-
return translated.join(' ');
|
|
1721
|
-
}
|
|
1722
|
-
|
|
1723
|
-
/**
|
|
1724
|
-
* Translate all elements in a parsed statement
|
|
1725
|
-
*/
|
|
1726
|
-
function translateElements(
|
|
1727
|
-
parsed: ParsedStatement,
|
|
1728
|
-
sourceLocale: string,
|
|
1729
|
-
targetLocale: string
|
|
1730
|
-
): void {
|
|
1731
|
-
for (const [_role, element] of parsed.roles) {
|
|
1732
|
-
// Always process possessive 's syntax, even for selectors
|
|
1733
|
-
// E.g., "#button's textContent" should translate the possessive
|
|
1734
|
-
if (element.value.includes("'s")) {
|
|
1735
|
-
element.translated = translateMultiWordValue(element.value, sourceLocale, targetLocale);
|
|
1736
|
-
} else if (!element.isSelector && !element.isLiteral) {
|
|
1737
|
-
element.translated = translateMultiWordValue(element.value, sourceLocale, targetLocale);
|
|
1738
|
-
} else {
|
|
1739
|
-
element.translated = element.value;
|
|
1740
|
-
}
|
|
1741
|
-
}
|
|
1742
|
-
}
|
|
1743
|
-
|
|
1744
|
-
// =============================================================================
|
|
1745
|
-
// Caret-scope masking (`^name on <selector>`)
|
|
1746
|
-
// =============================================================================
|
|
1747
|
-
|
|
1748
|
-
/** Private-use sentinels bracketing a masked caret-scope index. */
|
|
1749
|
-
const CARET_SCOPE_OPEN = '\uE000';
|
|
1750
|
-
const CARET_SCOPE_CLOSE = '\uE001';
|
|
1751
|
-
|
|
1752
|
-
/**
|
|
1753
|
-
* Match `^name on <selector>` and the scope's selector form (#id, .class,
|
|
1754
|
-
* <tag/>, [attr]). The `^name` is kept; only the ` on <selector>` scope is masked.
|
|
1755
|
-
*/
|
|
1756
|
-
const CARET_SCOPE_RE = /(\^[A-Za-z_][\w-]*)(\s+on\s+(?:[#.][\w-]+|<[^>]*\/>|\[[^\]]+\]))/g;
|
|
1757
|
-
|
|
1758
|
-
/**
|
|
1759
|
-
* Mask the ` on <selector>` scope of every `^name on <selector>` read behind an
|
|
1760
|
-
* opaque token attached to `^name`, so the overloaded `on` doesn't reach the
|
|
1761
|
-
* splitter / event-handler parser. Returns null when there's nothing to mask.
|
|
1762
|
-
*/
|
|
1763
|
-
function maskCaretScopes(input: string): { masked: string; scopes: string[] } | null {
|
|
1764
|
-
const scopes: string[] = [];
|
|
1765
|
-
const masked = input.replace(CARET_SCOPE_RE, (_m, varTok: string, scope: string) => {
|
|
1766
|
-
const idx = scopes.length;
|
|
1767
|
-
scopes.push(scope);
|
|
1768
|
-
return `${varTok}${CARET_SCOPE_OPEN}${idx}${CARET_SCOPE_CLOSE}`;
|
|
1769
|
-
});
|
|
1770
|
-
return scopes.length > 0 ? { masked, scopes } : null;
|
|
1771
|
-
}
|
|
1772
|
-
|
|
1773
|
-
/** Restore masked caret-scope tokens to their verbatim ` on <selector>` form. */
|
|
1774
|
-
function restoreCaretScopes(input: string, scopes: string[]): string {
|
|
1775
|
-
return input.replace(
|
|
1776
|
-
new RegExp(`${CARET_SCOPE_OPEN}(\\d+)${CARET_SCOPE_CLOSE}`, 'g'),
|
|
1777
|
-
(_m, n: string) => scopes[Number(n)] ?? ''
|
|
1778
|
-
);
|
|
1779
|
-
}
|
|
1780
|
-
|
|
1781
|
-
// =============================================================================
|
|
1782
|
-
// View-transition tail masking (`using view transition`)
|
|
1783
|
-
// =============================================================================
|
|
1784
|
-
|
|
1785
|
-
/** Private-use sentinels bracketing a masked view-transition-tail index. */
|
|
1786
|
-
const VIEW_TAIL_OPEN = '\uE002';
|
|
1787
|
-
const VIEW_TAIL_CLOSE = '\uE003';
|
|
1788
|
-
|
|
1789
|
-
/**
|
|
1790
|
-
* Match `swap`/`process`'s trailing `using view transition` modifier.
|
|
1791
|
-
*
|
|
1792
|
-
* `using` is in no dictionary and in no role table, so it sweeps into whatever
|
|
1793
|
-
* role phrase is open; `transition` IS a translated command keyword in every
|
|
1794
|
-
* dictionary (es `transición`, de `übergang`, ja `遷移`) and is in
|
|
1795
|
-
* `ENGLISH_COMMANDS`, so `splitOnCommandBoundaries` splits the clause there and
|
|
1796
|
-
* the rejoin plants a phantom translated `transition` COMMAND after the target's
|
|
1797
|
-
* `then`-connective (`intercambiar #a con #b using view entonces transición`).
|
|
1798
|
-
*
|
|
1799
|
-
* The phrase has no native form in any of the 24 languages — semantic's
|
|
1800
|
-
* `USING_VIEW_MARKER_ALL_LANGS` matches the literal English `using view` marker
|
|
1801
|
-
* everywhere — so it is a passthrough, matched on the English surface regardless
|
|
1802
|
-
* of source locale. The value word is optional so a bare `using view` still
|
|
1803
|
-
* masks rather than half-splitting; `then` is excluded so a clause boundary is
|
|
1804
|
-
* never swallowed into the tail.
|
|
1805
|
-
*/
|
|
1806
|
-
const VIEW_TAIL_RE = /\busing\s+view\b(?:\s+(?!then\b)[A-Za-z][\w-]*)?/gi;
|
|
1807
|
-
|
|
1808
|
-
/** Whether a token is a masked view-transition tail. */
|
|
1809
|
-
const VIEW_TAIL_TOKEN_RE = new RegExp(`^${VIEW_TAIL_OPEN}(\\d+)${VIEW_TAIL_CLOSE}$`);
|
|
1810
|
-
|
|
1811
|
-
/**
|
|
1812
|
-
* Mask every `using view <value>` tail behind an opaque token, so the phrase
|
|
1813
|
-
* never reaches the splitter, the word translator, or the role parser. Returns
|
|
1814
|
-
* null when there's nothing to mask.
|
|
1815
|
-
*/
|
|
1816
|
-
function maskViewTransitionTails(input: string): { masked: string; tails: string[] } | null {
|
|
1817
|
-
const tails: string[] = [];
|
|
1818
|
-
const masked = input.replace(VIEW_TAIL_RE, match => {
|
|
1819
|
-
const idx = tails.length;
|
|
1820
|
-
tails.push(match);
|
|
1821
|
-
return `${VIEW_TAIL_OPEN}${idx}${VIEW_TAIL_CLOSE}`;
|
|
1822
|
-
});
|
|
1823
|
-
return tails.length > 0 ? { masked, tails } : null;
|
|
1824
|
-
}
|
|
1825
|
-
|
|
1826
|
-
/** Restore masked view-transition tokens to their verbatim English phrase. */
|
|
1827
|
-
function restoreViewTransitionTails(input: string, tails: string[]): string {
|
|
1828
|
-
return input.replace(
|
|
1829
|
-
new RegExp(`${VIEW_TAIL_OPEN}(\\d+)${VIEW_TAIL_CLOSE}`, 'g'),
|
|
1830
|
-
(_m, n: string) => tails[Number(n)] ?? ''
|
|
1831
|
-
);
|
|
1832
|
-
}
|
|
1833
|
-
|
|
1834
|
-
// =============================================================================
|
|
1835
|
-
// Main Transformer
|
|
1836
|
-
// =============================================================================
|
|
1837
|
-
|
|
1838
|
-
export class GrammarTransformer {
|
|
1839
|
-
private sourceProfile: LanguageProfile;
|
|
1840
|
-
private targetProfile: LanguageProfile;
|
|
1841
|
-
|
|
1842
|
-
constructor(sourceLocale: string = 'en', targetLocale: string) {
|
|
1843
|
-
const source = getProfile(sourceLocale);
|
|
1844
|
-
const target = getProfile(targetLocale);
|
|
1845
|
-
|
|
1846
|
-
if (!source) throw new Error(`Unknown source locale: ${sourceLocale}`);
|
|
1847
|
-
if (!target) throw new Error(`Unknown target locale: ${targetLocale}`);
|
|
1848
|
-
|
|
1849
|
-
this.sourceProfile = source;
|
|
1850
|
-
this.targetProfile = target;
|
|
1851
|
-
}
|
|
1852
|
-
|
|
1853
|
-
/**
|
|
1854
|
-
* Transform a hyperscript statement from source to target language.
|
|
1855
|
-
* Handles compound statements with "then" by splitting, transforming each part,
|
|
1856
|
-
* and rejoining with the target language's "then" keyword.
|
|
1857
|
-
*
|
|
1858
|
-
* For multi-line input, preserves line structure (indentation, blank lines).
|
|
1859
|
-
*/
|
|
1860
|
-
transform(input: string): string {
|
|
1861
|
-
const out = this.transformInternal(input);
|
|
1862
|
-
// Hebrew: repair a fronted accusative marker the body-split heuristics can emit
|
|
1863
|
-
// (`… את הוסף .x …` → `… הוסף את .x …`). Applied to the assembled output; idempotent
|
|
1864
|
-
// across the internal recursion. See repairHebrewFrontedAccusative.
|
|
1865
|
-
return this.targetProfile.code === 'he' ? repairHebrewFrontedAccusative(out) : out;
|
|
1866
|
-
}
|
|
1867
|
-
|
|
1868
|
-
private transformInternal(input: string): string {
|
|
1869
|
-
// `using view transition` is a passthrough phrase, not translatable content:
|
|
1870
|
-
// `using` is in no dictionary and `transition` is a COMMAND keyword, so left
|
|
1871
|
-
// in place the splitter tears the clause apart there and the tail renders as
|
|
1872
|
-
// a phantom translated command. Mask it before any splitting/translation and
|
|
1873
|
-
// restore it verbatim; transformSingle re-appends the opaque token at the
|
|
1874
|
-
// clause tail so it lands after the reorder, not inside a role phrase.
|
|
1875
|
-
const viewTails = maskViewTransitionTails(input);
|
|
1876
|
-
if (viewTails) {
|
|
1877
|
-
return restoreViewTransitionTails(this.transformInternal(viewTails.masked), viewTails.tails);
|
|
1878
|
-
}
|
|
1879
|
-
|
|
1880
|
-
// Caret-scoped variable reads (`^name on <selector>`) carry a second,
|
|
1881
|
-
// overloaded `on` that the splitter/event-parser would mistake for an event
|
|
1882
|
-
// or command boundary — mangling `put ^count on #host into me`. Mask the
|
|
1883
|
-
// ` on <selector>` scope behind an opaque token attached to `^name` so the
|
|
1884
|
-
// command reorders as if the patient were a single value, then restore it.
|
|
1885
|
-
// `on` is kept verbatim (the semantic caret-scope matcher accepts it by raw
|
|
1886
|
-
// value across languages — passthrough-alignment).
|
|
1887
|
-
const caret = maskCaretScopes(input);
|
|
1888
|
-
if (caret) {
|
|
1889
|
-
return restoreCaretScopes(this.transform(caret.masked), caret.scopes);
|
|
1890
|
-
}
|
|
1891
|
-
|
|
1892
|
-
const targetThen = getTargetThenKeyword(this.targetProfile.code);
|
|
1893
|
-
|
|
1894
|
-
// Inline JS blocks (`... js <raw js> end`) must be masked BEFORE any
|
|
1895
|
-
// splitting/reordering: the body is raw JavaScript, not hyperscript, so it
|
|
1896
|
-
// must never be tokenized, translated, or word-order reordered. (Single-line
|
|
1897
|
-
// only here; multi-line js bodies are handled with the behavior work.)
|
|
1898
|
-
if (!input.includes('\n')) {
|
|
1899
|
-
const jsBlock = this.tryTransformJsBlock(input);
|
|
1900
|
-
if (jsBlock !== null) return jsBlock;
|
|
1901
|
-
|
|
1902
|
-
// Event handler whose body leads with a command-modifier
|
|
1903
|
-
// (`on click async fetch …`, `on click once add …`): lift the modifier out
|
|
1904
|
-
// so the real verb isn't mistaken for the action and the SOV reorder keeps
|
|
1905
|
-
// the body patient-first (recoverable by the parser's SOV event extraction).
|
|
1906
|
-
const eventModifier = this.tryTransformEventWithModifierBody(input);
|
|
1907
|
-
if (eventModifier !== null) return eventModifier;
|
|
1908
|
-
|
|
1909
|
-
// Event handler whose body is a block (`on <event> if/repeat/unless … end`):
|
|
1910
|
-
// the block body must be reordered as a self-contained unit, not shredded
|
|
1911
|
-
// across the event handler's role soup.
|
|
1912
|
-
const eventBlock = this.tryTransformEventWithBlockBody(input);
|
|
1913
|
-
if (eventBlock !== null) return eventBlock;
|
|
1914
|
-
|
|
1915
|
-
// Event handler whose body is an un-terminated inline `unless` guard
|
|
1916
|
-
// (`on click unless I match .disabled toggle .selected`): route the guard
|
|
1917
|
-
// through the standalone block path so Hebrew's accusative marker lands on
|
|
1918
|
-
// the body command, not the condition. Hebrew-only; null elsewhere.
|
|
1919
|
-
const eventGuard = this.tryTransformEventWithUnlessGuard(input);
|
|
1920
|
-
if (eventGuard !== null) return eventGuard;
|
|
1921
|
-
}
|
|
1922
|
-
|
|
1923
|
-
// Check if input has multi-line structure worth preserving
|
|
1924
|
-
const hasMultiLineStructure = input.includes('\n');
|
|
1925
|
-
|
|
1926
|
-
if (hasMultiLineStructure) {
|
|
1927
|
-
// Multi-line case - preserve structure (indentation, blank lines)
|
|
1928
|
-
const { parts, lineMetadata, partToLineIndex } = splitCompoundStatementWithMetadata(
|
|
1929
|
-
input,
|
|
1930
|
-
this.sourceProfile.code
|
|
1931
|
-
);
|
|
1932
|
-
|
|
1933
|
-
const transformedParts = parts.map(part => this.transformSingle(part));
|
|
1934
|
-
|
|
1935
|
-
return reconstructWithLineStructure(
|
|
1936
|
-
transformedParts,
|
|
1937
|
-
lineMetadata,
|
|
1938
|
-
partToLineIndex,
|
|
1939
|
-
targetThen
|
|
1940
|
-
);
|
|
1941
|
-
}
|
|
1942
|
-
|
|
1943
|
-
// Single-line case - use existing logic
|
|
1944
|
-
const parts = splitCompoundStatement(input, this.sourceProfile.code);
|
|
1945
|
-
|
|
1946
|
-
if (parts.length > 1) {
|
|
1947
|
-
const transformedParts = parts.map(part => this.transformSingle(part));
|
|
1948
|
-
return transformedParts.join(` ${targetThen} `);
|
|
1949
|
-
}
|
|
1950
|
-
|
|
1951
|
-
// Single statement (no "then" splitting needed)
|
|
1952
|
-
return this.transformSingle(input);
|
|
1953
|
-
}
|
|
1954
|
-
|
|
1955
|
-
/**
|
|
1956
|
-
* Transform a single hyperscript statement (no compound "then" chains).
|
|
1957
|
-
*/
|
|
1958
|
-
private transformSingle(input: string): string {
|
|
1959
|
-
// 0. Reactive block? Route around parseStatement entirely so
|
|
1960
|
-
// block-syntactic tokens (live/when/unless/end) aren't treated
|
|
1961
|
-
// as command verbs or swept into role values, and so SOV/VSO
|
|
1962
|
-
// reorder applies only inside the body.
|
|
1963
|
-
const block = extractBlockStructure(input, this.sourceProfile.code);
|
|
1964
|
-
if (block) {
|
|
1965
|
-
return this.transformBlock(block);
|
|
1966
|
-
}
|
|
1967
|
-
|
|
1968
|
-
// 0a. Fragment carrying a stranded trailing block terminator (`wait 200ms
|
|
1969
|
-
// end`, `set x to y end` — what the `then`-splitter leaves from
|
|
1970
|
-
// `if … then <cmd> end`). `end` is not a marker, so left in place it is
|
|
1971
|
-
// swept into the open role's VALUE and rendered inside that phrase
|
|
1972
|
-
// (bn `200ms শেষ কে অপেক্ষা`). Strip it, transform the clause alone, and
|
|
1973
|
-
// re-append the translated terminator after the verb — the same tail
|
|
1974
|
-
// position transformBlockBody emits for event-headed blocks.
|
|
1975
|
-
const strippedEnd = this.transformWithTrailingEnd(input);
|
|
1976
|
-
if (strippedEnd !== null) {
|
|
1977
|
-
return strippedEnd;
|
|
1978
|
-
}
|
|
1979
|
-
|
|
1980
|
-
// 0b. `set @attr to V on <scope>` (S1 tabs-aria): strip the trailing
|
|
1981
|
-
// `on <scope>`, transform the scope-less set normally, then re-insert
|
|
1982
|
-
// `on <scope>` where the semantic set patterns expect it.
|
|
1983
|
-
const setScope = this.transformSetWithScope(input);
|
|
1984
|
-
if (setScope !== null) {
|
|
1985
|
-
return setScope;
|
|
1986
|
-
}
|
|
1987
|
-
|
|
1988
|
-
// 0c. Masked `using view transition` tail (see maskViewTransitionTails):
|
|
1989
|
-
// strip the opaque token, transform the clause without it, and re-append
|
|
1990
|
-
// it at the clause tail. Left in the token stream it would be swept into
|
|
1991
|
-
// the open role's VALUE and rendered ahead of that role's marker
|
|
1992
|
-
// (ja `#b using view transition で` instead of `#b で using view
|
|
1993
|
-
// transition`) — the same failure mode transformWithTrailingEnd fixes
|
|
1994
|
-
// for a stranded `end`.
|
|
1995
|
-
const viewTail = this.transformWithViewTransitionTail(input);
|
|
1996
|
-
if (viewTail !== null) {
|
|
1997
|
-
return viewTail;
|
|
1998
|
-
}
|
|
1999
|
-
|
|
2000
|
-
// 1. Parse into semantic roles
|
|
2001
|
-
const parsed = parseStatement(input, this.sourceProfile.code);
|
|
2002
|
-
if (!parsed) {
|
|
2003
|
-
return input; // Return unchanged if parsing fails
|
|
2004
|
-
}
|
|
2005
|
-
|
|
2006
|
-
// 1b. Re-assign a mis-marked primary argument (e.g. `wait`'s duration) off the
|
|
2007
|
-
// default `patient` role so the target doesn't emit a spurious object-marker.
|
|
2008
|
-
// Runs before translation while `action` is still the English keyword.
|
|
2009
|
-
applyPrimaryRole(parsed, this.targetProfile);
|
|
2010
|
-
|
|
2011
|
-
// 2. Translate individual words
|
|
2012
|
-
translateElements(parsed, this.sourceProfile.code, this.targetProfile.code);
|
|
2013
|
-
|
|
2014
|
-
// 3. Find applicable rule
|
|
2015
|
-
const rule = this.findRule(parsed);
|
|
2016
|
-
|
|
2017
|
-
// 4. Apply transformation
|
|
2018
|
-
if (rule?.transform.custom) {
|
|
2019
|
-
return rule.transform.custom(parsed, this.targetProfile);
|
|
2020
|
-
}
|
|
2021
|
-
|
|
2022
|
-
// 5. Reorder according to target language's canonical order
|
|
2023
|
-
const roleOrder = rule?.transform.roleOrder || this.targetProfile.canonicalOrder;
|
|
2024
|
-
const reordered = reorderRoles(parsed.roles, roleOrder);
|
|
2025
|
-
|
|
2026
|
-
// 6. Insert grammatical markers
|
|
2027
|
-
const shouldInsertMarkers = rule?.transform.insertMarkers ?? true;
|
|
2028
|
-
if (shouldInsertMarkers) {
|
|
2029
|
-
const result = insertMarkers(
|
|
2030
|
-
reordered,
|
|
2031
|
-
this.targetProfile.markers,
|
|
2032
|
-
this.targetProfile.adpositionType
|
|
2033
|
-
);
|
|
2034
|
-
// Use joinTokens for proper suffix/prefix attachment (Turkish -i, Quechua -ta, etc.)
|
|
2035
|
-
return joinTokens(result);
|
|
2036
|
-
}
|
|
2037
|
-
|
|
2038
|
-
// 7. Join without markers (still use joinTokens for consistency)
|
|
2039
|
-
return joinTokens(reordered.map(e => e.translated || e.value));
|
|
2040
|
-
}
|
|
2041
|
-
|
|
2042
|
-
/**
|
|
2043
|
-
* Clause carrying a masked `using view transition` tail: strip the opaque
|
|
2044
|
-
* token, transform the clause alone, and re-append the token at the very end.
|
|
2045
|
-
*
|
|
2046
|
-
* The tail is a clause-final modifier in every word order the corpus emits:
|
|
2047
|
-
* the semantic side matches it as the literal `using view` marker plus a value
|
|
2048
|
-
* word, and the SOV/VSO event-handler patterns admit it as an optional
|
|
2049
|
-
* TRAILING group (after the with-marked operand). So the target position is
|
|
2050
|
-
* "end of the transformed clause" for all 24 languages — no per-profile
|
|
2051
|
-
* placement decision, which is what makes this a passthrough rather than a
|
|
2052
|
-
* role.
|
|
2053
|
-
*
|
|
2054
|
-
* Returns null when the clause carries no masked tail, or when the token is
|
|
2055
|
-
* not clause-final (nothing to reposition — leaving it in place still restores
|
|
2056
|
-
* verbatim English).
|
|
2057
|
-
*/
|
|
2058
|
-
private transformWithViewTransitionTail(input: string): string | null {
|
|
2059
|
-
const trimmed = input.trim();
|
|
2060
|
-
const tokens = trimmed.split(/\s+/);
|
|
2061
|
-
if (tokens.length < 2) {
|
|
2062
|
-
return null;
|
|
2063
|
-
}
|
|
2064
|
-
if (!VIEW_TAIL_TOKEN_RE.test(tokens[tokens.length - 1])) {
|
|
2065
|
-
return null;
|
|
2066
|
-
}
|
|
2067
|
-
const tail = tokens[tokens.length - 1];
|
|
2068
|
-
const head = tokens.slice(0, -1).join(' ');
|
|
2069
|
-
return `${this.transformSingle(head)} ${tail}`;
|
|
2070
|
-
}
|
|
2071
|
-
|
|
2072
|
-
/**
|
|
2073
|
-
* `<command …> end` fragments: transform the command without its stranded
|
|
2074
|
-
* terminator, then re-append the translated terminator as a standalone
|
|
2075
|
-
* trailing token. Fragments that open a block of their own (`if … end`,
|
|
2076
|
-
* `repeat … end`, `js … end`) bail — their terminator belongs to them and
|
|
2077
|
-
* their dedicated paths handle it.
|
|
2078
|
-
*/
|
|
2079
|
-
private transformWithTrailingEnd(input: string): string | null {
|
|
2080
|
-
const src = this.sourceProfile.code;
|
|
2081
|
-
const tokens = input.trim().split(/\s+/);
|
|
2082
|
-
if (tokens.length < 2) {
|
|
2083
|
-
return null;
|
|
2084
|
-
}
|
|
2085
|
-
const sourceEnd = translateWord('end', 'en', src).toLowerCase();
|
|
2086
|
-
if (tokens[tokens.length - 1].toLowerCase() !== sourceEnd) {
|
|
2087
|
-
return null;
|
|
2088
|
-
}
|
|
2089
|
-
const openers = new Set(
|
|
2090
|
-
['if', 'repeat', 'unless', 'while', 'when', 'live', 'js'].map(k =>
|
|
2091
|
-
translateWord(k, 'en', src).toLowerCase()
|
|
2092
|
-
)
|
|
2093
|
-
);
|
|
2094
|
-
if (tokens.slice(0, -1).some(t => openers.has(t.toLowerCase()))) {
|
|
2095
|
-
return null;
|
|
2096
|
-
}
|
|
2097
|
-
const inner = this.transformSingle(tokens.slice(0, -1).join(' '));
|
|
2098
|
-
const endT = translateWord(tokens[tokens.length - 1], src, this.targetProfile.code);
|
|
2099
|
-
return `${inner} ${endT}`;
|
|
2100
|
-
}
|
|
2101
|
-
|
|
2102
|
-
/**
|
|
2103
|
-
* Detect and transform an inline JS block (`[on <event>] js <raw js> end`).
|
|
2104
|
-
*
|
|
2105
|
-
* The `js ... end` body is raw JavaScript: it must not be tokenized,
|
|
2106
|
-
* translated, or word-order reordered. We mask the whole block with a single
|
|
2107
|
-
* opaque placeholder, run the surrounding statement (the event-handler head,
|
|
2108
|
-
* if any) through the normal reorder pipeline so the placeholder lands in the
|
|
2109
|
-
* correct action position, then substitute the translated `js`/`end` keywords
|
|
2110
|
-
* around the verbatim body.
|
|
2111
|
-
*
|
|
2112
|
-
* Returns `null` (fall through to the normal path) when there is no js block,
|
|
2113
|
-
* no matching `end`, or trailing content after `end` (kept tight on purpose).
|
|
2114
|
-
*/
|
|
2115
|
-
private tryTransformJsBlock(input: string): string | null {
|
|
2116
|
-
const src = this.sourceProfile.code;
|
|
2117
|
-
const dst = this.targetProfile.code;
|
|
2118
|
-
|
|
2119
|
-
// The `js` / `end` keyword forms in the SOURCE language.
|
|
2120
|
-
const sourceJs = translateWord('js', 'en', src);
|
|
2121
|
-
const sourceEnd = translateWord('end', 'en', src).toLowerCase();
|
|
2122
|
-
|
|
2123
|
-
const tokens = input.split(/\s+/).filter(t => t.length > 0);
|
|
2124
|
-
// The js command token, optionally with a `(locals)` suffix: `js`, `js(me)`.
|
|
2125
|
-
const escapedJs = sourceJs.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
2126
|
-
const jsRe = new RegExp(`^${escapedJs}(\\(.*\\))?$`, 'i');
|
|
2127
|
-
const jsIdx = tokens.findIndex(t => jsRe.test(t));
|
|
2128
|
-
if (jsIdx === -1) return null;
|
|
2129
|
-
|
|
2130
|
-
// First `end` after the js keyword closes the block (raw JS never contains a
|
|
2131
|
-
// bare hyperscript `end` token).
|
|
2132
|
-
let endIdx = -1;
|
|
2133
|
-
for (let i = jsIdx + 1; i < tokens.length; i++) {
|
|
2134
|
-
if (tokens[i].toLowerCase() === sourceEnd) {
|
|
2135
|
-
endIdx = i;
|
|
2136
|
-
break;
|
|
2137
|
-
}
|
|
2138
|
-
}
|
|
2139
|
-
if (endIdx === -1) return null;
|
|
2140
|
-
// Trailing content after `end` — leave for the normal path.
|
|
2141
|
-
if (endIdx !== tokens.length - 1) return null;
|
|
2142
|
-
|
|
2143
|
-
const jsToken = tokens[jsIdx];
|
|
2144
|
-
const jsParen = jsToken.match(jsRe)?.[1] ?? '';
|
|
2145
|
-
const jsKeywordRaw = jsParen ? jsToken.slice(0, jsToken.length - jsParen.length) : jsToken;
|
|
2146
|
-
|
|
2147
|
-
const body = tokens.slice(jsIdx + 1, endIdx).join(' ');
|
|
2148
|
-
const targetJs = translateWord(jsKeywordRaw, src, dst) + jsParen;
|
|
2149
|
-
const targetEnd = translateWord(tokens[endIdx], src, dst);
|
|
2150
|
-
const replacement = [targetJs, body, targetEnd].filter(s => s.length > 0).join(' ');
|
|
2151
|
-
|
|
2152
|
-
const before = tokens.slice(0, jsIdx);
|
|
2153
|
-
// Bare `js ... end` with no leading event-handler head: emit directly.
|
|
2154
|
-
if (before.length === 0) return replacement;
|
|
2155
|
-
|
|
2156
|
-
// Mask the block as one opaque action token, reorder the surrounding
|
|
2157
|
-
// statement, then restore the verbatim block.
|
|
2158
|
-
const placeholder = 'JSBLOCKPLACEHOLDER';
|
|
2159
|
-
const reordered = this.transformSingle([...before, placeholder].join(' '));
|
|
2160
|
-
if (!reordered.includes(placeholder)) return null; // unexpected — fall through
|
|
2161
|
-
return reordered.replace(placeholder, replacement);
|
|
2162
|
-
}
|
|
2163
|
-
|
|
2164
|
-
/**
|
|
2165
|
-
* Transform an event handler whose body is a block command
|
|
2166
|
-
* (`on <event> [from <src>] {if|repeat|unless|while|for} … end`).
|
|
2167
|
-
*
|
|
2168
|
-
* `parseEventHandler` would treat the block keyword as the action and sweep the
|
|
2169
|
-
* condition/body into role values, then reorder them — shredding the block
|
|
2170
|
-
* (`if event.shiftKey call submitAndContinue() end` → scattered tokens). Instead
|
|
2171
|
-
* we mask the whole block as an opaque action placeholder, reorder the event
|
|
2172
|
-
* head normally, transform the block as a self-contained unit, and restitch.
|
|
2173
|
-
*
|
|
2174
|
-
* Returns `null` (fall through) when the input isn't an event handler, has no
|
|
2175
|
-
* block-keyword body, or has no closing `end`.
|
|
2176
|
-
*/
|
|
2177
|
-
private tryTransformEventWithBlockBody(input: string): string | null {
|
|
2178
|
-
const tokens = tokenize(input, this.sourceProfile);
|
|
2179
|
-
if (tokens.length === 0) return null;
|
|
2180
|
-
if (!EVENT_KEYWORDS.has(tokens[0]?.toLowerCase())) return null;
|
|
2181
|
-
|
|
2182
|
-
let blockIdx = -1;
|
|
2183
|
-
for (let i = 1; i < tokens.length; i++) {
|
|
2184
|
-
if (BLOCK_BODY_KEYWORDS.has(tokens[i].toLowerCase())) {
|
|
2185
|
-
blockIdx = i;
|
|
2186
|
-
break;
|
|
2187
|
-
}
|
|
2188
|
-
}
|
|
2189
|
-
if (blockIdx <= 0) return null;
|
|
2190
|
-
// Only handle the explicitly-terminated form; un-terminated bodies
|
|
2191
|
-
// (`on click unless X toggle Y`) keep the existing path.
|
|
2192
|
-
if (tokens[tokens.length - 1].toLowerCase() !== 'end') return null;
|
|
2193
|
-
|
|
2194
|
-
const eventHead = tokens.slice(0, blockIdx);
|
|
2195
|
-
const blockTokens = tokens.slice(blockIdx);
|
|
2196
|
-
|
|
2197
|
-
// Event heads carrying a `from <source>` modifier (`on keydown[...] from
|
|
2198
|
-
// .modal if … end`): for SVO/SOV targets, route them through this block-body
|
|
2199
|
-
// path too — masking the block and emitting the event clause (incl. the
|
|
2200
|
-
// translated `from <source>`, which `transformSingle` → `parseEventHandler`
|
|
2201
|
-
// already reorders) first, then the transformed block. This keeps the if-block
|
|
2202
|
-
// body from being shredded across the event handler's roles and properly
|
|
2203
|
-
// translates its inner keywords (`in`/`focus`/`first`/positional); the old
|
|
2204
|
-
// exclusion left them English and mangled the order (this is what cleared
|
|
2205
|
-
// `focus-trap` in tr). VSO targets (ar, tl) are kept on the existing path:
|
|
2206
|
-
// there the event-first emission with a `from`-source reorders incorrectly and
|
|
2207
|
-
// regresses `focus-trap`/`window-keydown`, while the existing path already
|
|
2208
|
-
// parses them. So the `from`-source exclusion is scoped to VSO only.
|
|
2209
|
-
if (this.targetProfile.wordOrder === 'VSO' && eventHead.some(t => t.toLowerCase() === 'from')) {
|
|
2210
|
-
return null;
|
|
2211
|
-
}
|
|
2212
|
-
|
|
2213
|
-
const placeholder = 'EVENTBLOCKPLACEHOLDER';
|
|
2214
|
-
const headOut = this.transformSingle([...eventHead, placeholder].join(' '));
|
|
2215
|
-
if (!headOut.includes(placeholder)) return null;
|
|
2216
|
-
|
|
2217
|
-
const blockOut = this.transformBlockBody(blockTokens);
|
|
2218
|
-
|
|
2219
|
-
// Always emit the event clause first, then the block. An event handler's
|
|
2220
|
-
// event is a leading delimiter, and the semantic parser only matches a
|
|
2221
|
-
// block body when it follows the event — even in verb-first (VSO) languages
|
|
2222
|
-
// whose normal command order would push the event to the end. So strip the
|
|
2223
|
-
// placeholder out of the (possibly reordered) head and append the block,
|
|
2224
|
-
// rather than substituting in place.
|
|
2225
|
-
const eventClause = headOut.replace(placeholder, '').replace(/\s+/g, ' ').trim();
|
|
2226
|
-
return [eventClause, blockOut].filter(s => s.length > 0).join(' ');
|
|
2227
|
-
}
|
|
2228
|
-
|
|
2229
|
-
/**
|
|
2230
|
-
* Transform an event handler whose body is an inline `unless` guard with NO
|
|
2231
|
-
* `end` (`on <event> unless <cond> <body>` — the `unless-condition` shape).
|
|
2232
|
-
*
|
|
2233
|
-
* Object-marking SVO targets (he, zh). `parseEventHandler` reads `unless` as the
|
|
2234
|
-
* action and sweeps the whole `<cond> <body>` tail into a single `patient` blob;
|
|
2235
|
-
* the target then prefixes that blob with its object marker — Hebrew's accusative
|
|
2236
|
-
* את (`… אלא את I match .disabled מתג .selected`) or Chinese's BA particle 把
|
|
2237
|
-
* (`… 除非 把 I match .disabled 切换 .selected`) — and the inner toggle loses its
|
|
2238
|
-
* own marker. The semantic parser can't recover the guard from that: the marker
|
|
2239
|
-
* ahead of the condition blocks the `unless` pattern AND the now-markerless body
|
|
2240
|
-
* command fails its object-marked toggle pattern, so the body collapses (`unless`
|
|
2241
|
-
* dropped). Marker-less languages (de/it/ar/pl) tolerate the same role-blob and
|
|
2242
|
-
* stay faithful, so this is an object-marker artifact, not a general parse gap.
|
|
2243
|
-
*
|
|
2244
|
-
* The standalone `unless <cond> <body>` path already produces the correct shape
|
|
2245
|
-
* (`extractBlockStructure` → `transformBlock`: condition kept marker-free, body
|
|
2246
|
-
* command keeps its marker — he `אלא I match .disabled מתג את .selected`, zh
|
|
2247
|
-
* `除非 I match .disabled 切换 把 .selected`). So we split the event head off,
|
|
2248
|
-
* transform the guard through that path, and emit the event clause first (he and
|
|
2249
|
-
* zh are both SVO — event leads). Returns `null` (fall through) when the input
|
|
2250
|
-
* isn't an object-marking event handler with an un-terminated inline `unless`
|
|
2251
|
-
* guard.
|
|
2252
|
-
*/
|
|
2253
|
-
private tryTransformEventWithUnlessGuard(input: string): string | null {
|
|
2254
|
-
// SVO object-marking targets only — these front the unless tail with an object
|
|
2255
|
-
// marker (he את / zh 把) that breaks the parse. Event-leads emission below
|
|
2256
|
-
// assumes SVO, so SOV/VSO object-markers (ja/ko/tr/ar) are intentionally out.
|
|
2257
|
-
if (!UNLESS_GUARD_OBJECT_MARKING_LOCALES.has(this.targetProfile.code)) return null;
|
|
2258
|
-
|
|
2259
|
-
const tokens = tokenize(input, this.sourceProfile);
|
|
2260
|
-
if (tokens.length === 0) return null;
|
|
2261
|
-
if (!EVENT_KEYWORDS.has(tokens[0]?.toLowerCase())) return null;
|
|
2262
|
-
|
|
2263
|
-
let guardIdx = -1;
|
|
2264
|
-
for (let i = 1; i < tokens.length; i++) {
|
|
2265
|
-
if (tokens[i].toLowerCase() === 'unless') {
|
|
2266
|
-
guardIdx = i;
|
|
2267
|
-
break;
|
|
2268
|
-
}
|
|
2269
|
-
}
|
|
2270
|
-
if (guardIdx <= 0) return null;
|
|
2271
|
-
// The terminated form (`… unless … end`) is handled by
|
|
2272
|
-
// tryTransformEventWithBlockBody's masking path; only take the inline guard.
|
|
2273
|
-
if (tokens[tokens.length - 1].toLowerCase() === 'end') return null;
|
|
2274
|
-
|
|
2275
|
-
const eventHead = tokens.slice(0, guardIdx);
|
|
2276
|
-
const guard = tokens.slice(guardIdx).join(' ');
|
|
2277
|
-
|
|
2278
|
-
// `unless` is a BLOCK_HEAD keyword, so transform() routes the guard through
|
|
2279
|
-
// extractBlockStructure → transformBlock (the standalone shape that parses).
|
|
2280
|
-
const guardOut = this.transform(guard);
|
|
2281
|
-
if (!guardOut) return null;
|
|
2282
|
-
|
|
2283
|
-
const placeholder = 'EVENTGUARDPLACEHOLDER';
|
|
2284
|
-
const headOut = this.transformSingle([...eventHead, placeholder].join(' '));
|
|
2285
|
-
if (!headOut.includes(placeholder)) return null;
|
|
2286
|
-
const eventClause = headOut.replace(placeholder, '').replace(/\s+/g, ' ').trim();
|
|
2287
|
-
return [eventClause, guardOut].filter(s => s.length > 0).join(' ');
|
|
2288
|
-
}
|
|
2289
|
-
|
|
2290
|
-
/**
|
|
2291
|
-
* Transform an event handler whose body leads with a command-modifier
|
|
2292
|
-
* (`on <event> [from <src>] {async|once|debounced [at N]|throttled [at N]} <body>`).
|
|
2293
|
-
*
|
|
2294
|
-
* `parseEventHandler` reads the first token after the event as the **action**, so
|
|
2295
|
-
* a leading modifier is mistaken for the verb and the real verb (`fetch`/`add`) is
|
|
2296
|
-
* swept into the patient. For SOV targets the reorder then surfaces that verb
|
|
2297
|
-
* **first** (`取得 /api/data を クリック …`), and the semantic parser matches the
|
|
2298
|
-
* leading `<verb> <patient>` with the low-priority `*-generated-verb-first`
|
|
2299
|
-
* command pattern — returning a bare command and discarding the event + the rest
|
|
2300
|
-
* of the body (degenerate parse).
|
|
2301
|
-
*
|
|
2302
|
-
* Instead, lift the modifier out, transform the modifier-free handler through the
|
|
2303
|
-
* normal path (which keeps the body in canonical patient-first SOV order so the
|
|
2304
|
-
* event sits mid-stream and the existing SOV event-extraction recovers it), then
|
|
2305
|
-
* re-emit the modifier as a **leading English literal**. The semantic parser
|
|
2306
|
-
* strips a leading `once`/`debounced`/`throttled` (`extractStandaloneModifiers`)
|
|
2307
|
-
* and an `async` anywhere (`stripAsyncModifier`) before parsing, so the modifier
|
|
2308
|
-
* is consumed as handler metadata rather than shadowing the body.
|
|
2309
|
-
*
|
|
2310
|
-
* Returns `null` (fall through) when the input isn't an event handler or the body
|
|
2311
|
-
* doesn't lead with a modifier — leaving simple/Mode-B handlers byte-identical.
|
|
2312
|
-
*/
|
|
2313
|
-
private tryTransformEventWithModifierBody(input: string): string | null {
|
|
2314
|
-
// The verb-first degenerate parse this works around is specific to SOV
|
|
2315
|
-
// reorder: only there does a leading modifier displace the patient-first
|
|
2316
|
-
// order and surface the verb first. SVO/VSO/V2/other targets keep the body
|
|
2317
|
-
// in an order the parser already handles, so leave them byte-identical.
|
|
2318
|
-
if (this.targetProfile.wordOrder !== 'SOV') return null;
|
|
2319
|
-
|
|
2320
|
-
const tokens = tokenize(input, this.sourceProfile);
|
|
2321
|
-
if (tokens.length === 0) return null;
|
|
2322
|
-
if (!EVENT_KEYWORDS.has(tokens[0]?.toLowerCase())) return null;
|
|
2323
|
-
|
|
2324
|
-
// Walk past the event clause head: event keyword, event token, any
|
|
2325
|
-
// `or`-conjoined events, and an optional `from <source>` modifier — mirroring
|
|
2326
|
-
// parseEventHandler's head parsing — to find where the body begins.
|
|
2327
|
-
let i = 1; // past the event keyword
|
|
2328
|
-
if (!tokens[i]) return null;
|
|
2329
|
-
i++; // past the event token
|
|
2330
|
-
while (tokens[i] && EVENT_CONJUNCTIONS.has(tokens[i].toLowerCase()) && tokens[i + 1]) {
|
|
2331
|
-
i += 2;
|
|
2332
|
-
}
|
|
2333
|
-
if (tokens[i]?.toLowerCase() === 'from' && tokens[i + 1]) {
|
|
2334
|
-
i++; // skip 'from'
|
|
2335
|
-
// Collect source tokens until a command verb or a body modifier.
|
|
2336
|
-
while (
|
|
2337
|
-
tokens[i] &&
|
|
2338
|
-
!ENGLISH_COMMANDS.has(tokens[i].toLowerCase()) &&
|
|
2339
|
-
!BODY_MODIFIER_KEYWORDS.has(tokens[i].toLowerCase())
|
|
2340
|
-
) {
|
|
2341
|
-
i++;
|
|
2342
|
-
}
|
|
2343
|
-
}
|
|
2344
|
-
|
|
2345
|
-
const modWord = tokens[i]?.toLowerCase();
|
|
2346
|
-
if (!modWord || !BODY_MODIFIER_KEYWORDS.has(modWord)) return null;
|
|
2347
|
-
|
|
2348
|
-
// Consume the modifier phrase. `debounced`/`throttled` may carry an optional
|
|
2349
|
-
// `at <duration>` (or a bare duration). `async`/`once` are single tokens.
|
|
2350
|
-
const modStart = i;
|
|
2351
|
-
let modEnd = i + 1;
|
|
2352
|
-
if (modWord !== 'async' && modWord !== 'once') {
|
|
2353
|
-
if (tokens[modEnd]?.toLowerCase() === 'at') modEnd++;
|
|
2354
|
-
if (tokens[modEnd] && /^\d+(ms|s|m)?$/.test(tokens[modEnd])) modEnd++;
|
|
2355
|
-
}
|
|
2356
|
-
|
|
2357
|
-
const modifierPhrase = tokens.slice(modStart, modEnd).join(' ');
|
|
2358
|
-
const rebuilt = [...tokens.slice(0, modStart), ...tokens.slice(modEnd)].join(' ');
|
|
2359
|
-
|
|
2360
|
-
// A handler with only a modifier and no body has nothing to keep patient-first.
|
|
2361
|
-
if (tokens.length - (modEnd - modStart) <= 2) return null;
|
|
2362
|
-
|
|
2363
|
-
// Re-run the full transform on the modifier-free handler so then-chains and
|
|
2364
|
-
// juxtaposed bodies route through the existing (working) paths. The rebuilt
|
|
2365
|
-
// input no longer leads with a modifier, so this never re-enters here.
|
|
2366
|
-
const bodyOut = this.transform(rebuilt);
|
|
2367
|
-
return [modifierPhrase, bodyOut].filter(s => s.length > 0).join(' ');
|
|
2368
|
-
}
|
|
2369
|
-
|
|
2370
|
-
/**
|
|
2371
|
-
* Transform a `set <stuff> on <scope>` clause (S1 tabs-aria). The trailing
|
|
2372
|
-
* `on <scope>` is the element(s) the attribute is set on — kept attached by
|
|
2373
|
-
* splitOnCommandBoundaries. The semantic parser captures it as a `scope` role
|
|
2374
|
-
* via the passthrough literal `on` (setSchema's scope markerOverride is `on`
|
|
2375
|
-
* in every language), so `on` is emitted verbatim and only the scope *value*
|
|
2376
|
-
* is translated (selectors pass through; `me`/`it`/`you` translate to the
|
|
2377
|
-
* native reference, which the parser also accepts).
|
|
2378
|
-
*
|
|
2379
|
-
* Positioning matches where the set patterns expect the scope: at the clause
|
|
2380
|
-
* end for verb-first orders (SVO/VSO), and immediately before the clause-final
|
|
2381
|
-
* verb for SOV (the generated SOV pattern is `{dest} {patient} on {scope}
|
|
2382
|
-
* {verb}`). Returns null (fall through) when there is no trailing `on <scope>`.
|
|
2383
|
-
*
|
|
2384
|
-
* Source is English in the sync-translations pipeline, so the `set` verb and
|
|
2385
|
-
* `on` marker are matched as English literals.
|
|
2386
|
-
*/
|
|
2387
|
-
private transformSetWithScope(input: string): string | null {
|
|
2388
|
-
const src = this.sourceProfile.code;
|
|
2389
|
-
const dst = this.targetProfile.code;
|
|
2390
|
-
|
|
2391
|
-
const m = input.match(/^(.*\bset\b.*\S)\s+on\s+([#.<@[]\S*|me|it|you)\s*$/i);
|
|
2392
|
-
if (!m) return null;
|
|
2393
|
-
const head = m[1];
|
|
2394
|
-
const scopeRaw = m[2];
|
|
2395
|
-
|
|
2396
|
-
// Transform the scope-less clause via the normal path; `head` no longer ends
|
|
2397
|
-
// in `on <scope>`, so this never re-enters transformSetWithScope.
|
|
2398
|
-
const headOut = this.transformSingle(head);
|
|
2399
|
-
|
|
2400
|
-
const scopeT = /^[#.<@[]/.test(scopeRaw) ? scopeRaw : translateWord(scopeRaw, src, dst);
|
|
2401
|
-
|
|
2402
|
-
// SOV needs the scope positioned per how the semantic set patterns match:
|
|
2403
|
-
// - Event-handler set (`on <event> set …`): the dest-first SOV event-handler
|
|
2404
|
-
// pattern is verb-MEDIAL and carries an optional trailing `[on {scope}]`,
|
|
2405
|
-
// so append the scope at the clause end.
|
|
2406
|
-
// - Standalone then-clause set: SOV emits verb-MEDIAL (`{dest} {verb}
|
|
2407
|
-
// {patient}`), but the only command set pattern with scope is verb-LAST
|
|
2408
|
-
// (`{dest} {patient} on {scope} {verb}`). Move the medial verb to the end
|
|
2409
|
-
// and place `on {scope}` before it, so the generated command pattern matches.
|
|
2410
|
-
if (this.targetProfile.wordOrder === 'SOV') {
|
|
2411
|
-
const verb = translateWord('set', 'en', dst);
|
|
2412
|
-
const firstTok = input.trim().split(/\s+/)[0]?.toLowerCase();
|
|
2413
|
-
const isEventHandler = !!firstTok && EVENT_KEYWORDS.has(firstTok);
|
|
2414
|
-
const toks = headOut.split(/\s+/).filter(Boolean);
|
|
2415
|
-
if (!isEventHandler) {
|
|
2416
|
-
// Standalone then-clause set: SOV emits verb-MEDIAL; move the verb to the
|
|
2417
|
-
// end and place `on {scope}` before it so the verb-last command set
|
|
2418
|
-
// pattern (scope before verb) matches.
|
|
2419
|
-
const vIdx = toks.indexOf(verb);
|
|
2420
|
-
if (vIdx >= 0) {
|
|
2421
|
-
toks.splice(vIdx, 1);
|
|
2422
|
-
toks.push('on', scopeT, verb);
|
|
2423
|
-
return toks.join(' ');
|
|
2424
|
-
}
|
|
2425
|
-
} else if (toks.length > 0 && toks[toks.length - 1] === verb) {
|
|
2426
|
-
// Event-handler set whose verb is clause-final (e.g. qu
|
|
2427
|
-
// `{dest} ta {patient} man {event} pi {verb}`): the parser extracts the
|
|
2428
|
-
// event and matches the body as a verb-last command, so the scope must
|
|
2429
|
-
// sit before the verb. Verb-MEDIAL SOV event handlers (ja/ko/tr/bn/hi)
|
|
2430
|
-
// fall through to the append branch, where their fused event-handler set
|
|
2431
|
-
// pattern carries the trailing `[on {scope}]` group.
|
|
2432
|
-
toks.splice(toks.length - 1, 0, 'on', scopeT);
|
|
2433
|
-
return toks.join(' ');
|
|
2434
|
-
}
|
|
2435
|
-
return `${headOut} on ${scopeT}`;
|
|
2436
|
-
}
|
|
2437
|
-
|
|
2438
|
-
// Verb-first (SVO/VSO/V2): append `on <scope>` at the clause end, which the
|
|
2439
|
-
// trailing `[on {scope}]` group on the set patterns matches.
|
|
2440
|
-
return `${headOut} on ${scopeT}`;
|
|
2441
|
-
}
|
|
2442
|
-
|
|
2443
|
-
/**
|
|
2444
|
-
* Transform a self-contained block command (`{head} {clause?} {body} {end}`),
|
|
2445
|
-
* where head ∈ {if, repeat, unless, …}. The clause (condition / `until event …`)
|
|
2446
|
-
* runs up to the first command verb and is translated word-by-word; the body is
|
|
2447
|
-
* recursively transformed (so its inner commands reorder for the target); the
|
|
2448
|
-
* head/tail keywords are translated. The block is never word-order reordered as
|
|
2449
|
-
* a whole — delimiters stay at the edges regardless of target word order.
|
|
2450
|
-
*/
|
|
2451
|
-
private transformBlockBody(blockTokens: string[]): string {
|
|
2452
|
-
const src = this.sourceProfile.code;
|
|
2453
|
-
const dst = this.targetProfile.code;
|
|
2454
|
-
|
|
2455
|
-
const head = blockTokens[0];
|
|
2456
|
-
const hasEnd = blockTokens[blockTokens.length - 1]?.toLowerCase() === 'end';
|
|
2457
|
-
const tail = hasEnd ? blockTokens[blockTokens.length - 1] : '';
|
|
2458
|
-
const inner = blockTokens.slice(1, hasEnd ? -1 : undefined);
|
|
2459
|
-
|
|
2460
|
-
// Skip predicate-adjective positions (`… is empty`): the adjective is
|
|
2461
|
-
// part of the condition clause, not the body's first verb — cutting there
|
|
2462
|
-
// displaced it into the next command's argument zone (empty ×8 bn/hi/tr).
|
|
2463
|
-
const commands = getCommandKeywordsForLocale(src);
|
|
2464
|
-
const copulas = getCopulasForLocale(src);
|
|
2465
|
-
let bodyStart = inner.findIndex(
|
|
2466
|
-
(t, i) => commands.has(t.toLowerCase()) && !isPredicateAdjectivePosition(inner, i, copulas)
|
|
2467
|
-
);
|
|
2468
|
-
if (bodyStart < 0) bodyStart = inner.length;
|
|
2469
|
-
|
|
2470
|
-
const clause = inner.slice(0, bodyStart).join(' ');
|
|
2471
|
-
const bodyTokens = inner.slice(bodyStart);
|
|
2472
|
-
|
|
2473
|
-
const headT = translateWord(head, src, dst);
|
|
2474
|
-
const tailT = tail ? translateWord(tail, src, dst) : '';
|
|
2475
|
-
const clauseT = clause ? translateMultiWordValue(clause, src, dst) : '';
|
|
2476
|
-
const bodyT = this.transformConditionalBody(bodyTokens);
|
|
2477
|
-
|
|
2478
|
-
// SOV condition/branch boundary (R1 deferred-tail Family G): the SOV body
|
|
2479
|
-
// renders its first command operand-first (`最初 <button/> の中 .modal を
|
|
2480
|
-
// フォーカス`), so nothing marks where the condition ends and the branch
|
|
2481
|
-
// operand begins — the semantic fold's command-start detection needs a
|
|
2482
|
-
// `{value}{particle}` run followed by a verb, which a POSITIONAL-headed
|
|
2483
|
-
// operand (`first <button/> …`) never forms, and the condition scan
|
|
2484
|
-
// swallows the operand's head (focus-trap: ja focus.patient fell to the
|
|
2485
|
-
// `me` default, ko/qu to the `.modal` tail). Emit the target's
|
|
2486
|
-
// then-connective at the seam — the boundary the fold already respects
|
|
2487
|
-
// (isThenKeyword) — gated to exactly the blind shape: SOV target, a
|
|
2488
|
-
// positional keyword right after the branch's command verb, and no `then`
|
|
2489
|
-
// already ending the condition.
|
|
2490
|
-
const POSITIONAL_BRANCH_HEADS = new Set(['first', 'last', 'next', 'previous', 'closest']);
|
|
2491
|
-
const thenT =
|
|
2492
|
-
this.targetProfile.wordOrder === 'SOV' &&
|
|
2493
|
-
clauseT &&
|
|
2494
|
-
POSITIONAL_BRANCH_HEADS.has(bodyTokens[1]?.toLowerCase()) &&
|
|
2495
|
-
inner[bodyStart - 1]?.toLowerCase() !== 'then'
|
|
2496
|
-
? translateWord('then', src, dst)
|
|
2497
|
-
: '';
|
|
2498
|
-
|
|
2499
|
-
return [headT, clauseT, thenT, bodyT, tailT].filter(s => s.length > 0).join(' ');
|
|
2500
|
-
}
|
|
2501
|
-
|
|
2502
|
-
/**
|
|
2503
|
-
* Transform an `if`/`unless` block body, splitting it at a top-level `else` into
|
|
2504
|
-
* a then-branch and an else-branch so each is reordered as a self-contained unit
|
|
2505
|
-
* and the `else` keyword itself is translated. Without this, the body is reordered
|
|
2506
|
-
* as one stream: `else` rides along glued to the preceding clause (and, when that
|
|
2507
|
-
* clause begins with a selector, is marked a selector and left *untranslated*),
|
|
2508
|
-
* and a spurious `then` is inserted around it — both of which break the target
|
|
2509
|
-
* text and the downstream parse. The split is depth-aware so an `else` belonging
|
|
2510
|
-
* to a nested block is not mistaken for this block's separator. Bodies without an
|
|
2511
|
-
* `else` transform exactly as before.
|
|
2512
|
-
*/
|
|
2513
|
-
private transformConditionalBody(bodyTokens: string[]): string {
|
|
2514
|
-
const src = this.sourceProfile.code;
|
|
2515
|
-
const dst = this.targetProfile.code;
|
|
2516
|
-
|
|
2517
|
-
const sourceElse = translateWord('else', 'en', src).toLowerCase();
|
|
2518
|
-
let depth = 0;
|
|
2519
|
-
let elseIdx = -1;
|
|
2520
|
-
for (let i = 0; i < bodyTokens.length; i++) {
|
|
2521
|
-
const t = bodyTokens[i].toLowerCase();
|
|
2522
|
-
if (BLOCK_BODY_KEYWORDS.has(t)) depth++;
|
|
2523
|
-
else if (t === 'end' && depth > 0) depth--;
|
|
2524
|
-
else if (t === sourceElse && depth === 0) {
|
|
2525
|
-
elseIdx = i;
|
|
2526
|
-
break;
|
|
2527
|
-
}
|
|
2528
|
-
}
|
|
2529
|
-
|
|
2530
|
-
if (elseIdx === -1) {
|
|
2531
|
-
const body = bodyTokens.join(' ');
|
|
2532
|
-
return body ? this.transform(body) : '';
|
|
2533
|
-
}
|
|
2534
|
-
|
|
2535
|
-
const thenBranch = bodyTokens.slice(0, elseIdx).join(' ');
|
|
2536
|
-
const elseBranch = bodyTokens.slice(elseIdx + 1).join(' ');
|
|
2537
|
-
const elseT = translateWord(bodyTokens[elseIdx], src, dst);
|
|
2538
|
-
|
|
2539
|
-
return [
|
|
2540
|
-
thenBranch ? this.transform(thenBranch) : '',
|
|
2541
|
-
elseT,
|
|
2542
|
-
elseBranch ? this.transform(elseBranch) : '',
|
|
2543
|
-
]
|
|
2544
|
-
.filter(s => s.length > 0)
|
|
2545
|
-
.join(' ');
|
|
2546
|
-
}
|
|
2547
|
-
|
|
2548
|
-
/**
|
|
2549
|
-
* Translate a reactive block by translating the head/tail/connector
|
|
2550
|
-
* via the dictionary, recursively transforming the body through the
|
|
2551
|
-
* regular pipeline, and rejoining in source-language position order.
|
|
2552
|
-
* Block-syntactic tokens are never reordered: they're delimiters, not
|
|
2553
|
-
* arguments, and authors expect them at start/end positions
|
|
2554
|
-
* regardless of target word order.
|
|
2555
|
-
*/
|
|
2556
|
-
private transformBlock(block: BlockStructure): string {
|
|
2557
|
-
const src = this.sourceProfile.code;
|
|
2558
|
-
const dst = this.targetProfile.code;
|
|
2559
|
-
|
|
2560
|
-
const head = translateWord(block.headKeyword, src, dst);
|
|
2561
|
-
const tail = block.tailKeyword ? translateWord(block.tailKeyword, src, dst) : '';
|
|
2562
|
-
const connector = block.connector ? translateWord(block.connector, src, dst) : '';
|
|
2563
|
-
const prefix = block.prefixExpr ? translateMultiWordValue(block.prefixExpr, src, dst) : '';
|
|
2564
|
-
|
|
2565
|
-
// Recurse through `transform()` (not `transformSingle`) so the body
|
|
2566
|
-
// gets `then`-splitting and nested-block handling for free.
|
|
2567
|
-
const body = this.transform(block.body);
|
|
2568
|
-
|
|
2569
|
-
return [head, prefix, connector, body, tail].filter(s => s.length > 0).join(' ');
|
|
2570
|
-
}
|
|
2571
|
-
|
|
2572
|
-
/**
|
|
2573
|
-
* Find the best matching rule for this statement
|
|
2574
|
-
*/
|
|
2575
|
-
private findRule(parsed: ParsedStatement): GrammarRule | undefined {
|
|
2576
|
-
if (!this.targetProfile.rules) return undefined;
|
|
2577
|
-
|
|
2578
|
-
const matchingRules = this.targetProfile.rules
|
|
2579
|
-
.filter(rule => this.matchesRule(parsed, rule))
|
|
2580
|
-
.sort((a, b) => b.priority - a.priority);
|
|
2581
|
-
|
|
2582
|
-
return matchingRules[0];
|
|
2583
|
-
}
|
|
2584
|
-
|
|
2585
|
-
/**
|
|
2586
|
-
* Check if a parsed statement matches a rule
|
|
2587
|
-
*/
|
|
2588
|
-
private matchesRule(parsed: ParsedStatement, rule: GrammarRule): boolean {
|
|
2589
|
-
const { match } = rule;
|
|
2590
|
-
|
|
2591
|
-
// Check required roles
|
|
2592
|
-
for (const role of match.requiredRoles) {
|
|
2593
|
-
if (!parsed.roles.has(role)) {
|
|
2594
|
-
return false;
|
|
2595
|
-
}
|
|
2596
|
-
}
|
|
2597
|
-
|
|
2598
|
-
// Check command match if specified
|
|
2599
|
-
if (match.commands && match.commands.length > 0) {
|
|
2600
|
-
const action = parsed.roles.get('action');
|
|
2601
|
-
if (!action) return false;
|
|
2602
|
-
|
|
2603
|
-
const actionValue = action.value.toLowerCase();
|
|
2604
|
-
if (!match.commands.some(cmd => cmd.toLowerCase() === actionValue)) {
|
|
2605
|
-
return false;
|
|
2606
|
-
}
|
|
2607
|
-
}
|
|
2608
|
-
|
|
2609
|
-
// Check custom predicate
|
|
2610
|
-
if (match.predicate && !match.predicate(parsed)) {
|
|
2611
|
-
return false;
|
|
2612
|
-
}
|
|
2613
|
-
|
|
2614
|
-
return true;
|
|
2615
|
-
}
|
|
2616
|
-
}
|
|
2617
|
-
|
|
2618
|
-
// =============================================================================
|
|
2619
|
-
// Convenience Functions
|
|
2620
|
-
// =============================================================================
|
|
2621
|
-
|
|
2622
|
-
/**
|
|
2623
|
-
* Transform hyperscript from English to target language
|
|
2624
|
-
*/
|
|
2625
|
-
export function toLocale(input: string, targetLocale: string): string {
|
|
2626
|
-
const transformer = new GrammarTransformer('en', targetLocale);
|
|
2627
|
-
return transformer.transform(input);
|
|
2628
|
-
}
|
|
2629
|
-
|
|
2630
|
-
/**
|
|
2631
|
-
* Transform hyperscript from source language to English
|
|
2632
|
-
*/
|
|
2633
|
-
export function toEnglish(input: string, sourceLocale: string): string {
|
|
2634
|
-
const transformer = new GrammarTransformer(sourceLocale, 'en');
|
|
2635
|
-
return transformer.transform(input);
|
|
2636
|
-
}
|
|
2637
|
-
|
|
2638
|
-
/**
|
|
2639
|
-
* Transform between any two languages.
|
|
2640
|
-
*
|
|
2641
|
-
* Uses direct translation for supported language pairs (ja↔zh, es↔pt, ko↔ja),
|
|
2642
|
-
* falling back to English pivot for other pairs.
|
|
2643
|
-
*/
|
|
2644
|
-
export function translate(input: string, sourceLocale: string, targetLocale: string): string {
|
|
2645
|
-
if (sourceLocale === targetLocale) return input;
|
|
2646
|
-
if (sourceLocale === 'en') return toLocale(input, targetLocale);
|
|
2647
|
-
if (targetLocale === 'en') return toEnglish(input, sourceLocale);
|
|
2648
|
-
|
|
2649
|
-
// Try direct translation for supported pairs
|
|
2650
|
-
if (hasDirectMapping(sourceLocale, targetLocale)) {
|
|
2651
|
-
return translateDirect(input, sourceLocale, targetLocale);
|
|
2652
|
-
}
|
|
2653
|
-
|
|
2654
|
-
// Fallback: Via English pivot
|
|
2655
|
-
const english = toEnglish(input, sourceLocale);
|
|
2656
|
-
return toLocale(english, targetLocale);
|
|
2657
|
-
}
|
|
2658
|
-
|
|
2659
|
-
/**
|
|
2660
|
-
* Direct translation between language pairs without English pivot.
|
|
2661
|
-
* More accurate for closely related languages (ja↔zh, es↔pt).
|
|
2662
|
-
*/
|
|
2663
|
-
function translateDirect(input: string, sourceLocale: string, targetLocale: string): string {
|
|
2664
|
-
const mapping = getDirectMapping(sourceLocale, targetLocale);
|
|
2665
|
-
if (!mapping) {
|
|
2666
|
-
// Fallback to pivot translation
|
|
2667
|
-
return toLocale(toEnglish(input, sourceLocale), targetLocale);
|
|
2668
|
-
}
|
|
2669
|
-
|
|
2670
|
-
// Tokenize input
|
|
2671
|
-
const tokens = input.split(/\s+/);
|
|
2672
|
-
|
|
2673
|
-
// Translate each token using direct mapping
|
|
2674
|
-
const translated = tokens.map(token => {
|
|
2675
|
-
// Preserve CSS selectors and literals
|
|
2676
|
-
if (token.startsWith('#') || token.startsWith('.') || token.startsWith('@')) {
|
|
2677
|
-
return token;
|
|
2678
|
-
}
|
|
2679
|
-
if (token.startsWith('"') || token.startsWith("'")) {
|
|
2680
|
-
return token;
|
|
2681
|
-
}
|
|
2682
|
-
|
|
2683
|
-
// Look up in direct mapping
|
|
2684
|
-
const directTranslation = mapping.words[token];
|
|
2685
|
-
if (directTranslation) {
|
|
2686
|
-
return directTranslation;
|
|
2687
|
-
}
|
|
2688
|
-
|
|
2689
|
-
// Check for suffix-attached tokens (e.g., "#count-ta" in Quechua)
|
|
2690
|
-
const suffixMatch = token.match(/^(.+?)(-.+)$/);
|
|
2691
|
-
if (suffixMatch) {
|
|
2692
|
-
const [, base, suffix] = suffixMatch;
|
|
2693
|
-
const translatedBase = mapping.words[base] || base;
|
|
2694
|
-
return translatedBase + suffix;
|
|
2695
|
-
}
|
|
2696
|
-
|
|
2697
|
-
// Return unchanged if no mapping found
|
|
2698
|
-
return token;
|
|
2699
|
-
});
|
|
2700
|
-
|
|
2701
|
-
return translated.join(' ');
|
|
2702
|
-
}
|
|
2703
|
-
|
|
2704
|
-
// =============================================================================
|
|
2705
|
-
// Examples (for testing)
|
|
2706
|
-
// =============================================================================
|
|
2707
|
-
|
|
2708
|
-
export const examples = {
|
|
2709
|
-
english: {
|
|
2710
|
-
eventHandler: 'on click increment #count',
|
|
2711
|
-
putInto: 'put my value into #output',
|
|
2712
|
-
toggle: 'toggle .active',
|
|
2713
|
-
wait: 'wait 2 seconds',
|
|
2714
|
-
},
|
|
2715
|
-
|
|
2716
|
-
// Expected outputs (approximate, for reference)
|
|
2717
|
-
japanese: {
|
|
2718
|
-
eventHandler: '#count を クリック で 増加',
|
|
2719
|
-
putInto: '私の 値 を #output に 置く',
|
|
2720
|
-
toggle: '.active を 切り替え',
|
|
2721
|
-
wait: '2秒 待つ',
|
|
2722
|
-
},
|
|
2723
|
-
|
|
2724
|
-
chinese: {
|
|
2725
|
-
eventHandler: '当 点击 时 增加 #count',
|
|
2726
|
-
putInto: '把 我的值 放 到 #output',
|
|
2727
|
-
toggle: '切换 .active',
|
|
2728
|
-
wait: '等待 2秒',
|
|
2729
|
-
},
|
|
2730
|
-
|
|
2731
|
-
arabic: {
|
|
2732
|
-
eventHandler: 'زِد #count عند النقر',
|
|
2733
|
-
putInto: 'ضع قيمتي في #output',
|
|
2734
|
-
toggle: 'بدّل .active',
|
|
2735
|
-
wait: 'انتظر ثانيتين',
|
|
2736
|
-
},
|
|
2737
|
-
};
|