@lokascript/i18n 2.5.1 → 2.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/browser.cjs +1274 -154
- package/dist/browser.cjs.map +1 -1
- package/dist/browser.d.cts +3 -3
- package/dist/browser.d.ts +3 -3
- package/dist/browser.js +1274 -155
- package/dist/browser.js.map +1 -1
- package/dist/dictionaries/index.cjs +364 -133
- package/dist/dictionaries/index.cjs.map +1 -1
- package/dist/dictionaries/index.d.cts +1 -1
- package/dist/dictionaries/index.d.ts +1 -1
- package/dist/dictionaries/index.js +364 -133
- package/dist/dictionaries/index.js.map +1 -1
- package/dist/index.cjs +8860 -7740
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +3 -3
- package/dist/index.d.ts +3 -3
- package/dist/index.js +8860 -7741
- package/dist/index.js.map +1 -1
- package/dist/lokascript-i18n.min.js +1 -1
- package/dist/lokascript-i18n.min.js.map +1 -1
- package/dist/lokascript-i18n.mjs +1697 -146
- package/dist/lokascript-i18n.mjs.map +1 -1
- package/dist/plugins/vite.cjs +359 -132
- package/dist/plugins/vite.cjs.map +1 -1
- package/dist/plugins/vite.js +359 -132
- package/dist/plugins/vite.js.map +1 -1
- package/dist/plugins/webpack.cjs +359 -132
- package/dist/plugins/webpack.cjs.map +1 -1
- package/dist/plugins/webpack.js +359 -132
- package/dist/plugins/webpack.js.map +1 -1
- package/dist/{transformer-BLv389qz.d.ts → transformer-BA8YX9u1.d.ts} +120 -2
- package/dist/{transformer-PFsYELrc.d.cts → transformer-DM7iplCU.d.cts} +120 -2
- package/dist/{types-BcALAO6h.d.cts → types-BYtpqGq3.d.cts} +1 -1
- package/dist/{types-BcALAO6h.d.ts → types-BYtpqGq3.d.ts} +1 -1
- package/package.json +9 -6
- package/src/browser.ts +1 -0
- package/src/command-primary-roles.test.ts +59 -0
- package/src/constants.ts +51 -0
- package/src/dictionaries/ar.ts +8 -3
- package/src/dictionaries/bn.ts +3 -1
- package/src/dictionaries/de.ts +19 -4
- package/src/dictionaries/derive.ts +4 -0
- package/src/dictionaries/en.ts +9 -0
- package/src/dictionaries/es.ts +2 -2
- package/src/dictionaries/fr.ts +6 -2
- package/src/dictionaries/he.ts +7 -1
- package/src/dictionaries/hi.ts +17 -10
- package/src/dictionaries/id.ts +15 -8
- package/src/dictionaries/it.ts +6 -3
- package/src/dictionaries/ja.ts +10 -4
- package/src/dictionaries/ko.ts +9 -4
- package/src/dictionaries/ms.ts +2 -1
- package/src/dictionaries/pl.ts +11 -3
- package/src/dictionaries/pt.ts +10 -4
- package/src/dictionaries/qu.ts +46 -22
- package/src/dictionaries/ru.ts +14 -7
- package/src/dictionaries/sw.ts +37 -10
- package/src/dictionaries/th.ts +3 -1
- package/src/dictionaries/tl.ts +14 -7
- package/src/dictionaries/tr.ts +22 -9
- package/src/dictionaries/uk.ts +13 -7
- package/src/dictionaries/vi.ts +3 -3
- package/src/dictionaries/zh.ts +10 -2
- package/src/examples/new-languages.ts +1 -1
- package/src/grammar/grammar.test.ts +1242 -1
- package/src/grammar/index.ts +1 -0
- package/src/grammar/profiles/index.ts +358 -12
- package/src/grammar/transformer.ts +1053 -16
- package/src/grammar/types.ts +22 -1
- package/src/index.ts +1 -0
- package/src/new-languages.test.ts +7 -3
- package/src/parser/parser-integration.test.ts +5 -1
- package/src/positional-keyword-drift.test.ts +138 -0
package/src/grammar/types.ts
CHANGED
|
@@ -538,7 +538,28 @@ export function insertMarkers(
|
|
|
538
538
|
const result: string[] = [];
|
|
539
539
|
|
|
540
540
|
for (const element of elements) {
|
|
541
|
-
|
|
541
|
+
let marker = markers.find(m => m.role === element.role);
|
|
542
|
+
|
|
543
|
+
// A selector-shaped "event" is never a trigger. It's the dangling target of
|
|
544
|
+
// a locative `on` (`set @role to "alert" on #sr-announce`) that
|
|
545
|
+
// splitOnCommandBoundaries splits off as a headless pseudo-handler (set/put
|
|
546
|
+
// are deliberately NOT in ON_TARGET_COMMANDS — dual-destination collision).
|
|
547
|
+
// Emitting the event marker after it (ko 할 때, ja で, bn এ, tr de) plants a
|
|
548
|
+
// spurious mid-stream event anchor in the emission; suppress the marker and
|
|
549
|
+
// emit the bare selector, which body parsers skip harmlessly.
|
|
550
|
+
if (marker?.role === 'event' && /^[#.<@[]/.test(element.translated || element.value)) {
|
|
551
|
+
marker = undefined;
|
|
552
|
+
}
|
|
553
|
+
|
|
554
|
+
// A pronoun-valued "duration" is never a time expression. It's the target
|
|
555
|
+
// phrase of `take … for me` (en maps `for` → duration lexically), kept
|
|
556
|
+
// in-clause by splitOnCommandBoundaries' loop-head guard. Emitting the
|
|
557
|
+
// duration marker after it (bn জন্য — also bn's `for` loop keyword) anchors
|
|
558
|
+
// a spurious `for` command in the semantic parse; suppress the marker and
|
|
559
|
+
// emit the bare pronoun, which take-pattern matchers skip harmlessly.
|
|
560
|
+
if (marker?.role === 'duration' && /^(me|you|it)$/i.test(element.value)) {
|
|
561
|
+
marker = undefined;
|
|
562
|
+
}
|
|
542
563
|
|
|
543
564
|
if (marker) {
|
|
544
565
|
if (adpositionType === 'preposition') {
|
package/src/index.ts
CHANGED
|
@@ -80,7 +80,7 @@ describe('New Language Support', () => {
|
|
|
80
80
|
// TODO: Once grammar transformation is implemented, test for native word order
|
|
81
81
|
expect(result).toContain('klik'); // click → klik
|
|
82
82
|
expect(result).toContain('pada'); // on → pada
|
|
83
|
-
expect(result).toContain('
|
|
83
|
+
expect(result).toContain('alihkan'); // toggle → alihkan (aligned to the semantic profile primary; `ganti` is swap's alternative)
|
|
84
84
|
});
|
|
85
85
|
|
|
86
86
|
it('should handle form interactions', () => {
|
|
@@ -98,7 +98,9 @@ describe('New Language Support', () => {
|
|
|
98
98
|
from: 'en',
|
|
99
99
|
to: 'id',
|
|
100
100
|
});
|
|
101
|
-
|
|
101
|
+
// fetch → muat (profile primary; 'ambil' is take's word — the old
|
|
102
|
+
// emission made every id fetch parse as take)
|
|
103
|
+
expect(result).toContain('muat');
|
|
102
104
|
expect(result).toContain('taruh'); // put → taruh
|
|
103
105
|
expect(result).toContain('hasil'); // result → hasil
|
|
104
106
|
});
|
|
@@ -154,7 +156,9 @@ describe('New Language Support', () => {
|
|
|
154
156
|
to: 'qu',
|
|
155
157
|
});
|
|
156
158
|
expect(result).toContain('niy'); // tell → niy
|
|
157
|
-
|
|
159
|
+
// closest → kaylla (single token the qu tokenizer recognizes; the old
|
|
160
|
+
// aswan_kaylla compound split as as+wan+_+kaylla and never parsed)
|
|
161
|
+
expect(result).toContain('kaylla');
|
|
158
162
|
});
|
|
159
163
|
|
|
160
164
|
it('should support pluralization', () => {
|
|
@@ -384,7 +384,11 @@ describe('KeywordProvider Integration', () => {
|
|
|
384
384
|
expect(trKeywords.resolve('ve')).toBe('and');
|
|
385
385
|
expect(trKeywords.resolve('veya')).toBe('or');
|
|
386
386
|
expect(trKeywords.resolve('değil')).toBe('not');
|
|
387
|
-
|
|
387
|
+
// 'sonra' is the positional `after` (`put X sonra Y`); `then` uses 'ardından'
|
|
388
|
+
// (disambiguated so the put-after position word no longer collides with then —
|
|
389
|
+
// mirrors zh then:'那么' vs after:'之后'). See put-tr-after / the tr then-set.
|
|
390
|
+
expect(trKeywords.resolve('sonra')).toBe('after');
|
|
391
|
+
expect(trKeywords.resolve('ardından')).toBe('then');
|
|
388
392
|
expect(trKeywords.resolve('yoksa')).toBe('else');
|
|
389
393
|
});
|
|
390
394
|
|
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Dict positional emissions ↔ tokenizer recognition.
|
|
3
|
+
*
|
|
4
|
+
* The i18n dictionaries' `expressions` section declares what the transformer
|
|
5
|
+
* EMITS for positional/scope concepts (first/last/next/previous/closest/
|
|
6
|
+
* parent/random); the semantic tokenizers declare what the parser RECOGNIZES.
|
|
7
|
+
* Nothing else keeps the two in sync — Swahili's dict emitted `ijayo` for
|
|
8
|
+
* `next` while the tokenizer only knew `ifuatayo`, so `put X into next <sel>`
|
|
9
|
+
* failed to parse outright (fixed in #338).
|
|
10
|
+
*
|
|
11
|
+
* For every language and concept this test tokenizes the dict emission and
|
|
12
|
+
* requires it to normalize back to that concept. Same convention as
|
|
13
|
+
* command-primary-roles.test.ts: the test imports @lokascript/semantic, but
|
|
14
|
+
* tests don't ship in the bundle.
|
|
15
|
+
*
|
|
16
|
+
* KNOWN_DRIFT below is a burn-down list of the misalignments that existed when
|
|
17
|
+
* the test was introduced (mostly `random`, the `closest` compounds, and the
|
|
18
|
+
* ru/uk positional sets — see the table in the PR that added this). Each entry
|
|
19
|
+
* suppresses the exact-match requirement for one lang:concept pair. The test
|
|
20
|
+
* also fails when an entry STARTS passing, so fixes must remove their entry —
|
|
21
|
+
* the list can only shrink. Do NOT add new entries to silence a regression;
|
|
22
|
+
* new drift means a dict emission and a tokenizer disagree and one of them is
|
|
23
|
+
* wrong.
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
import { describe, it, expect } from 'vitest';
|
|
27
|
+
import { getTokenizer } from '@lokascript/semantic';
|
|
28
|
+
import { dictionaries } from './dictionaries';
|
|
29
|
+
|
|
30
|
+
const POSITIONAL_CONCEPTS = [
|
|
31
|
+
'first',
|
|
32
|
+
'last',
|
|
33
|
+
'next',
|
|
34
|
+
'previous',
|
|
35
|
+
'closest',
|
|
36
|
+
'parent',
|
|
37
|
+
'random',
|
|
38
|
+
] as const;
|
|
39
|
+
|
|
40
|
+
/**
|
|
41
|
+
* Burn-down list (lang:concept). State as of introduction:
|
|
42
|
+
* - `random` is unrecognized by most tokenizers (no extras entry).
|
|
43
|
+
* - The `closest` superlatives/compounds (es máscercano, fr plusproche,
|
|
44
|
+
* it piùvicino, pt mais_próximo, tr en_yakın, qu aswan_kaylla) split or
|
|
45
|
+
* missed entirely — the tokenizer never matches multi-word/underscore
|
|
46
|
+
* natives, so the old tokenizer entries were dead. Fixed by aligning each
|
|
47
|
+
* dict to a single token the tokenizer recognizes (es cercano, fr proche,
|
|
48
|
+
* it vicino, pt maispróximo, tr enyakın, qu kaylla).
|
|
49
|
+
* - ru/uk tokenizers carried only the feminine/neuter gendered variants — the
|
|
50
|
+
* masculine nominative forms the dict emits were never listed; fixed.
|
|
51
|
+
* - qu next/previous were CROSS-MAPPED (qhipantin→last, ñawpaqnin→first —
|
|
52
|
+
* morphology bound the prefixes); fixed with exact tokenizer entries.
|
|
53
|
+
* - de closest: FIXED (R2 wave 13). The dict used to emit `nächste` for both
|
|
54
|
+
* next and closest, and the tokenizer normalizes `nächste`→next by design, so
|
|
55
|
+
* closest was unrecoverable. The dict now emits the unambiguous
|
|
56
|
+
* `nächstgelegene` for closest, which the tokenizer maps →closest (distinct
|
|
57
|
+
* word, no shadowing of next).
|
|
58
|
+
* - bn last (শেষ) normalizes to `end` (the block terminator) — a polysemous
|
|
59
|
+
* word claimed by the structural keyword. (sw had the same collision via
|
|
60
|
+
* mwisho; fixed by emitting the distinct concatenated adjective `wamwisho`,
|
|
61
|
+
* which the tokenizer reads as `last` — the saufsi/wennnicht/enyakın class.)
|
|
62
|
+
*/
|
|
63
|
+
const KNOWN_DRIFT = new Set<string>([
|
|
64
|
+
'ar:parent',
|
|
65
|
+
'ar:random',
|
|
66
|
+
'bn:last',
|
|
67
|
+
'bn:random',
|
|
68
|
+
'de:parent',
|
|
69
|
+
'de:random',
|
|
70
|
+
'es:random',
|
|
71
|
+
'fr:random',
|
|
72
|
+
'hi:random',
|
|
73
|
+
'id:random',
|
|
74
|
+
'it:parent',
|
|
75
|
+
'it:random',
|
|
76
|
+
'ja:random',
|
|
77
|
+
'ko:random',
|
|
78
|
+
'ms:random',
|
|
79
|
+
'pl:random',
|
|
80
|
+
'pt:random',
|
|
81
|
+
'qu:parent',
|
|
82
|
+
'qu:random',
|
|
83
|
+
'sw:random',
|
|
84
|
+
'th:random',
|
|
85
|
+
'tr:random',
|
|
86
|
+
'zh:random',
|
|
87
|
+
]);
|
|
88
|
+
|
|
89
|
+
function normalizeEmission(lang: string, emission: string): string | null {
|
|
90
|
+
let tokenizer: ReturnType<typeof getTokenizer>;
|
|
91
|
+
try {
|
|
92
|
+
tokenizer = getTokenizer(lang);
|
|
93
|
+
} catch {
|
|
94
|
+
return null; // no tokenizer for this language — out of scope
|
|
95
|
+
}
|
|
96
|
+
try {
|
|
97
|
+
const stream = tokenizer.tokenize(emission);
|
|
98
|
+
const token = stream.peek() as { normalized?: string; value: string } | undefined | null;
|
|
99
|
+
if (!token) return '(no-token)';
|
|
100
|
+
return token.normalized ?? token.value;
|
|
101
|
+
} catch {
|
|
102
|
+
return '(tokenize-error)';
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
describe('dict positional emissions are recognized by the tokenizer', () => {
|
|
107
|
+
for (const [lang, dict] of Object.entries(dictionaries)) {
|
|
108
|
+
if (lang === 'en') continue;
|
|
109
|
+
const expressions = dict.expressions ?? {};
|
|
110
|
+
for (const concept of POSITIONAL_CONCEPTS) {
|
|
111
|
+
const emission = expressions[concept];
|
|
112
|
+
if (!emission) continue;
|
|
113
|
+
const key = `${lang}:${concept}`;
|
|
114
|
+
const expectedDrift = KNOWN_DRIFT.has(key);
|
|
115
|
+
|
|
116
|
+
it(`[${key}] '${emission}' ${expectedDrift ? 'is on the burn-down list' : `normalizes to '${concept}'`}`, () => {
|
|
117
|
+
const normalized = normalizeEmission(lang, emission);
|
|
118
|
+
if (normalized === null) return; // no tokenizer registered
|
|
119
|
+
const aligned = normalized === concept;
|
|
120
|
+
if (expectedDrift) {
|
|
121
|
+
expect(
|
|
122
|
+
aligned,
|
|
123
|
+
`[${key}] '${emission}' now normalizes to '${concept}' — the drift is fixed; ` +
|
|
124
|
+
`remove '${key}' from KNOWN_DRIFT in positional-keyword-drift.test.ts (the list only shrinks).`
|
|
125
|
+
).toBe(false);
|
|
126
|
+
} else {
|
|
127
|
+
expect(
|
|
128
|
+
aligned,
|
|
129
|
+
`[${key}] dict emits '${emission}' but the ${lang} tokenizer normalizes it to ` +
|
|
130
|
+
`'${normalized}', not '${concept}'. The transformer emits words the parser can't ` +
|
|
131
|
+
`read as positionals (the sw 'ijayo' bug class, #338). Align the dict emission to ` +
|
|
132
|
+
`a word the tokenizer recognizes, or teach the tokenizer the dict's word.`
|
|
133
|
+
).toBe(true);
|
|
134
|
+
}
|
|
135
|
+
});
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
});
|