@lokascript/framework 2.7.2 → 2.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. package/CHANGELOG.md +393 -0
  2. package/dist/api/create-dsl.d.ts +93 -1
  3. package/dist/api/create-dsl.d.ts.map +1 -1
  4. package/dist/api/domain-registry.d.ts +5 -3
  5. package/dist/api/domain-registry.d.ts.map +1 -1
  6. package/dist/api/index.js +238 -20
  7. package/dist/api/index.js.map +1 -1
  8. package/dist/core/index.js +67 -7
  9. package/dist/core/index.js.map +1 -1
  10. package/dist/core/tokenization/base-tokenizer.d.ts +39 -3
  11. package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
  12. package/dist/core/tokenization/extractors.d.ts +6 -0
  13. package/dist/core/tokenization/extractors.d.ts.map +1 -1
  14. package/dist/core/tokenization/index.js +67 -7
  15. package/dist/core/tokenization/index.js.map +1 -1
  16. package/dist/core/tokenization/token-utils.d.ts +15 -0
  17. package/dist/core/tokenization/token-utils.d.ts.map +1 -1
  18. package/dist/generation/index.js +75 -48
  19. package/dist/generation/index.js.map +1 -1
  20. package/dist/generation/pattern-generator.d.ts +8 -1
  21. package/dist/generation/pattern-generator.d.ts.map +1 -1
  22. package/dist/generation/renderer.d.ts +53 -1
  23. package/dist/generation/renderer.d.ts.map +1 -1
  24. package/dist/index.cjs +349 -118
  25. package/dist/index.cjs.map +1 -1
  26. package/dist/index.d.ts +4 -4
  27. package/dist/index.d.ts.map +1 -1
  28. package/dist/index.js +345 -118
  29. package/dist/index.js.map +1 -1
  30. package/dist/interfaces/value-extractor.d.ts +5 -0
  31. package/dist/interfaces/value-extractor.d.ts.map +1 -1
  32. package/dist/multilingual/index.js +66 -7
  33. package/dist/multilingual/index.js.map +1 -1
  34. package/dist/testing/index.js +4 -12
  35. package/dist/testing/index.js.map +1 -1
  36. package/package.json +4 -3
  37. package/src/api/create-dsl.test.ts +11 -0
  38. package/src/api/create-dsl.ts +278 -9
  39. package/src/api/domain-registry.ts +15 -10
  40. package/src/api/extensions.test.ts +322 -0
  41. package/src/core/tokenization/base-tokenizer.ts +78 -9
  42. package/src/core/tokenization/colon-qualifier.test.ts +129 -0
  43. package/src/core/tokenization/css-selector-extractor.test.ts +67 -0
  44. package/src/core/tokenization/extractors.ts +6 -0
  45. package/src/core/tokenization/token-utils.ts +18 -0
  46. package/src/generation/domain-renderer.test.ts +172 -0
  47. package/src/generation/pattern-generator.test.ts +102 -0
  48. package/src/generation/pattern-generator.ts +32 -20
  49. package/src/generation/renderer.test.ts +243 -4
  50. package/src/generation/renderer.ts +188 -45
  51. package/src/index.ts +9 -1
  52. package/src/interfaces/value-extractor.ts +50 -0
  53. package/src/ir/protocol-json.test.ts +21 -0
  54. package/src/ir/references.test.ts +5 -2
  55. package/src/prompts/prompt-generator.ts +4 -1
@@ -0,0 +1,322 @@
1
+ /**
2
+ * DomainExtension: adding a command to a DSL from outside its package.
3
+ *
4
+ * Every domain used to render through a hardcoded `switch (node.action)` and
5
+ * generate through another, so a downstream consumer could not add a command
6
+ * without editing the package. This is the supported path: a schema plus one
7
+ * vocabulary entry per language.
8
+ *
9
+ * The command exercised here ("research") is the one lokascript-learn built by
10
+ * hand against 2.8.0 to prove the underlying pieces worked.
11
+ */
12
+
13
+ import { describe, it, expect } from 'vitest';
14
+ import {
15
+ createMultilingualDSL,
16
+ createSimpleTokenizer,
17
+ defineCommand,
18
+ defineRole,
19
+ type DomainExtension,
20
+ type ExtractionResult,
21
+ type LanguageTokenizer,
22
+ type SemanticNode,
23
+ type ValueExtractor,
24
+ } from '../index';
25
+
26
+ /**
27
+ * Keeps `#id` / `.class` a single token, as every real domain tokenizer does.
28
+ * Without it the `#` splits off and an SOV source role followed by a particle
29
+ * never matches.
30
+ */
31
+ class CSSSelectorExtractor implements ValueExtractor {
32
+ readonly name = 'css-selector';
33
+
34
+ canExtract(input: string, position: number): boolean {
35
+ const char = input[position];
36
+ if (char !== '#' && char !== '.') return false;
37
+ return /[a-zA-Z_-]/.test(input[position + 1] ?? '');
38
+ }
39
+
40
+ extract(input: string, position: number): ExtractionResult | null {
41
+ let end = position + 1;
42
+ while (end < input.length && /[a-zA-Z0-9_-]/.test(input[end])) end++;
43
+ if (end === position + 1) return null;
44
+ return { value: input.slice(position, end), length: end - position };
45
+ }
46
+ }
47
+
48
+ // =============================================================================
49
+ // A three-word-order toy DSL
50
+ // =============================================================================
51
+
52
+ const askSchema = defineCommand({
53
+ action: 'ask',
54
+ description: 'Ask a question',
55
+ category: 'llm',
56
+ primaryRole: 'patient',
57
+ roles: [
58
+ defineRole({
59
+ role: 'patient',
60
+ description: 'The question',
61
+ required: true,
62
+ expectedTypes: ['expression'],
63
+ }),
64
+ defineRole({
65
+ role: 'source',
66
+ description: 'Where to look',
67
+ required: true,
68
+ expectedTypes: ['expression'],
69
+ markerOverride: { en: 'from', ja: 'から', ar: 'من' },
70
+ }),
71
+ ],
72
+ });
73
+
74
+ const researchSchema = defineCommand({
75
+ action: 'research',
76
+ description: 'Research a topic',
77
+ category: 'llm',
78
+ primaryRole: 'patient',
79
+ roles: [
80
+ defineRole({
81
+ role: 'patient',
82
+ description: 'The topic',
83
+ required: true,
84
+ expectedTypes: ['expression'],
85
+ }),
86
+ defineRole({
87
+ role: 'source',
88
+ description: 'Where to look',
89
+ required: true,
90
+ expectedTypes: ['expression'],
91
+ markerOverride: { en: 'from', ja: 'から', ar: 'من' },
92
+ }),
93
+ ],
94
+ });
95
+
96
+ const PROFILES = {
97
+ en: {
98
+ code: 'en',
99
+ wordOrder: 'SVO' as const,
100
+ keywords: { ask: { primary: 'ask' } },
101
+ roleMarkers: {},
102
+ },
103
+ ja: {
104
+ code: 'ja',
105
+ wordOrder: 'SOV' as const,
106
+ keywords: { ask: { primary: '聞く' } },
107
+ roleMarkers: {},
108
+ },
109
+ ar: {
110
+ code: 'ar',
111
+ wordOrder: 'VSO' as const,
112
+ keywords: { ask: { primary: 'اسأل' } },
113
+ roleMarkers: {},
114
+ },
115
+ };
116
+
117
+ const research: DomainExtension = {
118
+ schema: researchSchema,
119
+ vocabulary: {
120
+ en: { keyword: { primary: 'research' } },
121
+ ja: { keyword: { primary: '調査' } },
122
+ ar: { keyword: { primary: 'ابحث' } },
123
+ },
124
+ };
125
+
126
+ function tokenizerFor(code: 'en' | 'ja' | 'ar'): LanguageTokenizer {
127
+ return createSimpleTokenizer({
128
+ language: code,
129
+ // Both the built-in and the extension vocabulary: a tokenizer is configured
130
+ // once, so it must already know the words an extension may introduce.
131
+ keywords: [
132
+ ...Object.values(PROFILES[code].keywords).map(k => k.primary),
133
+ 'research',
134
+ '調査',
135
+ 'ابحث',
136
+ 'from',
137
+ 'から',
138
+ 'من',
139
+ ],
140
+ caseInsensitive: code === 'en',
141
+ customExtractors: [new CSSSelectorExtractor()],
142
+ });
143
+ }
144
+
145
+ function createToyDSL(options: { extensions?: readonly DomainExtension[] } = {}) {
146
+ return createMultilingualDSL({
147
+ name: 'Toy',
148
+ schemas: [askSchema],
149
+ languages: (['en', 'ja', 'ar'] as const).map(code => ({
150
+ code,
151
+ name: code,
152
+ nativeName: code,
153
+ tokenizer: tokenizerFor(code),
154
+ patternProfile: PROFILES[code],
155
+ })),
156
+ codeGenerator: {
157
+ generate: (node: SemanticNode) => `BASE:${node.action}`,
158
+ },
159
+ ...(options.extensions && { extensions: options.extensions }),
160
+ });
161
+ }
162
+
163
+ function makeNode(action: string, roles: Record<string, string>): SemanticNode {
164
+ const rolesMap = new Map<string, { type: 'expression'; raw: string }>();
165
+ for (const [k, v] of Object.entries(roles)) rolesMap.set(k, { type: 'expression', raw: v });
166
+ return { kind: 'command', action, roles: rolesMap };
167
+ }
168
+
169
+ // =============================================================================
170
+ // Tests
171
+ // =============================================================================
172
+
173
+ describe('DomainExtension', () => {
174
+ describe('parsing', () => {
175
+ it('parses the extension command in an SVO language', () => {
176
+ const dsl = createToyDSL({ extensions: [research] });
177
+ const node = dsl.parse('research "climate" from #wiki', 'en');
178
+ expect(node.action).toBe('research');
179
+ expect(node.roles.get('patient')).toBeDefined();
180
+ expect(node.roles.get('source')).toBeDefined();
181
+ });
182
+
183
+ it('parses the extension command in an SOV language', () => {
184
+ const dsl = createToyDSL({ extensions: [research] });
185
+ const node = dsl.parse('"climate" #wiki から 調査', 'ja');
186
+ expect(node.action).toBe('research');
187
+ });
188
+
189
+ it('parses the extension command in a VSO language', () => {
190
+ const dsl = createToyDSL({ extensions: [research] });
191
+ const node = dsl.parse('ابحث "climate" من #wiki', 'ar');
192
+ expect(node.action).toBe('research');
193
+ });
194
+
195
+ it('does not parse the extension command without the extension', () => {
196
+ const dsl = createToyDSL();
197
+ expect(() => dsl.parse('research "climate" from #wiki', 'en')).toThrow(/No pattern matched/);
198
+ });
199
+
200
+ it('reports the extension action as supported by explicit syntax', () => {
201
+ const dsl = createToyDSL({ extensions: [research] });
202
+ const node = dsl.parse('[research patient:climate source:#wiki]', 'en');
203
+ expect(node.action).toBe('research');
204
+ });
205
+ });
206
+
207
+ describe('rendering', () => {
208
+ const dsl = createToyDSL({ extensions: [research] });
209
+ const node = makeNode('research', { patient: '"climate"', source: '#wiki' });
210
+
211
+ it('renders SVO from the schema alone', () => {
212
+ expect(dsl.render?.(node, 'en')).toBe('research "climate" from #wiki');
213
+ });
214
+
215
+ it('renders verb-final for SOV', () => {
216
+ expect(dsl.render?.(node, 'ja')).toBe('"climate" #wiki から 調査');
217
+ });
218
+
219
+ it('renders verb-initial for VSO', () => {
220
+ expect(dsl.render?.(node, 'ar')).toBe('ابحث "climate" من #wiki');
221
+ });
222
+
223
+ it('prefers an extension-supplied renderer', () => {
224
+ const custom = createToyDSL({
225
+ extensions: [{ ...research, render: () => 'CUSTOM' }],
226
+ });
227
+ expect(custom.render?.(node, 'en')).toBe('CUSTOM');
228
+ });
229
+
230
+ it('returns null for an action with neither renderer nor schema', () => {
231
+ expect(dsl.render?.(makeNode('nonsense', {}), 'en')).toBeNull();
232
+ });
233
+
234
+ it('prefers the domain renderer over the schema fallback for built-ins', () => {
235
+ const withDomainRenderer = createMultilingualDSL({
236
+ name: 'Toy',
237
+ schemas: [askSchema],
238
+ languages: [
239
+ {
240
+ code: 'en',
241
+ name: 'en',
242
+ nativeName: 'en',
243
+ tokenizer: tokenizerFor('en'),
244
+ patternProfile: PROFILES.en,
245
+ },
246
+ ],
247
+ renderer: node => (node.action === 'ask' ? 'DOMAIN RENDERED' : null),
248
+ extensions: [{ ...research, vocabulary: { en: research.vocabulary.en } }],
249
+ });
250
+
251
+ expect(withDomainRenderer.render?.(makeNode('ask', { patient: 'q' }), 'en')).toBe(
252
+ 'DOMAIN RENDERED'
253
+ );
254
+ // …while the extension still falls through to the schema renderer
255
+ expect(withDomainRenderer.render?.(node, 'en')).toBe('research "climate" from #wiki');
256
+ });
257
+ });
258
+
259
+ describe('compilation', () => {
260
+ it('uses the extension generator for the extension action', () => {
261
+ const dsl = createToyDSL({
262
+ extensions: [{ ...research, generate: node => `EXT:${node.action}` }],
263
+ });
264
+ const result = dsl.compile('research "climate" from #wiki', 'en');
265
+ expect(result.ok).toBe(true);
266
+ expect(result.code).toBe('EXT:research');
267
+ });
268
+
269
+ it('leaves the base generator handling built-in actions', () => {
270
+ const dsl = createToyDSL({
271
+ extensions: [{ ...research, generate: node => `EXT:${node.action}` }],
272
+ });
273
+ const result = dsl.compile('ask "q" from #wiki', 'en');
274
+ expect(result.ok).toBe(true);
275
+ expect(result.code).toBe('BASE:ask');
276
+ });
277
+
278
+ it('falls back to the base generator when the extension supplies none', () => {
279
+ const dsl = createToyDSL({ extensions: [research] });
280
+ const result = dsl.compile('research "climate" from #wiki', 'en');
281
+ expect(result.ok).toBe(true);
282
+ expect(result.code).toBe('BASE:research');
283
+ });
284
+ });
285
+
286
+ describe('built-in commands are unaffected', () => {
287
+ it('parses and compiles identically with and without extensions', () => {
288
+ const plain = createToyDSL();
289
+ const extended = createToyDSL({ extensions: [research] });
290
+
291
+ for (const [input, language] of [
292
+ ['ask "q" from #wiki', 'en'],
293
+ ['"q" #wiki から 聞く', 'ja'],
294
+ ['اسأل "q" من #wiki', 'ar'],
295
+ ] as const) {
296
+ expect(extended.parse(input, language).action).toBe(plain.parse(input, language).action);
297
+ expect(extended.compile(input, language).code).toBe(plain.compile(input, language).code);
298
+ }
299
+ });
300
+ });
301
+
302
+ describe('configuration errors', () => {
303
+ it('rejects an extension whose action collides with a built-in', () => {
304
+ const collision: DomainExtension = {
305
+ schema: defineCommand({
306
+ action: 'ask',
307
+ roles: [defineRole({ role: 'patient', required: true, expectedTypes: ['expression'] })],
308
+ }),
309
+ vocabulary: { en: { keyword: { primary: 'inquire' } } },
310
+ };
311
+ expect(() => createToyDSL({ extensions: [collision] })).toThrow(/collides/);
312
+ });
313
+
314
+ it('rejects vocabulary for a language the DSL does not configure', () => {
315
+ const unknownLanguage: DomainExtension = {
316
+ ...research,
317
+ vocabulary: { ...research.vocabulary, xx: { keyword: { primary: 'x' } } },
318
+ };
319
+ expect(() => createToyDSL({ extensions: [unknownLanguage] })).toThrow(/not configured/);
320
+ });
321
+ });
322
+ });
@@ -20,6 +20,7 @@ import {
20
20
  isWhitespace,
21
21
  isDigit,
22
22
  isAsciiIdentifierChar,
23
+ stripOptionalDiacritics,
23
24
  TokenStreamImpl,
24
25
  type TimeUnitMapping,
25
26
  type CreateTokenOptions,
@@ -347,7 +348,61 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
347
348
  }
348
349
  }
349
350
 
350
- return new TokenStreamImpl(tokens, this.language);
351
+ return new TokenStreamImpl(this.mergeColonQualifiedNames(tokens), this.language);
352
+ }
353
+
354
+ /**
355
+ * ASCII word of the shape the English word-walker produces. Excludes `:`, so a
356
+ * token that already carries a qualifier never merges again — `a:b:c` yields
357
+ * `a:b` + `:c`, byte-matching the English extractor's single-segment merge.
358
+ */
359
+ private static readonly ASCII_WORD = /^[A-Za-z_][A-Za-z0-9_]*$/;
360
+
361
+ /** `:name` — only a variable-ref-style extractor ever emits this token shape. */
362
+ private static readonly COLON_QUALIFIER = /^:[A-Za-z_][A-Za-z0-9_]*$/;
363
+
364
+ /**
365
+ * Fuse `name` + `:qualifier` into ONE identifier (`draggable:start`).
366
+ *
367
+ * `:name` is hyperscript's local-variable sigil, but a colon IMMEDIATELY
368
+ * preceded by an identifier is a qualifier (custom event namespace), not a
369
+ * sigil. The English tokenizer already merges these inside
370
+ * EnglishKeywordExtractor; this post-pass gives the other 23 languages the
371
+ * same stream. Strict position adjacency is the discriminator: whitespace
372
+ * between the tokens (`trigger :start`) breaks `end === start`, so a spaced
373
+ * local-variable reference survives untouched.
374
+ *
375
+ * Self-gating for non-hyperscript tokenizers (domain DSLs): their extractor
376
+ * sets tokenize `:` as bare punctuation (length 1), which never matches
377
+ * COLON_QUALIFIER, so this pass is a no-op for them.
378
+ */
379
+ protected mergeColonQualifiedNames(tokens: LanguageToken[]): LanguageToken[] {
380
+ const out: LanguageToken[] = [];
381
+ for (const tok of tokens) {
382
+ const prev = out[out.length - 1];
383
+ // No kind gate: the English extractor merges before classification, so a
384
+ // word some language classifies as particle/keyword (es `a`, tr `i`)
385
+ // must fuse the same way. ASCII_WORD already excludes every non-word
386
+ // kind structurally (selectors, urls, numbers, strings, operators).
387
+ if (
388
+ prev &&
389
+ BaseTokenizer.ASCII_WORD.test(prev.value) &&
390
+ BaseTokenizer.COLON_QUALIFIER.test(tok.value) &&
391
+ prev.position.end === tok.position.start
392
+ ) {
393
+ const merged = prev.value + tok.value;
394
+ // Re-classify and drop normalized/stem/metadata — the merged word is no
395
+ // longer the keyword the pieces may have been (matches the en shape).
396
+ out[out.length - 1] = createToken(
397
+ merged,
398
+ this.classifyToken(merged),
399
+ createPosition(prev.position.start, tok.position.end)
400
+ );
401
+ continue;
402
+ }
403
+ out.push(tok);
404
+ }
405
+ return out;
351
406
  }
352
407
 
353
408
  /**
@@ -544,9 +599,7 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
544
599
  * @returns Word without diacritics
545
600
  */
546
601
  protected removeDiacritics(word: string): string {
547
- // Arabic diacritics: U+064B-U+0652 (fatha, kasra, damma, sukun, shadda, etc.)
548
- // U+0670 (superscript alif)
549
- return word.replace(/[\u064B-\u0652\u0670]/g, '');
602
+ return stripOptionalDiacritics(word);
550
603
  }
551
604
 
552
605
  /**
@@ -650,25 +703,41 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
650
703
  }
651
704
 
652
705
  /**
653
- * Look up a keyword by native word (case-insensitive).
706
+ * Look up a keyword by native word (case-insensitive, diacritic-insensitive).
654
707
  * O(1) lookup using the keyword map.
655
708
  *
709
+ * The map is INDEXED both with and without diacritics (see
710
+ * `initializeKeywordsFromProfile`), so a stripped QUERY is the other half of
711
+ * that: it lets a surface form carrying harakat the profile does not happen to
712
+ * spell still find its entry. Only consulted after the exact lookup misses, so
713
+ * every previously-matching word resolves byte-identically.
714
+ *
715
+ * Half-implementing this — indexing stripped but querying exact — is what made
716
+ * diacritized `بَدِّل` (toggle) tokenize as `kind=particle normalized=with`:
717
+ * `isKeyword` returned false, so the guard in `ArabicProcliticExtractor` that
718
+ * exists to prevent exactly that handed the word on, and the single-char `ب`
719
+ * bi- proclitic claimed it. A wrong CONCEPT, not a failed parse.
720
+ *
656
721
  * @param native - Native word to look up
657
722
  * @returns KeywordEntry if found, undefined otherwise
658
723
  */
659
724
  protected lookupKeyword(native: string): KeywordEntry | undefined {
660
- return this.profileKeywordMap.get(native.toLowerCase());
725
+ const exact = this.profileKeywordMap.get(native.toLowerCase());
726
+ if (exact) return exact;
727
+ const stripped = this.removeDiacritics(native);
728
+ if (stripped === native) return undefined;
729
+ return this.profileKeywordMap.get(stripped.toLowerCase());
661
730
  }
662
731
 
663
732
  /**
664
- * Check if a word is a known keyword (case-insensitive).
665
- * O(1) lookup using the keyword map.
733
+ * Check if a word is a known keyword (case-insensitive, diacritic-insensitive).
734
+ * O(1) lookup using the keyword map. See {@link lookupKeyword}.
666
735
  *
667
736
  * @param native - Native word to check
668
737
  * @returns true if the word is a keyword
669
738
  */
670
739
  protected isKeyword(native: string): boolean {
671
- return this.profileKeywordMap.has(native.toLowerCase());
740
+ return this.lookupKeyword(native) !== undefined;
672
741
  }
673
742
 
674
743
  /**
@@ -0,0 +1,129 @@
1
+ /**
2
+ * Contract tests for BaseTokenizer.mergeColonQualifiedNames().
3
+ *
4
+ * `:name` is hyperscript's local-variable sigil, but a colon immediately
5
+ * preceded by an identifier is a qualifier (custom event namespace:
6
+ * `draggable:start`). The English tokenizer merges these inside its keyword
7
+ * extractor; the post-pass gives every other language the same stream. The
8
+ * merge must fire only on strict adjacency and only on `:name`-shaped tokens —
9
+ * domain DSL tokenizers (SQL, BDD, …) emit `:` as bare punctuation, which the
10
+ * pass must never touch.
11
+ */
12
+
13
+ import { describe, it, expect } from 'vitest';
14
+ import { BaseTokenizer, createSimpleTokenizer } from './base-tokenizer';
15
+ import type { TokenKind } from '../types';
16
+ import type { ValueExtractor, ExtractionResult } from '../../interfaces/value-extractor';
17
+
18
+ /** Minimal `:name`/`$name`/`^name` extractor (shape of semantic's VariableRefExtractor). */
19
+ class SigilRefExtractor implements ValueExtractor {
20
+ readonly name = 'sigil-ref';
21
+
22
+ canExtract(input: string, position: number): boolean {
23
+ const ch = input[position];
24
+ return (
25
+ (ch === ':' || ch === '$' || ch === '^') &&
26
+ position + 1 < input.length &&
27
+ /[a-zA-Z_]/.test(input[position + 1])
28
+ );
29
+ }
30
+
31
+ extract(input: string, position: number): ExtractionResult | null {
32
+ if (!this.canExtract(input, position)) return null;
33
+ let length = 1;
34
+ while (position + length < input.length && /[a-zA-Z0-9_]/.test(input[position + length])) {
35
+ length++;
36
+ }
37
+ return { value: input.substring(position, position + length), length };
38
+ }
39
+ }
40
+
41
+ /** Minimal ASCII word extractor (shape of the per-language word walkers). */
42
+ class WordExtractor implements ValueExtractor {
43
+ readonly name = 'word';
44
+
45
+ canExtract(input: string, position: number): boolean {
46
+ return /[a-zA-Z_]/.test(input[position]);
47
+ }
48
+
49
+ extract(input: string, position: number): ExtractionResult | null {
50
+ let length = 0;
51
+ while (position + length < input.length && /[a-zA-Z0-9_]/.test(input[position + length])) {
52
+ length++;
53
+ }
54
+ return length > 0 ? { value: input.substring(position, position + length), length } : null;
55
+ }
56
+ }
57
+
58
+ const KEYWORDS = new Set(['trigger', 'click']);
59
+
60
+ class ProbeTokenizer extends BaseTokenizer {
61
+ readonly language = 'xx';
62
+ readonly direction = 'ltr' as const;
63
+
64
+ constructor() {
65
+ super();
66
+ this.registerExtractors([new SigilRefExtractor(), new WordExtractor()]);
67
+ }
68
+
69
+ classifyToken(token: string): TokenKind {
70
+ return KEYWORDS.has(token) ? 'keyword' : 'identifier';
71
+ }
72
+ }
73
+
74
+ function values(
75
+ tokenizer: { tokenize(input: string): { tokens: readonly { value: string }[] } },
76
+ input: string
77
+ ): string[] {
78
+ return tokenizer.tokenize(input).tokens.map(t => t.value);
79
+ }
80
+
81
+ describe('mergeColonQualifiedNames', () => {
82
+ const t = new ProbeTokenizer();
83
+
84
+ it('fuses identifier + :qualifier into one token', () => {
85
+ expect(values(t, 'trigger draggable:start')).toEqual(['trigger', 'draggable:start']);
86
+ });
87
+
88
+ it('re-classifies the merged token and spans both positions', () => {
89
+ const tokens = t.tokenize('trigger draggable:start').tokens;
90
+ const merged = tokens[1];
91
+ expect(merged.kind).toBe('identifier');
92
+ expect(merged.position.start).toBe('trigger '.length);
93
+ expect(merged.position.end).toBe('trigger draggable:start'.length);
94
+ });
95
+
96
+ it('fuses when the pre-colon word is a keyword (en parity: merge before lookup)', () => {
97
+ const tokens = t.tokenize('click:foo').tokens;
98
+ expect(tokens.map(x => x.value)).toEqual(['click:foo']);
99
+ expect(tokens[0].kind).toBe('identifier');
100
+ });
101
+
102
+ it('does not fuse across whitespace — spaced :name stays a local-variable ref', () => {
103
+ expect(values(t, 'trigger :start')).toEqual(['trigger', ':start']);
104
+ });
105
+
106
+ it('leaves a leading bare sigil untouched', () => {
107
+ expect(values(t, ':start')).toEqual([':start']);
108
+ });
109
+
110
+ it('merges a single segment only — a:b:c matches the English extractor', () => {
111
+ expect(values(t, 'a:b:c')).toEqual(['a:b', ':c']);
112
+ });
113
+
114
+ it('never touches $ and ^ sigils', () => {
115
+ expect(values(t, 'put $foo into ^bar')).toEqual(['put', '$foo', 'into', '^bar']);
116
+ });
117
+
118
+ it('is a no-op for domain-DSL streams where : is bare punctuation', () => {
119
+ const sql = createSimpleTokenizer({
120
+ language: 'en',
121
+ keywords: ['select', 'from', 'where'],
122
+ includeOperators: true,
123
+ });
124
+ // Named params never fuse: the colon tokenizes as length-1 punctuation,
125
+ // which can never match the :name qualifier shape.
126
+ expect(values(sql, 'WHERE x = :param')).toEqual(['WHERE', 'x', '=', ':', 'param']);
127
+ expect(values(sql, 'x=:param')).toEqual(['x', '=', ':', 'param']);
128
+ });
129
+ });
@@ -0,0 +1,67 @@
1
+ import { describe, it, expect } from 'vitest';
2
+ import { CssSelectorExtractor } from '../../interfaces/value-extractor';
3
+
4
+ /**
5
+ * `getDefaultExtractors()` has no CSS-selector extractor, so a DSL that does not
6
+ * register one splits the sigil off as its own token and the role capture keeps
7
+ * only that sigil — `add .active to #button` parsed with patient `"."` and
8
+ * destination `"#"` in domain-learn, -todo, -sql and -jsx, silently, in every
9
+ * language. Five other domains each carried a private copy of this class; this
10
+ * is the shared one.
11
+ */
12
+ describe('CssSelectorExtractor', () => {
13
+ const ex = new CssSelectorExtractor();
14
+
15
+ const extract = (input: string, pos = 0) =>
16
+ ex.canExtract(input, pos) ? ex.extract(input, pos) : null;
17
+
18
+ it.each([
19
+ ['.active', '.active'],
20
+ ['#button', '#button'],
21
+ ['.btn-primary', '.btn-primary'],
22
+ ['._private', '._private'],
23
+ ['#a1', '#a1'],
24
+ ['.-leading-hyphen', '.-leading-hyphen'],
25
+ ])('extracts %s whole', (input, expected) => {
26
+ expect(extract(input)).toEqual({ value: expected, length: expected.length });
27
+ });
28
+
29
+ it('stops at the first character that cannot be in a selector', () => {
30
+ expect(extract('.active to #button')).toEqual({ value: '.active', length: 7 });
31
+ });
32
+
33
+ it('keeps diacritics, so Latin-script class names survive', () => {
34
+ expect(extract('.año')).toEqual({ value: '.año', length: 4 });
35
+ expect(extract('#botón')).toEqual({ value: '#botón', length: 6 });
36
+ });
37
+
38
+ // The SOV languages write their particles flush against the value, so a
39
+ // Unicode-wide body turns `#buttonに` into a single token and carries the role
40
+ // marker inside the value — which is worse than truncating, because the marker
41
+ // is then missing from where the pattern expects it.
42
+ it.each([
43
+ ['#buttonに', '#button'],
44
+ ['.activeを', '.active'],
45
+ ['#button에', '#button'],
46
+ ['.active를', '.active'],
47
+ ['#button的', '#button'],
48
+ ])('stops at the particle in %s', (input, expected) => {
49
+ expect(extract(input)).toEqual({ value: expected, length: expected.length });
50
+ });
51
+
52
+ it.each([
53
+ ['.', 'a bare sigil'],
54
+ ['#', 'a bare sigil'],
55
+ ['. active', 'a sigil followed by a space'],
56
+ ['.1col', 'a digit — CSS identifiers cannot start with one'],
57
+ ['button', 'no sigil at all'],
58
+ ])('declines %s (%s)', input => {
59
+ expect(ex.canExtract(input, 0)).toBe(false);
60
+ });
61
+
62
+ it('declines a CJK-only class name rather than claiming the sigil alone', () => {
63
+ // `.追加` would otherwise extract as a bare `.`; leaving it unclaimed lets
64
+ // the keyword/identifier extractors see the verb.
65
+ expect(ex.canExtract('.追加', 0)).toBe(false);
66
+ });
67
+ });
@@ -34,6 +34,12 @@ import {
34
34
  * Method call handling:
35
35
  * - #dialog.showModal() → stops after #dialog (method call, not compound selector)
36
36
  * - #box.active → compound selector (no parens)
37
+ *
38
+ * NOTE: intentionally diverges from the semantic package's copy
39
+ * (packages/semantic/src/tokenizers/extractors/css-selector.ts), which also
40
+ * consumes pseudo-class/pseudo-element segments (#x:hover, .a:not(.b)). This
41
+ * legacy version is only used by BaseTokenizer.trySelector (no semantic call
42
+ * sites) and stays as-is.
37
43
  */
38
44
  export function extractCssSelector(input: string, startPos: number): string | null {
39
45
  if (startPos >= input.length) return null;
@@ -237,6 +237,24 @@ export function isDigit(char: string): boolean {
237
237
  return /\d/.test(char);
238
238
  }
239
239
 
240
+ /**
241
+ * Strip diacritical marks that are OPTIONAL in their script — currently Arabic
242
+ * harakat (U+064B–U+0652: fatha, kasra, damma, sukun, shadda…) and the
243
+ * superscript alif (U+0670).
244
+ *
245
+ * `بدّل` and `بَدِّل` are the same word; Arabic prose writes either. So any place
246
+ * that compares an Arabic surface form against a declared keyword has to compare
247
+ * stripped, or the same word fails to match itself.
248
+ *
249
+ * Deliberately Arabic-only. Latin diacritics are NOT optional — `obtén`, `récupère`
250
+ * and `vá` differ from their unaccented spellings in meaning or validity — and
251
+ * Hebrew niqqud (U+05B0–U+05BC) is out of range too, so this is inert for every
252
+ * other script.
253
+ */
254
+ export function stripOptionalDiacritics(word: string): string {
255
+ return word.replace(/[ً-ْٰ]/g, '');
256
+ }
257
+
240
258
  /**
241
259
  * Check if a character is an ASCII letter.
242
260
  */