@lokascript/framework 2.7.2 → 2.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +393 -0
- package/dist/api/create-dsl.d.ts +93 -1
- package/dist/api/create-dsl.d.ts.map +1 -1
- package/dist/api/domain-registry.d.ts +5 -3
- package/dist/api/domain-registry.d.ts.map +1 -1
- package/dist/api/index.js +238 -20
- package/dist/api/index.js.map +1 -1
- package/dist/core/index.js +67 -7
- package/dist/core/index.js.map +1 -1
- package/dist/core/tokenization/base-tokenizer.d.ts +39 -3
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
- package/dist/core/tokenization/extractors.d.ts +6 -0
- package/dist/core/tokenization/extractors.d.ts.map +1 -1
- package/dist/core/tokenization/index.js +67 -7
- package/dist/core/tokenization/index.js.map +1 -1
- package/dist/core/tokenization/token-utils.d.ts +15 -0
- package/dist/core/tokenization/token-utils.d.ts.map +1 -1
- package/dist/generation/index.js +75 -48
- package/dist/generation/index.js.map +1 -1
- package/dist/generation/pattern-generator.d.ts +8 -1
- package/dist/generation/pattern-generator.d.ts.map +1 -1
- package/dist/generation/renderer.d.ts +53 -1
- package/dist/generation/renderer.d.ts.map +1 -1
- package/dist/index.cjs +349 -118
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.ts +4 -4
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +345 -118
- package/dist/index.js.map +1 -1
- package/dist/interfaces/value-extractor.d.ts +5 -0
- package/dist/interfaces/value-extractor.d.ts.map +1 -1
- package/dist/multilingual/index.js +66 -7
- package/dist/multilingual/index.js.map +1 -1
- package/dist/testing/index.js +4 -12
- package/dist/testing/index.js.map +1 -1
- package/package.json +4 -3
- package/src/api/create-dsl.test.ts +11 -0
- package/src/api/create-dsl.ts +278 -9
- package/src/api/domain-registry.ts +15 -10
- package/src/api/extensions.test.ts +322 -0
- package/src/core/tokenization/base-tokenizer.ts +78 -9
- package/src/core/tokenization/colon-qualifier.test.ts +129 -0
- package/src/core/tokenization/css-selector-extractor.test.ts +67 -0
- package/src/core/tokenization/extractors.ts +6 -0
- package/src/core/tokenization/token-utils.ts +18 -0
- package/src/generation/domain-renderer.test.ts +172 -0
- package/src/generation/pattern-generator.test.ts +102 -0
- package/src/generation/pattern-generator.ts +32 -20
- package/src/generation/renderer.test.ts +243 -4
- package/src/generation/renderer.ts +188 -45
- package/src/index.ts +9 -1
- package/src/interfaces/value-extractor.ts +50 -0
- package/src/ir/protocol-json.test.ts +21 -0
- package/src/ir/references.test.ts +5 -2
- package/src/prompts/prompt-generator.ts +4 -1
|
@@ -0,0 +1,322 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* DomainExtension: adding a command to a DSL from outside its package.
|
|
3
|
+
*
|
|
4
|
+
* Every domain used to render through a hardcoded `switch (node.action)` and
|
|
5
|
+
* generate through another, so a downstream consumer could not add a command
|
|
6
|
+
* without editing the package. This is the supported path: a schema plus one
|
|
7
|
+
* vocabulary entry per language.
|
|
8
|
+
*
|
|
9
|
+
* The command exercised here ("research") is the one lokascript-learn built by
|
|
10
|
+
* hand against 2.8.0 to prove the underlying pieces worked.
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import { describe, it, expect } from 'vitest';
|
|
14
|
+
import {
|
|
15
|
+
createMultilingualDSL,
|
|
16
|
+
createSimpleTokenizer,
|
|
17
|
+
defineCommand,
|
|
18
|
+
defineRole,
|
|
19
|
+
type DomainExtension,
|
|
20
|
+
type ExtractionResult,
|
|
21
|
+
type LanguageTokenizer,
|
|
22
|
+
type SemanticNode,
|
|
23
|
+
type ValueExtractor,
|
|
24
|
+
} from '../index';
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* Keeps `#id` / `.class` a single token, as every real domain tokenizer does.
|
|
28
|
+
* Without it the `#` splits off and an SOV source role followed by a particle
|
|
29
|
+
* never matches.
|
|
30
|
+
*/
|
|
31
|
+
class CSSSelectorExtractor implements ValueExtractor {
|
|
32
|
+
readonly name = 'css-selector';
|
|
33
|
+
|
|
34
|
+
canExtract(input: string, position: number): boolean {
|
|
35
|
+
const char = input[position];
|
|
36
|
+
if (char !== '#' && char !== '.') return false;
|
|
37
|
+
return /[a-zA-Z_-]/.test(input[position + 1] ?? '');
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
extract(input: string, position: number): ExtractionResult | null {
|
|
41
|
+
let end = position + 1;
|
|
42
|
+
while (end < input.length && /[a-zA-Z0-9_-]/.test(input[end])) end++;
|
|
43
|
+
if (end === position + 1) return null;
|
|
44
|
+
return { value: input.slice(position, end), length: end - position };
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
// =============================================================================
|
|
49
|
+
// A three-word-order toy DSL
|
|
50
|
+
// =============================================================================
|
|
51
|
+
|
|
52
|
+
const askSchema = defineCommand({
|
|
53
|
+
action: 'ask',
|
|
54
|
+
description: 'Ask a question',
|
|
55
|
+
category: 'llm',
|
|
56
|
+
primaryRole: 'patient',
|
|
57
|
+
roles: [
|
|
58
|
+
defineRole({
|
|
59
|
+
role: 'patient',
|
|
60
|
+
description: 'The question',
|
|
61
|
+
required: true,
|
|
62
|
+
expectedTypes: ['expression'],
|
|
63
|
+
}),
|
|
64
|
+
defineRole({
|
|
65
|
+
role: 'source',
|
|
66
|
+
description: 'Where to look',
|
|
67
|
+
required: true,
|
|
68
|
+
expectedTypes: ['expression'],
|
|
69
|
+
markerOverride: { en: 'from', ja: 'から', ar: 'من' },
|
|
70
|
+
}),
|
|
71
|
+
],
|
|
72
|
+
});
|
|
73
|
+
|
|
74
|
+
const researchSchema = defineCommand({
|
|
75
|
+
action: 'research',
|
|
76
|
+
description: 'Research a topic',
|
|
77
|
+
category: 'llm',
|
|
78
|
+
primaryRole: 'patient',
|
|
79
|
+
roles: [
|
|
80
|
+
defineRole({
|
|
81
|
+
role: 'patient',
|
|
82
|
+
description: 'The topic',
|
|
83
|
+
required: true,
|
|
84
|
+
expectedTypes: ['expression'],
|
|
85
|
+
}),
|
|
86
|
+
defineRole({
|
|
87
|
+
role: 'source',
|
|
88
|
+
description: 'Where to look',
|
|
89
|
+
required: true,
|
|
90
|
+
expectedTypes: ['expression'],
|
|
91
|
+
markerOverride: { en: 'from', ja: 'から', ar: 'من' },
|
|
92
|
+
}),
|
|
93
|
+
],
|
|
94
|
+
});
|
|
95
|
+
|
|
96
|
+
const PROFILES = {
|
|
97
|
+
en: {
|
|
98
|
+
code: 'en',
|
|
99
|
+
wordOrder: 'SVO' as const,
|
|
100
|
+
keywords: { ask: { primary: 'ask' } },
|
|
101
|
+
roleMarkers: {},
|
|
102
|
+
},
|
|
103
|
+
ja: {
|
|
104
|
+
code: 'ja',
|
|
105
|
+
wordOrder: 'SOV' as const,
|
|
106
|
+
keywords: { ask: { primary: '聞く' } },
|
|
107
|
+
roleMarkers: {},
|
|
108
|
+
},
|
|
109
|
+
ar: {
|
|
110
|
+
code: 'ar',
|
|
111
|
+
wordOrder: 'VSO' as const,
|
|
112
|
+
keywords: { ask: { primary: 'اسأل' } },
|
|
113
|
+
roleMarkers: {},
|
|
114
|
+
},
|
|
115
|
+
};
|
|
116
|
+
|
|
117
|
+
const research: DomainExtension = {
|
|
118
|
+
schema: researchSchema,
|
|
119
|
+
vocabulary: {
|
|
120
|
+
en: { keyword: { primary: 'research' } },
|
|
121
|
+
ja: { keyword: { primary: '調査' } },
|
|
122
|
+
ar: { keyword: { primary: 'ابحث' } },
|
|
123
|
+
},
|
|
124
|
+
};
|
|
125
|
+
|
|
126
|
+
function tokenizerFor(code: 'en' | 'ja' | 'ar'): LanguageTokenizer {
|
|
127
|
+
return createSimpleTokenizer({
|
|
128
|
+
language: code,
|
|
129
|
+
// Both the built-in and the extension vocabulary: a tokenizer is configured
|
|
130
|
+
// once, so it must already know the words an extension may introduce.
|
|
131
|
+
keywords: [
|
|
132
|
+
...Object.values(PROFILES[code].keywords).map(k => k.primary),
|
|
133
|
+
'research',
|
|
134
|
+
'調査',
|
|
135
|
+
'ابحث',
|
|
136
|
+
'from',
|
|
137
|
+
'から',
|
|
138
|
+
'من',
|
|
139
|
+
],
|
|
140
|
+
caseInsensitive: code === 'en',
|
|
141
|
+
customExtractors: [new CSSSelectorExtractor()],
|
|
142
|
+
});
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
function createToyDSL(options: { extensions?: readonly DomainExtension[] } = {}) {
|
|
146
|
+
return createMultilingualDSL({
|
|
147
|
+
name: 'Toy',
|
|
148
|
+
schemas: [askSchema],
|
|
149
|
+
languages: (['en', 'ja', 'ar'] as const).map(code => ({
|
|
150
|
+
code,
|
|
151
|
+
name: code,
|
|
152
|
+
nativeName: code,
|
|
153
|
+
tokenizer: tokenizerFor(code),
|
|
154
|
+
patternProfile: PROFILES[code],
|
|
155
|
+
})),
|
|
156
|
+
codeGenerator: {
|
|
157
|
+
generate: (node: SemanticNode) => `BASE:${node.action}`,
|
|
158
|
+
},
|
|
159
|
+
...(options.extensions && { extensions: options.extensions }),
|
|
160
|
+
});
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
function makeNode(action: string, roles: Record<string, string>): SemanticNode {
|
|
164
|
+
const rolesMap = new Map<string, { type: 'expression'; raw: string }>();
|
|
165
|
+
for (const [k, v] of Object.entries(roles)) rolesMap.set(k, { type: 'expression', raw: v });
|
|
166
|
+
return { kind: 'command', action, roles: rolesMap };
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
// =============================================================================
|
|
170
|
+
// Tests
|
|
171
|
+
// =============================================================================
|
|
172
|
+
|
|
173
|
+
describe('DomainExtension', () => {
|
|
174
|
+
describe('parsing', () => {
|
|
175
|
+
it('parses the extension command in an SVO language', () => {
|
|
176
|
+
const dsl = createToyDSL({ extensions: [research] });
|
|
177
|
+
const node = dsl.parse('research "climate" from #wiki', 'en');
|
|
178
|
+
expect(node.action).toBe('research');
|
|
179
|
+
expect(node.roles.get('patient')).toBeDefined();
|
|
180
|
+
expect(node.roles.get('source')).toBeDefined();
|
|
181
|
+
});
|
|
182
|
+
|
|
183
|
+
it('parses the extension command in an SOV language', () => {
|
|
184
|
+
const dsl = createToyDSL({ extensions: [research] });
|
|
185
|
+
const node = dsl.parse('"climate" #wiki から 調査', 'ja');
|
|
186
|
+
expect(node.action).toBe('research');
|
|
187
|
+
});
|
|
188
|
+
|
|
189
|
+
it('parses the extension command in a VSO language', () => {
|
|
190
|
+
const dsl = createToyDSL({ extensions: [research] });
|
|
191
|
+
const node = dsl.parse('ابحث "climate" من #wiki', 'ar');
|
|
192
|
+
expect(node.action).toBe('research');
|
|
193
|
+
});
|
|
194
|
+
|
|
195
|
+
it('does not parse the extension command without the extension', () => {
|
|
196
|
+
const dsl = createToyDSL();
|
|
197
|
+
expect(() => dsl.parse('research "climate" from #wiki', 'en')).toThrow(/No pattern matched/);
|
|
198
|
+
});
|
|
199
|
+
|
|
200
|
+
it('reports the extension action as supported by explicit syntax', () => {
|
|
201
|
+
const dsl = createToyDSL({ extensions: [research] });
|
|
202
|
+
const node = dsl.parse('[research patient:climate source:#wiki]', 'en');
|
|
203
|
+
expect(node.action).toBe('research');
|
|
204
|
+
});
|
|
205
|
+
});
|
|
206
|
+
|
|
207
|
+
describe('rendering', () => {
|
|
208
|
+
const dsl = createToyDSL({ extensions: [research] });
|
|
209
|
+
const node = makeNode('research', { patient: '"climate"', source: '#wiki' });
|
|
210
|
+
|
|
211
|
+
it('renders SVO from the schema alone', () => {
|
|
212
|
+
expect(dsl.render?.(node, 'en')).toBe('research "climate" from #wiki');
|
|
213
|
+
});
|
|
214
|
+
|
|
215
|
+
it('renders verb-final for SOV', () => {
|
|
216
|
+
expect(dsl.render?.(node, 'ja')).toBe('"climate" #wiki から 調査');
|
|
217
|
+
});
|
|
218
|
+
|
|
219
|
+
it('renders verb-initial for VSO', () => {
|
|
220
|
+
expect(dsl.render?.(node, 'ar')).toBe('ابحث "climate" من #wiki');
|
|
221
|
+
});
|
|
222
|
+
|
|
223
|
+
it('prefers an extension-supplied renderer', () => {
|
|
224
|
+
const custom = createToyDSL({
|
|
225
|
+
extensions: [{ ...research, render: () => 'CUSTOM' }],
|
|
226
|
+
});
|
|
227
|
+
expect(custom.render?.(node, 'en')).toBe('CUSTOM');
|
|
228
|
+
});
|
|
229
|
+
|
|
230
|
+
it('returns null for an action with neither renderer nor schema', () => {
|
|
231
|
+
expect(dsl.render?.(makeNode('nonsense', {}), 'en')).toBeNull();
|
|
232
|
+
});
|
|
233
|
+
|
|
234
|
+
it('prefers the domain renderer over the schema fallback for built-ins', () => {
|
|
235
|
+
const withDomainRenderer = createMultilingualDSL({
|
|
236
|
+
name: 'Toy',
|
|
237
|
+
schemas: [askSchema],
|
|
238
|
+
languages: [
|
|
239
|
+
{
|
|
240
|
+
code: 'en',
|
|
241
|
+
name: 'en',
|
|
242
|
+
nativeName: 'en',
|
|
243
|
+
tokenizer: tokenizerFor('en'),
|
|
244
|
+
patternProfile: PROFILES.en,
|
|
245
|
+
},
|
|
246
|
+
],
|
|
247
|
+
renderer: node => (node.action === 'ask' ? 'DOMAIN RENDERED' : null),
|
|
248
|
+
extensions: [{ ...research, vocabulary: { en: research.vocabulary.en } }],
|
|
249
|
+
});
|
|
250
|
+
|
|
251
|
+
expect(withDomainRenderer.render?.(makeNode('ask', { patient: 'q' }), 'en')).toBe(
|
|
252
|
+
'DOMAIN RENDERED'
|
|
253
|
+
);
|
|
254
|
+
// …while the extension still falls through to the schema renderer
|
|
255
|
+
expect(withDomainRenderer.render?.(node, 'en')).toBe('research "climate" from #wiki');
|
|
256
|
+
});
|
|
257
|
+
});
|
|
258
|
+
|
|
259
|
+
describe('compilation', () => {
|
|
260
|
+
it('uses the extension generator for the extension action', () => {
|
|
261
|
+
const dsl = createToyDSL({
|
|
262
|
+
extensions: [{ ...research, generate: node => `EXT:${node.action}` }],
|
|
263
|
+
});
|
|
264
|
+
const result = dsl.compile('research "climate" from #wiki', 'en');
|
|
265
|
+
expect(result.ok).toBe(true);
|
|
266
|
+
expect(result.code).toBe('EXT:research');
|
|
267
|
+
});
|
|
268
|
+
|
|
269
|
+
it('leaves the base generator handling built-in actions', () => {
|
|
270
|
+
const dsl = createToyDSL({
|
|
271
|
+
extensions: [{ ...research, generate: node => `EXT:${node.action}` }],
|
|
272
|
+
});
|
|
273
|
+
const result = dsl.compile('ask "q" from #wiki', 'en');
|
|
274
|
+
expect(result.ok).toBe(true);
|
|
275
|
+
expect(result.code).toBe('BASE:ask');
|
|
276
|
+
});
|
|
277
|
+
|
|
278
|
+
it('falls back to the base generator when the extension supplies none', () => {
|
|
279
|
+
const dsl = createToyDSL({ extensions: [research] });
|
|
280
|
+
const result = dsl.compile('research "climate" from #wiki', 'en');
|
|
281
|
+
expect(result.ok).toBe(true);
|
|
282
|
+
expect(result.code).toBe('BASE:research');
|
|
283
|
+
});
|
|
284
|
+
});
|
|
285
|
+
|
|
286
|
+
describe('built-in commands are unaffected', () => {
|
|
287
|
+
it('parses and compiles identically with and without extensions', () => {
|
|
288
|
+
const plain = createToyDSL();
|
|
289
|
+
const extended = createToyDSL({ extensions: [research] });
|
|
290
|
+
|
|
291
|
+
for (const [input, language] of [
|
|
292
|
+
['ask "q" from #wiki', 'en'],
|
|
293
|
+
['"q" #wiki から 聞く', 'ja'],
|
|
294
|
+
['اسأل "q" من #wiki', 'ar'],
|
|
295
|
+
] as const) {
|
|
296
|
+
expect(extended.parse(input, language).action).toBe(plain.parse(input, language).action);
|
|
297
|
+
expect(extended.compile(input, language).code).toBe(plain.compile(input, language).code);
|
|
298
|
+
}
|
|
299
|
+
});
|
|
300
|
+
});
|
|
301
|
+
|
|
302
|
+
describe('configuration errors', () => {
|
|
303
|
+
it('rejects an extension whose action collides with a built-in', () => {
|
|
304
|
+
const collision: DomainExtension = {
|
|
305
|
+
schema: defineCommand({
|
|
306
|
+
action: 'ask',
|
|
307
|
+
roles: [defineRole({ role: 'patient', required: true, expectedTypes: ['expression'] })],
|
|
308
|
+
}),
|
|
309
|
+
vocabulary: { en: { keyword: { primary: 'inquire' } } },
|
|
310
|
+
};
|
|
311
|
+
expect(() => createToyDSL({ extensions: [collision] })).toThrow(/collides/);
|
|
312
|
+
});
|
|
313
|
+
|
|
314
|
+
it('rejects vocabulary for a language the DSL does not configure', () => {
|
|
315
|
+
const unknownLanguage: DomainExtension = {
|
|
316
|
+
...research,
|
|
317
|
+
vocabulary: { ...research.vocabulary, xx: { keyword: { primary: 'x' } } },
|
|
318
|
+
};
|
|
319
|
+
expect(() => createToyDSL({ extensions: [unknownLanguage] })).toThrow(/not configured/);
|
|
320
|
+
});
|
|
321
|
+
});
|
|
322
|
+
});
|
|
@@ -20,6 +20,7 @@ import {
|
|
|
20
20
|
isWhitespace,
|
|
21
21
|
isDigit,
|
|
22
22
|
isAsciiIdentifierChar,
|
|
23
|
+
stripOptionalDiacritics,
|
|
23
24
|
TokenStreamImpl,
|
|
24
25
|
type TimeUnitMapping,
|
|
25
26
|
type CreateTokenOptions,
|
|
@@ -347,7 +348,61 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
|
|
|
347
348
|
}
|
|
348
349
|
}
|
|
349
350
|
|
|
350
|
-
return new TokenStreamImpl(tokens, this.language);
|
|
351
|
+
return new TokenStreamImpl(this.mergeColonQualifiedNames(tokens), this.language);
|
|
352
|
+
}
|
|
353
|
+
|
|
354
|
+
/**
|
|
355
|
+
* ASCII word of the shape the English word-walker produces. Excludes `:`, so a
|
|
356
|
+
* token that already carries a qualifier never merges again — `a:b:c` yields
|
|
357
|
+
* `a:b` + `:c`, byte-matching the English extractor's single-segment merge.
|
|
358
|
+
*/
|
|
359
|
+
private static readonly ASCII_WORD = /^[A-Za-z_][A-Za-z0-9_]*$/;
|
|
360
|
+
|
|
361
|
+
/** `:name` — only a variable-ref-style extractor ever emits this token shape. */
|
|
362
|
+
private static readonly COLON_QUALIFIER = /^:[A-Za-z_][A-Za-z0-9_]*$/;
|
|
363
|
+
|
|
364
|
+
/**
|
|
365
|
+
* Fuse `name` + `:qualifier` into ONE identifier (`draggable:start`).
|
|
366
|
+
*
|
|
367
|
+
* `:name` is hyperscript's local-variable sigil, but a colon IMMEDIATELY
|
|
368
|
+
* preceded by an identifier is a qualifier (custom event namespace), not a
|
|
369
|
+
* sigil. The English tokenizer already merges these inside
|
|
370
|
+
* EnglishKeywordExtractor; this post-pass gives the other 23 languages the
|
|
371
|
+
* same stream. Strict position adjacency is the discriminator: whitespace
|
|
372
|
+
* between the tokens (`trigger :start`) breaks `end === start`, so a spaced
|
|
373
|
+
* local-variable reference survives untouched.
|
|
374
|
+
*
|
|
375
|
+
* Self-gating for non-hyperscript tokenizers (domain DSLs): their extractor
|
|
376
|
+
* sets tokenize `:` as bare punctuation (length 1), which never matches
|
|
377
|
+
* COLON_QUALIFIER, so this pass is a no-op for them.
|
|
378
|
+
*/
|
|
379
|
+
protected mergeColonQualifiedNames(tokens: LanguageToken[]): LanguageToken[] {
|
|
380
|
+
const out: LanguageToken[] = [];
|
|
381
|
+
for (const tok of tokens) {
|
|
382
|
+
const prev = out[out.length - 1];
|
|
383
|
+
// No kind gate: the English extractor merges before classification, so a
|
|
384
|
+
// word some language classifies as particle/keyword (es `a`, tr `i`)
|
|
385
|
+
// must fuse the same way. ASCII_WORD already excludes every non-word
|
|
386
|
+
// kind structurally (selectors, urls, numbers, strings, operators).
|
|
387
|
+
if (
|
|
388
|
+
prev &&
|
|
389
|
+
BaseTokenizer.ASCII_WORD.test(prev.value) &&
|
|
390
|
+
BaseTokenizer.COLON_QUALIFIER.test(tok.value) &&
|
|
391
|
+
prev.position.end === tok.position.start
|
|
392
|
+
) {
|
|
393
|
+
const merged = prev.value + tok.value;
|
|
394
|
+
// Re-classify and drop normalized/stem/metadata — the merged word is no
|
|
395
|
+
// longer the keyword the pieces may have been (matches the en shape).
|
|
396
|
+
out[out.length - 1] = createToken(
|
|
397
|
+
merged,
|
|
398
|
+
this.classifyToken(merged),
|
|
399
|
+
createPosition(prev.position.start, tok.position.end)
|
|
400
|
+
);
|
|
401
|
+
continue;
|
|
402
|
+
}
|
|
403
|
+
out.push(tok);
|
|
404
|
+
}
|
|
405
|
+
return out;
|
|
351
406
|
}
|
|
352
407
|
|
|
353
408
|
/**
|
|
@@ -544,9 +599,7 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
|
|
|
544
599
|
* @returns Word without diacritics
|
|
545
600
|
*/
|
|
546
601
|
protected removeDiacritics(word: string): string {
|
|
547
|
-
|
|
548
|
-
// U+0670 (superscript alif)
|
|
549
|
-
return word.replace(/[\u064B-\u0652\u0670]/g, '');
|
|
602
|
+
return stripOptionalDiacritics(word);
|
|
550
603
|
}
|
|
551
604
|
|
|
552
605
|
/**
|
|
@@ -650,25 +703,41 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
|
|
|
650
703
|
}
|
|
651
704
|
|
|
652
705
|
/**
|
|
653
|
-
* Look up a keyword by native word (case-insensitive).
|
|
706
|
+
* Look up a keyword by native word (case-insensitive, diacritic-insensitive).
|
|
654
707
|
* O(1) lookup using the keyword map.
|
|
655
708
|
*
|
|
709
|
+
* The map is INDEXED both with and without diacritics (see
|
|
710
|
+
* `initializeKeywordsFromProfile`), so a stripped QUERY is the other half of
|
|
711
|
+
* that: it lets a surface form carrying harakat the profile does not happen to
|
|
712
|
+
* spell still find its entry. Only consulted after the exact lookup misses, so
|
|
713
|
+
* every previously-matching word resolves byte-identically.
|
|
714
|
+
*
|
|
715
|
+
* Half-implementing this — indexing stripped but querying exact — is what made
|
|
716
|
+
* diacritized `بَدِّل` (toggle) tokenize as `kind=particle normalized=with`:
|
|
717
|
+
* `isKeyword` returned false, so the guard in `ArabicProcliticExtractor` that
|
|
718
|
+
* exists to prevent exactly that handed the word on, and the single-char `ب`
|
|
719
|
+
* bi- proclitic claimed it. A wrong CONCEPT, not a failed parse.
|
|
720
|
+
*
|
|
656
721
|
* @param native - Native word to look up
|
|
657
722
|
* @returns KeywordEntry if found, undefined otherwise
|
|
658
723
|
*/
|
|
659
724
|
protected lookupKeyword(native: string): KeywordEntry | undefined {
|
|
660
|
-
|
|
725
|
+
const exact = this.profileKeywordMap.get(native.toLowerCase());
|
|
726
|
+
if (exact) return exact;
|
|
727
|
+
const stripped = this.removeDiacritics(native);
|
|
728
|
+
if (stripped === native) return undefined;
|
|
729
|
+
return this.profileKeywordMap.get(stripped.toLowerCase());
|
|
661
730
|
}
|
|
662
731
|
|
|
663
732
|
/**
|
|
664
|
-
* Check if a word is a known keyword (case-insensitive).
|
|
665
|
-
* O(1) lookup using the keyword map.
|
|
733
|
+
* Check if a word is a known keyword (case-insensitive, diacritic-insensitive).
|
|
734
|
+
* O(1) lookup using the keyword map. See {@link lookupKeyword}.
|
|
666
735
|
*
|
|
667
736
|
* @param native - Native word to check
|
|
668
737
|
* @returns true if the word is a keyword
|
|
669
738
|
*/
|
|
670
739
|
protected isKeyword(native: string): boolean {
|
|
671
|
-
return this.
|
|
740
|
+
return this.lookupKeyword(native) !== undefined;
|
|
672
741
|
}
|
|
673
742
|
|
|
674
743
|
/**
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Contract tests for BaseTokenizer.mergeColonQualifiedNames().
|
|
3
|
+
*
|
|
4
|
+
* `:name` is hyperscript's local-variable sigil, but a colon immediately
|
|
5
|
+
* preceded by an identifier is a qualifier (custom event namespace:
|
|
6
|
+
* `draggable:start`). The English tokenizer merges these inside its keyword
|
|
7
|
+
* extractor; the post-pass gives every other language the same stream. The
|
|
8
|
+
* merge must fire only on strict adjacency and only on `:name`-shaped tokens —
|
|
9
|
+
* domain DSL tokenizers (SQL, BDD, …) emit `:` as bare punctuation, which the
|
|
10
|
+
* pass must never touch.
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import { describe, it, expect } from 'vitest';
|
|
14
|
+
import { BaseTokenizer, createSimpleTokenizer } from './base-tokenizer';
|
|
15
|
+
import type { TokenKind } from '../types';
|
|
16
|
+
import type { ValueExtractor, ExtractionResult } from '../../interfaces/value-extractor';
|
|
17
|
+
|
|
18
|
+
/** Minimal `:name`/`$name`/`^name` extractor (shape of semantic's VariableRefExtractor). */
|
|
19
|
+
class SigilRefExtractor implements ValueExtractor {
|
|
20
|
+
readonly name = 'sigil-ref';
|
|
21
|
+
|
|
22
|
+
canExtract(input: string, position: number): boolean {
|
|
23
|
+
const ch = input[position];
|
|
24
|
+
return (
|
|
25
|
+
(ch === ':' || ch === '$' || ch === '^') &&
|
|
26
|
+
position + 1 < input.length &&
|
|
27
|
+
/[a-zA-Z_]/.test(input[position + 1])
|
|
28
|
+
);
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
extract(input: string, position: number): ExtractionResult | null {
|
|
32
|
+
if (!this.canExtract(input, position)) return null;
|
|
33
|
+
let length = 1;
|
|
34
|
+
while (position + length < input.length && /[a-zA-Z0-9_]/.test(input[position + length])) {
|
|
35
|
+
length++;
|
|
36
|
+
}
|
|
37
|
+
return { value: input.substring(position, position + length), length };
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/** Minimal ASCII word extractor (shape of the per-language word walkers). */
|
|
42
|
+
class WordExtractor implements ValueExtractor {
|
|
43
|
+
readonly name = 'word';
|
|
44
|
+
|
|
45
|
+
canExtract(input: string, position: number): boolean {
|
|
46
|
+
return /[a-zA-Z_]/.test(input[position]);
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
extract(input: string, position: number): ExtractionResult | null {
|
|
50
|
+
let length = 0;
|
|
51
|
+
while (position + length < input.length && /[a-zA-Z0-9_]/.test(input[position + length])) {
|
|
52
|
+
length++;
|
|
53
|
+
}
|
|
54
|
+
return length > 0 ? { value: input.substring(position, position + length), length } : null;
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
const KEYWORDS = new Set(['trigger', 'click']);
|
|
59
|
+
|
|
60
|
+
class ProbeTokenizer extends BaseTokenizer {
|
|
61
|
+
readonly language = 'xx';
|
|
62
|
+
readonly direction = 'ltr' as const;
|
|
63
|
+
|
|
64
|
+
constructor() {
|
|
65
|
+
super();
|
|
66
|
+
this.registerExtractors([new SigilRefExtractor(), new WordExtractor()]);
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
classifyToken(token: string): TokenKind {
|
|
70
|
+
return KEYWORDS.has(token) ? 'keyword' : 'identifier';
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
function values(
|
|
75
|
+
tokenizer: { tokenize(input: string): { tokens: readonly { value: string }[] } },
|
|
76
|
+
input: string
|
|
77
|
+
): string[] {
|
|
78
|
+
return tokenizer.tokenize(input).tokens.map(t => t.value);
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
describe('mergeColonQualifiedNames', () => {
|
|
82
|
+
const t = new ProbeTokenizer();
|
|
83
|
+
|
|
84
|
+
it('fuses identifier + :qualifier into one token', () => {
|
|
85
|
+
expect(values(t, 'trigger draggable:start')).toEqual(['trigger', 'draggable:start']);
|
|
86
|
+
});
|
|
87
|
+
|
|
88
|
+
it('re-classifies the merged token and spans both positions', () => {
|
|
89
|
+
const tokens = t.tokenize('trigger draggable:start').tokens;
|
|
90
|
+
const merged = tokens[1];
|
|
91
|
+
expect(merged.kind).toBe('identifier');
|
|
92
|
+
expect(merged.position.start).toBe('trigger '.length);
|
|
93
|
+
expect(merged.position.end).toBe('trigger draggable:start'.length);
|
|
94
|
+
});
|
|
95
|
+
|
|
96
|
+
it('fuses when the pre-colon word is a keyword (en parity: merge before lookup)', () => {
|
|
97
|
+
const tokens = t.tokenize('click:foo').tokens;
|
|
98
|
+
expect(tokens.map(x => x.value)).toEqual(['click:foo']);
|
|
99
|
+
expect(tokens[0].kind).toBe('identifier');
|
|
100
|
+
});
|
|
101
|
+
|
|
102
|
+
it('does not fuse across whitespace — spaced :name stays a local-variable ref', () => {
|
|
103
|
+
expect(values(t, 'trigger :start')).toEqual(['trigger', ':start']);
|
|
104
|
+
});
|
|
105
|
+
|
|
106
|
+
it('leaves a leading bare sigil untouched', () => {
|
|
107
|
+
expect(values(t, ':start')).toEqual([':start']);
|
|
108
|
+
});
|
|
109
|
+
|
|
110
|
+
it('merges a single segment only — a:b:c matches the English extractor', () => {
|
|
111
|
+
expect(values(t, 'a:b:c')).toEqual(['a:b', ':c']);
|
|
112
|
+
});
|
|
113
|
+
|
|
114
|
+
it('never touches $ and ^ sigils', () => {
|
|
115
|
+
expect(values(t, 'put $foo into ^bar')).toEqual(['put', '$foo', 'into', '^bar']);
|
|
116
|
+
});
|
|
117
|
+
|
|
118
|
+
it('is a no-op for domain-DSL streams where : is bare punctuation', () => {
|
|
119
|
+
const sql = createSimpleTokenizer({
|
|
120
|
+
language: 'en',
|
|
121
|
+
keywords: ['select', 'from', 'where'],
|
|
122
|
+
includeOperators: true,
|
|
123
|
+
});
|
|
124
|
+
// Named params never fuse: the colon tokenizes as length-1 punctuation,
|
|
125
|
+
// which can never match the :name qualifier shape.
|
|
126
|
+
expect(values(sql, 'WHERE x = :param')).toEqual(['WHERE', 'x', '=', ':', 'param']);
|
|
127
|
+
expect(values(sql, 'x=:param')).toEqual(['x', '=', ':', 'param']);
|
|
128
|
+
});
|
|
129
|
+
});
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
import { describe, it, expect } from 'vitest';
|
|
2
|
+
import { CssSelectorExtractor } from '../../interfaces/value-extractor';
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* `getDefaultExtractors()` has no CSS-selector extractor, so a DSL that does not
|
|
6
|
+
* register one splits the sigil off as its own token and the role capture keeps
|
|
7
|
+
* only that sigil — `add .active to #button` parsed with patient `"."` and
|
|
8
|
+
* destination `"#"` in domain-learn, -todo, -sql and -jsx, silently, in every
|
|
9
|
+
* language. Five other domains each carried a private copy of this class; this
|
|
10
|
+
* is the shared one.
|
|
11
|
+
*/
|
|
12
|
+
describe('CssSelectorExtractor', () => {
|
|
13
|
+
const ex = new CssSelectorExtractor();
|
|
14
|
+
|
|
15
|
+
const extract = (input: string, pos = 0) =>
|
|
16
|
+
ex.canExtract(input, pos) ? ex.extract(input, pos) : null;
|
|
17
|
+
|
|
18
|
+
it.each([
|
|
19
|
+
['.active', '.active'],
|
|
20
|
+
['#button', '#button'],
|
|
21
|
+
['.btn-primary', '.btn-primary'],
|
|
22
|
+
['._private', '._private'],
|
|
23
|
+
['#a1', '#a1'],
|
|
24
|
+
['.-leading-hyphen', '.-leading-hyphen'],
|
|
25
|
+
])('extracts %s whole', (input, expected) => {
|
|
26
|
+
expect(extract(input)).toEqual({ value: expected, length: expected.length });
|
|
27
|
+
});
|
|
28
|
+
|
|
29
|
+
it('stops at the first character that cannot be in a selector', () => {
|
|
30
|
+
expect(extract('.active to #button')).toEqual({ value: '.active', length: 7 });
|
|
31
|
+
});
|
|
32
|
+
|
|
33
|
+
it('keeps diacritics, so Latin-script class names survive', () => {
|
|
34
|
+
expect(extract('.año')).toEqual({ value: '.año', length: 4 });
|
|
35
|
+
expect(extract('#botón')).toEqual({ value: '#botón', length: 6 });
|
|
36
|
+
});
|
|
37
|
+
|
|
38
|
+
// The SOV languages write their particles flush against the value, so a
|
|
39
|
+
// Unicode-wide body turns `#buttonに` into a single token and carries the role
|
|
40
|
+
// marker inside the value — which is worse than truncating, because the marker
|
|
41
|
+
// is then missing from where the pattern expects it.
|
|
42
|
+
it.each([
|
|
43
|
+
['#buttonに', '#button'],
|
|
44
|
+
['.activeを', '.active'],
|
|
45
|
+
['#button에', '#button'],
|
|
46
|
+
['.active를', '.active'],
|
|
47
|
+
['#button的', '#button'],
|
|
48
|
+
])('stops at the particle in %s', (input, expected) => {
|
|
49
|
+
expect(extract(input)).toEqual({ value: expected, length: expected.length });
|
|
50
|
+
});
|
|
51
|
+
|
|
52
|
+
it.each([
|
|
53
|
+
['.', 'a bare sigil'],
|
|
54
|
+
['#', 'a bare sigil'],
|
|
55
|
+
['. active', 'a sigil followed by a space'],
|
|
56
|
+
['.1col', 'a digit — CSS identifiers cannot start with one'],
|
|
57
|
+
['button', 'no sigil at all'],
|
|
58
|
+
])('declines %s (%s)', input => {
|
|
59
|
+
expect(ex.canExtract(input, 0)).toBe(false);
|
|
60
|
+
});
|
|
61
|
+
|
|
62
|
+
it('declines a CJK-only class name rather than claiming the sigil alone', () => {
|
|
63
|
+
// `.追加` would otherwise extract as a bare `.`; leaving it unclaimed lets
|
|
64
|
+
// the keyword/identifier extractors see the verb.
|
|
65
|
+
expect(ex.canExtract('.追加', 0)).toBe(false);
|
|
66
|
+
});
|
|
67
|
+
});
|
|
@@ -34,6 +34,12 @@ import {
|
|
|
34
34
|
* Method call handling:
|
|
35
35
|
* - #dialog.showModal() → stops after #dialog (method call, not compound selector)
|
|
36
36
|
* - #box.active → compound selector (no parens)
|
|
37
|
+
*
|
|
38
|
+
* NOTE: intentionally diverges from the semantic package's copy
|
|
39
|
+
* (packages/semantic/src/tokenizers/extractors/css-selector.ts), which also
|
|
40
|
+
* consumes pseudo-class/pseudo-element segments (#x:hover, .a:not(.b)). This
|
|
41
|
+
* legacy version is only used by BaseTokenizer.trySelector (no semantic call
|
|
42
|
+
* sites) and stays as-is.
|
|
37
43
|
*/
|
|
38
44
|
export function extractCssSelector(input: string, startPos: number): string | null {
|
|
39
45
|
if (startPos >= input.length) return null;
|
|
@@ -237,6 +237,24 @@ export function isDigit(char: string): boolean {
|
|
|
237
237
|
return /\d/.test(char);
|
|
238
238
|
}
|
|
239
239
|
|
|
240
|
+
/**
|
|
241
|
+
* Strip diacritical marks that are OPTIONAL in their script — currently Arabic
|
|
242
|
+
* harakat (U+064B–U+0652: fatha, kasra, damma, sukun, shadda…) and the
|
|
243
|
+
* superscript alif (U+0670).
|
|
244
|
+
*
|
|
245
|
+
* `بدّل` and `بَدِّل` are the same word; Arabic prose writes either. So any place
|
|
246
|
+
* that compares an Arabic surface form against a declared keyword has to compare
|
|
247
|
+
* stripped, or the same word fails to match itself.
|
|
248
|
+
*
|
|
249
|
+
* Deliberately Arabic-only. Latin diacritics are NOT optional — `obtén`, `récupère`
|
|
250
|
+
* and `vá` differ from their unaccented spellings in meaning or validity — and
|
|
251
|
+
* Hebrew niqqud (U+05B0–U+05BC) is out of range too, so this is inert for every
|
|
252
|
+
* other script.
|
|
253
|
+
*/
|
|
254
|
+
export function stripOptionalDiacritics(word: string): string {
|
|
255
|
+
return word.replace(/[ً-ْٰ]/g, '');
|
|
256
|
+
}
|
|
257
|
+
|
|
240
258
|
/**
|
|
241
259
|
* Check if a character is an ASCII letter.
|
|
242
260
|
*/
|