@lokascript/framework 2.7.1 → 2.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/api/index.js +6 -2
- package/dist/api/index.js.map +1 -1
- package/dist/core/index.js +41 -1
- package/dist/core/index.js.map +1 -1
- package/dist/core/tokenization/base-tokenizer.d.ts +24 -0
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
- package/dist/core/tokenization/extractors.d.ts +6 -0
- package/dist/core/tokenization/extractors.d.ts.map +1 -1
- package/dist/core/tokenization/index.js +41 -1
- package/dist/core/tokenization/index.js.map +1 -1
- package/dist/generation/index.js +2 -1
- package/dist/generation/index.js.map +1 -1
- package/dist/generation/pattern-generator.d.ts.map +1 -1
- package/dist/index.cjs +47 -3
- package/dist/index.cjs.map +1 -1
- package/dist/index.js +47 -3
- package/dist/index.js.map +1 -1
- package/dist/multilingual/index.js +41 -1
- package/dist/multilingual/index.js.map +1 -1
- package/dist/testing/index.js +4 -12
- package/dist/testing/index.js.map +1 -1
- package/package.json +2 -2
- package/src/core/tokenization/base-tokenizer.ts +55 -1
- package/src/core/tokenization/colon-qualifier.test.ts +129 -0
- package/src/core/tokenization/extractors.ts +6 -0
- package/src/generation/pattern-generator.ts +5 -1
- package/src/ir/protocol-json.test.ts +21 -0
- package/src/ir/references.test.ts +5 -2
- package/src/prompts/prompt-generator.ts +4 -1
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@lokascript/framework",
|
|
3
|
-
"version": "2.
|
|
3
|
+
"version": "2.8.0",
|
|
4
4
|
"description": "Generic framework for building multilingual DSLs with semantic parsing and grammar transformation",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "dist/index.cjs",
|
|
@@ -96,7 +96,7 @@
|
|
|
96
96
|
"author": "LokaScript Contributors",
|
|
97
97
|
"license": "MIT",
|
|
98
98
|
"dependencies": {
|
|
99
|
-
"@lokascript/intent": "^2.
|
|
99
|
+
"@lokascript/intent": "^2.8.0"
|
|
100
100
|
},
|
|
101
101
|
"devDependencies": {
|
|
102
102
|
"@types/node": "^20.0.0",
|
|
@@ -347,7 +347,61 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
|
|
|
347
347
|
}
|
|
348
348
|
}
|
|
349
349
|
|
|
350
|
-
return new TokenStreamImpl(tokens, this.language);
|
|
350
|
+
return new TokenStreamImpl(this.mergeColonQualifiedNames(tokens), this.language);
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
/**
|
|
354
|
+
* ASCII word of the shape the English word-walker produces. Excludes `:`, so a
|
|
355
|
+
* token that already carries a qualifier never merges again — `a:b:c` yields
|
|
356
|
+
* `a:b` + `:c`, byte-matching the English extractor's single-segment merge.
|
|
357
|
+
*/
|
|
358
|
+
private static readonly ASCII_WORD = /^[A-Za-z_][A-Za-z0-9_]*$/;
|
|
359
|
+
|
|
360
|
+
/** `:name` — only a variable-ref-style extractor ever emits this token shape. */
|
|
361
|
+
private static readonly COLON_QUALIFIER = /^:[A-Za-z_][A-Za-z0-9_]*$/;
|
|
362
|
+
|
|
363
|
+
/**
|
|
364
|
+
* Fuse `name` + `:qualifier` into ONE identifier (`draggable:start`).
|
|
365
|
+
*
|
|
366
|
+
* `:name` is hyperscript's local-variable sigil, but a colon IMMEDIATELY
|
|
367
|
+
* preceded by an identifier is a qualifier (custom event namespace), not a
|
|
368
|
+
* sigil. The English tokenizer already merges these inside
|
|
369
|
+
* EnglishKeywordExtractor; this post-pass gives the other 23 languages the
|
|
370
|
+
* same stream. Strict position adjacency is the discriminator: whitespace
|
|
371
|
+
* between the tokens (`trigger :start`) breaks `end === start`, so a spaced
|
|
372
|
+
* local-variable reference survives untouched.
|
|
373
|
+
*
|
|
374
|
+
* Self-gating for non-hyperscript tokenizers (domain DSLs): their extractor
|
|
375
|
+
* sets tokenize `:` as bare punctuation (length 1), which never matches
|
|
376
|
+
* COLON_QUALIFIER, so this pass is a no-op for them.
|
|
377
|
+
*/
|
|
378
|
+
protected mergeColonQualifiedNames(tokens: LanguageToken[]): LanguageToken[] {
|
|
379
|
+
const out: LanguageToken[] = [];
|
|
380
|
+
for (const tok of tokens) {
|
|
381
|
+
const prev = out[out.length - 1];
|
|
382
|
+
// No kind gate: the English extractor merges before classification, so a
|
|
383
|
+
// word some language classifies as particle/keyword (es `a`, tr `i`)
|
|
384
|
+
// must fuse the same way. ASCII_WORD already excludes every non-word
|
|
385
|
+
// kind structurally (selectors, urls, numbers, strings, operators).
|
|
386
|
+
if (
|
|
387
|
+
prev &&
|
|
388
|
+
BaseTokenizer.ASCII_WORD.test(prev.value) &&
|
|
389
|
+
BaseTokenizer.COLON_QUALIFIER.test(tok.value) &&
|
|
390
|
+
prev.position.end === tok.position.start
|
|
391
|
+
) {
|
|
392
|
+
const merged = prev.value + tok.value;
|
|
393
|
+
// Re-classify and drop normalized/stem/metadata — the merged word is no
|
|
394
|
+
// longer the keyword the pieces may have been (matches the en shape).
|
|
395
|
+
out[out.length - 1] = createToken(
|
|
396
|
+
merged,
|
|
397
|
+
this.classifyToken(merged),
|
|
398
|
+
createPosition(prev.position.start, tok.position.end)
|
|
399
|
+
);
|
|
400
|
+
continue;
|
|
401
|
+
}
|
|
402
|
+
out.push(tok);
|
|
403
|
+
}
|
|
404
|
+
return out;
|
|
351
405
|
}
|
|
352
406
|
|
|
353
407
|
/**
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Contract tests for BaseTokenizer.mergeColonQualifiedNames().
|
|
3
|
+
*
|
|
4
|
+
* `:name` is hyperscript's local-variable sigil, but a colon immediately
|
|
5
|
+
* preceded by an identifier is a qualifier (custom event namespace:
|
|
6
|
+
* `draggable:start`). The English tokenizer merges these inside its keyword
|
|
7
|
+
* extractor; the post-pass gives every other language the same stream. The
|
|
8
|
+
* merge must fire only on strict adjacency and only on `:name`-shaped tokens —
|
|
9
|
+
* domain DSL tokenizers (SQL, BDD, …) emit `:` as bare punctuation, which the
|
|
10
|
+
* pass must never touch.
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import { describe, it, expect } from 'vitest';
|
|
14
|
+
import { BaseTokenizer, createSimpleTokenizer } from './base-tokenizer';
|
|
15
|
+
import type { TokenKind } from '../types';
|
|
16
|
+
import type { ValueExtractor, ExtractionResult } from '../../interfaces/value-extractor';
|
|
17
|
+
|
|
18
|
+
/** Minimal `:name`/`$name`/`^name` extractor (shape of semantic's VariableRefExtractor). */
|
|
19
|
+
class SigilRefExtractor implements ValueExtractor {
|
|
20
|
+
readonly name = 'sigil-ref';
|
|
21
|
+
|
|
22
|
+
canExtract(input: string, position: number): boolean {
|
|
23
|
+
const ch = input[position];
|
|
24
|
+
return (
|
|
25
|
+
(ch === ':' || ch === '$' || ch === '^') &&
|
|
26
|
+
position + 1 < input.length &&
|
|
27
|
+
/[a-zA-Z_]/.test(input[position + 1])
|
|
28
|
+
);
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
extract(input: string, position: number): ExtractionResult | null {
|
|
32
|
+
if (!this.canExtract(input, position)) return null;
|
|
33
|
+
let length = 1;
|
|
34
|
+
while (position + length < input.length && /[a-zA-Z0-9_]/.test(input[position + length])) {
|
|
35
|
+
length++;
|
|
36
|
+
}
|
|
37
|
+
return { value: input.substring(position, position + length), length };
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/** Minimal ASCII word extractor (shape of the per-language word walkers). */
|
|
42
|
+
class WordExtractor implements ValueExtractor {
|
|
43
|
+
readonly name = 'word';
|
|
44
|
+
|
|
45
|
+
canExtract(input: string, position: number): boolean {
|
|
46
|
+
return /[a-zA-Z_]/.test(input[position]);
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
extract(input: string, position: number): ExtractionResult | null {
|
|
50
|
+
let length = 0;
|
|
51
|
+
while (position + length < input.length && /[a-zA-Z0-9_]/.test(input[position + length])) {
|
|
52
|
+
length++;
|
|
53
|
+
}
|
|
54
|
+
return length > 0 ? { value: input.substring(position, position + length), length } : null;
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
const KEYWORDS = new Set(['trigger', 'click']);
|
|
59
|
+
|
|
60
|
+
class ProbeTokenizer extends BaseTokenizer {
|
|
61
|
+
readonly language = 'xx';
|
|
62
|
+
readonly direction = 'ltr' as const;
|
|
63
|
+
|
|
64
|
+
constructor() {
|
|
65
|
+
super();
|
|
66
|
+
this.registerExtractors([new SigilRefExtractor(), new WordExtractor()]);
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
classifyToken(token: string): TokenKind {
|
|
70
|
+
return KEYWORDS.has(token) ? 'keyword' : 'identifier';
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
function values(
|
|
75
|
+
tokenizer: { tokenize(input: string): { tokens: readonly { value: string }[] } },
|
|
76
|
+
input: string
|
|
77
|
+
): string[] {
|
|
78
|
+
return tokenizer.tokenize(input).tokens.map(t => t.value);
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
describe('mergeColonQualifiedNames', () => {
|
|
82
|
+
const t = new ProbeTokenizer();
|
|
83
|
+
|
|
84
|
+
it('fuses identifier + :qualifier into one token', () => {
|
|
85
|
+
expect(values(t, 'trigger draggable:start')).toEqual(['trigger', 'draggable:start']);
|
|
86
|
+
});
|
|
87
|
+
|
|
88
|
+
it('re-classifies the merged token and spans both positions', () => {
|
|
89
|
+
const tokens = t.tokenize('trigger draggable:start').tokens;
|
|
90
|
+
const merged = tokens[1];
|
|
91
|
+
expect(merged.kind).toBe('identifier');
|
|
92
|
+
expect(merged.position.start).toBe('trigger '.length);
|
|
93
|
+
expect(merged.position.end).toBe('trigger draggable:start'.length);
|
|
94
|
+
});
|
|
95
|
+
|
|
96
|
+
it('fuses when the pre-colon word is a keyword (en parity: merge before lookup)', () => {
|
|
97
|
+
const tokens = t.tokenize('click:foo').tokens;
|
|
98
|
+
expect(tokens.map(x => x.value)).toEqual(['click:foo']);
|
|
99
|
+
expect(tokens[0].kind).toBe('identifier');
|
|
100
|
+
});
|
|
101
|
+
|
|
102
|
+
it('does not fuse across whitespace — spaced :name stays a local-variable ref', () => {
|
|
103
|
+
expect(values(t, 'trigger :start')).toEqual(['trigger', ':start']);
|
|
104
|
+
});
|
|
105
|
+
|
|
106
|
+
it('leaves a leading bare sigil untouched', () => {
|
|
107
|
+
expect(values(t, ':start')).toEqual([':start']);
|
|
108
|
+
});
|
|
109
|
+
|
|
110
|
+
it('merges a single segment only — a:b:c matches the English extractor', () => {
|
|
111
|
+
expect(values(t, 'a:b:c')).toEqual(['a:b', ':c']);
|
|
112
|
+
});
|
|
113
|
+
|
|
114
|
+
it('never touches $ and ^ sigils', () => {
|
|
115
|
+
expect(values(t, 'put $foo into ^bar')).toEqual(['put', '$foo', 'into', '^bar']);
|
|
116
|
+
});
|
|
117
|
+
|
|
118
|
+
it('is a no-op for domain-DSL streams where : is bare punctuation', () => {
|
|
119
|
+
const sql = createSimpleTokenizer({
|
|
120
|
+
language: 'en',
|
|
121
|
+
keywords: ['select', 'from', 'where'],
|
|
122
|
+
includeOperators: true,
|
|
123
|
+
});
|
|
124
|
+
// Named params never fuse: the colon tokenizes as length-1 punctuation,
|
|
125
|
+
// which can never match the :name qualifier shape.
|
|
126
|
+
expect(values(sql, 'WHERE x = :param')).toEqual(['WHERE', 'x', '=', ':', 'param']);
|
|
127
|
+
expect(values(sql, 'x=:param')).toEqual(['x', '=', ':', 'param']);
|
|
128
|
+
});
|
|
129
|
+
});
|
|
@@ -34,6 +34,12 @@ import {
|
|
|
34
34
|
* Method call handling:
|
|
35
35
|
* - #dialog.showModal() → stops after #dialog (method call, not compound selector)
|
|
36
36
|
* - #box.active → compound selector (no parens)
|
|
37
|
+
*
|
|
38
|
+
* NOTE: intentionally diverges from the semantic package's copy
|
|
39
|
+
* (packages/semantic/src/tokenizers/extractors/css-selector.ts), which also
|
|
40
|
+
* consumes pseudo-class/pseudo-element segments (#x:hover, .a:not(.b)). This
|
|
41
|
+
* legacy version is only used by BaseTokenizer.trySelector (no semantic call
|
|
42
|
+
* sites) and stays as-is.
|
|
37
43
|
*/
|
|
38
44
|
export function extractCssSelector(input: string, startPos: number): string | null {
|
|
39
45
|
if (startPos >= input.length) return null;
|
|
@@ -272,7 +272,11 @@ function buildFormatString(
|
|
|
272
272
|
parts.push(keyword);
|
|
273
273
|
}
|
|
274
274
|
|
|
275
|
-
|
|
275
|
+
// Order roles the same way buildTokens does (descending position). Iterating
|
|
276
|
+
// declaration order here would disagree with the token order whenever a
|
|
277
|
+
// schema's roles are declared in a different order than their positions imply.
|
|
278
|
+
const sortedRoles = sortRolesByWordOrder(schema.roles, profile.wordOrder);
|
|
279
|
+
for (const role of sortedRoles) {
|
|
276
280
|
const marker = getMarkerForRole(role, profile);
|
|
277
281
|
const roleName = `{${role.role}}`;
|
|
278
282
|
|
|
@@ -1036,6 +1036,27 @@ describe('v1.2: diagnostics', () => {
|
|
|
1036
1036
|
expect(node.diagnostics).toHaveLength(1);
|
|
1037
1037
|
expect(node.diagnostics![0].code).toBe('SCHEMA_VALUE_TYPE_MISMATCH');
|
|
1038
1038
|
expect(node.diagnostics![0].severity).toBe('error');
|
|
1039
|
+
expect(node.diagnostics![0].role).toBe('patient');
|
|
1040
|
+
});
|
|
1041
|
+
|
|
1042
|
+
// `role` is required by the wire format's Diagnostic shape. Both directions
|
|
1043
|
+
// used to drop it, and the fixtures happened to omit it, so nothing noticed.
|
|
1044
|
+
it('preserves a diagnostic `role` across a protocol-JSON round-trip', () => {
|
|
1045
|
+
const json: ProtocolNodeJSON = {
|
|
1046
|
+
kind: 'command',
|
|
1047
|
+
action: 'toggle',
|
|
1048
|
+
roles: { patient: { type: 'literal', value: 'hello', dataType: 'string' } },
|
|
1049
|
+
diagnostics: [
|
|
1050
|
+
{
|
|
1051
|
+
level: 'error',
|
|
1052
|
+
role: 'patient',
|
|
1053
|
+
message: "toggle.patient expects type [selector], got 'literal'",
|
|
1054
|
+
code: 'SCHEMA_VALUE_TYPE_MISMATCH',
|
|
1055
|
+
},
|
|
1056
|
+
],
|
|
1057
|
+
};
|
|
1058
|
+
expect(fromProtocolJSON(json).diagnostics![0].role).toBe('patient');
|
|
1059
|
+
expect(toProtocolJSON(fromProtocolJSON(json))).toEqual(json);
|
|
1039
1060
|
});
|
|
1040
1061
|
|
|
1041
1062
|
describe('fixture conformance: type-constraints.json', () => {
|
|
@@ -10,6 +10,9 @@ describe('isValidReference', () => {
|
|
|
10
10
|
expect(isValidReference('event')).toBe(true);
|
|
11
11
|
expect(isValidReference('target')).toBe(true);
|
|
12
12
|
expect(isValidReference('body')).toBe(true);
|
|
13
|
+
expect(isValidReference('document')).toBe(true);
|
|
14
|
+
expect(isValidReference('window')).toBe(true);
|
|
15
|
+
expect(isValidReference('detail')).toBe(true);
|
|
13
16
|
});
|
|
14
17
|
|
|
15
18
|
it('rejects non-references', () => {
|
|
@@ -32,7 +35,7 @@ describe('isValidReference', () => {
|
|
|
32
35
|
expect(isValidReference('it', custom)).toBe(false);
|
|
33
36
|
});
|
|
34
37
|
|
|
35
|
-
it('DEFAULT_REFERENCES contains exactly
|
|
36
|
-
expect(DEFAULT_REFERENCES.size).toBe(
|
|
38
|
+
it('DEFAULT_REFERENCES contains exactly 10 entries', () => {
|
|
39
|
+
expect(DEFAULT_REFERENCES.size).toBe(10);
|
|
37
40
|
});
|
|
38
41
|
});
|
|
@@ -165,7 +165,9 @@ function buildProtocolSection(): PromptSection {
|
|
|
165
165
|
3. No spaces around the colon in role:value pairs
|
|
166
166
|
4. Strings with spaces must be quoted: \`patient:"hello world"\`
|
|
167
167
|
5. Selectors start with \`#\`, \`.\`, \`[\`, \`@\`, or \`*\`
|
|
168
|
-
6.
|
|
168
|
+
6. A selector containing a space, a combinator (\`>\` \`+\` \`~\`), or a comma must use a selector literal: \`patient:<ul > li/>\`, \`patient:<.a, .b/>\`
|
|
169
|
+
7. Inside a structural role (\`body\`, \`then\`, \`else\`, \`condition\`, \`loop-body\`, \`variable\`, \`catch\`, \`finally\`) a \`[...]\` value is always a nested command. Write an attribute selector there as \`condition:<[data-active]/>\`
|
|
170
|
+
8. Output must be valid bracket syntax: \`[action role:value ...]\``;
|
|
169
171
|
|
|
170
172
|
return {
|
|
171
173
|
id: 'protocol',
|
|
@@ -180,6 +182,7 @@ function buildValueTypeSection(): PromptSection {
|
|
|
180
182
|
|
|
181
183
|
| Type | Syntax | Example |
|
|
182
184
|
|------|--------|---------|
|
|
185
|
+
| Selector literal | Delimited by \`<\` and \`/>\` | \`<ul > li/>\`, \`<.a, .b/>\`, \`<[data-id]/>\` |
|
|
183
186
|
| Selector | Starts with \`#\` \`.\` \`[\` \`@\` \`*\` | \`#button\`, \`.active\`, \`[data-id]\` |
|
|
184
187
|
| String | Quoted with \`"\` or \`'\` | \`"hello world"\`, \`'json'\` |
|
|
185
188
|
| Boolean | Exact: \`true\` / \`false\` | \`visible:true\` |
|