@lokascript/framework 2.7.1 → 2.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@lokascript/framework",
3
- "version": "2.7.1",
3
+ "version": "2.8.0",
4
4
  "description": "Generic framework for building multilingual DSLs with semantic parsing and grammar transformation",
5
5
  "type": "module",
6
6
  "main": "dist/index.cjs",
@@ -96,7 +96,7 @@
96
96
  "author": "LokaScript Contributors",
97
97
  "license": "MIT",
98
98
  "dependencies": {
99
- "@lokascript/intent": "^2.7.1"
99
+ "@lokascript/intent": "^2.8.0"
100
100
  },
101
101
  "devDependencies": {
102
102
  "@types/node": "^20.0.0",
@@ -347,7 +347,61 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
347
347
  }
348
348
  }
349
349
 
350
- return new TokenStreamImpl(tokens, this.language);
350
+ return new TokenStreamImpl(this.mergeColonQualifiedNames(tokens), this.language);
351
+ }
352
+
353
+ /**
354
+ * ASCII word of the shape the English word-walker produces. Excludes `:`, so a
355
+ * token that already carries a qualifier never merges again — `a:b:c` yields
356
+ * `a:b` + `:c`, byte-matching the English extractor's single-segment merge.
357
+ */
358
+ private static readonly ASCII_WORD = /^[A-Za-z_][A-Za-z0-9_]*$/;
359
+
360
+ /** `:name` — only a variable-ref-style extractor ever emits this token shape. */
361
+ private static readonly COLON_QUALIFIER = /^:[A-Za-z_][A-Za-z0-9_]*$/;
362
+
363
+ /**
364
+ * Fuse `name` + `:qualifier` into ONE identifier (`draggable:start`).
365
+ *
366
+ * `:name` is hyperscript's local-variable sigil, but a colon IMMEDIATELY
367
+ * preceded by an identifier is a qualifier (custom event namespace), not a
368
+ * sigil. The English tokenizer already merges these inside
369
+ * EnglishKeywordExtractor; this post-pass gives the other 23 languages the
370
+ * same stream. Strict position adjacency is the discriminator: whitespace
371
+ * between the tokens (`trigger :start`) breaks `end === start`, so a spaced
372
+ * local-variable reference survives untouched.
373
+ *
374
+ * Self-gating for non-hyperscript tokenizers (domain DSLs): their extractor
375
+ * sets tokenize `:` as bare punctuation (length 1), which never matches
376
+ * COLON_QUALIFIER, so this pass is a no-op for them.
377
+ */
378
+ protected mergeColonQualifiedNames(tokens: LanguageToken[]): LanguageToken[] {
379
+ const out: LanguageToken[] = [];
380
+ for (const tok of tokens) {
381
+ const prev = out[out.length - 1];
382
+ // No kind gate: the English extractor merges before classification, so a
383
+ // word some language classifies as particle/keyword (es `a`, tr `i`)
384
+ // must fuse the same way. ASCII_WORD already excludes every non-word
385
+ // kind structurally (selectors, urls, numbers, strings, operators).
386
+ if (
387
+ prev &&
388
+ BaseTokenizer.ASCII_WORD.test(prev.value) &&
389
+ BaseTokenizer.COLON_QUALIFIER.test(tok.value) &&
390
+ prev.position.end === tok.position.start
391
+ ) {
392
+ const merged = prev.value + tok.value;
393
+ // Re-classify and drop normalized/stem/metadata — the merged word is no
394
+ // longer the keyword the pieces may have been (matches the en shape).
395
+ out[out.length - 1] = createToken(
396
+ merged,
397
+ this.classifyToken(merged),
398
+ createPosition(prev.position.start, tok.position.end)
399
+ );
400
+ continue;
401
+ }
402
+ out.push(tok);
403
+ }
404
+ return out;
351
405
  }
352
406
 
353
407
  /**
@@ -0,0 +1,129 @@
1
+ /**
2
+ * Contract tests for BaseTokenizer.mergeColonQualifiedNames().
3
+ *
4
+ * `:name` is hyperscript's local-variable sigil, but a colon immediately
5
+ * preceded by an identifier is a qualifier (custom event namespace:
6
+ * `draggable:start`). The English tokenizer merges these inside its keyword
7
+ * extractor; the post-pass gives every other language the same stream. The
8
+ * merge must fire only on strict adjacency and only on `:name`-shaped tokens —
9
+ * domain DSL tokenizers (SQL, BDD, …) emit `:` as bare punctuation, which the
10
+ * pass must never touch.
11
+ */
12
+
13
+ import { describe, it, expect } from 'vitest';
14
+ import { BaseTokenizer, createSimpleTokenizer } from './base-tokenizer';
15
+ import type { TokenKind } from '../types';
16
+ import type { ValueExtractor, ExtractionResult } from '../../interfaces/value-extractor';
17
+
18
+ /** Minimal `:name`/`$name`/`^name` extractor (shape of semantic's VariableRefExtractor). */
19
+ class SigilRefExtractor implements ValueExtractor {
20
+ readonly name = 'sigil-ref';
21
+
22
+ canExtract(input: string, position: number): boolean {
23
+ const ch = input[position];
24
+ return (
25
+ (ch === ':' || ch === '$' || ch === '^') &&
26
+ position + 1 < input.length &&
27
+ /[a-zA-Z_]/.test(input[position + 1])
28
+ );
29
+ }
30
+
31
+ extract(input: string, position: number): ExtractionResult | null {
32
+ if (!this.canExtract(input, position)) return null;
33
+ let length = 1;
34
+ while (position + length < input.length && /[a-zA-Z0-9_]/.test(input[position + length])) {
35
+ length++;
36
+ }
37
+ return { value: input.substring(position, position + length), length };
38
+ }
39
+ }
40
+
41
+ /** Minimal ASCII word extractor (shape of the per-language word walkers). */
42
+ class WordExtractor implements ValueExtractor {
43
+ readonly name = 'word';
44
+
45
+ canExtract(input: string, position: number): boolean {
46
+ return /[a-zA-Z_]/.test(input[position]);
47
+ }
48
+
49
+ extract(input: string, position: number): ExtractionResult | null {
50
+ let length = 0;
51
+ while (position + length < input.length && /[a-zA-Z0-9_]/.test(input[position + length])) {
52
+ length++;
53
+ }
54
+ return length > 0 ? { value: input.substring(position, position + length), length } : null;
55
+ }
56
+ }
57
+
58
+ const KEYWORDS = new Set(['trigger', 'click']);
59
+
60
+ class ProbeTokenizer extends BaseTokenizer {
61
+ readonly language = 'xx';
62
+ readonly direction = 'ltr' as const;
63
+
64
+ constructor() {
65
+ super();
66
+ this.registerExtractors([new SigilRefExtractor(), new WordExtractor()]);
67
+ }
68
+
69
+ classifyToken(token: string): TokenKind {
70
+ return KEYWORDS.has(token) ? 'keyword' : 'identifier';
71
+ }
72
+ }
73
+
74
+ function values(
75
+ tokenizer: { tokenize(input: string): { tokens: readonly { value: string }[] } },
76
+ input: string
77
+ ): string[] {
78
+ return tokenizer.tokenize(input).tokens.map(t => t.value);
79
+ }
80
+
81
+ describe('mergeColonQualifiedNames', () => {
82
+ const t = new ProbeTokenizer();
83
+
84
+ it('fuses identifier + :qualifier into one token', () => {
85
+ expect(values(t, 'trigger draggable:start')).toEqual(['trigger', 'draggable:start']);
86
+ });
87
+
88
+ it('re-classifies the merged token and spans both positions', () => {
89
+ const tokens = t.tokenize('trigger draggable:start').tokens;
90
+ const merged = tokens[1];
91
+ expect(merged.kind).toBe('identifier');
92
+ expect(merged.position.start).toBe('trigger '.length);
93
+ expect(merged.position.end).toBe('trigger draggable:start'.length);
94
+ });
95
+
96
+ it('fuses when the pre-colon word is a keyword (en parity: merge before lookup)', () => {
97
+ const tokens = t.tokenize('click:foo').tokens;
98
+ expect(tokens.map(x => x.value)).toEqual(['click:foo']);
99
+ expect(tokens[0].kind).toBe('identifier');
100
+ });
101
+
102
+ it('does not fuse across whitespace — spaced :name stays a local-variable ref', () => {
103
+ expect(values(t, 'trigger :start')).toEqual(['trigger', ':start']);
104
+ });
105
+
106
+ it('leaves a leading bare sigil untouched', () => {
107
+ expect(values(t, ':start')).toEqual([':start']);
108
+ });
109
+
110
+ it('merges a single segment only — a:b:c matches the English extractor', () => {
111
+ expect(values(t, 'a:b:c')).toEqual(['a:b', ':c']);
112
+ });
113
+
114
+ it('never touches $ and ^ sigils', () => {
115
+ expect(values(t, 'put $foo into ^bar')).toEqual(['put', '$foo', 'into', '^bar']);
116
+ });
117
+
118
+ it('is a no-op for domain-DSL streams where : is bare punctuation', () => {
119
+ const sql = createSimpleTokenizer({
120
+ language: 'en',
121
+ keywords: ['select', 'from', 'where'],
122
+ includeOperators: true,
123
+ });
124
+ // Named params never fuse: the colon tokenizes as length-1 punctuation,
125
+ // which can never match the :name qualifier shape.
126
+ expect(values(sql, 'WHERE x = :param')).toEqual(['WHERE', 'x', '=', ':', 'param']);
127
+ expect(values(sql, 'x=:param')).toEqual(['x', '=', ':', 'param']);
128
+ });
129
+ });
@@ -34,6 +34,12 @@ import {
34
34
  * Method call handling:
35
35
  * - #dialog.showModal() → stops after #dialog (method call, not compound selector)
36
36
  * - #box.active → compound selector (no parens)
37
+ *
38
+ * NOTE: intentionally diverges from the semantic package's copy
39
+ * (packages/semantic/src/tokenizers/extractors/css-selector.ts), which also
40
+ * consumes pseudo-class/pseudo-element segments (#x:hover, .a:not(.b)). This
41
+ * legacy version is only used by BaseTokenizer.trySelector (no semantic call
42
+ * sites) and stays as-is.
37
43
  */
38
44
  export function extractCssSelector(input: string, startPos: number): string | null {
39
45
  if (startPos >= input.length) return null;
@@ -272,7 +272,11 @@ function buildFormatString(
272
272
  parts.push(keyword);
273
273
  }
274
274
 
275
- for (const role of schema.roles) {
275
+ // Order roles the same way buildTokens does (descending position). Iterating
276
+ // declaration order here would disagree with the token order whenever a
277
+ // schema's roles are declared in a different order than their positions imply.
278
+ const sortedRoles = sortRolesByWordOrder(schema.roles, profile.wordOrder);
279
+ for (const role of sortedRoles) {
276
280
  const marker = getMarkerForRole(role, profile);
277
281
  const roleName = `{${role.role}}`;
278
282
 
@@ -1036,6 +1036,27 @@ describe('v1.2: diagnostics', () => {
1036
1036
  expect(node.diagnostics).toHaveLength(1);
1037
1037
  expect(node.diagnostics![0].code).toBe('SCHEMA_VALUE_TYPE_MISMATCH');
1038
1038
  expect(node.diagnostics![0].severity).toBe('error');
1039
+ expect(node.diagnostics![0].role).toBe('patient');
1040
+ });
1041
+
1042
+ // `role` is required by the wire format's Diagnostic shape. Both directions
1043
+ // used to drop it, and the fixtures happened to omit it, so nothing noticed.
1044
+ it('preserves a diagnostic `role` across a protocol-JSON round-trip', () => {
1045
+ const json: ProtocolNodeJSON = {
1046
+ kind: 'command',
1047
+ action: 'toggle',
1048
+ roles: { patient: { type: 'literal', value: 'hello', dataType: 'string' } },
1049
+ diagnostics: [
1050
+ {
1051
+ level: 'error',
1052
+ role: 'patient',
1053
+ message: "toggle.patient expects type [selector], got 'literal'",
1054
+ code: 'SCHEMA_VALUE_TYPE_MISMATCH',
1055
+ },
1056
+ ],
1057
+ };
1058
+ expect(fromProtocolJSON(json).diagnostics![0].role).toBe('patient');
1059
+ expect(toProtocolJSON(fromProtocolJSON(json))).toEqual(json);
1039
1060
  });
1040
1061
 
1041
1062
  describe('fixture conformance: type-constraints.json', () => {
@@ -10,6 +10,9 @@ describe('isValidReference', () => {
10
10
  expect(isValidReference('event')).toBe(true);
11
11
  expect(isValidReference('target')).toBe(true);
12
12
  expect(isValidReference('body')).toBe(true);
13
+ expect(isValidReference('document')).toBe(true);
14
+ expect(isValidReference('window')).toBe(true);
15
+ expect(isValidReference('detail')).toBe(true);
13
16
  });
14
17
 
15
18
  it('rejects non-references', () => {
@@ -32,7 +35,7 @@ describe('isValidReference', () => {
32
35
  expect(isValidReference('it', custom)).toBe(false);
33
36
  });
34
37
 
35
- it('DEFAULT_REFERENCES contains exactly 7 entries', () => {
36
- expect(DEFAULT_REFERENCES.size).toBe(7);
38
+ it('DEFAULT_REFERENCES contains exactly 10 entries', () => {
39
+ expect(DEFAULT_REFERENCES.size).toBe(10);
37
40
  });
38
41
  });
@@ -165,7 +165,9 @@ function buildProtocolSection(): PromptSection {
165
165
  3. No spaces around the colon in role:value pairs
166
166
  4. Strings with spaces must be quoted: \`patient:"hello world"\`
167
167
  5. Selectors start with \`#\`, \`.\`, \`[\`, \`@\`, or \`*\`
168
- 6. Output must be valid bracket syntax: \`[action role:value ...]\``;
168
+ 6. A selector containing a space, a combinator (\`>\` \`+\` \`~\`), or a comma must use a selector literal: \`patient:<ul > li/>\`, \`patient:<.a, .b/>\`
169
+ 7. Inside a structural role (\`body\`, \`then\`, \`else\`, \`condition\`, \`loop-body\`, \`variable\`, \`catch\`, \`finally\`) a \`[...]\` value is always a nested command. Write an attribute selector there as \`condition:<[data-active]/>\`
170
+ 8. Output must be valid bracket syntax: \`[action role:value ...]\``;
169
171
 
170
172
  return {
171
173
  id: 'protocol',
@@ -180,6 +182,7 @@ function buildValueTypeSection(): PromptSection {
180
182
 
181
183
  | Type | Syntax | Example |
182
184
  |------|--------|---------|
185
+ | Selector literal | Delimited by \`<\` and \`/>\` | \`<ul > li/>\`, \`<.a, .b/>\`, \`<[data-id]/>\` |
183
186
  | Selector | Starts with \`#\` \`.\` \`[\` \`@\` \`*\` | \`#button\`, \`.active\`, \`[data-id]\` |
184
187
  | String | Quoted with \`"\` or \`'\` | \`"hello world"\`, \`'json'\` |
185
188
  | Boolean | Exact: \`true\` / \`false\` | \`visible:true\` |