@lokascript/framework 2.5.1 → 2.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@lokascript/framework",
3
- "version": "2.5.1",
3
+ "version": "2.6.0",
4
4
  "description": "Generic framework for building multilingual DSLs with semantic parsing and grammar transformation",
5
5
  "type": "module",
6
6
  "main": "dist/index.cjs",
@@ -69,7 +69,7 @@
69
69
  "test": "vitest",
70
70
  "test:run": "vitest run",
71
71
  "test:watch": "vitest watch",
72
- "test:check": "vitest run --reporter=dot 2>&1 | tail -5",
72
+ "test:check": "VITEST_QUIET=1 bash ../../scripts/vitest-run.sh --reporter=dot",
73
73
  "test:coverage": "vitest run --coverage",
74
74
  "test:unit": "vitest run src/core/**/*.test.ts src/generation/**/*.test.ts",
75
75
  "test:integration": "vitest run src/__test__/**/*.test.ts src/api/**/*.test.ts",
@@ -96,7 +96,7 @@
96
96
  "author": "LokaScript Contributors",
97
97
  "license": "MIT",
98
98
  "dependencies": {
99
- "@lokascript/intent": "^2.5.1"
99
+ "@lokascript/intent": "^2.6.0"
100
100
  },
101
101
  "devDependencies": {
102
102
  "@types/node": "^20.0.0",
@@ -32,6 +32,61 @@ import { getDefaultExtractors } from './default-extractors';
32
32
  // Uses the canonical list from OperatorExtractor to avoid duplication.
33
33
  const SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
34
34
 
35
+ /**
36
+ * Normalized concepts that are matched via the pattern matcher's ROLE-MARKER
37
+ * MECHANISM (the source/destination/event clause matchers in
38
+ * `packages/semantic/src/parser/pattern-matcher.ts`), which peeks/advances a
39
+ * SINGLE token and checks `.value`/`.normalized`. A multi-word phrase carrying
40
+ * one of these must NOT be pre-matched as a single keyword token — doing so
41
+ * shadows the single-word marker those clause matchers expect (e.g. id
42
+ * `ke dalam`=into hides the `ke` destination marker; ko `할 때`=eventMarker
43
+ * pre-empts SOV event extraction).
44
+ *
45
+ * NOTE — what is *not* here. Prepositional modifiers that the generated patterns
46
+ * expose as ordinary pattern LITERALS (`before`/`after` in put-before/after,
47
+ * `until` in repeat-until) are deliberately absent: those are read by
48
+ * `matchLiteralToken`, which compares the whole token by exact value OR
49
+ * normalized form (`getMatchType`), so a multi-word marker token (`से पहले`,
50
+ * `cho đến khi`) matches the literal's value/alternatives directly with no
51
+ * special handling. Keeping them out lets `tryMultiWordKeyword` emit them as one
52
+ * token — the profile-driven replacement for the per-language hardcoded compound
53
+ * lists (Task #10). `into` stays excluded because it IS consumed by the role
54
+ * mechanism in some languages (id destination `ke`), and `from`/`to`/`with`/
55
+ * `on`/`at`/`of`/`as`/`by`/`in` are genuine role markers. Command verbs /
56
+ * control-flow / event names were always absent. See `multiWordKeywords` /
57
+ * `tryMultiWordKeyword`.
58
+ */
59
+ const MARKER_CONCEPT_NORMALIZEDS: ReadonlySet<string> = new Set([
60
+ // Role-marker role names (profile.roleMarkers normalizeds)
61
+ 'patient',
62
+ 'destination',
63
+ 'source',
64
+ 'style',
65
+ 'event',
66
+ 'eventMarker',
67
+ 'agent',
68
+ 'goal',
69
+ 'manner',
70
+ // Prepositional / positional modifier concepts matched via the role mechanism
71
+ // (profile.keywords "Modifiers"). `before`/`after`/`until` are intentionally
72
+ // NOT here — they are pattern literals (see the note above).
73
+ 'into',
74
+ 'from',
75
+ 'to',
76
+ 'with',
77
+ 'at',
78
+ 'of',
79
+ 'as',
80
+ 'by',
81
+ 'in',
82
+ 'on',
83
+ 'over',
84
+ 'under',
85
+ 'between',
86
+ 'through',
87
+ 'without',
88
+ ]);
89
+
35
90
  // =============================================================================
36
91
  // Types
37
92
  // =============================================================================
@@ -40,6 +95,39 @@ const SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
40
95
  // for backward compatibility with code importing from this module.
41
96
  export type { KeywordEntry };
42
97
 
98
+ /**
99
+ * Standard DOM event names recognized in every language as universal fallbacks.
100
+ * The i18n grammar transformer emits these verbatim (no native dictionary form),
101
+ * so each tokenizer must accept them or English-named event handlers won't parse.
102
+ * Kept to genuine DOM event names (not command verbs) to minimize collisions; the
103
+ * registration is `!has`-guarded so any native keyword of the same spelling wins.
104
+ */
105
+ const ENGLISH_DOM_EVENT_NAMES: readonly string[] = [
106
+ 'click',
107
+ 'dblclick',
108
+ 'input',
109
+ 'change',
110
+ 'submit',
111
+ 'keydown',
112
+ 'keyup',
113
+ 'keypress',
114
+ 'mousedown',
115
+ 'mouseup',
116
+ 'mouseover',
117
+ 'mouseout',
118
+ 'mouseenter',
119
+ 'mouseleave',
120
+ 'mousemove',
121
+ 'pointerdown',
122
+ 'pointerup',
123
+ 'pointermove',
124
+ 'focus',
125
+ 'blur',
126
+ 'load',
127
+ 'resize',
128
+ 'scroll',
129
+ ];
130
+
43
131
  /**
44
132
  * Profile interface for keyword derivation.
45
133
  * Matches the structure of LanguageProfile but only includes fields needed for tokenization.
@@ -81,9 +169,33 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
81
169
  /** Keywords derived from profile, sorted longest-first for greedy matching */
82
170
  protected profileKeywords: KeywordEntry[] = [];
83
171
 
172
+ /**
173
+ * Space-containing profile keywords (multi-word phrases), longest-first.
174
+ * Used by `tryMultiWordKeyword` so natural spaced forms (hi `मेल खाता`,
175
+ * vi `chuyển đổi`, es `tecla abajo`, …) tokenize as ONE keyword — the
176
+ * profile-driven replacement for the per-language hardcoded compound lists.
177
+ * Empty for no-space (CJK) languages, so they are unaffected.
178
+ */
179
+ protected multiWordKeywords: KeywordEntry[] = [];
180
+
84
181
  /** Map for O(1) keyword lookups by lowercase native word */
85
182
  protected profileKeywordMap: Map<string, KeywordEntry> = new Map();
86
183
 
184
+ /**
185
+ * The raw EXTRAS list passed to initializeKeywordsFromProfile, kept pre-dedup.
186
+ * The keyword map is keyed by native word with last-wins insertion, so a
187
+ * duplicate native word inside the extras silently shadows the earlier entry
188
+ * (e.g. a `nächste→closest` entry shadowing `nächste→next` broke German
189
+ * positional expressions). Exposed so consistency tests can detect such
190
+ * intra-extras collisions, which are invisible in the deduplicated map.
191
+ */
192
+ private rawExtraEntries: KeywordEntry[] = [];
193
+
194
+ /** Raw extras as passed in, pre-dedup — for consistency tests. */
195
+ getExtraKeywordEntries(): readonly KeywordEntry[] {
196
+ return this.rawExtraEntries;
197
+ }
198
+
87
199
  /**
88
200
  * Pluggable value extractors for domain-specific syntax.
89
201
  * When registered, BaseTokenizer will use extractor-based tokenization instead of legacy methods.
@@ -171,6 +283,19 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
171
283
  }
172
284
  if (pos >= input.length) break;
173
285
 
286
+ // Multi-word keyword pre-match: a profile keyword containing a space
287
+ // (e.g. hi `मेल खाता`, vi `chuyển đổi`, es `tecla abajo`) is matched as ONE
288
+ // keyword token at a word boundary, longest-first. Runs before the
289
+ // per-language extractors so natural spaced multi-word keywords tokenize
290
+ // without each tokenizer hardcoding a compound list. No-op for single-word
291
+ // and no-space (CJK) languages (multiWordKeywords is empty).
292
+ const multiWord = this.tryMultiWordKeyword(input, pos);
293
+ if (multiWord) {
294
+ tokens.push(multiWord);
295
+ pos = multiWord.position.end;
296
+ continue;
297
+ }
298
+
174
299
  // Try registered extractors in order
175
300
  let extracted = false;
176
301
  for (const extractor of this.extractors) {
@@ -297,6 +422,7 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
297
422
  ): void {
298
423
  // Use a Map to deduplicate, with extras taking precedence
299
424
  const keywordMap = new Map<string, KeywordEntry>();
425
+ this.rawExtraEntries = extras;
300
426
 
301
427
  // Extract from keywords (command translations)
302
428
  if (profile.keywords) {
@@ -356,6 +482,20 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
356
482
  }
357
483
  }
358
484
 
485
+ // Register English DOM event names as universal fallbacks. The i18n grammar
486
+ // transformer has no native form for most DOM events, so it passes them
487
+ // through verbatim (`on keyup …` → `<on-marker> keyup …`). Without these,
488
+ // non-English token streams treat `keyup`/`keydown`/`resize`/… as bare
489
+ // identifiers, and event handlers using them (often with `[key==…]` guards)
490
+ // fail to parse. Guarded by `!has` so any native mapping wins (same policy
491
+ // as the English-reference fallbacks above). Generalizes the per-language
492
+ // registration introduced for Hebrew in #272.
493
+ for (const evt of ENGLISH_DOM_EVENT_NAMES) {
494
+ if (!keywordMap.has(evt)) {
495
+ keywordMap.set(evt, { native: evt, normalized: evt });
496
+ }
497
+ }
498
+
359
499
  // Add extra entries (literals, positional, events) - these OVERRIDE profile entries
360
500
  for (const extra of extras) {
361
501
  keywordMap.set(extra.native, extra);
@@ -366,6 +506,20 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
366
506
  (a, b) => b.native.length - a.native.length
367
507
  );
368
508
 
509
+ // Multi-word (space-containing) keywords, for longest-phrase matching at a
510
+ // token boundary. Already longest-first (profileKeywords is sorted above).
511
+ // Marker/modifier concepts are EXCLUDED: those are matched positionally by
512
+ // the pattern matcher (role markers), and greedily consuming a multi-word
513
+ // marker phrase shadows the single-word marker patterns rely on — e.g. id
514
+ // `ke dalam` (into) would swallow the `ke` destination marker, and ko `할 때`
515
+ // (eventMarker) would pre-empt the SOV event extraction. Command verbs,
516
+ // control-flow, and event-name keywords (vi `với mỗi`=for, es `tecla abajo`=
517
+ // keydown, bn `তৈরি করুন`=make) are kept — the pattern matcher treats those
518
+ // as keyword literals, so one-token matching is strictly better.
519
+ this.multiWordKeywords = this.profileKeywords.filter(
520
+ k => k.native.includes(' ') && !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
521
+ );
522
+
369
523
  // Build Map for O(1) lookups (case-insensitive + diacritic-insensitive)
370
524
  // This allows matching both 'بدّل' (with shadda) and 'بدل' (without) to the same entry
371
525
  this.profileKeywordMap = new Map();
@@ -417,6 +571,40 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
417
571
  return null;
418
572
  }
419
573
 
574
+ /**
575
+ * Match the longest multi-word (space-containing) profile keyword at `pos`,
576
+ * requiring the match to end at a word boundary. The profile-driven
577
+ * counterpart of the per-language hardcoded compound lists (the hindi and
578
+ * vietnamese keyword extractors). Returns a keyword token (with the normalized
579
+ * form) or null. Case-sensitive against the stored native form, mirroring
580
+ * `tryProfileKeyword`/`isKeywordStart` (the i18n dicts emit a fixed surface
581
+ * case). No-op when `multiWordKeywords` is empty (no-space/CJK languages).
582
+ *
583
+ * @param input - Input string
584
+ * @param pos - Current position (must be a token-start boundary)
585
+ * @param isWordChar - End-boundary predicate (defaults to Unicode letter/digit/_)
586
+ */
587
+ protected tryMultiWordKeyword(
588
+ input: string,
589
+ pos: number,
590
+ isWordChar: (char: string) => boolean = ch => /[\p{L}\p{N}_]/u.test(ch)
591
+ ): LanguageToken | null {
592
+ if (this.multiWordKeywords.length === 0) return null;
593
+ const rest = input.slice(pos);
594
+ for (const entry of this.multiWordKeywords) {
595
+ if (!rest.startsWith(entry.native)) continue;
596
+ const after = input[pos + entry.native.length];
597
+ if (after !== undefined && isWordChar(after)) continue; // not a word boundary
598
+ return createToken(
599
+ entry.native,
600
+ 'keyword',
601
+ createPosition(pos, pos + entry.native.length),
602
+ entry.normalized
603
+ );
604
+ }
605
+ return null;
606
+ }
607
+
420
608
  /**
421
609
  * Check if the remaining input starts with any known keyword.
422
610
  * Useful for non-space languages to detect word boundaries.
@@ -430,6 +618,37 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
430
618
  return this.profileKeywords.some(entry => remaining.startsWith(entry.native));
431
619
  }
432
620
 
621
+ /**
622
+ * Check if a known keyword starts at the given position AND ends at a word
623
+ * boundary (end of input or a non-word character).
624
+ *
625
+ * Space-delimited languages must use this (not `isKeywordStart`) for
626
+ * word-walk break checks: the keyword table includes English canonical
627
+ * fallbacks (me, it, you, …), so a raw `startsWith` check splits any native
628
+ * word with an embedded fallback mid-word (e.g. Quechua ñit'iy contains
629
+ * "it"). CJK/no-space tokenizers rely on mid-text keyword starts and must
630
+ * keep using `isKeywordStart`.
631
+ *
632
+ * @param input - Input string
633
+ * @param pos - Current position
634
+ * @param isWordChar - Language-specific word-character predicate; pass the
635
+ * tokenizer's letter classifier so e.g. the Quechua glottal apostrophe
636
+ * counts as part of a word. Defaults to Unicode letters/digits/underscore.
637
+ * @returns true if a keyword starts here and is not followed by a word char
638
+ */
639
+ protected isKeywordStartAtBoundary(
640
+ input: string,
641
+ pos: number,
642
+ isWordChar: (char: string) => boolean = ch => /[\p{L}\p{N}_]/u.test(ch)
643
+ ): boolean {
644
+ const remaining = input.slice(pos);
645
+ return this.profileKeywords.some(entry => {
646
+ if (!remaining.startsWith(entry.native)) return false;
647
+ const after = input[pos + entry.native.length];
648
+ return after === undefined || !isWordChar(after);
649
+ });
650
+ }
651
+
433
652
  /**
434
653
  * Look up a keyword by native word (case-insensitive).
435
654
  * O(1) lookup using the keyword map.
@@ -0,0 +1,73 @@
1
+ /**
2
+ * Contract tests for BaseTokenizer.isKeywordStartAtBoundary().
3
+ *
4
+ * The keyword table injects English canonical reference words (me, it, you, …)
5
+ * for every language. Space-delimited tokenizers that break their word-walk on
6
+ * a raw isKeywordStart() therefore split native words around embedded English
7
+ * fallbacks (e.g. Quechua ñit'iy contains "it"). The boundary-aware variant
8
+ * only fires when the keyword match ends at a word boundary.
9
+ */
10
+
11
+ import { describe, it, expect } from 'vitest';
12
+ import { BaseTokenizer } from './base-tokenizer';
13
+ import type { TokenKind } from '../types';
14
+
15
+ class ProbeTokenizer extends BaseTokenizer {
16
+ readonly language = 'xx';
17
+ readonly direction = 'ltr' as const;
18
+
19
+ constructor() {
20
+ super();
21
+ this.initializeKeywordsFromProfile({
22
+ keywords: { toggle: { primary: 'tikray' } },
23
+ // references inject the English canonical fallbacks "it" and "me" too
24
+ references: { it: 'pay', me: 'nuqa' },
25
+ });
26
+ }
27
+
28
+ classifyToken(): TokenKind {
29
+ return 'identifier';
30
+ }
31
+
32
+ probeRaw(input: string, pos: number): boolean {
33
+ return this.isKeywordStart(input, pos);
34
+ }
35
+
36
+ probeBoundary(input: string, pos: number, isWordChar?: (ch: string) => boolean): boolean {
37
+ return this.isKeywordStartAtBoundary(input, pos, isWordChar);
38
+ }
39
+ }
40
+
41
+ describe('isKeywordStartAtBoundary', () => {
42
+ const t = new ProbeTokenizer();
43
+
44
+ it('does not fire on an English fallback embedded mid-word', () => {
45
+ // "it" inside "umitaq" — raw check fires, boundary check must not
46
+ expect(t.probeRaw('umitaq', 2)).toBe(true);
47
+ expect(t.probeBoundary('umitaq', 2)).toBe(false);
48
+ // "me" inside "umema"
49
+ expect(t.probeRaw('umema', 1)).toBe(true);
50
+ expect(t.probeBoundary('umema', 1)).toBe(false);
51
+ });
52
+
53
+ it('fires when the keyword match ends at end of input', () => {
54
+ expect(t.probeBoundary('umit', 2)).toBe(true);
55
+ });
56
+
57
+ it('fires when the keyword match is followed by a non-word char', () => {
58
+ expect(t.probeBoundary('um it goes', 3)).toBe(true);
59
+ expect(t.probeBoundary('xit.', 1)).toBe(true);
60
+ });
61
+
62
+ it('respects a language-specific word-char predicate', () => {
63
+ // Default predicate: apostrophe is not a word char → boundary fires
64
+ expect(t.probeBoundary("xit'y", 1)).toBe(true);
65
+ // Quechua-style predicate counting the glottal apostrophe → no boundary
66
+ const quechuaLike = (ch: string) => /[a-zñ'’]/i.test(ch);
67
+ expect(t.probeBoundary("xit'y", 1, quechuaLike)).toBe(false);
68
+ });
69
+
70
+ it('returns false when no keyword starts at the position at all', () => {
71
+ expect(t.probeBoundary('umitaq', 0)).toBe(false);
72
+ });
73
+ });
@@ -406,6 +406,21 @@ export interface TokenizerContext {
406
406
  */
407
407
  isKeywordStart(input: string, position: number): boolean;
408
408
 
409
+ /**
410
+ * Like `isKeywordStart`, but only true when the keyword match ends at a
411
+ * word boundary (end of input or a char rejected by `isWordChar`).
412
+ * Space-delimited languages must use this for word-walk break checks —
413
+ * the keyword table includes English canonical fallbacks (me, it, you, …),
414
+ * so the raw check splits native words mid-word (e.g. Quechua ñit'iy
415
+ * contains "it"). Optional for backward compatibility with hand-rolled
416
+ * contexts; callers should treat absence as "no boundary keyword here".
417
+ */
418
+ isKeywordStartAtBoundary?(
419
+ input: string,
420
+ position: number,
421
+ isWordChar?: (char: string) => boolean
422
+ ): boolean;
423
+
409
424
  /**
410
425
  * Optional morphological normalizer for this language.
411
426
  */
@@ -449,6 +464,11 @@ export function createTokenizerContext(tokenizer: {
449
464
  lookupKeyword(native: string): KeywordEntry | undefined;
450
465
  isKeyword(native: string): boolean;
451
466
  isKeywordStart(input: string, position: number): boolean;
467
+ isKeywordStartAtBoundary?(
468
+ input: string,
469
+ position: number,
470
+ isWordChar?: (char: string) => boolean
471
+ ): boolean;
452
472
  normalizer?: MorphologicalNormalizer;
453
473
  }): TokenizerContext {
454
474
  const ctx: TokenizerContext = {
@@ -457,6 +477,9 @@ export function createTokenizerContext(tokenizer: {
457
477
  lookupKeyword: tokenizer.lookupKeyword.bind(tokenizer),
458
478
  isKeyword: tokenizer.isKeyword.bind(tokenizer),
459
479
  isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer),
480
+ ...(tokenizer.isKeywordStartAtBoundary
481
+ ? { isKeywordStartAtBoundary: tokenizer.isKeywordStartAtBoundary.bind(tokenizer) }
482
+ : {}),
460
483
  };
461
484
 
462
485
  if (tokenizer.normalizer) {