@lokascript/framework 2.5.1 → 2.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/core/index.js +148 -1
- package/dist/core/index.js.map +1 -1
- package/dist/core/tokenization/base-tokenizer.d.ts +52 -0
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
- package/dist/core/tokenization/index.js +148 -1
- package/dist/core/tokenization/index.js.map +1 -1
- package/dist/index.cjs +148 -1
- package/dist/index.cjs.map +1 -1
- package/dist/index.js +148 -1
- package/dist/index.js.map +1 -1
- package/dist/interfaces/value-extractor.d.ts +11 -0
- package/dist/interfaces/value-extractor.d.ts.map +1 -1
- package/dist/testing/index.js +12 -4
- package/dist/testing/index.js.map +1 -1
- package/package.json +3 -3
- package/src/core/tokenization/base-tokenizer.ts +219 -0
- package/src/core/tokenization/keyword-boundary.test.ts +73 -0
- package/src/interfaces/value-extractor.ts +23 -0
- package/src/core/pattern-matching/pattern-matcher.ts.backup +0 -1267
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@lokascript/framework",
|
|
3
|
-
"version": "2.
|
|
3
|
+
"version": "2.6.0",
|
|
4
4
|
"description": "Generic framework for building multilingual DSLs with semantic parsing and grammar transformation",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "dist/index.cjs",
|
|
@@ -69,7 +69,7 @@
|
|
|
69
69
|
"test": "vitest",
|
|
70
70
|
"test:run": "vitest run",
|
|
71
71
|
"test:watch": "vitest watch",
|
|
72
|
-
"test:check": "vitest
|
|
72
|
+
"test:check": "VITEST_QUIET=1 bash ../../scripts/vitest-run.sh --reporter=dot",
|
|
73
73
|
"test:coverage": "vitest run --coverage",
|
|
74
74
|
"test:unit": "vitest run src/core/**/*.test.ts src/generation/**/*.test.ts",
|
|
75
75
|
"test:integration": "vitest run src/__test__/**/*.test.ts src/api/**/*.test.ts",
|
|
@@ -96,7 +96,7 @@
|
|
|
96
96
|
"author": "LokaScript Contributors",
|
|
97
97
|
"license": "MIT",
|
|
98
98
|
"dependencies": {
|
|
99
|
-
"@lokascript/intent": "^2.
|
|
99
|
+
"@lokascript/intent": "^2.6.0"
|
|
100
100
|
},
|
|
101
101
|
"devDependencies": {
|
|
102
102
|
"@types/node": "^20.0.0",
|
|
@@ -32,6 +32,61 @@ import { getDefaultExtractors } from './default-extractors';
|
|
|
32
32
|
// Uses the canonical list from OperatorExtractor to avoid duplication.
|
|
33
33
|
const SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
|
|
34
34
|
|
|
35
|
+
/**
|
|
36
|
+
* Normalized concepts that are matched via the pattern matcher's ROLE-MARKER
|
|
37
|
+
* MECHANISM (the source/destination/event clause matchers in
|
|
38
|
+
* `packages/semantic/src/parser/pattern-matcher.ts`), which peeks/advances a
|
|
39
|
+
* SINGLE token and checks `.value`/`.normalized`. A multi-word phrase carrying
|
|
40
|
+
* one of these must NOT be pre-matched as a single keyword token — doing so
|
|
41
|
+
* shadows the single-word marker those clause matchers expect (e.g. id
|
|
42
|
+
* `ke dalam`=into hides the `ke` destination marker; ko `할 때`=eventMarker
|
|
43
|
+
* pre-empts SOV event extraction).
|
|
44
|
+
*
|
|
45
|
+
* NOTE — what is *not* here. Prepositional modifiers that the generated patterns
|
|
46
|
+
* expose as ordinary pattern LITERALS (`before`/`after` in put-before/after,
|
|
47
|
+
* `until` in repeat-until) are deliberately absent: those are read by
|
|
48
|
+
* `matchLiteralToken`, which compares the whole token by exact value OR
|
|
49
|
+
* normalized form (`getMatchType`), so a multi-word marker token (`से पहले`,
|
|
50
|
+
* `cho đến khi`) matches the literal's value/alternatives directly with no
|
|
51
|
+
* special handling. Keeping them out lets `tryMultiWordKeyword` emit them as one
|
|
52
|
+
* token — the profile-driven replacement for the per-language hardcoded compound
|
|
53
|
+
* lists (Task #10). `into` stays excluded because it IS consumed by the role
|
|
54
|
+
* mechanism in some languages (id destination `ke`), and `from`/`to`/`with`/
|
|
55
|
+
* `on`/`at`/`of`/`as`/`by`/`in` are genuine role markers. Command verbs /
|
|
56
|
+
* control-flow / event names were always absent. See `multiWordKeywords` /
|
|
57
|
+
* `tryMultiWordKeyword`.
|
|
58
|
+
*/
|
|
59
|
+
const MARKER_CONCEPT_NORMALIZEDS: ReadonlySet<string> = new Set([
|
|
60
|
+
// Role-marker role names (profile.roleMarkers normalizeds)
|
|
61
|
+
'patient',
|
|
62
|
+
'destination',
|
|
63
|
+
'source',
|
|
64
|
+
'style',
|
|
65
|
+
'event',
|
|
66
|
+
'eventMarker',
|
|
67
|
+
'agent',
|
|
68
|
+
'goal',
|
|
69
|
+
'manner',
|
|
70
|
+
// Prepositional / positional modifier concepts matched via the role mechanism
|
|
71
|
+
// (profile.keywords "Modifiers"). `before`/`after`/`until` are intentionally
|
|
72
|
+
// NOT here — they are pattern literals (see the note above).
|
|
73
|
+
'into',
|
|
74
|
+
'from',
|
|
75
|
+
'to',
|
|
76
|
+
'with',
|
|
77
|
+
'at',
|
|
78
|
+
'of',
|
|
79
|
+
'as',
|
|
80
|
+
'by',
|
|
81
|
+
'in',
|
|
82
|
+
'on',
|
|
83
|
+
'over',
|
|
84
|
+
'under',
|
|
85
|
+
'between',
|
|
86
|
+
'through',
|
|
87
|
+
'without',
|
|
88
|
+
]);
|
|
89
|
+
|
|
35
90
|
// =============================================================================
|
|
36
91
|
// Types
|
|
37
92
|
// =============================================================================
|
|
@@ -40,6 +95,39 @@ const SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
|
|
|
40
95
|
// for backward compatibility with code importing from this module.
|
|
41
96
|
export type { KeywordEntry };
|
|
42
97
|
|
|
98
|
+
/**
|
|
99
|
+
* Standard DOM event names recognized in every language as universal fallbacks.
|
|
100
|
+
* The i18n grammar transformer emits these verbatim (no native dictionary form),
|
|
101
|
+
* so each tokenizer must accept them or English-named event handlers won't parse.
|
|
102
|
+
* Kept to genuine DOM event names (not command verbs) to minimize collisions; the
|
|
103
|
+
* registration is `!has`-guarded so any native keyword of the same spelling wins.
|
|
104
|
+
*/
|
|
105
|
+
const ENGLISH_DOM_EVENT_NAMES: readonly string[] = [
|
|
106
|
+
'click',
|
|
107
|
+
'dblclick',
|
|
108
|
+
'input',
|
|
109
|
+
'change',
|
|
110
|
+
'submit',
|
|
111
|
+
'keydown',
|
|
112
|
+
'keyup',
|
|
113
|
+
'keypress',
|
|
114
|
+
'mousedown',
|
|
115
|
+
'mouseup',
|
|
116
|
+
'mouseover',
|
|
117
|
+
'mouseout',
|
|
118
|
+
'mouseenter',
|
|
119
|
+
'mouseleave',
|
|
120
|
+
'mousemove',
|
|
121
|
+
'pointerdown',
|
|
122
|
+
'pointerup',
|
|
123
|
+
'pointermove',
|
|
124
|
+
'focus',
|
|
125
|
+
'blur',
|
|
126
|
+
'load',
|
|
127
|
+
'resize',
|
|
128
|
+
'scroll',
|
|
129
|
+
];
|
|
130
|
+
|
|
43
131
|
/**
|
|
44
132
|
* Profile interface for keyword derivation.
|
|
45
133
|
* Matches the structure of LanguageProfile but only includes fields needed for tokenization.
|
|
@@ -81,9 +169,33 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
|
|
|
81
169
|
/** Keywords derived from profile, sorted longest-first for greedy matching */
|
|
82
170
|
protected profileKeywords: KeywordEntry[] = [];
|
|
83
171
|
|
|
172
|
+
/**
|
|
173
|
+
* Space-containing profile keywords (multi-word phrases), longest-first.
|
|
174
|
+
* Used by `tryMultiWordKeyword` so natural spaced forms (hi `मेल खाता`,
|
|
175
|
+
* vi `chuyển đổi`, es `tecla abajo`, …) tokenize as ONE keyword — the
|
|
176
|
+
* profile-driven replacement for the per-language hardcoded compound lists.
|
|
177
|
+
* Empty for no-space (CJK) languages, so they are unaffected.
|
|
178
|
+
*/
|
|
179
|
+
protected multiWordKeywords: KeywordEntry[] = [];
|
|
180
|
+
|
|
84
181
|
/** Map for O(1) keyword lookups by lowercase native word */
|
|
85
182
|
protected profileKeywordMap: Map<string, KeywordEntry> = new Map();
|
|
86
183
|
|
|
184
|
+
/**
|
|
185
|
+
* The raw EXTRAS list passed to initializeKeywordsFromProfile, kept pre-dedup.
|
|
186
|
+
* The keyword map is keyed by native word with last-wins insertion, so a
|
|
187
|
+
* duplicate native word inside the extras silently shadows the earlier entry
|
|
188
|
+
* (e.g. a `nächste→closest` entry shadowing `nächste→next` broke German
|
|
189
|
+
* positional expressions). Exposed so consistency tests can detect such
|
|
190
|
+
* intra-extras collisions, which are invisible in the deduplicated map.
|
|
191
|
+
*/
|
|
192
|
+
private rawExtraEntries: KeywordEntry[] = [];
|
|
193
|
+
|
|
194
|
+
/** Raw extras as passed in, pre-dedup — for consistency tests. */
|
|
195
|
+
getExtraKeywordEntries(): readonly KeywordEntry[] {
|
|
196
|
+
return this.rawExtraEntries;
|
|
197
|
+
}
|
|
198
|
+
|
|
87
199
|
/**
|
|
88
200
|
* Pluggable value extractors for domain-specific syntax.
|
|
89
201
|
* When registered, BaseTokenizer will use extractor-based tokenization instead of legacy methods.
|
|
@@ -171,6 +283,19 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
|
|
|
171
283
|
}
|
|
172
284
|
if (pos >= input.length) break;
|
|
173
285
|
|
|
286
|
+
// Multi-word keyword pre-match: a profile keyword containing a space
|
|
287
|
+
// (e.g. hi `मेल खाता`, vi `chuyển đổi`, es `tecla abajo`) is matched as ONE
|
|
288
|
+
// keyword token at a word boundary, longest-first. Runs before the
|
|
289
|
+
// per-language extractors so natural spaced multi-word keywords tokenize
|
|
290
|
+
// without each tokenizer hardcoding a compound list. No-op for single-word
|
|
291
|
+
// and no-space (CJK) languages (multiWordKeywords is empty).
|
|
292
|
+
const multiWord = this.tryMultiWordKeyword(input, pos);
|
|
293
|
+
if (multiWord) {
|
|
294
|
+
tokens.push(multiWord);
|
|
295
|
+
pos = multiWord.position.end;
|
|
296
|
+
continue;
|
|
297
|
+
}
|
|
298
|
+
|
|
174
299
|
// Try registered extractors in order
|
|
175
300
|
let extracted = false;
|
|
176
301
|
for (const extractor of this.extractors) {
|
|
@@ -297,6 +422,7 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
|
|
|
297
422
|
): void {
|
|
298
423
|
// Use a Map to deduplicate, with extras taking precedence
|
|
299
424
|
const keywordMap = new Map<string, KeywordEntry>();
|
|
425
|
+
this.rawExtraEntries = extras;
|
|
300
426
|
|
|
301
427
|
// Extract from keywords (command translations)
|
|
302
428
|
if (profile.keywords) {
|
|
@@ -356,6 +482,20 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
|
|
|
356
482
|
}
|
|
357
483
|
}
|
|
358
484
|
|
|
485
|
+
// Register English DOM event names as universal fallbacks. The i18n grammar
|
|
486
|
+
// transformer has no native form for most DOM events, so it passes them
|
|
487
|
+
// through verbatim (`on keyup …` → `<on-marker> keyup …`). Without these,
|
|
488
|
+
// non-English token streams treat `keyup`/`keydown`/`resize`/… as bare
|
|
489
|
+
// identifiers, and event handlers using them (often with `[key==…]` guards)
|
|
490
|
+
// fail to parse. Guarded by `!has` so any native mapping wins (same policy
|
|
491
|
+
// as the English-reference fallbacks above). Generalizes the per-language
|
|
492
|
+
// registration introduced for Hebrew in #272.
|
|
493
|
+
for (const evt of ENGLISH_DOM_EVENT_NAMES) {
|
|
494
|
+
if (!keywordMap.has(evt)) {
|
|
495
|
+
keywordMap.set(evt, { native: evt, normalized: evt });
|
|
496
|
+
}
|
|
497
|
+
}
|
|
498
|
+
|
|
359
499
|
// Add extra entries (literals, positional, events) - these OVERRIDE profile entries
|
|
360
500
|
for (const extra of extras) {
|
|
361
501
|
keywordMap.set(extra.native, extra);
|
|
@@ -366,6 +506,20 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
|
|
|
366
506
|
(a, b) => b.native.length - a.native.length
|
|
367
507
|
);
|
|
368
508
|
|
|
509
|
+
// Multi-word (space-containing) keywords, for longest-phrase matching at a
|
|
510
|
+
// token boundary. Already longest-first (profileKeywords is sorted above).
|
|
511
|
+
// Marker/modifier concepts are EXCLUDED: those are matched positionally by
|
|
512
|
+
// the pattern matcher (role markers), and greedily consuming a multi-word
|
|
513
|
+
// marker phrase shadows the single-word marker patterns rely on — e.g. id
|
|
514
|
+
// `ke dalam` (into) would swallow the `ke` destination marker, and ko `할 때`
|
|
515
|
+
// (eventMarker) would pre-empt the SOV event extraction. Command verbs,
|
|
516
|
+
// control-flow, and event-name keywords (vi `với mỗi`=for, es `tecla abajo`=
|
|
517
|
+
// keydown, bn `তৈরি করুন`=make) are kept — the pattern matcher treats those
|
|
518
|
+
// as keyword literals, so one-token matching is strictly better.
|
|
519
|
+
this.multiWordKeywords = this.profileKeywords.filter(
|
|
520
|
+
k => k.native.includes(' ') && !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
|
|
521
|
+
);
|
|
522
|
+
|
|
369
523
|
// Build Map for O(1) lookups (case-insensitive + diacritic-insensitive)
|
|
370
524
|
// This allows matching both 'بدّل' (with shadda) and 'بدل' (without) to the same entry
|
|
371
525
|
this.profileKeywordMap = new Map();
|
|
@@ -417,6 +571,40 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
|
|
|
417
571
|
return null;
|
|
418
572
|
}
|
|
419
573
|
|
|
574
|
+
/**
|
|
575
|
+
* Match the longest multi-word (space-containing) profile keyword at `pos`,
|
|
576
|
+
* requiring the match to end at a word boundary. The profile-driven
|
|
577
|
+
* counterpart of the per-language hardcoded compound lists (the hindi and
|
|
578
|
+
* vietnamese keyword extractors). Returns a keyword token (with the normalized
|
|
579
|
+
* form) or null. Case-sensitive against the stored native form, mirroring
|
|
580
|
+
* `tryProfileKeyword`/`isKeywordStart` (the i18n dicts emit a fixed surface
|
|
581
|
+
* case). No-op when `multiWordKeywords` is empty (no-space/CJK languages).
|
|
582
|
+
*
|
|
583
|
+
* @param input - Input string
|
|
584
|
+
* @param pos - Current position (must be a token-start boundary)
|
|
585
|
+
* @param isWordChar - End-boundary predicate (defaults to Unicode letter/digit/_)
|
|
586
|
+
*/
|
|
587
|
+
protected tryMultiWordKeyword(
|
|
588
|
+
input: string,
|
|
589
|
+
pos: number,
|
|
590
|
+
isWordChar: (char: string) => boolean = ch => /[\p{L}\p{N}_]/u.test(ch)
|
|
591
|
+
): LanguageToken | null {
|
|
592
|
+
if (this.multiWordKeywords.length === 0) return null;
|
|
593
|
+
const rest = input.slice(pos);
|
|
594
|
+
for (const entry of this.multiWordKeywords) {
|
|
595
|
+
if (!rest.startsWith(entry.native)) continue;
|
|
596
|
+
const after = input[pos + entry.native.length];
|
|
597
|
+
if (after !== undefined && isWordChar(after)) continue; // not a word boundary
|
|
598
|
+
return createToken(
|
|
599
|
+
entry.native,
|
|
600
|
+
'keyword',
|
|
601
|
+
createPosition(pos, pos + entry.native.length),
|
|
602
|
+
entry.normalized
|
|
603
|
+
);
|
|
604
|
+
}
|
|
605
|
+
return null;
|
|
606
|
+
}
|
|
607
|
+
|
|
420
608
|
/**
|
|
421
609
|
* Check if the remaining input starts with any known keyword.
|
|
422
610
|
* Useful for non-space languages to detect word boundaries.
|
|
@@ -430,6 +618,37 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
|
|
|
430
618
|
return this.profileKeywords.some(entry => remaining.startsWith(entry.native));
|
|
431
619
|
}
|
|
432
620
|
|
|
621
|
+
/**
|
|
622
|
+
* Check if a known keyword starts at the given position AND ends at a word
|
|
623
|
+
* boundary (end of input or a non-word character).
|
|
624
|
+
*
|
|
625
|
+
* Space-delimited languages must use this (not `isKeywordStart`) for
|
|
626
|
+
* word-walk break checks: the keyword table includes English canonical
|
|
627
|
+
* fallbacks (me, it, you, …), so a raw `startsWith` check splits any native
|
|
628
|
+
* word with an embedded fallback mid-word (e.g. Quechua ñit'iy contains
|
|
629
|
+
* "it"). CJK/no-space tokenizers rely on mid-text keyword starts and must
|
|
630
|
+
* keep using `isKeywordStart`.
|
|
631
|
+
*
|
|
632
|
+
* @param input - Input string
|
|
633
|
+
* @param pos - Current position
|
|
634
|
+
* @param isWordChar - Language-specific word-character predicate; pass the
|
|
635
|
+
* tokenizer's letter classifier so e.g. the Quechua glottal apostrophe
|
|
636
|
+
* counts as part of a word. Defaults to Unicode letters/digits/underscore.
|
|
637
|
+
* @returns true if a keyword starts here and is not followed by a word char
|
|
638
|
+
*/
|
|
639
|
+
protected isKeywordStartAtBoundary(
|
|
640
|
+
input: string,
|
|
641
|
+
pos: number,
|
|
642
|
+
isWordChar: (char: string) => boolean = ch => /[\p{L}\p{N}_]/u.test(ch)
|
|
643
|
+
): boolean {
|
|
644
|
+
const remaining = input.slice(pos);
|
|
645
|
+
return this.profileKeywords.some(entry => {
|
|
646
|
+
if (!remaining.startsWith(entry.native)) return false;
|
|
647
|
+
const after = input[pos + entry.native.length];
|
|
648
|
+
return after === undefined || !isWordChar(after);
|
|
649
|
+
});
|
|
650
|
+
}
|
|
651
|
+
|
|
433
652
|
/**
|
|
434
653
|
* Look up a keyword by native word (case-insensitive).
|
|
435
654
|
* O(1) lookup using the keyword map.
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Contract tests for BaseTokenizer.isKeywordStartAtBoundary().
|
|
3
|
+
*
|
|
4
|
+
* The keyword table injects English canonical reference words (me, it, you, …)
|
|
5
|
+
* for every language. Space-delimited tokenizers that break their word-walk on
|
|
6
|
+
* a raw isKeywordStart() therefore split native words around embedded English
|
|
7
|
+
* fallbacks (e.g. Quechua ñit'iy contains "it"). The boundary-aware variant
|
|
8
|
+
* only fires when the keyword match ends at a word boundary.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
import { describe, it, expect } from 'vitest';
|
|
12
|
+
import { BaseTokenizer } from './base-tokenizer';
|
|
13
|
+
import type { TokenKind } from '../types';
|
|
14
|
+
|
|
15
|
+
class ProbeTokenizer extends BaseTokenizer {
|
|
16
|
+
readonly language = 'xx';
|
|
17
|
+
readonly direction = 'ltr' as const;
|
|
18
|
+
|
|
19
|
+
constructor() {
|
|
20
|
+
super();
|
|
21
|
+
this.initializeKeywordsFromProfile({
|
|
22
|
+
keywords: { toggle: { primary: 'tikray' } },
|
|
23
|
+
// references inject the English canonical fallbacks "it" and "me" too
|
|
24
|
+
references: { it: 'pay', me: 'nuqa' },
|
|
25
|
+
});
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
classifyToken(): TokenKind {
|
|
29
|
+
return 'identifier';
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
probeRaw(input: string, pos: number): boolean {
|
|
33
|
+
return this.isKeywordStart(input, pos);
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
probeBoundary(input: string, pos: number, isWordChar?: (ch: string) => boolean): boolean {
|
|
37
|
+
return this.isKeywordStartAtBoundary(input, pos, isWordChar);
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
describe('isKeywordStartAtBoundary', () => {
|
|
42
|
+
const t = new ProbeTokenizer();
|
|
43
|
+
|
|
44
|
+
it('does not fire on an English fallback embedded mid-word', () => {
|
|
45
|
+
// "it" inside "umitaq" — raw check fires, boundary check must not
|
|
46
|
+
expect(t.probeRaw('umitaq', 2)).toBe(true);
|
|
47
|
+
expect(t.probeBoundary('umitaq', 2)).toBe(false);
|
|
48
|
+
// "me" inside "umema"
|
|
49
|
+
expect(t.probeRaw('umema', 1)).toBe(true);
|
|
50
|
+
expect(t.probeBoundary('umema', 1)).toBe(false);
|
|
51
|
+
});
|
|
52
|
+
|
|
53
|
+
it('fires when the keyword match ends at end of input', () => {
|
|
54
|
+
expect(t.probeBoundary('umit', 2)).toBe(true);
|
|
55
|
+
});
|
|
56
|
+
|
|
57
|
+
it('fires when the keyword match is followed by a non-word char', () => {
|
|
58
|
+
expect(t.probeBoundary('um it goes', 3)).toBe(true);
|
|
59
|
+
expect(t.probeBoundary('xit.', 1)).toBe(true);
|
|
60
|
+
});
|
|
61
|
+
|
|
62
|
+
it('respects a language-specific word-char predicate', () => {
|
|
63
|
+
// Default predicate: apostrophe is not a word char → boundary fires
|
|
64
|
+
expect(t.probeBoundary("xit'y", 1)).toBe(true);
|
|
65
|
+
// Quechua-style predicate counting the glottal apostrophe → no boundary
|
|
66
|
+
const quechuaLike = (ch: string) => /[a-zñ'’]/i.test(ch);
|
|
67
|
+
expect(t.probeBoundary("xit'y", 1, quechuaLike)).toBe(false);
|
|
68
|
+
});
|
|
69
|
+
|
|
70
|
+
it('returns false when no keyword starts at the position at all', () => {
|
|
71
|
+
expect(t.probeBoundary('umitaq', 0)).toBe(false);
|
|
72
|
+
});
|
|
73
|
+
});
|
|
@@ -406,6 +406,21 @@ export interface TokenizerContext {
|
|
|
406
406
|
*/
|
|
407
407
|
isKeywordStart(input: string, position: number): boolean;
|
|
408
408
|
|
|
409
|
+
/**
|
|
410
|
+
* Like `isKeywordStart`, but only true when the keyword match ends at a
|
|
411
|
+
* word boundary (end of input or a char rejected by `isWordChar`).
|
|
412
|
+
* Space-delimited languages must use this for word-walk break checks —
|
|
413
|
+
* the keyword table includes English canonical fallbacks (me, it, you, …),
|
|
414
|
+
* so the raw check splits native words mid-word (e.g. Quechua ñit'iy
|
|
415
|
+
* contains "it"). Optional for backward compatibility with hand-rolled
|
|
416
|
+
* contexts; callers should treat absence as "no boundary keyword here".
|
|
417
|
+
*/
|
|
418
|
+
isKeywordStartAtBoundary?(
|
|
419
|
+
input: string,
|
|
420
|
+
position: number,
|
|
421
|
+
isWordChar?: (char: string) => boolean
|
|
422
|
+
): boolean;
|
|
423
|
+
|
|
409
424
|
/**
|
|
410
425
|
* Optional morphological normalizer for this language.
|
|
411
426
|
*/
|
|
@@ -449,6 +464,11 @@ export function createTokenizerContext(tokenizer: {
|
|
|
449
464
|
lookupKeyword(native: string): KeywordEntry | undefined;
|
|
450
465
|
isKeyword(native: string): boolean;
|
|
451
466
|
isKeywordStart(input: string, position: number): boolean;
|
|
467
|
+
isKeywordStartAtBoundary?(
|
|
468
|
+
input: string,
|
|
469
|
+
position: number,
|
|
470
|
+
isWordChar?: (char: string) => boolean
|
|
471
|
+
): boolean;
|
|
452
472
|
normalizer?: MorphologicalNormalizer;
|
|
453
473
|
}): TokenizerContext {
|
|
454
474
|
const ctx: TokenizerContext = {
|
|
@@ -457,6 +477,9 @@ export function createTokenizerContext(tokenizer: {
|
|
|
457
477
|
lookupKeyword: tokenizer.lookupKeyword.bind(tokenizer),
|
|
458
478
|
isKeyword: tokenizer.isKeyword.bind(tokenizer),
|
|
459
479
|
isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer),
|
|
480
|
+
...(tokenizer.isKeywordStartAtBoundary
|
|
481
|
+
? { isKeywordStartAtBoundary: tokenizer.isKeywordStartAtBoundary.bind(tokenizer) }
|
|
482
|
+
: {}),
|
|
460
483
|
};
|
|
461
484
|
|
|
462
485
|
if (tokenizer.normalizer) {
|