@lokascript/framework 2.6.0 → 2.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1 +1,1550 @@
1
+ // src/interfaces/value-extractor.ts
2
+ var StringLiteralExtractor = class {
3
+ constructor() {
4
+ this.name = "string-literal";
5
+ }
6
+ canExtract(input, position) {
7
+ const char = input[position];
8
+ return char === '"' || char === "'" || char === "`" || char === "\u201C" || // Chinese double quote open "
9
+ char === "\u2018";
10
+ }
11
+ extract(input, position) {
12
+ const quote = input[position];
13
+ if (quote === "\u201C") {
14
+ let length2 = 1;
15
+ while (position + length2 < input.length) {
16
+ if (input[position + length2] === "\u201D") {
17
+ length2++;
18
+ return { value: input.substring(position, position + length2), length: length2 };
19
+ }
20
+ length2++;
21
+ }
22
+ return null;
23
+ }
24
+ if (quote === "\u2018") {
25
+ let length2 = 1;
26
+ while (position + length2 < input.length) {
27
+ if (input[position + length2] === "\u2019") {
28
+ length2++;
29
+ return { value: input.substring(position, position + length2), length: length2 };
30
+ }
31
+ length2++;
32
+ }
33
+ return null;
34
+ }
35
+ let length = 1;
36
+ let escaped = false;
37
+ while (position + length < input.length) {
38
+ const char = input[position + length];
39
+ if (escaped) {
40
+ escaped = false;
41
+ length++;
42
+ continue;
43
+ }
44
+ if (char === "\\") {
45
+ escaped = true;
46
+ length++;
47
+ continue;
48
+ }
49
+ if (char === quote) {
50
+ length++;
51
+ return {
52
+ value: input.substring(position, position + length),
53
+ length
54
+ };
55
+ }
56
+ length++;
57
+ }
58
+ return null;
59
+ }
60
+ };
61
+ var NumberExtractor = class {
62
+ constructor() {
63
+ this.name = "number";
64
+ }
65
+ canExtract(input, position) {
66
+ return /\d/.test(input[position]);
67
+ }
68
+ extract(input, position) {
69
+ let length = 0;
70
+ let hasDecimal = false;
71
+ while (position + length < input.length) {
72
+ const char = input[position + length];
73
+ if (/\d/.test(char)) {
74
+ length++;
75
+ } else if (char === "." && !hasDecimal) {
76
+ hasDecimal = true;
77
+ length++;
78
+ } else {
79
+ break;
80
+ }
81
+ }
82
+ if (length === 0) return null;
83
+ const numValue = input.substring(position, position + length);
84
+ const afterNum = position + length;
85
+ if (afterNum < input.length) {
86
+ const remaining = input.slice(afterNum);
87
+ const cjkMultiUnits = [
88
+ { pattern: "\u6BEB\u79D2", suffix: "ms" },
89
+ // Chinese milliseconds
90
+ { pattern: "\u5206\u949F", suffix: "m" },
91
+ // Chinese minutes
92
+ { pattern: "\u5C0F\u65F6", suffix: "h" },
93
+ // Chinese hours
94
+ { pattern: "\u30DF\u30EA\u79D2", suffix: "ms" },
95
+ // Japanese milliseconds
96
+ { pattern: "\u6642\u9593", suffix: "h" }
97
+ // Japanese hours
98
+ ];
99
+ for (const unit of cjkMultiUnits) {
100
+ if (remaining.startsWith(unit.pattern)) {
101
+ return {
102
+ value: numValue + unit.suffix,
103
+ length: length + unit.pattern.length,
104
+ metadata: { hasTimeUnit: true }
105
+ };
106
+ }
107
+ }
108
+ if (remaining.startsWith("ms")) {
109
+ return {
110
+ value: numValue + "ms",
111
+ length: length + 2,
112
+ metadata: { hasTimeUnit: true }
113
+ };
114
+ }
115
+ const cjkSingleUnits = [
116
+ { pattern: "\u79D2", suffix: "s" },
117
+ // CJK seconds
118
+ { pattern: "\u5206", suffix: "m" }
119
+ // CJK minutes
120
+ ];
121
+ for (const unit of cjkSingleUnits) {
122
+ if (remaining.startsWith(unit.pattern)) {
123
+ return {
124
+ value: numValue + unit.suffix,
125
+ length: length + 1,
126
+ metadata: { hasTimeUnit: true }
127
+ };
128
+ }
129
+ }
130
+ if (/^[smh](?![a-zA-Z])/.test(remaining)) {
131
+ return {
132
+ value: numValue + remaining[0],
133
+ length: length + 1,
134
+ metadata: { hasTimeUnit: true }
135
+ };
136
+ }
137
+ }
138
+ return { value: numValue, length };
139
+ }
140
+ };
141
+ var IdentifierExtractor = class {
142
+ constructor() {
143
+ this.name = "identifier";
144
+ }
145
+ canExtract(input, position) {
146
+ return /[a-zA-Z_]/.test(input[position]);
147
+ }
148
+ extract(input, position) {
149
+ let length = 0;
150
+ while (position + length < input.length) {
151
+ const char = input[position + length];
152
+ if (/[a-zA-Z0-9_]/.test(char)) {
153
+ length++;
154
+ } else {
155
+ break;
156
+ }
157
+ }
158
+ return length > 0 ? {
159
+ value: input.substring(position, position + length),
160
+ length
161
+ } : null;
162
+ }
163
+ };
164
+ var UnicodeIdentifierExtractor = class {
165
+ constructor() {
166
+ this.name = "unicode-identifier";
167
+ }
168
+ canExtract(input, position) {
169
+ const code = input.charCodeAt(position);
170
+ if (code < 128) return false;
171
+ return /\p{L}/u.test(input[position]);
172
+ }
173
+ extract(input, position) {
174
+ let length = 0;
175
+ while (position + length < input.length) {
176
+ const char = input[position + length];
177
+ if (/[\p{L}\p{N}\p{M}]/u.test(char)) {
178
+ length++;
179
+ } else {
180
+ break;
181
+ }
182
+ }
183
+ return length > 0 ? { value: input.substring(position, position + length), length } : null;
184
+ }
185
+ };
186
+ var LatinExtendedIdentifierExtractor = class {
187
+ constructor() {
188
+ this.name = "latin-extended-identifier";
189
+ }
190
+ canExtract(input, position) {
191
+ return /\p{L}/u.test(input[position]);
192
+ }
193
+ extract(input, position) {
194
+ let end = position;
195
+ while (end < input.length && /[\p{L}\p{N}_-]/u.test(input[end])) {
196
+ end++;
197
+ }
198
+ if (end === position) return null;
199
+ return { value: input.slice(position, end), length: end - position };
200
+ }
201
+ };
202
+ function isContextAwareExtractor(extractor) {
203
+ return "setContext" in extractor && typeof extractor.setContext === "function";
204
+ }
205
+ function createTokenizerContext(tokenizer) {
206
+ const ctx = {
207
+ language: tokenizer.language,
208
+ direction: tokenizer.direction,
209
+ lookupKeyword: tokenizer.lookupKeyword.bind(tokenizer),
210
+ isKeyword: tokenizer.isKeyword.bind(tokenizer),
211
+ isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer),
212
+ ...tokenizer.isKeywordStartAtBoundary ? { isKeywordStartAtBoundary: tokenizer.isKeywordStartAtBoundary.bind(tokenizer) } : {}
213
+ };
214
+ if (tokenizer.normalizer) {
215
+ return { ...ctx, normalizer: tokenizer.normalizer };
216
+ }
217
+ return ctx;
218
+ }
219
+
220
+ // src/core/tokenization/token-utils.ts
221
+ var TokenStreamImpl = class {
222
+ constructor(tokens, language) {
223
+ this.pos = 0;
224
+ this.tokens = tokens;
225
+ this.language = language;
226
+ }
227
+ peek(offset = 0) {
228
+ const index = this.pos + offset;
229
+ if (index < 0 || index >= this.tokens.length) {
230
+ return null;
231
+ }
232
+ return this.tokens[index];
233
+ }
234
+ advance() {
235
+ if (this.isAtEnd()) {
236
+ throw new Error("Unexpected end of token stream");
237
+ }
238
+ return this.tokens[this.pos++];
239
+ }
240
+ isAtEnd() {
241
+ return this.pos >= this.tokens.length;
242
+ }
243
+ mark() {
244
+ return { position: this.pos };
245
+ }
246
+ reset(mark) {
247
+ this.pos = mark.position;
248
+ }
249
+ position() {
250
+ return this.pos;
251
+ }
252
+ /**
253
+ * Get remaining tokens as an array.
254
+ */
255
+ remaining() {
256
+ return this.tokens.slice(this.pos);
257
+ }
258
+ /**
259
+ * Consume tokens while predicate is true.
260
+ */
261
+ takeWhile(predicate) {
262
+ const result = [];
263
+ while (!this.isAtEnd() && predicate(this.peek())) {
264
+ result.push(this.advance());
265
+ }
266
+ return result;
267
+ }
268
+ /**
269
+ * Skip tokens while predicate is true.
270
+ */
271
+ skipWhile(predicate) {
272
+ while (!this.isAtEnd() && predicate(this.peek())) {
273
+ this.advance();
274
+ }
275
+ }
276
+ };
277
+ function createPosition(start, end) {
278
+ return { start, end };
279
+ }
280
+ function createToken(valueOrParams, kind, position, normalizedOrOptions) {
281
+ if (typeof valueOrParams === "object") {
282
+ const { value: value2, kind: kind2, position: position2, normalized, stem, stemConfidence, metadata } = valueOrParams;
283
+ return {
284
+ value: value2,
285
+ kind: kind2,
286
+ position: position2,
287
+ ...normalized !== void 0 && { normalized },
288
+ ...stem !== void 0 && { stem },
289
+ ...stemConfidence !== void 0 && { stemConfidence },
290
+ ...metadata !== void 0 && { metadata }
291
+ };
292
+ }
293
+ const value = valueOrParams;
294
+ if (!kind || !position) {
295
+ throw new Error("createToken requires kind and position parameters");
296
+ }
297
+ if (typeof normalizedOrOptions === "string") {
298
+ return { value, kind, position, normalized: normalizedOrOptions };
299
+ }
300
+ if (normalizedOrOptions) {
301
+ const { normalized, stem, stemConfidence, metadata } = normalizedOrOptions;
302
+ return {
303
+ value,
304
+ kind,
305
+ position,
306
+ ...normalized !== void 0 && { normalized },
307
+ ...stem !== void 0 && { stem },
308
+ ...stemConfidence !== void 0 && { stemConfidence },
309
+ ...metadata !== void 0 && { metadata }
310
+ };
311
+ }
312
+ return { value, kind, position };
313
+ }
314
+ function isWhitespace(char) {
315
+ return /\s/.test(char);
316
+ }
317
+ function isSelectorStart(char) {
318
+ return char === "#" || char === "." || char === "[" || char === "@" || char === "*" || char === "<";
319
+ }
320
+ function isQuote(char) {
321
+ return char === '"' || char === "'" || char === "`" || char === "\u300C" || char === "\u300D";
322
+ }
323
+ function isDigit(char) {
324
+ return /\d/.test(char);
325
+ }
326
+ function isAsciiLetter(char) {
327
+ return /[a-zA-Z]/.test(char);
328
+ }
329
+ function isAsciiIdentifierChar(char) {
330
+ return /[a-zA-Z0-9_-]/.test(char);
331
+ }
332
+
333
+ // src/core/tokenization/extractors.ts
334
+ function extractCssSelector(input, startPos) {
335
+ if (startPos >= input.length) return null;
336
+ const char = input[startPos];
337
+ if (!isSelectorStart(char)) return null;
338
+ let pos = startPos;
339
+ let selector = "";
340
+ if (char === "#" || char === ".") {
341
+ selector += input[pos++];
342
+ while (pos < input.length && isAsciiIdentifierChar(input[pos])) {
343
+ selector += input[pos++];
344
+ }
345
+ if (selector.length <= 1) return null;
346
+ if (pos < input.length && input[pos] === "." && char === "#") {
347
+ const methodStart = pos + 1;
348
+ let methodEnd = methodStart;
349
+ while (methodEnd < input.length && isAsciiIdentifierChar(input[methodEnd])) {
350
+ methodEnd++;
351
+ }
352
+ if (methodEnd < input.length && input[methodEnd] === "(") {
353
+ return selector;
354
+ }
355
+ }
356
+ } else if (char === "[") {
357
+ let depth = 1;
358
+ let inQuote = false;
359
+ let quoteChar = null;
360
+ let escaped = false;
361
+ selector += input[pos++];
362
+ while (pos < input.length && depth > 0) {
363
+ const c = input[pos];
364
+ selector += c;
365
+ if (escaped) {
366
+ escaped = false;
367
+ } else if (c === "\\") {
368
+ escaped = true;
369
+ } else if (inQuote) {
370
+ if (c === quoteChar) {
371
+ inQuote = false;
372
+ quoteChar = null;
373
+ }
374
+ } else {
375
+ if (c === '"' || c === "'" || c === "`") {
376
+ inQuote = true;
377
+ quoteChar = c;
378
+ } else if (c === "[") {
379
+ depth++;
380
+ } else if (c === "]") {
381
+ depth--;
382
+ }
383
+ }
384
+ pos++;
385
+ }
386
+ if (depth !== 0) return null;
387
+ } else if (char === "@") {
388
+ selector += input[pos++];
389
+ while (pos < input.length && isAsciiIdentifierChar(input[pos])) {
390
+ selector += input[pos++];
391
+ }
392
+ if (selector.length <= 1) return null;
393
+ } else if (char === "*") {
394
+ selector += input[pos++];
395
+ while (pos < input.length && isAsciiIdentifierChar(input[pos])) {
396
+ selector += input[pos++];
397
+ }
398
+ if (selector.length <= 1) return null;
399
+ } else if (char === "<") {
400
+ selector += input[pos++];
401
+ if (pos >= input.length || !isAsciiLetter(input[pos])) return null;
402
+ while (pos < input.length && isAsciiIdentifierChar(input[pos])) {
403
+ selector += input[pos++];
404
+ }
405
+ while (pos < input.length) {
406
+ const modChar = input[pos];
407
+ if (modChar === ".") {
408
+ selector += input[pos++];
409
+ if (pos >= input.length || !isAsciiIdentifierChar(input[pos])) {
410
+ return null;
411
+ }
412
+ while (pos < input.length && isAsciiIdentifierChar(input[pos])) {
413
+ selector += input[pos++];
414
+ }
415
+ } else if (modChar === "#") {
416
+ selector += input[pos++];
417
+ if (pos >= input.length || !isAsciiIdentifierChar(input[pos])) {
418
+ return null;
419
+ }
420
+ while (pos < input.length && isAsciiIdentifierChar(input[pos])) {
421
+ selector += input[pos++];
422
+ }
423
+ } else if (modChar === "[") {
424
+ let depth = 1;
425
+ let inQuote = false;
426
+ let quoteChar = null;
427
+ let escaped = false;
428
+ selector += input[pos++];
429
+ while (pos < input.length && depth > 0) {
430
+ const c = input[pos];
431
+ selector += c;
432
+ if (escaped) {
433
+ escaped = false;
434
+ } else if (c === "\\") {
435
+ escaped = true;
436
+ } else if (inQuote) {
437
+ if (c === quoteChar) {
438
+ inQuote = false;
439
+ quoteChar = null;
440
+ }
441
+ } else {
442
+ if (c === '"' || c === "'" || c === "`") {
443
+ inQuote = true;
444
+ quoteChar = c;
445
+ } else if (c === "[") {
446
+ depth++;
447
+ } else if (c === "]") {
448
+ depth--;
449
+ }
450
+ }
451
+ pos++;
452
+ }
453
+ if (depth !== 0) return null;
454
+ } else {
455
+ break;
456
+ }
457
+ }
458
+ while (pos < input.length && isWhitespace(input[pos])) {
459
+ selector += input[pos++];
460
+ }
461
+ if (pos < input.length && input[pos] === "/") {
462
+ selector += input[pos++];
463
+ while (pos < input.length && isWhitespace(input[pos])) {
464
+ selector += input[pos++];
465
+ }
466
+ }
467
+ if (pos >= input.length || input[pos] !== ">") return null;
468
+ selector += input[pos++];
469
+ }
470
+ return selector || null;
471
+ }
472
+ function isPossessiveMarker(input, pos) {
473
+ if (pos >= input.length || input[pos] !== "'") return false;
474
+ if (pos + 1 >= input.length) return false;
475
+ const nextChar = input[pos + 1].toLowerCase();
476
+ if (nextChar !== "s") return false;
477
+ if (pos + 2 >= input.length) return true;
478
+ const afterS = input[pos + 2];
479
+ return isWhitespace(afterS) || afterS === "*" || !isAsciiIdentifierChar(afterS);
480
+ }
481
+ function extractStringLiteral(input, startPos) {
482
+ if (startPos >= input.length) return null;
483
+ const openQuote = input[startPos];
484
+ if (!isQuote(openQuote)) return null;
485
+ if (openQuote === "'" && isPossessiveMarker(input, startPos)) {
486
+ return null;
487
+ }
488
+ const closeQuoteMap = {
489
+ '"': '"',
490
+ "'": "'",
491
+ "`": "`",
492
+ "\u300C": "\u300D"
493
+ };
494
+ const closeQuote = closeQuoteMap[openQuote];
495
+ if (!closeQuote) return null;
496
+ let pos = startPos + 1;
497
+ let literal = openQuote;
498
+ let escaped = false;
499
+ while (pos < input.length) {
500
+ const char = input[pos];
501
+ literal += char;
502
+ if (escaped) {
503
+ escaped = false;
504
+ } else if (char === "\\") {
505
+ escaped = true;
506
+ } else if (char === closeQuote) {
507
+ return literal;
508
+ }
509
+ pos++;
510
+ }
511
+ return literal;
512
+ }
513
+ function isUrlStart(input, pos) {
514
+ if (pos >= input.length) return false;
515
+ const char = input[pos];
516
+ const next = input[pos + 1] || "";
517
+ const third = input[pos + 2] || "";
518
+ if (char === "/" && next !== "/" && /[a-zA-Z0-9._-]/.test(next)) {
519
+ return true;
520
+ }
521
+ if (char === "/" && next === "/" && /[a-zA-Z]/.test(third)) {
522
+ return true;
523
+ }
524
+ if (char === "." && (next === "/" || next === "." && third === "/")) {
525
+ return true;
526
+ }
527
+ const slice = input.slice(pos, pos + 8).toLowerCase();
528
+ if (slice.startsWith("http://") || slice.startsWith("https://")) {
529
+ return true;
530
+ }
531
+ return false;
532
+ }
533
+ function extractUrl(input, startPos) {
534
+ if (!isUrlStart(input, startPos)) return null;
535
+ let pos = startPos;
536
+ let url = "";
537
+ const urlChars = /[a-zA-Z0-9/:._\-?&=%@+~!$'()*,;[\]]/;
538
+ while (pos < input.length) {
539
+ const char = input[pos];
540
+ if (char === "#") {
541
+ if (url.length > 0 && /[a-zA-Z0-9/.]$/.test(url)) {
542
+ url += char;
543
+ pos++;
544
+ while (pos < input.length && /[a-zA-Z0-9_-]/.test(input[pos])) {
545
+ url += input[pos++];
546
+ }
547
+ }
548
+ break;
549
+ }
550
+ if (urlChars.test(char)) {
551
+ url += char;
552
+ pos++;
553
+ } else {
554
+ break;
555
+ }
556
+ }
557
+ if (url.length < 2) return null;
558
+ return url;
559
+ }
560
+ function extractNumber(input, startPos) {
561
+ if (startPos >= input.length) return null;
562
+ const char = input[startPos];
563
+ if (!isDigit(char) && char !== "-" && char !== "+") return null;
564
+ let pos = startPos;
565
+ let number = "";
566
+ if (input[pos] === "-" || input[pos] === "+") {
567
+ number += input[pos++];
568
+ }
569
+ if (pos >= input.length || !isDigit(input[pos])) {
570
+ return null;
571
+ }
572
+ while (pos < input.length && isDigit(input[pos])) {
573
+ number += input[pos++];
574
+ }
575
+ if (pos < input.length && input[pos] === ".") {
576
+ number += input[pos++];
577
+ while (pos < input.length && isDigit(input[pos])) {
578
+ number += input[pos++];
579
+ }
580
+ }
581
+ if (pos < input.length) {
582
+ const suffix = input.slice(pos, pos + 2);
583
+ if (suffix === "ms") {
584
+ number += "ms";
585
+ } else if (input[pos] === "s" || input[pos] === "m" || input[pos] === "h") {
586
+ number += input[pos];
587
+ }
588
+ }
589
+ return number;
590
+ }
591
+
592
+ // src/core/tokenization/extractors/operator.ts
593
+ var DEFAULT_OPERATORS = [
594
+ // Three-character operators
595
+ "===",
596
+ "!==",
597
+ "->",
598
+ // Two-character operators
599
+ "==",
600
+ "!=",
601
+ "<=",
602
+ ">=",
603
+ "&&",
604
+ "||",
605
+ "**",
606
+ "+=",
607
+ "-=",
608
+ "*=",
609
+ "/=",
610
+ // Single-character operators
611
+ "+",
612
+ "-",
613
+ "*",
614
+ "/",
615
+ "=",
616
+ ">",
617
+ "<",
618
+ "!",
619
+ "&",
620
+ "|",
621
+ "%",
622
+ "^",
623
+ "~"
624
+ ];
625
+ var OperatorExtractor = class {
626
+ constructor(operators = DEFAULT_OPERATORS) {
627
+ this.operators = operators;
628
+ this.name = "operator";
629
+ this.operators = [...operators].sort((a, b) => b.length - a.length);
630
+ }
631
+ canExtract(input, position) {
632
+ return this.operators.some((op) => input.startsWith(op, position));
633
+ }
634
+ extract(input, position) {
635
+ for (const op of this.operators) {
636
+ if (input.startsWith(op, position)) {
637
+ return {
638
+ value: op,
639
+ length: op.length
640
+ };
641
+ }
642
+ }
643
+ return null;
644
+ }
645
+ };
646
+
647
+ // src/core/tokenization/extractors/punctuation.ts
648
+ var DEFAULT_PUNCTUATION = "()[]{},:;";
649
+ var PunctuationExtractor = class {
650
+ constructor(punctuation = DEFAULT_PUNCTUATION) {
651
+ this.punctuation = punctuation;
652
+ this.name = "punctuation";
653
+ }
654
+ canExtract(input, position) {
655
+ return this.punctuation.includes(input[position]);
656
+ }
657
+ extract(input, position) {
658
+ const char = input[position];
659
+ if (this.punctuation.includes(char)) {
660
+ return {
661
+ value: char,
662
+ length: 1
663
+ };
664
+ }
665
+ return null;
666
+ }
667
+ };
668
+
669
+ // src/core/tokenization/default-extractors.ts
670
+ function getDefaultExtractors() {
671
+ return [
672
+ new StringLiteralExtractor(),
673
+ // "strings", 'strings', `strings`
674
+ new NumberExtractor(),
675
+ // 123, 45.67
676
+ new OperatorExtractor(),
677
+ // +, -, *, /, =, >, <, etc.
678
+ new PunctuationExtractor(),
679
+ // ( ) [ ] { } , : ;
680
+ new IdentifierExtractor(),
681
+ // variable_names, functionNames (ASCII)
682
+ new UnicodeIdentifierExtractor()
683
+ // CJK, Arabic, Cyrillic, etc.
684
+ ];
685
+ }
686
+
687
+ // src/core/tokenization/base-tokenizer.ts
688
+ var SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
689
+ var MARKER_CONCEPT_NORMALIZEDS = /* @__PURE__ */ new Set([
690
+ // Role-marker role names (profile.roleMarkers normalizeds)
691
+ "patient",
692
+ "destination",
693
+ "source",
694
+ "style",
695
+ "event",
696
+ "eventMarker",
697
+ "agent",
698
+ "goal",
699
+ "manner",
700
+ // Prepositional / positional modifier concepts matched via the role mechanism
701
+ // (profile.keywords "Modifiers"). `before`/`after`/`until` are intentionally
702
+ // NOT here — they are pattern literals (see the note above).
703
+ "into",
704
+ "from",
705
+ "to",
706
+ "with",
707
+ "at",
708
+ "of",
709
+ "as",
710
+ "by",
711
+ "in",
712
+ "on",
713
+ "over",
714
+ "under",
715
+ "between",
716
+ "through",
717
+ "without"
718
+ ]);
719
+ var ENGLISH_DOM_EVENT_NAMES = [
720
+ "click",
721
+ "dblclick",
722
+ "input",
723
+ "change",
724
+ "submit",
725
+ "keydown",
726
+ "keyup",
727
+ "keypress",
728
+ "mousedown",
729
+ "mouseup",
730
+ "mouseover",
731
+ "mouseout",
732
+ "mouseenter",
733
+ "mouseleave",
734
+ "mousemove",
735
+ "pointerdown",
736
+ "pointerup",
737
+ "pointermove",
738
+ "focus",
739
+ "blur",
740
+ "load",
741
+ "resize",
742
+ "scroll"
743
+ ];
744
+ var _BaseTokenizer = class _BaseTokenizer {
745
+ constructor() {
746
+ /** Keywords derived from profile, sorted longest-first for greedy matching */
747
+ this.profileKeywords = [];
748
+ /**
749
+ * Space-containing profile keywords (multi-word phrases), longest-first.
750
+ * Used by `tryMultiWordKeyword` so natural spaced forms (hi `मेल खाता`,
751
+ * vi `chuyển đổi`, es `tecla abajo`, …) tokenize as ONE keyword — the
752
+ * profile-driven replacement for the per-language hardcoded compound lists.
753
+ * Empty for no-space (CJK) languages, so they are unaffected.
754
+ */
755
+ this.multiWordKeywords = [];
756
+ /** Map for O(1) keyword lookups by lowercase native word */
757
+ this.profileKeywordMap = /* @__PURE__ */ new Map();
758
+ /**
759
+ * The raw EXTRAS list passed to initializeKeywordsFromProfile, kept pre-dedup.
760
+ * The keyword map is keyed by native word with last-wins insertion, so a
761
+ * duplicate native word inside the extras silently shadows the earlier entry
762
+ * (e.g. a `nächste→closest` entry shadowing `nächste→next` broke German
763
+ * positional expressions). Exposed so consistency tests can detect such
764
+ * intra-extras collisions, which are invisible in the deduplicated map.
765
+ */
766
+ this.rawExtraEntries = [];
767
+ /**
768
+ * Pluggable value extractors for domain-specific syntax.
769
+ * When registered, BaseTokenizer will use extractor-based tokenization instead of legacy methods.
770
+ */
771
+ this.extractors = [];
772
+ }
773
+ /** Raw extras as passed in, pre-dedup — for consistency tests. */
774
+ getExtraKeywordEntries() {
775
+ return this.rawExtraEntries;
776
+ }
777
+ /**
778
+ * Tokenize input string to token stream.
779
+ * Delegates to extractor-based tokenization if extractors are registered,
780
+ * otherwise subclass must override this method.
781
+ *
782
+ * @param input - Input string to tokenize
783
+ * @returns Token stream
784
+ */
785
+ tokenize(input) {
786
+ if (this.isUsingExtractors()) {
787
+ return this.tokenizeWithExtractors(input);
788
+ }
789
+ throw new Error(
790
+ `${this.constructor.name}: tokenize() not implemented and no extractors registered. Either register extractors or override tokenize() method.`
791
+ );
792
+ }
793
+ /**
794
+ * Register a value extractor for domain-specific syntax.
795
+ * Extractors are tried in registration order during tokenization.
796
+ * Context-aware extractors automatically receive the tokenizer context.
797
+ *
798
+ * @param extractor - Value extractor to register
799
+ */
800
+ registerExtractor(extractor) {
801
+ if (isContextAwareExtractor(extractor)) {
802
+ extractor.setContext(createTokenizerContext(this));
803
+ }
804
+ this.extractors.push(extractor);
805
+ }
806
+ /**
807
+ * Register multiple value extractors at once.
808
+ *
809
+ * @param extractors - Array of value extractors to register
810
+ */
811
+ registerExtractors(extractors) {
812
+ for (const extractor of extractors) {
813
+ this.registerExtractor(extractor);
814
+ }
815
+ }
816
+ /**
817
+ * Clear all registered extractors.
818
+ * Returns tokenizer to legacy mode.
819
+ */
820
+ clearExtractors() {
821
+ this.extractors = [];
822
+ }
823
+ /**
824
+ * Check if this tokenizer is using extractor-based tokenization.
825
+ * Returns true if any extractors are registered.
826
+ */
827
+ isUsingExtractors() {
828
+ return this.extractors.length > 0;
829
+ }
830
+ /**
831
+ * Tokenize input using registered value extractors.
832
+ * This is the new path - extractors handle all syntax detection.
833
+ *
834
+ * @param input - Input string to tokenize
835
+ * @returns Token stream
836
+ */
837
+ tokenizeWithExtractors(input) {
838
+ const tokens = [];
839
+ let pos = 0;
840
+ while (pos < input.length) {
841
+ while (pos < input.length && isWhitespace(input[pos])) {
842
+ pos++;
843
+ }
844
+ if (pos >= input.length) break;
845
+ const multiWord = this.tryMultiWordKeyword(input, pos);
846
+ if (multiWord) {
847
+ tokens.push(multiWord);
848
+ pos = multiWord.position.end;
849
+ continue;
850
+ }
851
+ let extracted = false;
852
+ for (const extractor of this.extractors) {
853
+ if (extractor.canExtract(input, pos)) {
854
+ const result = extractor.extract(input, pos);
855
+ if (result) {
856
+ const normalized = result.metadata?.normalized;
857
+ const stem = result.metadata?.stem;
858
+ const stemConfidence = result.metadata?.stemConfidence;
859
+ const cleanMetadata = {};
860
+ if (result.metadata) {
861
+ for (const [key, value] of Object.entries(result.metadata)) {
862
+ if (key !== "normalized" && key !== "stem" && key !== "stemConfidence") {
863
+ cleanMetadata[key] = value;
864
+ }
865
+ }
866
+ }
867
+ const options = {};
868
+ if (normalized) options.normalized = normalized;
869
+ if (stem) options.stem = stem;
870
+ if (stemConfidence !== void 0) options.stemConfidence = stemConfidence;
871
+ if (Object.keys(cleanMetadata).length > 0) options.metadata = cleanMetadata;
872
+ tokens.push(
873
+ createToken(
874
+ result.value,
875
+ this.classifyToken(result.value),
876
+ createPosition(pos, pos + result.length),
877
+ Object.keys(options).length > 0 ? options : void 0
878
+ )
879
+ );
880
+ pos += result.length;
881
+ extracted = true;
882
+ break;
883
+ }
884
+ }
885
+ }
886
+ if (!extracted) {
887
+ const char = input[pos];
888
+ const kind = this.classifyUnknownChar(char);
889
+ tokens.push(createToken(char, kind, createPosition(pos, pos + 1)));
890
+ pos++;
891
+ }
892
+ }
893
+ return new TokenStreamImpl(tokens, this.language);
894
+ }
895
+ /**
896
+ * Classify an unknown character when no extractor matches.
897
+ * Provides sensible defaults for common syntax.
898
+ *
899
+ * @param char - Character to classify
900
+ * @returns Token kind
901
+ */
902
+ classifyUnknownChar(char) {
903
+ if ("()[]{},:;".includes(char)) return "punctuation";
904
+ if ("+-*/<>=!&|".includes(char)) return "operator";
905
+ return "identifier";
906
+ }
907
+ /**
908
+ * Check if current position is a property access (obj.prop) vs CSS selector (.active).
909
+ * Property access: no whitespace before '.', previous token is identifier/keyword/selector.
910
+ * Also detects standalone method calls: .identifier( pattern.
911
+ *
912
+ * Returns true if '.' was emitted as an operator token and pos should advance by 1.
913
+ * Returns false if this is a CSS selector and should be handled by trySelector().
914
+ */
915
+ tryPropertyAccess(input, pos, tokens) {
916
+ if (input[pos] !== ".") return false;
917
+ const lastToken = tokens[tokens.length - 1];
918
+ const hasWhitespaceBefore = lastToken && lastToken.position.end < pos;
919
+ const isPropertyAccess = lastToken && !hasWhitespaceBefore && (lastToken.kind === "identifier" || lastToken.kind === "keyword" || lastToken.kind === "selector");
920
+ if (isPropertyAccess) {
921
+ tokens.push(createToken(".", "operator", createPosition(pos, pos + 1)));
922
+ return true;
923
+ }
924
+ const methodStart = pos + 1;
925
+ let methodEnd = methodStart;
926
+ while (methodEnd < input.length && isAsciiIdentifierChar(input[methodEnd])) {
927
+ methodEnd++;
928
+ }
929
+ if (methodEnd < input.length && input[methodEnd] === "(") {
930
+ tokens.push(createToken(".", "operator", createPosition(pos, pos + 1)));
931
+ return true;
932
+ }
933
+ return false;
934
+ }
935
+ /**
936
+ * Initialize keyword mappings from a language profile.
937
+ * Builds a list of native→english mappings from:
938
+ * - profile.keywords (primary + alternatives)
939
+ * - profile.references (me, it, you, etc.)
940
+ * - profile.roleMarkers (into, from, with, etc.)
941
+ *
942
+ * Results are sorted longest-first for greedy matching (important for non-space languages).
943
+ * Extras take precedence over profile entries when there are duplicates.
944
+ *
945
+ * @param profile - Language profile containing keyword translations
946
+ * @param extras - Additional keyword entries to include (literals, positional, events)
947
+ */
948
+ initializeKeywordsFromProfile(profile, extras = []) {
949
+ const keywordMap = /* @__PURE__ */ new Map();
950
+ this.rawExtraEntries = extras;
951
+ if (profile.keywords) {
952
+ for (const [normalized, translation] of Object.entries(profile.keywords)) {
953
+ keywordMap.set(translation.primary, {
954
+ native: translation.primary,
955
+ normalized: translation.normalized || normalized
956
+ });
957
+ if (translation.alternatives) {
958
+ for (const alt of translation.alternatives) {
959
+ keywordMap.set(alt, {
960
+ native: alt,
961
+ normalized: translation.normalized || normalized
962
+ });
963
+ }
964
+ }
965
+ }
966
+ }
967
+ if (profile.references) {
968
+ for (const [normalized, native] of Object.entries(profile.references)) {
969
+ keywordMap.set(native, { native, normalized });
970
+ }
971
+ for (const canonical of Object.keys(profile.references)) {
972
+ if (!keywordMap.has(canonical)) {
973
+ keywordMap.set(canonical, { native: canonical, normalized: canonical });
974
+ }
975
+ }
976
+ }
977
+ if (profile.roleMarkers) {
978
+ for (const [role, marker] of Object.entries(profile.roleMarkers)) {
979
+ if (marker.primary) {
980
+ keywordMap.set(marker.primary, { native: marker.primary, normalized: role });
981
+ }
982
+ if (marker.alternatives) {
983
+ for (const alt of marker.alternatives) {
984
+ keywordMap.set(alt, { native: alt, normalized: role });
985
+ }
986
+ }
987
+ }
988
+ }
989
+ if (profile.possessive?.keywords) {
990
+ for (const [native, normalized] of Object.entries(profile.possessive.keywords)) {
991
+ keywordMap.set(native, { native, normalized });
992
+ }
993
+ }
994
+ for (const evt of ENGLISH_DOM_EVENT_NAMES) {
995
+ if (!keywordMap.has(evt)) {
996
+ keywordMap.set(evt, { native: evt, normalized: evt });
997
+ }
998
+ }
999
+ for (const extra of extras) {
1000
+ keywordMap.set(extra.native, extra);
1001
+ }
1002
+ this.profileKeywords = Array.from(keywordMap.values()).sort(
1003
+ (a, b) => b.native.length - a.native.length
1004
+ );
1005
+ this.multiWordKeywords = this.profileKeywords.filter(
1006
+ (k) => k.native.includes(" ") && !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
1007
+ );
1008
+ this.profileKeywordMap = /* @__PURE__ */ new Map();
1009
+ for (const keyword of this.profileKeywords) {
1010
+ this.profileKeywordMap.set(keyword.native.toLowerCase(), keyword);
1011
+ const normalized = this.removeDiacritics(keyword.native);
1012
+ if (normalized !== keyword.native && !this.profileKeywordMap.has(normalized.toLowerCase())) {
1013
+ this.profileKeywordMap.set(normalized.toLowerCase(), keyword);
1014
+ }
1015
+ }
1016
+ }
1017
+ /**
1018
+ * Remove diacritical marks from a word for normalization.
1019
+ * Primarily for Arabic (shadda, fatha, kasra, damma, sukun, etc.)
1020
+ * but could be extended for other languages.
1021
+ *
1022
+ * @param word - Word to normalize
1023
+ * @returns Word without diacritics
1024
+ */
1025
+ removeDiacritics(word) {
1026
+ return word.replace(/[\u064B-\u0652\u0670]/g, "");
1027
+ }
1028
+ /**
1029
+ * Try to match a keyword from profile at the current position.
1030
+ * Uses longest-first greedy matching (important for non-space languages).
1031
+ *
1032
+ * @param input - Input string
1033
+ * @param pos - Current position
1034
+ * @returns Token if matched, null otherwise
1035
+ */
1036
+ tryProfileKeyword(input, pos) {
1037
+ for (const entry of this.profileKeywords) {
1038
+ if (input.slice(pos).startsWith(entry.native)) {
1039
+ return createToken(
1040
+ entry.native,
1041
+ "keyword",
1042
+ createPosition(pos, pos + entry.native.length),
1043
+ entry.normalized
1044
+ );
1045
+ }
1046
+ }
1047
+ return null;
1048
+ }
1049
+ /**
1050
+ * Match the longest multi-word (space-containing) profile keyword at `pos`,
1051
+ * requiring the match to end at a word boundary. The profile-driven
1052
+ * counterpart of the per-language hardcoded compound lists (the hindi and
1053
+ * vietnamese keyword extractors). Returns a keyword token (with the normalized
1054
+ * form) or null. Case-sensitive against the stored native form, mirroring
1055
+ * `tryProfileKeyword`/`isKeywordStart` (the i18n dicts emit a fixed surface
1056
+ * case). No-op when `multiWordKeywords` is empty (no-space/CJK languages).
1057
+ *
1058
+ * @param input - Input string
1059
+ * @param pos - Current position (must be a token-start boundary)
1060
+ * @param isWordChar - End-boundary predicate (defaults to Unicode letter/digit/_)
1061
+ */
1062
+ tryMultiWordKeyword(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
1063
+ if (this.multiWordKeywords.length === 0) return null;
1064
+ const rest = input.slice(pos);
1065
+ for (const entry of this.multiWordKeywords) {
1066
+ if (!rest.startsWith(entry.native)) continue;
1067
+ const after = input[pos + entry.native.length];
1068
+ if (after !== void 0 && isWordChar(after)) continue;
1069
+ return createToken(
1070
+ entry.native,
1071
+ "keyword",
1072
+ createPosition(pos, pos + entry.native.length),
1073
+ entry.normalized
1074
+ );
1075
+ }
1076
+ return null;
1077
+ }
1078
+ /**
1079
+ * Check if the remaining input starts with any known keyword.
1080
+ * Useful for non-space languages to detect word boundaries.
1081
+ *
1082
+ * @param input - Input string
1083
+ * @param pos - Current position
1084
+ * @returns true if a keyword starts at this position
1085
+ */
1086
+ isKeywordStart(input, pos) {
1087
+ const remaining = input.slice(pos);
1088
+ return this.profileKeywords.some((entry) => remaining.startsWith(entry.native));
1089
+ }
1090
+ /**
1091
+ * Check if a known keyword starts at the given position AND ends at a word
1092
+ * boundary (end of input or a non-word character).
1093
+ *
1094
+ * Space-delimited languages must use this (not `isKeywordStart`) for
1095
+ * word-walk break checks: the keyword table includes English canonical
1096
+ * fallbacks (me, it, you, …), so a raw `startsWith` check splits any native
1097
+ * word with an embedded fallback mid-word (e.g. Quechua ñit'iy contains
1098
+ * "it"). CJK/no-space tokenizers rely on mid-text keyword starts and must
1099
+ * keep using `isKeywordStart`.
1100
+ *
1101
+ * @param input - Input string
1102
+ * @param pos - Current position
1103
+ * @param isWordChar - Language-specific word-character predicate; pass the
1104
+ * tokenizer's letter classifier so e.g. the Quechua glottal apostrophe
1105
+ * counts as part of a word. Defaults to Unicode letters/digits/underscore.
1106
+ * @returns true if a keyword starts here and is not followed by a word char
1107
+ */
1108
+ isKeywordStartAtBoundary(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
1109
+ const remaining = input.slice(pos);
1110
+ return this.profileKeywords.some((entry) => {
1111
+ if (!remaining.startsWith(entry.native)) return false;
1112
+ const after = input[pos + entry.native.length];
1113
+ return after === void 0 || !isWordChar(after);
1114
+ });
1115
+ }
1116
+ /**
1117
+ * Look up a keyword by native word (case-insensitive).
1118
+ * O(1) lookup using the keyword map.
1119
+ *
1120
+ * @param native - Native word to look up
1121
+ * @returns KeywordEntry if found, undefined otherwise
1122
+ */
1123
+ lookupKeyword(native) {
1124
+ return this.profileKeywordMap.get(native.toLowerCase());
1125
+ }
1126
+ /**
1127
+ * Check if a word is a known keyword (case-insensitive).
1128
+ * O(1) lookup using the keyword map.
1129
+ *
1130
+ * @param native - Native word to check
1131
+ * @returns true if the word is a keyword
1132
+ */
1133
+ isKeyword(native) {
1134
+ return this.profileKeywordMap.has(native.toLowerCase());
1135
+ }
1136
+ /**
1137
+ * Set the morphological normalizer for this tokenizer.
1138
+ */
1139
+ setNormalizer(normalizer) {
1140
+ this.normalizer = normalizer;
1141
+ }
1142
+ /**
1143
+ * Try to normalize a word using the morphological normalizer.
1144
+ * Returns null if no normalizer is set or normalization fails.
1145
+ *
1146
+ * Note: We don't check isNormalizable() here because the individual tokenizers
1147
+ * historically called normalize() directly without that check. The normalize()
1148
+ * method itself handles returning noChange() for words that can't be normalized.
1149
+ */
1150
+ tryNormalize(word) {
1151
+ if (!this.normalizer) return null;
1152
+ const result = this.normalizer.normalize(word);
1153
+ if (result.stem !== word && result.confidence >= 0.7) {
1154
+ return result;
1155
+ }
1156
+ return null;
1157
+ }
1158
+ /**
1159
+ * Try morphological normalization and keyword lookup.
1160
+ *
1161
+ * If the word can be normalized to a stem that matches a known keyword,
1162
+ * returns a keyword token with morphological metadata (stem, stemConfidence).
1163
+ *
1164
+ * This is the common pattern for handling conjugated verbs across languages:
1165
+ * 1. Normalize the word (e.g., "toggled" → "toggle")
1166
+ * 2. Look up the stem in the keyword map
1167
+ * 3. Create a token with both the original form and stem metadata
1168
+ *
1169
+ * @param word - The word to normalize and look up
1170
+ * @param startPos - Start position for the token
1171
+ * @param endPos - End position for the token
1172
+ * @returns Token if stem matches a keyword, null otherwise
1173
+ */
1174
+ tryMorphKeywordMatch(word, startPos, endPos) {
1175
+ const result = this.tryNormalize(word);
1176
+ if (!result) return null;
1177
+ const stemEntry = this.lookupKeyword(result.stem);
1178
+ if (!stemEntry) return null;
1179
+ const tokenOptions = {
1180
+ normalized: stemEntry.normalized,
1181
+ stem: result.stem,
1182
+ stemConfidence: result.confidence
1183
+ };
1184
+ return createToken(word, "keyword", createPosition(startPos, endPos), tokenOptions);
1185
+ }
1186
+ /**
1187
+ * Try to extract a CSS selector at the current position.
1188
+ */
1189
+ trySelector(input, pos) {
1190
+ const selector = extractCssSelector(input, pos);
1191
+ if (selector) {
1192
+ return createToken(selector, "selector", createPosition(pos, pos + selector.length));
1193
+ }
1194
+ return null;
1195
+ }
1196
+ /**
1197
+ * Try to extract an event modifier at the current position.
1198
+ * Event modifiers are .once, .debounce(N), .throttle(N), .queue(strategy)
1199
+ */
1200
+ tryEventModifier(input, pos) {
1201
+ if (input[pos] !== ".") {
1202
+ return null;
1203
+ }
1204
+ const match = input.slice(pos).match(/^\.(?:once|debounce|throttle|queue)(?:\(([^)]+)\))?(?:\s|$|\.)/);
1205
+ if (!match) {
1206
+ return null;
1207
+ }
1208
+ const fullMatch = match[0].replace(/(\s|\.)$/, "");
1209
+ const modifierName = fullMatch.slice(1).split("(")[0];
1210
+ const value = match[1];
1211
+ const token = createToken(
1212
+ fullMatch,
1213
+ "event-modifier",
1214
+ createPosition(pos, pos + fullMatch.length)
1215
+ );
1216
+ return {
1217
+ ...token,
1218
+ metadata: {
1219
+ modifierName,
1220
+ value: value ? modifierName === "queue" ? value : parseInt(value, 10) : void 0
1221
+ }
1222
+ };
1223
+ }
1224
+ /**
1225
+ * Try to extract a string literal at the current position.
1226
+ */
1227
+ tryString(input, pos) {
1228
+ const literal = extractStringLiteral(input, pos);
1229
+ if (literal) {
1230
+ return createToken(literal, "literal", createPosition(pos, pos + literal.length));
1231
+ }
1232
+ return null;
1233
+ }
1234
+ /**
1235
+ * Try to extract a number at the current position.
1236
+ */
1237
+ tryNumber(input, pos) {
1238
+ const number = extractNumber(input, pos);
1239
+ if (number) {
1240
+ return createToken(number, "literal", createPosition(pos, pos + number.length));
1241
+ }
1242
+ return null;
1243
+ }
1244
+ /**
1245
+ * Try to match a time unit from a list of patterns.
1246
+ *
1247
+ * @param input - Input string
1248
+ * @param pos - Position after the number
1249
+ * @param timeUnits - Array of time unit mappings (native pattern → standard suffix)
1250
+ * @param skipWhitespace - Whether to skip whitespace before time unit (default: false)
1251
+ * @returns Object with matched suffix and new position, or null if no match
1252
+ */
1253
+ tryMatchTimeUnit(input, pos, timeUnits, skipWhitespace = false) {
1254
+ let unitPos = pos;
1255
+ if (skipWhitespace) {
1256
+ while (unitPos < input.length && isWhitespace(input[unitPos])) {
1257
+ unitPos++;
1258
+ }
1259
+ }
1260
+ const remaining = input.slice(unitPos);
1261
+ for (const unit of timeUnits) {
1262
+ const candidate = remaining.slice(0, unit.length);
1263
+ const matches = unit.caseInsensitive ? candidate.toLowerCase() === unit.pattern.toLowerCase() : candidate === unit.pattern;
1264
+ if (matches) {
1265
+ if (unit.notFollowedBy) {
1266
+ const nextChar = remaining[unit.length] || "";
1267
+ if (nextChar === unit.notFollowedBy) continue;
1268
+ }
1269
+ if (unit.checkBoundary) {
1270
+ const nextChar = remaining[unit.length] || "";
1271
+ if (isAsciiIdentifierChar(nextChar)) continue;
1272
+ }
1273
+ return { suffix: unit.suffix, endPos: unitPos + unit.length };
1274
+ }
1275
+ }
1276
+ return null;
1277
+ }
1278
+ /**
1279
+ * Parse a base number (sign, integer, decimal) without time units.
1280
+ * Returns the number string and end position.
1281
+ *
1282
+ * @param input - Input string
1283
+ * @param startPos - Start position
1284
+ * @param allowSign - Whether to allow +/- sign (default: true)
1285
+ * @returns Object with number string and end position, or null
1286
+ */
1287
+ parseBaseNumber(input, startPos, allowSign = true) {
1288
+ let pos = startPos;
1289
+ let number = "";
1290
+ if (allowSign && (input[pos] === "-" || input[pos] === "+")) {
1291
+ number += input[pos++];
1292
+ }
1293
+ if (pos >= input.length || !isDigit(input[pos])) {
1294
+ return null;
1295
+ }
1296
+ while (pos < input.length && isDigit(input[pos])) {
1297
+ number += input[pos++];
1298
+ }
1299
+ if (pos < input.length && input[pos] === ".") {
1300
+ number += input[pos++];
1301
+ while (pos < input.length && isDigit(input[pos])) {
1302
+ number += input[pos++];
1303
+ }
1304
+ }
1305
+ if (!number || number === "-" || number === "+") return null;
1306
+ return { number, endPos: pos };
1307
+ }
1308
+ /**
1309
+ * Try to extract a number with native language time units.
1310
+ *
1311
+ * This is a template method that handles the common pattern:
1312
+ * 1. Parse the base number (sign, integer, decimal)
1313
+ * 2. Try to match native language time units
1314
+ * 3. Fall back to standard time units (ms, s, m, h)
1315
+ *
1316
+ * @param input - Input string
1317
+ * @param pos - Start position
1318
+ * @param nativeTimeUnits - Language-specific time unit mappings
1319
+ * @param options - Configuration options
1320
+ * @returns Token if number found, null otherwise
1321
+ */
1322
+ tryNumberWithTimeUnits(input, pos, nativeTimeUnits, options = {}) {
1323
+ const { allowSign = true, skipWhitespace = false } = options;
1324
+ const baseResult = this.parseBaseNumber(input, pos, allowSign);
1325
+ if (!baseResult) return null;
1326
+ let { number, endPos } = baseResult;
1327
+ const allUnits = [...nativeTimeUnits, ..._BaseTokenizer.STANDARD_TIME_UNITS];
1328
+ const timeMatch = this.tryMatchTimeUnit(input, endPos, allUnits, skipWhitespace);
1329
+ if (timeMatch) {
1330
+ number += timeMatch.suffix;
1331
+ endPos = timeMatch.endPos;
1332
+ }
1333
+ return createToken(number, "literal", createPosition(pos, endPos));
1334
+ }
1335
+ /**
1336
+ * Try to extract a URL at the current position.
1337
+ * Handles /path, ./path, ../path, //domain.com, http://, https://
1338
+ */
1339
+ tryUrl(input, pos) {
1340
+ const url = extractUrl(input, pos);
1341
+ if (url) {
1342
+ return createToken(url, "url", createPosition(pos, pos + url.length));
1343
+ }
1344
+ return null;
1345
+ }
1346
+ /**
1347
+ * Try to extract a variable reference (:varname) at the current position.
1348
+ * In hyperscript, :x refers to a local variable named x.
1349
+ */
1350
+ tryVariableRef(input, pos) {
1351
+ if (input[pos] !== ":") return null;
1352
+ if (pos + 1 >= input.length) return null;
1353
+ if (!isAsciiIdentifierChar(input[pos + 1])) return null;
1354
+ let endPos = pos + 1;
1355
+ while (endPos < input.length && isAsciiIdentifierChar(input[endPos])) {
1356
+ endPos++;
1357
+ }
1358
+ const varRef = input.slice(pos, endPos);
1359
+ return createToken(varRef, "identifier", createPosition(pos, endPos));
1360
+ }
1361
+ /**
1362
+ * Try to extract an operator or punctuation token at the current position.
1363
+ * Handles two-character operators (==, !=, etc.) and single-character operators.
1364
+ */
1365
+ tryOperator(input, pos) {
1366
+ const twoChar = input.slice(pos, pos + 2);
1367
+ if (["==", "!=", "<=", ">=", "&&", "||", "->"].includes(twoChar)) {
1368
+ return createToken(twoChar, "operator", createPosition(pos, pos + 2));
1369
+ }
1370
+ const oneChar = input[pos];
1371
+ if (["<", ">", "!", "+", "-", "*", "/", "="].includes(oneChar)) {
1372
+ return createToken(oneChar, "operator", createPosition(pos, pos + 1));
1373
+ }
1374
+ if (["(", ")", "{", "}", ",", ";", ":"].includes(oneChar)) {
1375
+ return createToken(oneChar, "punctuation", createPosition(pos, pos + 1));
1376
+ }
1377
+ return null;
1378
+ }
1379
+ /**
1380
+ * Try to match a multi-character particle from a list.
1381
+ *
1382
+ * Used by languages like Japanese, Korean, and Chinese that have
1383
+ * multi-character particles (e.g., Japanese から, まで, より).
1384
+ *
1385
+ * @param input - Input string
1386
+ * @param pos - Current position
1387
+ * @param particles - Array of multi-character particles to match
1388
+ * @returns Token if matched, null otherwise
1389
+ */
1390
+ tryMultiCharParticle(input, pos, particles) {
1391
+ for (const particle of particles) {
1392
+ if (input.slice(pos, pos + particle.length) === particle) {
1393
+ return createToken(particle, "particle", createPosition(pos, pos + particle.length));
1394
+ }
1395
+ }
1396
+ return null;
1397
+ }
1398
+ };
1399
+ /**
1400
+ * Configuration for native language time units.
1401
+ * Maps patterns to their standard suffix (ms, s, m, h).
1402
+ */
1403
+ _BaseTokenizer.STANDARD_TIME_UNITS = [
1404
+ { pattern: "ms", suffix: "ms", length: 2 },
1405
+ { pattern: "s", suffix: "s", length: 1, checkBoundary: true },
1406
+ { pattern: "m", suffix: "m", length: 1, checkBoundary: true, notFollowedBy: "s" },
1407
+ { pattern: "h", suffix: "h", length: 1, checkBoundary: true }
1408
+ ];
1409
+ var BaseTokenizer = _BaseTokenizer;
1410
+ function createSimpleTokenizer(config) {
1411
+ const {
1412
+ language,
1413
+ direction = "ltr",
1414
+ keywords,
1415
+ keywordExtras,
1416
+ keywordProfile,
1417
+ includeOperators = false,
1418
+ caseInsensitive = true,
1419
+ customExtractors
1420
+ } = config;
1421
+ const keywordSet = new Set(caseInsensitive ? keywords.map((k) => k.toLowerCase()) : keywords);
1422
+ class SimpleTokenizer extends BaseTokenizer {
1423
+ constructor() {
1424
+ super();
1425
+ this.language = language;
1426
+ this.direction = direction;
1427
+ if (customExtractors) {
1428
+ this.registerExtractors(customExtractors);
1429
+ }
1430
+ this.registerExtractors(getDefaultExtractors());
1431
+ if (keywordProfile) {
1432
+ this.initializeKeywordsFromProfile(keywordProfile, keywordExtras);
1433
+ }
1434
+ }
1435
+ classifyToken(token) {
1436
+ const lookup = caseInsensitive ? token.toLowerCase() : token;
1437
+ if (keywordSet.has(lookup)) return "keyword";
1438
+ if (this.isKeyword(token)) return "keyword";
1439
+ if (/^\d/.test(token)) return "literal";
1440
+ if (/^['"]/.test(token)) return "literal";
1441
+ if (includeOperators && SIMPLE_TOKENIZER_OPERATOR_SET.has(token)) return "operator";
1442
+ return "identifier";
1443
+ }
1444
+ }
1445
+ return new SimpleTokenizer();
1446
+ }
1447
+
1448
+ // src/multilingual/builders.ts
1449
+ function mergeRoleMarkers(slice, vocab) {
1450
+ const merged = {};
1451
+ const add = (role, marker) => {
1452
+ if (!marker?.primary) {
1453
+ delete merged[role];
1454
+ return;
1455
+ }
1456
+ merged[role] = {
1457
+ primary: marker.primary,
1458
+ ...marker.alternatives?.length && { alternatives: [...marker.alternatives] },
1459
+ ...marker.position && { position: marker.position }
1460
+ };
1461
+ };
1462
+ for (const [role, marker] of Object.entries(slice.roleMarkers ?? {})) add(role, marker);
1463
+ for (const [role, marker] of Object.entries(vocab.roleMarkerOverrides ?? {})) add(role, marker);
1464
+ return merged;
1465
+ }
1466
+ function buildPatternProfile(slice, vocab) {
1467
+ const keywords = {};
1468
+ for (const [action, translation] of Object.entries(vocab.keywords)) {
1469
+ keywords[action] = {
1470
+ primary: translation.primary,
1471
+ ...translation.alternatives?.length && { alternatives: [...translation.alternatives] }
1472
+ };
1473
+ }
1474
+ const roleMarkers = mergeRoleMarkers(slice, vocab);
1475
+ return {
1476
+ code: slice.code,
1477
+ wordOrder: slice.wordOrder,
1478
+ keywords,
1479
+ ...Object.keys(roleMarkers).length > 0 && { roleMarkers }
1480
+ };
1481
+ }
1482
+ function defaultCaseInsensitive(script) {
1483
+ return script === void 0 || script === "latin" || script === "cyrillic";
1484
+ }
1485
+ function buildDomainTokenizer(slice, vocab, options = {}) {
1486
+ const roleMarkers = mergeRoleMarkers(slice, vocab);
1487
+ const keywords = /* @__PURE__ */ new Set();
1488
+ for (const translation of Object.values(vocab.keywords)) {
1489
+ keywords.add(translation.primary);
1490
+ for (const alt of translation.alternatives ?? []) keywords.add(alt);
1491
+ }
1492
+ for (const marker of Object.values(roleMarkers)) {
1493
+ keywords.add(marker.primary);
1494
+ for (const alt of marker.alternatives ?? []) keywords.add(alt);
1495
+ }
1496
+ for (const particle of slice.tokenization?.particles ?? []) keywords.add(particle);
1497
+ for (const extra of vocab.tokenizerKeywords ?? []) keywords.add(extra);
1498
+ const profileKeywords = {};
1499
+ for (const [action, translation] of Object.entries(vocab.keywords)) {
1500
+ profileKeywords[action] = {
1501
+ primary: translation.primary,
1502
+ ...translation.alternatives?.length && { alternatives: [...translation.alternatives] },
1503
+ normalized: translation.normalized ?? action
1504
+ };
1505
+ }
1506
+ const keywordProfile = {
1507
+ keywords: profileKeywords,
1508
+ ...Object.keys(roleMarkers).length > 0 && { roleMarkers }
1509
+ };
1510
+ const customExtractors = [
1511
+ ...options.customExtractors ?? [],
1512
+ ...slice.script === "latin" ? [new LatinExtendedIdentifierExtractor()] : []
1513
+ ];
1514
+ return createSimpleTokenizer({
1515
+ language: slice.code,
1516
+ direction: slice.direction ?? "ltr",
1517
+ keywords: [...keywords],
1518
+ ...vocab.keywordExtras?.length && { keywordExtras: vocab.keywordExtras.map((e) => ({ ...e })) },
1519
+ keywordProfile,
1520
+ includeOperators: options.includeOperators ?? false,
1521
+ caseInsensitive: options.caseInsensitive ?? defaultCaseInsensitive(slice.script),
1522
+ ...customExtractors.length > 0 && { customExtractors }
1523
+ });
1524
+ }
1525
+ function buildLanguageConfig(slice, vocab, meta = {}) {
1526
+ const name = meta.name ?? slice.name ?? slice.code;
1527
+ return {
1528
+ code: slice.code,
1529
+ name,
1530
+ nativeName: meta.nativeName ?? slice.nativeName ?? name,
1531
+ tokenizer: meta.tokenizer ?? buildDomainTokenizer(slice, vocab, meta.tokenizerOptions),
1532
+ patternProfile: buildPatternProfile(slice, vocab),
1533
+ ...meta.grammarProfile && { grammarProfile: meta.grammarProfile }
1534
+ };
1535
+ }
1536
+ function deriveRoleMarkers(slice, roleMapping) {
1537
+ const derived = {};
1538
+ for (const [domainRole, semanticRole] of Object.entries(roleMapping)) {
1539
+ const marker = slice.roleMarkers?.[semanticRole];
1540
+ if (marker?.primary) derived[domainRole] = marker.primary;
1541
+ }
1542
+ return derived;
1543
+ }
1544
+ export {
1545
+ buildDomainTokenizer,
1546
+ buildLanguageConfig,
1547
+ buildPatternProfile,
1548
+ deriveRoleMarkers
1549
+ };
1
1550
  //# sourceMappingURL=index.js.map