@lokascript/framework 2.7.1 → 2.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/api/index.js +6 -2
- package/dist/api/index.js.map +1 -1
- package/dist/core/index.js +41 -1
- package/dist/core/index.js.map +1 -1
- package/dist/core/tokenization/base-tokenizer.d.ts +24 -0
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
- package/dist/core/tokenization/extractors.d.ts +6 -0
- package/dist/core/tokenization/extractors.d.ts.map +1 -1
- package/dist/core/tokenization/index.js +41 -1
- package/dist/core/tokenization/index.js.map +1 -1
- package/dist/generation/index.js +2 -1
- package/dist/generation/index.js.map +1 -1
- package/dist/generation/pattern-generator.d.ts.map +1 -1
- package/dist/index.cjs +47 -3
- package/dist/index.cjs.map +1 -1
- package/dist/index.js +47 -3
- package/dist/index.js.map +1 -1
- package/dist/multilingual/index.js +41 -1
- package/dist/multilingual/index.js.map +1 -1
- package/dist/testing/index.js +4 -12
- package/dist/testing/index.js.map +1 -1
- package/package.json +2 -2
- package/src/core/tokenization/base-tokenizer.ts +55 -1
- package/src/core/tokenization/colon-qualifier.test.ts +129 -0
- package/src/core/tokenization/extractors.ts +6 -0
- package/src/generation/pattern-generator.ts +5 -1
- package/src/ir/protocol-json.test.ts +21 -0
- package/src/ir/references.test.ts +5 -2
- package/src/prompts/prompt-generator.ts +4 -1
|
@@ -890,7 +890,39 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
890
890
|
pos++;
|
|
891
891
|
}
|
|
892
892
|
}
|
|
893
|
-
return new TokenStreamImpl(tokens, this.language);
|
|
893
|
+
return new TokenStreamImpl(this.mergeColonQualifiedNames(tokens), this.language);
|
|
894
|
+
}
|
|
895
|
+
/**
|
|
896
|
+
* Fuse `name` + `:qualifier` into ONE identifier (`draggable:start`).
|
|
897
|
+
*
|
|
898
|
+
* `:name` is hyperscript's local-variable sigil, but a colon IMMEDIATELY
|
|
899
|
+
* preceded by an identifier is a qualifier (custom event namespace), not a
|
|
900
|
+
* sigil. The English tokenizer already merges these inside
|
|
901
|
+
* EnglishKeywordExtractor; this post-pass gives the other 23 languages the
|
|
902
|
+
* same stream. Strict position adjacency is the discriminator: whitespace
|
|
903
|
+
* between the tokens (`trigger :start`) breaks `end === start`, so a spaced
|
|
904
|
+
* local-variable reference survives untouched.
|
|
905
|
+
*
|
|
906
|
+
* Self-gating for non-hyperscript tokenizers (domain DSLs): their extractor
|
|
907
|
+
* sets tokenize `:` as bare punctuation (length 1), which never matches
|
|
908
|
+
* COLON_QUALIFIER, so this pass is a no-op for them.
|
|
909
|
+
*/
|
|
910
|
+
mergeColonQualifiedNames(tokens) {
|
|
911
|
+
const out = [];
|
|
912
|
+
for (const tok of tokens) {
|
|
913
|
+
const prev = out[out.length - 1];
|
|
914
|
+
if (prev && _BaseTokenizer.ASCII_WORD.test(prev.value) && _BaseTokenizer.COLON_QUALIFIER.test(tok.value) && prev.position.end === tok.position.start) {
|
|
915
|
+
const merged = prev.value + tok.value;
|
|
916
|
+
out[out.length - 1] = createToken(
|
|
917
|
+
merged,
|
|
918
|
+
this.classifyToken(merged),
|
|
919
|
+
createPosition(prev.position.start, tok.position.end)
|
|
920
|
+
);
|
|
921
|
+
continue;
|
|
922
|
+
}
|
|
923
|
+
out.push(tok);
|
|
924
|
+
}
|
|
925
|
+
return out;
|
|
894
926
|
}
|
|
895
927
|
/**
|
|
896
928
|
* Classify an unknown character when no extractor matches.
|
|
@@ -1396,6 +1428,14 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
1396
1428
|
return null;
|
|
1397
1429
|
}
|
|
1398
1430
|
};
|
|
1431
|
+
/**
|
|
1432
|
+
* ASCII word of the shape the English word-walker produces. Excludes `:`, so a
|
|
1433
|
+
* token that already carries a qualifier never merges again — `a:b:c` yields
|
|
1434
|
+
* `a:b` + `:c`, byte-matching the English extractor's single-segment merge.
|
|
1435
|
+
*/
|
|
1436
|
+
_BaseTokenizer.ASCII_WORD = /^[A-Za-z_][A-Za-z0-9_]*$/;
|
|
1437
|
+
/** `:name` — only a variable-ref-style extractor ever emits this token shape. */
|
|
1438
|
+
_BaseTokenizer.COLON_QUALIFIER = /^:[A-Za-z_][A-Za-z0-9_]*$/;
|
|
1399
1439
|
/**
|
|
1400
1440
|
* Configuration for native language time units.
|
|
1401
1441
|
* Maps patterns to their standard suffix (ms, s, m, h).
|