@lokascript/framework 2.7.1 → 2.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/api/index.js +6 -2
- package/dist/api/index.js.map +1 -1
- package/dist/core/index.js +41 -1
- package/dist/core/index.js.map +1 -1
- package/dist/core/tokenization/base-tokenizer.d.ts +24 -0
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
- package/dist/core/tokenization/extractors.d.ts +6 -0
- package/dist/core/tokenization/extractors.d.ts.map +1 -1
- package/dist/core/tokenization/index.js +41 -1
- package/dist/core/tokenization/index.js.map +1 -1
- package/dist/generation/index.js +2 -1
- package/dist/generation/index.js.map +1 -1
- package/dist/generation/pattern-generator.d.ts.map +1 -1
- package/dist/index.cjs +47 -3
- package/dist/index.cjs.map +1 -1
- package/dist/index.js +47 -3
- package/dist/index.js.map +1 -1
- package/dist/multilingual/index.js +41 -1
- package/dist/multilingual/index.js.map +1 -1
- package/dist/testing/index.js +4 -12
- package/dist/testing/index.js.map +1 -1
- package/package.json +2 -2
- package/src/core/tokenization/base-tokenizer.ts +55 -1
- package/src/core/tokenization/colon-qualifier.test.ts +129 -0
- package/src/core/tokenization/extractors.ts +6 -0
- package/src/generation/pattern-generator.ts +5 -1
- package/src/ir/protocol-json.test.ts +21 -0
- package/src/ir/references.test.ts +5 -2
- package/src/prompts/prompt-generator.ts +4 -1
package/dist/core/index.js
CHANGED
|
@@ -915,7 +915,39 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
915
915
|
pos++;
|
|
916
916
|
}
|
|
917
917
|
}
|
|
918
|
-
return new TokenStreamImpl(tokens, this.language);
|
|
918
|
+
return new TokenStreamImpl(this.mergeColonQualifiedNames(tokens), this.language);
|
|
919
|
+
}
|
|
920
|
+
/**
|
|
921
|
+
* Fuse `name` + `:qualifier` into ONE identifier (`draggable:start`).
|
|
922
|
+
*
|
|
923
|
+
* `:name` is hyperscript's local-variable sigil, but a colon IMMEDIATELY
|
|
924
|
+
* preceded by an identifier is a qualifier (custom event namespace), not a
|
|
925
|
+
* sigil. The English tokenizer already merges these inside
|
|
926
|
+
* EnglishKeywordExtractor; this post-pass gives the other 23 languages the
|
|
927
|
+
* same stream. Strict position adjacency is the discriminator: whitespace
|
|
928
|
+
* between the tokens (`trigger :start`) breaks `end === start`, so a spaced
|
|
929
|
+
* local-variable reference survives untouched.
|
|
930
|
+
*
|
|
931
|
+
* Self-gating for non-hyperscript tokenizers (domain DSLs): their extractor
|
|
932
|
+
* sets tokenize `:` as bare punctuation (length 1), which never matches
|
|
933
|
+
* COLON_QUALIFIER, so this pass is a no-op for them.
|
|
934
|
+
*/
|
|
935
|
+
mergeColonQualifiedNames(tokens) {
|
|
936
|
+
const out = [];
|
|
937
|
+
for (const tok of tokens) {
|
|
938
|
+
const prev = out[out.length - 1];
|
|
939
|
+
if (prev && _BaseTokenizer.ASCII_WORD.test(prev.value) && _BaseTokenizer.COLON_QUALIFIER.test(tok.value) && prev.position.end === tok.position.start) {
|
|
940
|
+
const merged = prev.value + tok.value;
|
|
941
|
+
out[out.length - 1] = createToken(
|
|
942
|
+
merged,
|
|
943
|
+
this.classifyToken(merged),
|
|
944
|
+
createPosition(prev.position.start, tok.position.end)
|
|
945
|
+
);
|
|
946
|
+
continue;
|
|
947
|
+
}
|
|
948
|
+
out.push(tok);
|
|
949
|
+
}
|
|
950
|
+
return out;
|
|
919
951
|
}
|
|
920
952
|
/**
|
|
921
953
|
* Classify an unknown character when no extractor matches.
|
|
@@ -1421,6 +1453,14 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
1421
1453
|
return null;
|
|
1422
1454
|
}
|
|
1423
1455
|
};
|
|
1456
|
+
/**
|
|
1457
|
+
* ASCII word of the shape the English word-walker produces. Excludes `:`, so a
|
|
1458
|
+
* token that already carries a qualifier never merges again — `a:b:c` yields
|
|
1459
|
+
* `a:b` + `:c`, byte-matching the English extractor's single-segment merge.
|
|
1460
|
+
*/
|
|
1461
|
+
_BaseTokenizer.ASCII_WORD = /^[A-Za-z_][A-Za-z0-9_]*$/;
|
|
1462
|
+
/** `:name` — only a variable-ref-style extractor ever emits this token shape. */
|
|
1463
|
+
_BaseTokenizer.COLON_QUALIFIER = /^:[A-Za-z_][A-Za-z0-9_]*$/;
|
|
1424
1464
|
/**
|
|
1425
1465
|
* Configuration for native language time units.
|
|
1426
1466
|
* Maps patterns to their standard suffix (ms, s, m, h).
|