@lokascript/language-server 2.5.0 → 2.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -3152,7 +3152,8 @@ function createTokenizerContext(tokenizer) {
|
|
|
3152
3152
|
direction: tokenizer.direction,
|
|
3153
3153
|
lookupKeyword: tokenizer.lookupKeyword.bind(tokenizer),
|
|
3154
3154
|
isKeyword: tokenizer.isKeyword.bind(tokenizer),
|
|
3155
|
-
isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer)
|
|
3155
|
+
isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer),
|
|
3156
|
+
...tokenizer.isKeywordStartAtBoundary ? { isKeywordStartAtBoundary: tokenizer.isKeywordStartAtBoundary.bind(tokenizer) } : {}
|
|
3156
3157
|
};
|
|
3157
3158
|
if (tokenizer.normalizer) {
|
|
3158
3159
|
return { ...ctx, normalizer: tokenizer.normalizer };
|
|
@@ -6068,12 +6069,73 @@ function createLatinCharClassifiers(letterPattern) {
|
|
|
6068
6069
|
return { isLetter, isIdentifierChar };
|
|
6069
6070
|
}
|
|
6070
6071
|
var SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
|
|
6072
|
+
var MARKER_CONCEPT_NORMALIZEDS = /* @__PURE__ */ new Set([
|
|
6073
|
+
// Role-marker role names (profile.roleMarkers normalizeds)
|
|
6074
|
+
"patient",
|
|
6075
|
+
"destination",
|
|
6076
|
+
"source",
|
|
6077
|
+
"style",
|
|
6078
|
+
"event",
|
|
6079
|
+
"eventMarker",
|
|
6080
|
+
"agent",
|
|
6081
|
+
"goal",
|
|
6082
|
+
"manner",
|
|
6083
|
+
// Prepositional / positional modifier concepts matched via the role mechanism
|
|
6084
|
+
// (profile.keywords "Modifiers"). `before`/`after`/`until` are intentionally
|
|
6085
|
+
// NOT here — they are pattern literals (see the note above).
|
|
6086
|
+
"into",
|
|
6087
|
+
"from",
|
|
6088
|
+
"to",
|
|
6089
|
+
"with",
|
|
6090
|
+
"at",
|
|
6091
|
+
"of",
|
|
6092
|
+
"as",
|
|
6093
|
+
"by",
|
|
6094
|
+
"in",
|
|
6095
|
+
"on",
|
|
6096
|
+
"over",
|
|
6097
|
+
"under",
|
|
6098
|
+
"between",
|
|
6099
|
+
"through",
|
|
6100
|
+
"without"
|
|
6101
|
+
]);
|
|
6102
|
+
var ENGLISH_DOM_EVENT_NAMES = [
|
|
6103
|
+
"click",
|
|
6104
|
+
"dblclick",
|
|
6105
|
+
"input",
|
|
6106
|
+
"change",
|
|
6107
|
+
"submit",
|
|
6108
|
+
"keydown",
|
|
6109
|
+
"keyup",
|
|
6110
|
+
"keypress",
|
|
6111
|
+
"mousedown",
|
|
6112
|
+
"mouseup",
|
|
6113
|
+
"mouseover",
|
|
6114
|
+
"mouseout",
|
|
6115
|
+
"mouseenter",
|
|
6116
|
+
"mouseleave",
|
|
6117
|
+
"mousemove",
|
|
6118
|
+
"pointerdown",
|
|
6119
|
+
"pointerup",
|
|
6120
|
+
"pointermove",
|
|
6121
|
+
"focus",
|
|
6122
|
+
"blur",
|
|
6123
|
+
"load",
|
|
6124
|
+
"resize",
|
|
6125
|
+
"scroll"
|
|
6126
|
+
];
|
|
6071
6127
|
var _BaseTokenizer = class _BaseTokenizer2 {
|
|
6072
6128
|
constructor() {
|
|
6073
6129
|
this.profileKeywords = [];
|
|
6130
|
+
this.multiWordKeywords = [];
|
|
6074
6131
|
this.profileKeywordMap = /* @__PURE__ */ new Map();
|
|
6132
|
+
this.rawExtraEntries = [];
|
|
6075
6133
|
this.extractors = [];
|
|
6076
6134
|
}
|
|
6135
|
+
/** Raw extras as passed in, pre-dedup — for consistency tests. */
|
|
6136
|
+
getExtraKeywordEntries() {
|
|
6137
|
+
return this.rawExtraEntries;
|
|
6138
|
+
}
|
|
6077
6139
|
/**
|
|
6078
6140
|
* Tokenize input string to token stream.
|
|
6079
6141
|
* Delegates to extractor-based tokenization if extractors are registered,
|
|
@@ -6142,6 +6204,12 @@ var _BaseTokenizer = class _BaseTokenizer2 {
|
|
|
6142
6204
|
pos++;
|
|
6143
6205
|
}
|
|
6144
6206
|
if (pos >= input.length) break;
|
|
6207
|
+
const multiWord = this.tryMultiWordKeyword(input, pos);
|
|
6208
|
+
if (multiWord) {
|
|
6209
|
+
tokens.push(multiWord);
|
|
6210
|
+
pos = multiWord.position.end;
|
|
6211
|
+
continue;
|
|
6212
|
+
}
|
|
6145
6213
|
let extracted = false;
|
|
6146
6214
|
for (const extractor of this.extractors) {
|
|
6147
6215
|
if (extractor.canExtract(input, pos)) {
|
|
@@ -6241,6 +6309,7 @@ var _BaseTokenizer = class _BaseTokenizer2 {
|
|
|
6241
6309
|
*/
|
|
6242
6310
|
initializeKeywordsFromProfile(profile, extras = []) {
|
|
6243
6311
|
const keywordMap = /* @__PURE__ */ new Map();
|
|
6312
|
+
this.rawExtraEntries = extras;
|
|
6244
6313
|
if (profile.keywords) {
|
|
6245
6314
|
for (const [normalized2, translation] of Object.entries(profile.keywords)) {
|
|
6246
6315
|
keywordMap.set(translation.primary, {
|
|
@@ -6284,12 +6353,20 @@ var _BaseTokenizer = class _BaseTokenizer2 {
|
|
|
6284
6353
|
keywordMap.set(native, { native, normalized: normalized2 });
|
|
6285
6354
|
}
|
|
6286
6355
|
}
|
|
6356
|
+
for (const evt of ENGLISH_DOM_EVENT_NAMES) {
|
|
6357
|
+
if (!keywordMap.has(evt)) {
|
|
6358
|
+
keywordMap.set(evt, { native: evt, normalized: evt });
|
|
6359
|
+
}
|
|
6360
|
+
}
|
|
6287
6361
|
for (const extra of extras) {
|
|
6288
6362
|
keywordMap.set(extra.native, extra);
|
|
6289
6363
|
}
|
|
6290
6364
|
this.profileKeywords = Array.from(keywordMap.values()).sort(
|
|
6291
6365
|
(a, b) => b.native.length - a.native.length
|
|
6292
6366
|
);
|
|
6367
|
+
this.multiWordKeywords = this.profileKeywords.filter(
|
|
6368
|
+
(k) => k.native.includes(" ") && !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
|
|
6369
|
+
);
|
|
6293
6370
|
this.profileKeywordMap = /* @__PURE__ */ new Map();
|
|
6294
6371
|
for (const keyword of this.profileKeywords) {
|
|
6295
6372
|
this.profileKeywordMap.set(keyword.native.toLowerCase(), keyword);
|
|
@@ -6331,6 +6408,35 @@ var _BaseTokenizer = class _BaseTokenizer2 {
|
|
|
6331
6408
|
}
|
|
6332
6409
|
return null;
|
|
6333
6410
|
}
|
|
6411
|
+
/**
|
|
6412
|
+
* Match the longest multi-word (space-containing) profile keyword at `pos`,
|
|
6413
|
+
* requiring the match to end at a word boundary. The profile-driven
|
|
6414
|
+
* counterpart of the per-language hardcoded compound lists (the hindi and
|
|
6415
|
+
* vietnamese keyword extractors). Returns a keyword token (with the normalized
|
|
6416
|
+
* form) or null. Case-sensitive against the stored native form, mirroring
|
|
6417
|
+
* `tryProfileKeyword`/`isKeywordStart` (the i18n dicts emit a fixed surface
|
|
6418
|
+
* case). No-op when `multiWordKeywords` is empty (no-space/CJK languages).
|
|
6419
|
+
*
|
|
6420
|
+
* @param input - Input string
|
|
6421
|
+
* @param pos - Current position (must be a token-start boundary)
|
|
6422
|
+
* @param isWordChar - End-boundary predicate (defaults to Unicode letter/digit/_)
|
|
6423
|
+
*/
|
|
6424
|
+
tryMultiWordKeyword(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
|
|
6425
|
+
if (this.multiWordKeywords.length === 0) return null;
|
|
6426
|
+
const rest = input.slice(pos);
|
|
6427
|
+
for (const entry of this.multiWordKeywords) {
|
|
6428
|
+
if (!rest.startsWith(entry.native)) continue;
|
|
6429
|
+
const after = input[pos + entry.native.length];
|
|
6430
|
+
if (after !== void 0 && isWordChar(after)) continue;
|
|
6431
|
+
return createToken(
|
|
6432
|
+
entry.native,
|
|
6433
|
+
"keyword",
|
|
6434
|
+
createPosition(pos, pos + entry.native.length),
|
|
6435
|
+
entry.normalized
|
|
6436
|
+
);
|
|
6437
|
+
}
|
|
6438
|
+
return null;
|
|
6439
|
+
}
|
|
6334
6440
|
/**
|
|
6335
6441
|
* Check if the remaining input starts with any known keyword.
|
|
6336
6442
|
* Useful for non-space languages to detect word boundaries.
|
|
@@ -6343,6 +6449,32 @@ var _BaseTokenizer = class _BaseTokenizer2 {
|
|
|
6343
6449
|
const remaining = input.slice(pos);
|
|
6344
6450
|
return this.profileKeywords.some((entry) => remaining.startsWith(entry.native));
|
|
6345
6451
|
}
|
|
6452
|
+
/**
|
|
6453
|
+
* Check if a known keyword starts at the given position AND ends at a word
|
|
6454
|
+
* boundary (end of input or a non-word character).
|
|
6455
|
+
*
|
|
6456
|
+
* Space-delimited languages must use this (not `isKeywordStart`) for
|
|
6457
|
+
* word-walk break checks: the keyword table includes English canonical
|
|
6458
|
+
* fallbacks (me, it, you, …), so a raw `startsWith` check splits any native
|
|
6459
|
+
* word with an embedded fallback mid-word (e.g. Quechua ñit'iy contains
|
|
6460
|
+
* "it"). CJK/no-space tokenizers rely on mid-text keyword starts and must
|
|
6461
|
+
* keep using `isKeywordStart`.
|
|
6462
|
+
*
|
|
6463
|
+
* @param input - Input string
|
|
6464
|
+
* @param pos - Current position
|
|
6465
|
+
* @param isWordChar - Language-specific word-character predicate; pass the
|
|
6466
|
+
* tokenizer's letter classifier so e.g. the Quechua glottal apostrophe
|
|
6467
|
+
* counts as part of a word. Defaults to Unicode letters/digits/underscore.
|
|
6468
|
+
* @returns true if a keyword starts here and is not followed by a word char
|
|
6469
|
+
*/
|
|
6470
|
+
isKeywordStartAtBoundary(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
|
|
6471
|
+
const remaining = input.slice(pos);
|
|
6472
|
+
return this.profileKeywords.some((entry) => {
|
|
6473
|
+
if (!remaining.startsWith(entry.native)) return false;
|
|
6474
|
+
const after = input[pos + entry.native.length];
|
|
6475
|
+
return after === void 0 || !isWordChar(after);
|
|
6476
|
+
});
|
|
6477
|
+
}
|
|
6346
6478
|
/**
|
|
6347
6479
|
* Look up a keyword by native word (case-insensitive).
|
|
6348
6480
|
* O(1) lookup using the keyword map.
|
package/dist/server.js
CHANGED
|
@@ -1137,7 +1137,7 @@ try {
|
|
|
1137
1137
|
}
|
|
1138
1138
|
var frameworkIR = null;
|
|
1139
1139
|
try {
|
|
1140
|
-
const fw = await import("./dist-
|
|
1140
|
+
const fw = await import("./dist-7HGHK5EJ.js");
|
|
1141
1141
|
frameworkIR = { fromInterchangeNode: fw.fromInterchangeNode, renderExplicit: fw.renderExplicit };
|
|
1142
1142
|
console.error(
|
|
1143
1143
|
"[lokascript-ls] @lokascript/framework loaded \u2014 LSE bracket notation in hover enabled"
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@lokascript/language-server",
|
|
3
|
-
"version": "2.
|
|
3
|
+
"version": "2.6.0",
|
|
4
4
|
"description": "Language Server Protocol implementation for LokaScript/hyperscript with 21 language support",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "dist/server.js",
|
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
"dev": "tsx src/server.ts --stdio",
|
|
14
14
|
"test": "vitest",
|
|
15
15
|
"typecheck": "tsc --noEmit",
|
|
16
|
-
"test:check": "vitest
|
|
16
|
+
"test:check": "VITEST_QUIET=1 bash ../../scripts/vitest-run.sh --reporter=dot"
|
|
17
17
|
},
|
|
18
18
|
"keywords": [
|
|
19
19
|
"lsp",
|
|
@@ -31,8 +31,8 @@
|
|
|
31
31
|
"vscode-languageserver-textdocument": "^1.0.11"
|
|
32
32
|
},
|
|
33
33
|
"peerDependencies": {
|
|
34
|
-
"@hyperfixi/core": "^2.
|
|
35
|
-
"@lokascript/semantic": "^2.
|
|
34
|
+
"@hyperfixi/core": "^2.6.0",
|
|
35
|
+
"@lokascript/semantic": "^2.6.0"
|
|
36
36
|
},
|
|
37
37
|
"peerDependenciesMeta": {
|
|
38
38
|
"@lokascript/semantic": {
|
|
@@ -43,8 +43,8 @@
|
|
|
43
43
|
}
|
|
44
44
|
},
|
|
45
45
|
"devDependencies": {
|
|
46
|
-
"@hyperfixi/core": "^2.
|
|
47
|
-
"@lokascript/semantic": "^2.
|
|
46
|
+
"@hyperfixi/core": "^2.6.0",
|
|
47
|
+
"@lokascript/semantic": "^2.6.0",
|
|
48
48
|
"@types/node": "^20.0.0",
|
|
49
49
|
"@vitest/coverage-v8": "^4.0.0",
|
|
50
50
|
"tsup": "^8.0.0",
|