@lokascript/language-server 2.5.0 → 2.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3152,7 +3152,8 @@ function createTokenizerContext(tokenizer) {
3152
3152
  direction: tokenizer.direction,
3153
3153
  lookupKeyword: tokenizer.lookupKeyword.bind(tokenizer),
3154
3154
  isKeyword: tokenizer.isKeyword.bind(tokenizer),
3155
- isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer)
3155
+ isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer),
3156
+ ...tokenizer.isKeywordStartAtBoundary ? { isKeywordStartAtBoundary: tokenizer.isKeywordStartAtBoundary.bind(tokenizer) } : {}
3156
3157
  };
3157
3158
  if (tokenizer.normalizer) {
3158
3159
  return { ...ctx, normalizer: tokenizer.normalizer };
@@ -6068,12 +6069,73 @@ function createLatinCharClassifiers(letterPattern) {
6068
6069
  return { isLetter, isIdentifierChar };
6069
6070
  }
6070
6071
  var SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
6072
+ var MARKER_CONCEPT_NORMALIZEDS = /* @__PURE__ */ new Set([
6073
+ // Role-marker role names (profile.roleMarkers normalizeds)
6074
+ "patient",
6075
+ "destination",
6076
+ "source",
6077
+ "style",
6078
+ "event",
6079
+ "eventMarker",
6080
+ "agent",
6081
+ "goal",
6082
+ "manner",
6083
+ // Prepositional / positional modifier concepts matched via the role mechanism
6084
+ // (profile.keywords "Modifiers"). `before`/`after`/`until` are intentionally
6085
+ // NOT here — they are pattern literals (see the note above).
6086
+ "into",
6087
+ "from",
6088
+ "to",
6089
+ "with",
6090
+ "at",
6091
+ "of",
6092
+ "as",
6093
+ "by",
6094
+ "in",
6095
+ "on",
6096
+ "over",
6097
+ "under",
6098
+ "between",
6099
+ "through",
6100
+ "without"
6101
+ ]);
6102
+ var ENGLISH_DOM_EVENT_NAMES = [
6103
+ "click",
6104
+ "dblclick",
6105
+ "input",
6106
+ "change",
6107
+ "submit",
6108
+ "keydown",
6109
+ "keyup",
6110
+ "keypress",
6111
+ "mousedown",
6112
+ "mouseup",
6113
+ "mouseover",
6114
+ "mouseout",
6115
+ "mouseenter",
6116
+ "mouseleave",
6117
+ "mousemove",
6118
+ "pointerdown",
6119
+ "pointerup",
6120
+ "pointermove",
6121
+ "focus",
6122
+ "blur",
6123
+ "load",
6124
+ "resize",
6125
+ "scroll"
6126
+ ];
6071
6127
  var _BaseTokenizer = class _BaseTokenizer2 {
6072
6128
  constructor() {
6073
6129
  this.profileKeywords = [];
6130
+ this.multiWordKeywords = [];
6074
6131
  this.profileKeywordMap = /* @__PURE__ */ new Map();
6132
+ this.rawExtraEntries = [];
6075
6133
  this.extractors = [];
6076
6134
  }
6135
+ /** Raw extras as passed in, pre-dedup — for consistency tests. */
6136
+ getExtraKeywordEntries() {
6137
+ return this.rawExtraEntries;
6138
+ }
6077
6139
  /**
6078
6140
  * Tokenize input string to token stream.
6079
6141
  * Delegates to extractor-based tokenization if extractors are registered,
@@ -6142,6 +6204,12 @@ var _BaseTokenizer = class _BaseTokenizer2 {
6142
6204
  pos++;
6143
6205
  }
6144
6206
  if (pos >= input.length) break;
6207
+ const multiWord = this.tryMultiWordKeyword(input, pos);
6208
+ if (multiWord) {
6209
+ tokens.push(multiWord);
6210
+ pos = multiWord.position.end;
6211
+ continue;
6212
+ }
6145
6213
  let extracted = false;
6146
6214
  for (const extractor of this.extractors) {
6147
6215
  if (extractor.canExtract(input, pos)) {
@@ -6241,6 +6309,7 @@ var _BaseTokenizer = class _BaseTokenizer2 {
6241
6309
  */
6242
6310
  initializeKeywordsFromProfile(profile, extras = []) {
6243
6311
  const keywordMap = /* @__PURE__ */ new Map();
6312
+ this.rawExtraEntries = extras;
6244
6313
  if (profile.keywords) {
6245
6314
  for (const [normalized2, translation] of Object.entries(profile.keywords)) {
6246
6315
  keywordMap.set(translation.primary, {
@@ -6284,12 +6353,20 @@ var _BaseTokenizer = class _BaseTokenizer2 {
6284
6353
  keywordMap.set(native, { native, normalized: normalized2 });
6285
6354
  }
6286
6355
  }
6356
+ for (const evt of ENGLISH_DOM_EVENT_NAMES) {
6357
+ if (!keywordMap.has(evt)) {
6358
+ keywordMap.set(evt, { native: evt, normalized: evt });
6359
+ }
6360
+ }
6287
6361
  for (const extra of extras) {
6288
6362
  keywordMap.set(extra.native, extra);
6289
6363
  }
6290
6364
  this.profileKeywords = Array.from(keywordMap.values()).sort(
6291
6365
  (a, b) => b.native.length - a.native.length
6292
6366
  );
6367
+ this.multiWordKeywords = this.profileKeywords.filter(
6368
+ (k) => k.native.includes(" ") && !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
6369
+ );
6293
6370
  this.profileKeywordMap = /* @__PURE__ */ new Map();
6294
6371
  for (const keyword of this.profileKeywords) {
6295
6372
  this.profileKeywordMap.set(keyword.native.toLowerCase(), keyword);
@@ -6331,6 +6408,35 @@ var _BaseTokenizer = class _BaseTokenizer2 {
6331
6408
  }
6332
6409
  return null;
6333
6410
  }
6411
+ /**
6412
+ * Match the longest multi-word (space-containing) profile keyword at `pos`,
6413
+ * requiring the match to end at a word boundary. The profile-driven
6414
+ * counterpart of the per-language hardcoded compound lists (the hindi and
6415
+ * vietnamese keyword extractors). Returns a keyword token (with the normalized
6416
+ * form) or null. Case-sensitive against the stored native form, mirroring
6417
+ * `tryProfileKeyword`/`isKeywordStart` (the i18n dicts emit a fixed surface
6418
+ * case). No-op when `multiWordKeywords` is empty (no-space/CJK languages).
6419
+ *
6420
+ * @param input - Input string
6421
+ * @param pos - Current position (must be a token-start boundary)
6422
+ * @param isWordChar - End-boundary predicate (defaults to Unicode letter/digit/_)
6423
+ */
6424
+ tryMultiWordKeyword(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
6425
+ if (this.multiWordKeywords.length === 0) return null;
6426
+ const rest = input.slice(pos);
6427
+ for (const entry of this.multiWordKeywords) {
6428
+ if (!rest.startsWith(entry.native)) continue;
6429
+ const after = input[pos + entry.native.length];
6430
+ if (after !== void 0 && isWordChar(after)) continue;
6431
+ return createToken(
6432
+ entry.native,
6433
+ "keyword",
6434
+ createPosition(pos, pos + entry.native.length),
6435
+ entry.normalized
6436
+ );
6437
+ }
6438
+ return null;
6439
+ }
6334
6440
  /**
6335
6441
  * Check if the remaining input starts with any known keyword.
6336
6442
  * Useful for non-space languages to detect word boundaries.
@@ -6343,6 +6449,32 @@ var _BaseTokenizer = class _BaseTokenizer2 {
6343
6449
  const remaining = input.slice(pos);
6344
6450
  return this.profileKeywords.some((entry) => remaining.startsWith(entry.native));
6345
6451
  }
6452
+ /**
6453
+ * Check if a known keyword starts at the given position AND ends at a word
6454
+ * boundary (end of input or a non-word character).
6455
+ *
6456
+ * Space-delimited languages must use this (not `isKeywordStart`) for
6457
+ * word-walk break checks: the keyword table includes English canonical
6458
+ * fallbacks (me, it, you, …), so a raw `startsWith` check splits any native
6459
+ * word with an embedded fallback mid-word (e.g. Quechua ñit'iy contains
6460
+ * "it"). CJK/no-space tokenizers rely on mid-text keyword starts and must
6461
+ * keep using `isKeywordStart`.
6462
+ *
6463
+ * @param input - Input string
6464
+ * @param pos - Current position
6465
+ * @param isWordChar - Language-specific word-character predicate; pass the
6466
+ * tokenizer's letter classifier so e.g. the Quechua glottal apostrophe
6467
+ * counts as part of a word. Defaults to Unicode letters/digits/underscore.
6468
+ * @returns true if a keyword starts here and is not followed by a word char
6469
+ */
6470
+ isKeywordStartAtBoundary(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
6471
+ const remaining = input.slice(pos);
6472
+ return this.profileKeywords.some((entry) => {
6473
+ if (!remaining.startsWith(entry.native)) return false;
6474
+ const after = input[pos + entry.native.length];
6475
+ return after === void 0 || !isWordChar(after);
6476
+ });
6477
+ }
6346
6478
  /**
6347
6479
  * Look up a keyword by native word (case-insensitive).
6348
6480
  * O(1) lookup using the keyword map.
package/dist/server.js CHANGED
@@ -1137,7 +1137,7 @@ try {
1137
1137
  }
1138
1138
  var frameworkIR = null;
1139
1139
  try {
1140
- const fw = await import("./dist-NRPF26M4.js");
1140
+ const fw = await import("./dist-7HGHK5EJ.js");
1141
1141
  frameworkIR = { fromInterchangeNode: fw.fromInterchangeNode, renderExplicit: fw.renderExplicit };
1142
1142
  console.error(
1143
1143
  "[lokascript-ls] @lokascript/framework loaded \u2014 LSE bracket notation in hover enabled"
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@lokascript/language-server",
3
- "version": "2.5.0",
3
+ "version": "2.6.0",
4
4
  "description": "Language Server Protocol implementation for LokaScript/hyperscript with 21 language support",
5
5
  "type": "module",
6
6
  "main": "dist/server.js",
@@ -13,7 +13,7 @@
13
13
  "dev": "tsx src/server.ts --stdio",
14
14
  "test": "vitest",
15
15
  "typecheck": "tsc --noEmit",
16
- "test:check": "vitest run --reporter=dot 2>&1 | tail -5"
16
+ "test:check": "VITEST_QUIET=1 bash ../../scripts/vitest-run.sh --reporter=dot"
17
17
  },
18
18
  "keywords": [
19
19
  "lsp",
@@ -31,8 +31,8 @@
31
31
  "vscode-languageserver-textdocument": "^1.0.11"
32
32
  },
33
33
  "peerDependencies": {
34
- "@hyperfixi/core": "^2.5.0",
35
- "@lokascript/semantic": "^2.5.0"
34
+ "@hyperfixi/core": "^2.6.0",
35
+ "@lokascript/semantic": "^2.6.0"
36
36
  },
37
37
  "peerDependenciesMeta": {
38
38
  "@lokascript/semantic": {
@@ -43,8 +43,8 @@
43
43
  }
44
44
  },
45
45
  "devDependencies": {
46
- "@hyperfixi/core": "^2.5.0",
47
- "@lokascript/semantic": "^2.5.0",
46
+ "@hyperfixi/core": "^2.6.0",
47
+ "@lokascript/semantic": "^2.6.0",
48
48
  "@types/node": "^20.0.0",
49
49
  "@vitest/coverage-v8": "^4.0.0",
50
50
  "tsup": "^8.0.0",