@lokascript/language-server 2.5.1 → 2.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3152,7 +3152,8 @@ function createTokenizerContext(tokenizer) {
3152
3152
  direction: tokenizer.direction,
3153
3153
  lookupKeyword: tokenizer.lookupKeyword.bind(tokenizer),
3154
3154
  isKeyword: tokenizer.isKeyword.bind(tokenizer),
3155
- isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer)
3155
+ isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer),
3156
+ ...tokenizer.isKeywordStartAtBoundary ? { isKeywordStartAtBoundary: tokenizer.isKeywordStartAtBoundary.bind(tokenizer) } : {}
3156
3157
  };
3157
3158
  if (tokenizer.normalizer) {
3158
3159
  return { ...ctx, normalizer: tokenizer.normalizer };
@@ -6053,27 +6054,74 @@ function withDefaultExtractors(tokenizer) {
6053
6054
  tokenizer.registerExtractors(getDefaultExtractors());
6054
6055
  return tokenizer;
6055
6056
  }
6056
- function createUnicodeRangeClassifier(ranges) {
6057
- return (char) => {
6058
- const code = char.charCodeAt(0);
6059
- return ranges.some(([start, end]) => code >= start && code <= end);
6060
- };
6061
- }
6062
- function combineClassifiers(...classifiers) {
6063
- return (char) => classifiers.some((fn) => fn(char));
6064
- }
6065
- function createLatinCharClassifiers(letterPattern) {
6066
- const isLetter = (char) => letterPattern.test(char);
6067
- const isIdentifierChar = (char) => isLetter(char) || /[0-9_-]/.test(char);
6068
- return { isLetter, isIdentifierChar };
6069
- }
6070
6057
  var SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
6058
+ var MARKER_CONCEPT_NORMALIZEDS = /* @__PURE__ */ new Set([
6059
+ // Role-marker role names (profile.roleMarkers normalizeds)
6060
+ "patient",
6061
+ "destination",
6062
+ "source",
6063
+ "style",
6064
+ "event",
6065
+ "eventMarker",
6066
+ "agent",
6067
+ "goal",
6068
+ "manner",
6069
+ // Prepositional / positional modifier concepts matched via the role mechanism
6070
+ // (profile.keywords "Modifiers"). `before`/`after`/`until` are intentionally
6071
+ // NOT here — they are pattern literals (see the note above).
6072
+ "into",
6073
+ "from",
6074
+ "to",
6075
+ "with",
6076
+ "at",
6077
+ "of",
6078
+ "as",
6079
+ "by",
6080
+ "in",
6081
+ "on",
6082
+ "over",
6083
+ "under",
6084
+ "between",
6085
+ "through",
6086
+ "without"
6087
+ ]);
6088
+ var ENGLISH_DOM_EVENT_NAMES = [
6089
+ "click",
6090
+ "dblclick",
6091
+ "input",
6092
+ "change",
6093
+ "submit",
6094
+ "keydown",
6095
+ "keyup",
6096
+ "keypress",
6097
+ "mousedown",
6098
+ "mouseup",
6099
+ "mouseover",
6100
+ "mouseout",
6101
+ "mouseenter",
6102
+ "mouseleave",
6103
+ "mousemove",
6104
+ "pointerdown",
6105
+ "pointerup",
6106
+ "pointermove",
6107
+ "focus",
6108
+ "blur",
6109
+ "load",
6110
+ "resize",
6111
+ "scroll"
6112
+ ];
6071
6113
  var _BaseTokenizer = class _BaseTokenizer2 {
6072
6114
  constructor() {
6073
6115
  this.profileKeywords = [];
6116
+ this.multiWordKeywords = [];
6074
6117
  this.profileKeywordMap = /* @__PURE__ */ new Map();
6118
+ this.rawExtraEntries = [];
6075
6119
  this.extractors = [];
6076
6120
  }
6121
+ /** Raw extras as passed in, pre-dedup — for consistency tests. */
6122
+ getExtraKeywordEntries() {
6123
+ return this.rawExtraEntries;
6124
+ }
6077
6125
  /**
6078
6126
  * Tokenize input string to token stream.
6079
6127
  * Delegates to extractor-based tokenization if extractors are registered,
@@ -6142,6 +6190,12 @@ var _BaseTokenizer = class _BaseTokenizer2 {
6142
6190
  pos++;
6143
6191
  }
6144
6192
  if (pos >= input.length) break;
6193
+ const multiWord = this.tryMultiWordKeyword(input, pos);
6194
+ if (multiWord) {
6195
+ tokens.push(multiWord);
6196
+ pos = multiWord.position.end;
6197
+ continue;
6198
+ }
6145
6199
  let extracted = false;
6146
6200
  for (const extractor of this.extractors) {
6147
6201
  if (extractor.canExtract(input, pos)) {
@@ -6241,6 +6295,7 @@ var _BaseTokenizer = class _BaseTokenizer2 {
6241
6295
  */
6242
6296
  initializeKeywordsFromProfile(profile, extras = []) {
6243
6297
  const keywordMap = /* @__PURE__ */ new Map();
6298
+ this.rawExtraEntries = extras;
6244
6299
  if (profile.keywords) {
6245
6300
  for (const [normalized2, translation] of Object.entries(profile.keywords)) {
6246
6301
  keywordMap.set(translation.primary, {
@@ -6284,12 +6339,20 @@ var _BaseTokenizer = class _BaseTokenizer2 {
6284
6339
  keywordMap.set(native, { native, normalized: normalized2 });
6285
6340
  }
6286
6341
  }
6342
+ for (const evt of ENGLISH_DOM_EVENT_NAMES) {
6343
+ if (!keywordMap.has(evt)) {
6344
+ keywordMap.set(evt, { native: evt, normalized: evt });
6345
+ }
6346
+ }
6287
6347
  for (const extra of extras) {
6288
6348
  keywordMap.set(extra.native, extra);
6289
6349
  }
6290
6350
  this.profileKeywords = Array.from(keywordMap.values()).sort(
6291
6351
  (a, b) => b.native.length - a.native.length
6292
6352
  );
6353
+ this.multiWordKeywords = this.profileKeywords.filter(
6354
+ (k) => k.native.includes(" ") && !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
6355
+ );
6293
6356
  this.profileKeywordMap = /* @__PURE__ */ new Map();
6294
6357
  for (const keyword of this.profileKeywords) {
6295
6358
  this.profileKeywordMap.set(keyword.native.toLowerCase(), keyword);
@@ -6331,6 +6394,35 @@ var _BaseTokenizer = class _BaseTokenizer2 {
6331
6394
  }
6332
6395
  return null;
6333
6396
  }
6397
+ /**
6398
+ * Match the longest multi-word (space-containing) profile keyword at `pos`,
6399
+ * requiring the match to end at a word boundary. The profile-driven
6400
+ * counterpart of the per-language hardcoded compound lists (the hindi and
6401
+ * vietnamese keyword extractors). Returns a keyword token (with the normalized
6402
+ * form) or null. Case-sensitive against the stored native form, mirroring
6403
+ * `tryProfileKeyword`/`isKeywordStart` (the i18n dicts emit a fixed surface
6404
+ * case). No-op when `multiWordKeywords` is empty (no-space/CJK languages).
6405
+ *
6406
+ * @param input - Input string
6407
+ * @param pos - Current position (must be a token-start boundary)
6408
+ * @param isWordChar - End-boundary predicate (defaults to Unicode letter/digit/_)
6409
+ */
6410
+ tryMultiWordKeyword(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
6411
+ if (this.multiWordKeywords.length === 0) return null;
6412
+ const rest = input.slice(pos);
6413
+ for (const entry of this.multiWordKeywords) {
6414
+ if (!rest.startsWith(entry.native)) continue;
6415
+ const after = input[pos + entry.native.length];
6416
+ if (after !== void 0 && isWordChar(after)) continue;
6417
+ return createToken(
6418
+ entry.native,
6419
+ "keyword",
6420
+ createPosition(pos, pos + entry.native.length),
6421
+ entry.normalized
6422
+ );
6423
+ }
6424
+ return null;
6425
+ }
6334
6426
  /**
6335
6427
  * Check if the remaining input starts with any known keyword.
6336
6428
  * Useful for non-space languages to detect word boundaries.
@@ -6343,6 +6435,32 @@ var _BaseTokenizer = class _BaseTokenizer2 {
6343
6435
  const remaining = input.slice(pos);
6344
6436
  return this.profileKeywords.some((entry) => remaining.startsWith(entry.native));
6345
6437
  }
6438
+ /**
6439
+ * Check if a known keyword starts at the given position AND ends at a word
6440
+ * boundary (end of input or a non-word character).
6441
+ *
6442
+ * Space-delimited languages must use this (not `isKeywordStart`) for
6443
+ * word-walk break checks: the keyword table includes English canonical
6444
+ * fallbacks (me, it, you, …), so a raw `startsWith` check splits any native
6445
+ * word with an embedded fallback mid-word (e.g. Quechua ñit'iy contains
6446
+ * "it"). CJK/no-space tokenizers rely on mid-text keyword starts and must
6447
+ * keep using `isKeywordStart`.
6448
+ *
6449
+ * @param input - Input string
6450
+ * @param pos - Current position
6451
+ * @param isWordChar - Language-specific word-character predicate; pass the
6452
+ * tokenizer's letter classifier so e.g. the Quechua glottal apostrophe
6453
+ * counts as part of a word. Defaults to Unicode letters/digits/underscore.
6454
+ * @returns true if a keyword starts here and is not followed by a word char
6455
+ */
6456
+ isKeywordStartAtBoundary(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
6457
+ const remaining = input.slice(pos);
6458
+ return this.profileKeywords.some((entry) => {
6459
+ if (!remaining.startsWith(entry.native)) return false;
6460
+ const after = input[pos + entry.native.length];
6461
+ return after === void 0 || !isWordChar(after);
6462
+ });
6463
+ }
6346
6464
  /**
6347
6465
  * Look up a keyword by native word (case-insensitive).
6348
6466
  * O(1) lookup using the keyword map.
@@ -6670,6 +6788,115 @@ function createSimpleTokenizer(config) {
6670
6788
  }
6671
6789
  return new SimpleTokenizer();
6672
6790
  }
6791
+ function mergeRoleMarkers(slice, vocab) {
6792
+ const merged = {};
6793
+ const add = (role, marker) => {
6794
+ if (!marker?.primary) {
6795
+ delete merged[role];
6796
+ return;
6797
+ }
6798
+ merged[role] = {
6799
+ primary: marker.primary,
6800
+ ...marker.alternatives?.length && { alternatives: [...marker.alternatives] },
6801
+ ...marker.position && { position: marker.position }
6802
+ };
6803
+ };
6804
+ for (const [role, marker] of Object.entries(slice.roleMarkers ?? {})) add(role, marker);
6805
+ for (const [role, marker] of Object.entries(vocab.roleMarkerOverrides ?? {})) add(role, marker);
6806
+ return merged;
6807
+ }
6808
+ function buildPatternProfile(slice, vocab) {
6809
+ const keywords = {};
6810
+ for (const [action, translation] of Object.entries(vocab.keywords)) {
6811
+ keywords[action] = {
6812
+ primary: translation.primary,
6813
+ ...translation.alternatives?.length && { alternatives: [...translation.alternatives] }
6814
+ };
6815
+ }
6816
+ const roleMarkers = mergeRoleMarkers(slice, vocab);
6817
+ return {
6818
+ code: slice.code,
6819
+ wordOrder: slice.wordOrder,
6820
+ keywords,
6821
+ ...Object.keys(roleMarkers).length > 0 && { roleMarkers }
6822
+ };
6823
+ }
6824
+ function defaultCaseInsensitive(script) {
6825
+ return script === void 0 || script === "latin" || script === "cyrillic";
6826
+ }
6827
+ function buildDomainTokenizer(slice, vocab, options = {}) {
6828
+ const roleMarkers = mergeRoleMarkers(slice, vocab);
6829
+ const keywords = /* @__PURE__ */ new Set();
6830
+ for (const translation of Object.values(vocab.keywords)) {
6831
+ keywords.add(translation.primary);
6832
+ for (const alt of translation.alternatives ?? []) keywords.add(alt);
6833
+ }
6834
+ for (const marker of Object.values(roleMarkers)) {
6835
+ keywords.add(marker.primary);
6836
+ for (const alt of marker.alternatives ?? []) keywords.add(alt);
6837
+ }
6838
+ for (const particle of slice.tokenization?.particles ?? []) keywords.add(particle);
6839
+ for (const extra of vocab.tokenizerKeywords ?? []) keywords.add(extra);
6840
+ const profileKeywords = {};
6841
+ for (const [action, translation] of Object.entries(vocab.keywords)) {
6842
+ profileKeywords[action] = {
6843
+ primary: translation.primary,
6844
+ ...translation.alternatives?.length && { alternatives: [...translation.alternatives] },
6845
+ normalized: translation.normalized ?? action
6846
+ };
6847
+ }
6848
+ const keywordProfile = {
6849
+ keywords: profileKeywords,
6850
+ ...Object.keys(roleMarkers).length > 0 && { roleMarkers }
6851
+ };
6852
+ const customExtractors = [
6853
+ ...options.customExtractors ?? [],
6854
+ ...slice.script === "latin" ? [new LatinExtendedIdentifierExtractor()] : []
6855
+ ];
6856
+ return createSimpleTokenizer({
6857
+ language: slice.code,
6858
+ direction: slice.direction ?? "ltr",
6859
+ keywords: [...keywords],
6860
+ ...vocab.keywordExtras?.length && { keywordExtras: vocab.keywordExtras.map((e) => ({ ...e })) },
6861
+ keywordProfile,
6862
+ includeOperators: options.includeOperators ?? false,
6863
+ caseInsensitive: options.caseInsensitive ?? defaultCaseInsensitive(slice.script),
6864
+ ...customExtractors.length > 0 && { customExtractors }
6865
+ });
6866
+ }
6867
+ function buildLanguageConfig(slice, vocab, meta = {}) {
6868
+ const name = meta.name ?? slice.name ?? slice.code;
6869
+ return {
6870
+ code: slice.code,
6871
+ name,
6872
+ nativeName: meta.nativeName ?? slice.nativeName ?? name,
6873
+ tokenizer: meta.tokenizer ?? buildDomainTokenizer(slice, vocab, meta.tokenizerOptions),
6874
+ patternProfile: buildPatternProfile(slice, vocab),
6875
+ ...meta.grammarProfile && { grammarProfile: meta.grammarProfile }
6876
+ };
6877
+ }
6878
+ function deriveRoleMarkers(slice, roleMapping) {
6879
+ const derived = {};
6880
+ for (const [domainRole, semanticRole] of Object.entries(roleMapping)) {
6881
+ const marker = slice.roleMarkers?.[semanticRole];
6882
+ if (marker?.primary) derived[domainRole] = marker.primary;
6883
+ }
6884
+ return derived;
6885
+ }
6886
+ function createUnicodeRangeClassifier(ranges) {
6887
+ return (char) => {
6888
+ const code = char.charCodeAt(0);
6889
+ return ranges.some(([start, end]) => code >= start && code <= end);
6890
+ };
6891
+ }
6892
+ function combineClassifiers(...classifiers) {
6893
+ return (char) => classifiers.some((fn) => fn(char));
6894
+ }
6895
+ function createLatinCharClassifiers(letterPattern) {
6896
+ const isLetter = (char) => letterPattern.test(char);
6897
+ const isIdentifierChar = (char) => isLetter(char) || /[0-9_-]/.test(char);
6898
+ return { isLetter, isIdentifierChar };
6899
+ }
6673
6900
  function noChange(word) {
6674
6901
  return { stem: word, confidence: 1 };
6675
6902
  }
@@ -7224,7 +7451,10 @@ export {
7224
7451
  WhitespaceExtractor,
7225
7452
  accumulateBlocks,
7226
7453
  buildDisambiguation,
7454
+ buildDomainTokenizer,
7227
7455
  buildFeedback,
7456
+ buildLanguageConfig,
7457
+ buildPatternProfile,
7228
7458
  buildPhrase,
7229
7459
  buildTablesFromProfiles,
7230
7460
  combineClassifiers,
@@ -7255,6 +7485,7 @@ export {
7255
7485
  createUnicodeRangeClassifier,
7256
7486
  defineCommand,
7257
7487
  defineRole,
7488
+ deriveRoleMarkers,
7258
7489
  detectWordOrders,
7259
7490
  extractCssSelector,
7260
7491
  extractNumber,
package/dist/server.js CHANGED
@@ -140,9 +140,6 @@ function detectLokascriptFeatures(code) {
140
140
  }
141
141
  return detected;
142
142
  }
143
- function getCommandsForMode(mode) {
144
- return mode === "hyperscript" ? HYPERSCRIPT_COMMANDS : ALL_COMMANDS;
145
- }
146
143
 
147
144
  // src/extraction.ts
148
145
  function isHtmlDocument(uri, content) {
@@ -1137,7 +1134,7 @@ try {
1137
1134
  }
1138
1135
  var frameworkIR = null;
1139
1136
  try {
1140
- const fw = await import("./dist-NRPF26M4.js");
1137
+ const fw = await import("./dist-MRVDB2YM.js");
1141
1138
  frameworkIR = { fromInterchangeNode: fw.fromInterchangeNode, renderExplicit: fw.renderExplicit };
1142
1139
  console.error(
1143
1140
  "[lokascript-ls] @lokascript/framework loaded \u2014 LSE bracket notation in hover enabled"
@@ -1441,7 +1438,7 @@ function getSemanticAnalyzer() {
1441
1438
  return cachedAnalyzer;
1442
1439
  }
1443
1440
  var globalSettings = envDefaultMode ? { ...defaultSettings, mode: envDefaultMode } : defaultSettings;
1444
- connection.onInitialize((params) => {
1441
+ connection.onInitialize((_params) => {
1445
1442
  resolvedMode = resolveMode(globalSettings);
1446
1443
  const brand = getBranding(resolvedMode);
1447
1444
  connection.console.log(`${brand} Language Server initializing (mode: ${resolvedMode})...`);
@@ -1813,15 +1810,12 @@ function getContextualCompletions(context, language) {
1813
1810
  }
1814
1811
  return eng;
1815
1812
  };
1816
- const commandMode = isHyperscriptCompatMode(resolvedMode) ? "hyperscript" : "lokascript";
1817
- const availableCommands = getCommandsForMode(commandMode);
1818
- const isCommandAvailable = (cmd) => availableCommands.includes(cmd);
1819
1813
  const getDetail = (command, fallback) => {
1820
1814
  const desc = getCommandDescription(command, effectiveLanguage);
1821
1815
  return desc?.detail ?? fallback;
1822
1816
  };
1823
1817
  switch (context) {
1824
- case "event":
1818
+ case "event": {
1825
1819
  const eventNames = lspMetadata?.EVENT_NAMES ?? FALLBACK_EVENT_NAMES;
1826
1820
  for (const eventName of eventNames) {
1827
1821
  completions.push({
@@ -1831,6 +1825,7 @@ function getContextualCompletions(context, language) {
1831
1825
  });
1832
1826
  }
1833
1827
  break;
1828
+ }
1834
1829
  case "command":
1835
1830
  completions.push(
1836
1831
  {
@@ -2174,7 +2169,7 @@ ${doc.example}
2174
2169
  var translationCache = null;
2175
2170
  function buildTranslationCache() {
2176
2171
  const cache2 = /* @__PURE__ */ new Map();
2177
- if (!semanticPackage) return cache2;
2172
+ if (!semanticPackage?.getKeywordTranslations) return cache2;
2178
2173
  const langMap = {
2179
2174
  es: "Spanish",
2180
2175
  ja: "Japanese",
@@ -2208,19 +2203,15 @@ function buildTranslationCache() {
2208
2203
  "result"
2209
2204
  ];
2210
2205
  for (const canonicalKey of canonicalKeywords) {
2206
+ const byLang = semanticPackage.getKeywordTranslations(
2207
+ canonicalKey,
2208
+ Object.keys(langMap)
2209
+ );
2211
2210
  const translations = [];
2212
2211
  for (const [lang, langName] of Object.entries(langMap)) {
2213
- try {
2214
- const profile = semanticPackage.tryGetProfile(lang);
2215
- const kw = profile?.keywords?.[canonicalKey];
2216
- if (kw) {
2217
- const trans = kw;
2218
- if (trans.primary && trans.primary !== canonicalKey) {
2219
- translations.push(`${langName}: \`${trans.primary}\``);
2220
- }
2221
- }
2222
- } catch (e) {
2223
- connection.console.log(`[lokascript-ls] Translation cache: profile '${lang}' not found`);
2212
+ const primary = byLang[lang]?.primary;
2213
+ if (primary && primary !== canonicalKey) {
2214
+ translations.push(`${langName}: \`${primary}\``);
2224
2215
  }
2225
2216
  }
2226
2217
  if (translations.length > 0) {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@lokascript/language-server",
3
- "version": "2.5.1",
3
+ "version": "2.7.0",
4
4
  "description": "Language Server Protocol implementation for LokaScript/hyperscript with 21 language support",
5
5
  "type": "module",
6
6
  "main": "dist/server.js",
@@ -13,7 +13,7 @@
13
13
  "dev": "tsx src/server.ts --stdio",
14
14
  "test": "vitest",
15
15
  "typecheck": "tsc --noEmit",
16
- "test:check": "vitest run --reporter=dot 2>&1 | tail -5"
16
+ "test:check": "VITEST_QUIET=1 bash ../../scripts/vitest-run.sh --reporter=dot"
17
17
  },
18
18
  "keywords": [
19
19
  "lsp",
@@ -31,8 +31,8 @@
31
31
  "vscode-languageserver-textdocument": "^1.0.11"
32
32
  },
33
33
  "peerDependencies": {
34
- "@hyperfixi/core": "^2.5.1",
35
- "@lokascript/semantic": "^2.5.1"
34
+ "@hyperfixi/core": "^2.7.0",
35
+ "@lokascript/semantic": "^2.7.0"
36
36
  },
37
37
  "peerDependenciesMeta": {
38
38
  "@lokascript/semantic": {
@@ -43,8 +43,8 @@
43
43
  }
44
44
  },
45
45
  "devDependencies": {
46
- "@hyperfixi/core": "^2.5.1",
47
- "@lokascript/semantic": "^2.5.1",
46
+ "@hyperfixi/core": "^2.7.0",
47
+ "@lokascript/semantic": "^2.7.0",
48
48
  "@types/node": "^20.0.0",
49
49
  "@vitest/coverage-v8": "^4.0.0",
50
50
  "tsup": "^8.0.0",