@atlaskit/editor-plugin-autocomplete 3.0.0 → 3.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -161,20 +161,21 @@ function applyGrammarFilter(candidates, previousWord) {
161
161
  dropped.push(entry.candidate.word);
162
162
  }
163
163
  }
164
+
165
+ // Grammar is authoritative.
164
166
  } catch (err) {
165
167
  _iterator2.e(err);
166
168
  } finally {
167
169
  _iterator2.f();
168
170
  }
169
- var finalFiltered = filtered.length > 0 ? filtered : candidates;
170
171
  return {
171
- filtered: finalFiltered,
172
+ filtered: filtered,
172
173
  grammarMeta: {
173
174
  prevWord: lowerPrev,
174
175
  prevTags: prevTags,
175
176
  before: candidates.length,
176
- after: finalFiltered.length,
177
- dropped: filtered.length > 0 ? dropped : []
177
+ after: filtered.length,
178
+ dropped: dropped
178
179
  }
179
180
  };
180
181
  }
@@ -1,4 +1,5 @@
1
1
  import _asyncToGenerator from "@babel/runtime/helpers/asyncToGenerator";
2
+ import _typeof from "@babel/runtime/helpers/typeof";
2
3
  import _slicedToArray from "@babel/runtime/helpers/slicedToArray";
3
4
  import _createClass from "@babel/runtime/helpers/createClass";
4
5
  import _classCallCheck from "@babel/runtime/helpers/classCallCheck";
@@ -26,11 +27,9 @@ function _arrayLikeToArray(r, a) { (null == a || a > r.length) && (a = r.length)
26
27
 
27
28
  import { EXPERIENCE_NAME, failExp, startExp, succeedExp } from '../analytics/ufo';
28
29
 
29
- // import bigramsData from './data/bigrams.json';
30
- import l3VocabularyData from './data/l3_vocabulary.json';
31
- import vocabularyData from './data/vocabulary_10k.json';
32
- import wordIndexData from './data/word_index_10k.json';
33
- // import { rankCandidates, isGrammarAllowed } from './scoring-pipeline';
30
+ // The vocabulary, L3 and word-index JSON payloads are dynamically imported in
31
+ // loadDefaultVocabulary / loadVectorsAsync so their (large) contents stay out of
32
+ // the editor's main chunk and only load when autocomplete is initialised.
34
33
  import { isAutocompleteDebugEnabled } from './debug-mode';
35
34
  import { rankCandidates, STAGE1_WEIGHT, STAGE2_WEIGHT, MIN_STAGE1_SCORE } from './scoring-pipeline';
36
35
  import { getStoredContextVector, getStoredLmLogits } from './slow-lane-client';
@@ -42,7 +41,7 @@ var PUNCTUATION_BOUNDARY_REGEX = /^[.,;:!?()\[\]{}"'`]+|[.,;:!?()\[\]{}"'`]+$/g;
42
41
  var MIN_PREFIX_LENGTH = 3;
43
42
  var MAX_CANDIDATES = 200;
44
43
  var CONTEXT_WORDS = 10;
45
- var MIN_SCORE_THRESHOLD = 0.2;
44
+ var MIN_SCORE_THRESHOLD = 0.35;
46
45
  var L3_BASELINE_FREQ = 0.001;
47
46
 
48
47
  // ─── Types ───────────────────────────────────────────────────────────────────
@@ -206,7 +205,6 @@ var wordTrie = new WeightedWordTrie();
206
205
  // L3 Trie (General English Fallback)
207
206
  var l3Trie = new WeightedWordTrie();
208
207
 
209
- // --- Initialization Function ---
210
208
  /**
211
209
  * Loads the General English vocabulary.
212
210
  * expects a simple array of strings: ["about", "above", "actually", ...]
@@ -330,14 +328,11 @@ var tokenize = function tokenize(text) {
330
328
  return tokens;
331
329
  };
332
330
  var extractPreviousWord = function extractPreviousWord(text) {
333
- // 1. Split the text by newlines or punctuation (. ? !)
331
+ // Only consider the current sentence/line the user is typing in.
334
332
  // eslint-disable-next-line require-unicode-regexp
335
333
  var sentences = text.split(/[\n.?!]+/);
336
-
337
- // 2. Only look at the current sentence/line the user is typing in
338
334
  var currentSentence = sentences[sentences.length - 1];
339
335
 
340
- // 3. Extract the previous word as normal
341
336
  // eslint-disable-next-line require-unicode-regexp
342
337
  var words = currentSentence.trimEnd().split(/\s+/);
343
338
  return words.length >= 2 ? words[words.length - 2] : '';
@@ -400,7 +395,6 @@ export var incrementSessionFreq = function incrementSessionFreq(word) {
400
395
  * Pass `undefined` (or omit the argument) to skip priming — useful when the
401
396
  * calling context does not yet have a page value available.
402
397
  */
403
- // NOTE: We ingest full page context here
404
398
  export var ingestDocumentPage = function ingestDocumentPage(pageContent) {
405
399
  if (!pageContent) {
406
400
  return;
@@ -433,7 +427,11 @@ export var ingestDocumentPage = function ingestDocumentPage(pageContent) {
433
427
  };
434
428
  export var predict = function predict(textBefore) {
435
429
  if (!isInitialized) {
436
- loadDefaultVocabulary();
430
+ // Vocabulary JSON is code-split and loads asynchronously. Kick off the load
431
+ // and skip this keystroke; the plugin also primes it on focus, so the tries
432
+ // are usually ready before the user types.
433
+ void loadDefaultVocabulary().catch(function () {});
434
+ return null;
437
435
  }
438
436
  var t0 = performance.now();
439
437
 
@@ -485,14 +483,11 @@ export var predict = function predict(textBefore) {
485
483
  if (currentWord.length < MIN_PREFIX_LENGTH) {
486
484
  return null;
487
485
  }
488
-
489
- // 1. Primary Query: Ask the L2 Domain Trie
490
486
  var candidates = wordTrie.getCandidates(currentWord, MAX_CANDIDATES);
491
487
 
492
- // 2. Fallback Query: Gap-fill with the L3 General English Trie
488
+ // Gap-fill from the L3 general-English trie, requesting a full buffer so
489
+ // enough survive de-duplication against the L2 results.
493
490
  if (candidates.length < MAX_CANDIDATES) {
494
- // Ask L3 for MAX_CANDIDATES to guarantee we have enough buffer
495
- // to survive the deduplication process.
496
491
  var l3Candidates = l3Trie.getCandidates(currentWord, MAX_CANDIDATES);
497
492
  var existingWords = new Set(candidates.map(function (c) {
498
493
  return c.word;
@@ -502,8 +497,7 @@ export var predict = function predict(textBefore) {
502
497
  try {
503
498
  for (_iterator0.s(); !(_step0 = _iterator0.n()).done;) {
504
499
  var l3c = _step0.value;
505
- if (candidates.length >= MAX_CANDIDATES) break; // Stop exactly at the limit
506
-
500
+ if (candidates.length >= MAX_CANDIDATES) break;
507
501
  if (!existingWords.has(l3c.word)) {
508
502
  candidates.push(l3c);
509
503
  }
@@ -687,9 +681,55 @@ export var predict = function predict(textBefore) {
687
681
 
688
682
  // ─── Data Loading ────────────────────────────────────────────────────────────
689
683
 
684
+ /**
685
+ * Unwrap a dynamically imported JSON module to its parsed value, handling both
686
+ * interop modes AFM's bundler chain emits: a `.default`-wrapped namespace
687
+ * (classic webpack) and a named-exports namespace (webpack 5 / atlaspack JSON
688
+ * modules, where `default` can be a misleading scalar). Named exports are
689
+ * preferred when present. The caller declares the JSON `shape` because a dense
690
+ * array and a sparse numeric-keyed object are emitted identically as named
691
+ * exports. Kept in lock-step with the matching helper in local-slow-lane-client.ts.
692
+ */
693
+ var unwrapJsonModule = function unwrapJsonModule(mod, shape) {
694
+ if (mod == null || _typeof(mod) !== 'object') {
695
+ return null;
696
+ }
697
+ var namespace = mod;
698
+ var ownKeys = Object.keys(namespace).filter(function (k) {
699
+ return k !== 'default' && k !== '__esModule';
700
+ });
701
+ if (ownKeys.length > 0) {
702
+ if (shape === 'array') {
703
+ var len = ownKeys.length;
704
+ var arr = new Array(len);
705
+ for (var i = 0; i < len; i++) {
706
+ arr[i] = namespace[String(i)];
707
+ }
708
+ return arr;
709
+ }
710
+ var obj = {};
711
+ var _iterator1 = _createForOfIteratorHelper(ownKeys),
712
+ _step1;
713
+ try {
714
+ for (_iterator1.s(); !(_step1 = _iterator1.n()).done;) {
715
+ var k = _step1.value;
716
+ obj[k] = namespace[k];
717
+ }
718
+ } catch (err) {
719
+ _iterator1.e(err);
720
+ } finally {
721
+ _iterator1.f();
722
+ }
723
+ return obj;
724
+ }
725
+ if ('default' in namespace && namespace.default != null) {
726
+ return namespace.default;
727
+ }
728
+ return null;
729
+ };
690
730
  export var loadVectorsAsync = /*#__PURE__*/function () {
691
731
  var _ref6 = _asyncToGenerator( /*#__PURE__*/_regeneratorRuntime.mark(function _callee(options) {
692
- var url, res, buffer, float32, wordIndex, nWords, dim, _t, _t2;
732
+ var url, _wordIndexOuter$index, res, buffer, float32, wordIndexModule, wordIndexOuter, wordIndex, nWords, dim, _t, _t2;
693
733
  return _regeneratorRuntime.wrap(function (_context) {
694
734
  while (1) switch (_context.prev = _context.next) {
695
735
  case 0:
@@ -749,9 +789,20 @@ export var loadVectorsAsync = /*#__PURE__*/function () {
749
789
  return res.arrayBuffer();
750
790
  case 9:
751
791
  buffer = _context.sent;
752
- float32 = new Float32Array(buffer);
753
- wordIndex = wordIndexData;
792
+ float32 = new Float32Array(buffer); // word_index_10k.json is wrapped as `{ "index": {…} }` so no real entry
793
+ // (e.g. the word "default") can shadow the synthetic ESM `default` export
794
+ // the bundler creates for dynamically-imported JSON.
795
+ _context.next = 10;
796
+ return import( /* webpackChunkName: "@atlaskit-internal_editor-plugin-autocomplete-word-index-10k" */'./data/word_index_10k.json');
797
+ case 10:
798
+ wordIndexModule = _context.sent;
799
+ wordIndexOuter = unwrapJsonModule(wordIndexModule, 'object');
800
+ wordIndex = (_wordIndexOuter$index = wordIndexOuter === null || wordIndexOuter === void 0 ? void 0 : wordIndexOuter.index) !== null && _wordIndexOuter$index !== void 0 ? _wordIndexOuter$index : {};
754
801
  nWords = Object.keys(wordIndex).length;
802
+ if (nWords === 0) {
803
+ // eslint-disable-next-line no-console
804
+ console.warn('[text-predictor] word_index_10k.json missing its `index` wrapper — wordIndex is empty, semantic scoring will be a no-op.');
805
+ }
755
806
  dim = float32.length / nWords;
756
807
  vectorStore = {
757
808
  float32: float32,
@@ -771,10 +822,10 @@ export var loadVectorsAsync = /*#__PURE__*/function () {
771
822
  sizeBytes: float32.byteLength
772
823
  });
773
824
  }
774
- _context.next = 11;
825
+ _context.next = 12;
775
826
  break;
776
- case 10:
777
- _context.prev = 10;
827
+ case 11:
828
+ _context.prev = 11;
778
829
  _t2 = _context["catch"](6);
779
830
  vectorsLoadStarted = false;
780
831
  failExp(EXPERIENCE_NAME.LOAD_VECTORS, 'singleton', {
@@ -782,11 +833,11 @@ export var loadVectorsAsync = /*#__PURE__*/function () {
782
833
  });
783
834
  // eslint-disable-next-line no-console
784
835
  console.warn('[text-predictor] Failed to load vectors:', _t2);
785
- case 11:
836
+ case 12:
786
837
  case "end":
787
838
  return _context.stop();
788
839
  }
789
- }, _callee, null, [[3, 5], [6, 10]]);
840
+ }, _callee, null, [[3, 5], [6, 11]]);
790
841
  }));
791
842
  return function loadVectorsAsync(_x) {
792
843
  return _ref6.apply(this, arguments);
@@ -795,40 +846,73 @@ export var loadVectorsAsync = /*#__PURE__*/function () {
795
846
  export var initVectors = function initVectors(store) {
796
847
  vectorStore = store;
797
848
  };
849
+ var vocabularyLoadPromise;
798
850
  export var loadDefaultVocabulary = function loadDefaultVocabulary() {
799
851
  if (isInitialized) {
800
- return;
852
+ return Promise.resolve();
801
853
  }
802
- startExp(EXPERIENCE_NAME.LOAD_VOCABULARY, 'singleton');
803
- try {
804
- // 1. Load the Atlassian Domain (L2)
805
- var data = vocabularyData;
806
- var terms = Object.entries(data.words).map(function (_ref7) {
807
- var _ref8 = _slicedToArray(_ref7, 2),
808
- word = _ref8[0],
809
- stats = _ref8[1];
810
- return {
811
- word: word,
812
- freq: stats.freq,
813
- docFreq: stats.doc_freq,
814
- authorFreq: stats.author_freq
815
- };
816
- });
817
- initVocabulary({
818
- terms: terms
819
- });
820
-
821
- // 2. Load General English (L3)
822
- var l3Words = l3VocabularyData;
823
- initL3Vocabulary(l3Words);
824
- succeedExp(EXPERIENCE_NAME.LOAD_VOCABULARY, 'singleton', {
825
- l2WordCount: terms.length,
826
- l3WordCount: l3Words.length
827
- });
828
- } catch (e) {
829
- failExp(EXPERIENCE_NAME.LOAD_VOCABULARY, 'singleton', {
830
- errorType: 'parse_error'
831
- });
832
- throw e;
854
+ if (vocabularyLoadPromise) {
855
+ return vocabularyLoadPromise;
833
856
  }
857
+ vocabularyLoadPromise = _asyncToGenerator( /*#__PURE__*/_regeneratorRuntime.mark(function _callee2() {
858
+ var _yield$Promise$all, _yield$Promise$all2, vocabularyModule, l3VocabularyModule, vocabularyData, l3VocabularyData, terms, _t3;
859
+ return _regeneratorRuntime.wrap(function (_context2) {
860
+ while (1) switch (_context2.prev = _context2.next) {
861
+ case 0:
862
+ startExp(EXPERIENCE_NAME.LOAD_VOCABULARY, 'singleton');
863
+ _context2.prev = 1;
864
+ _context2.next = 2;
865
+ return Promise.all([import( /* webpackChunkName: "@atlaskit-internal_editor-plugin-autocomplete-vocabulary-10k" */'./data/vocabulary_10k.json'), import( /* webpackChunkName: "@atlaskit-internal_editor-plugin-autocomplete-l3-vocabulary" */'./data/l3_vocabulary.json')]);
866
+ case 2:
867
+ _yield$Promise$all = _context2.sent;
868
+ _yield$Promise$all2 = _slicedToArray(_yield$Promise$all, 2);
869
+ vocabularyModule = _yield$Promise$all2[0];
870
+ l3VocabularyModule = _yield$Promise$all2[1];
871
+ vocabularyData = unwrapJsonModule(vocabularyModule, 'object');
872
+ l3VocabularyData = unwrapJsonModule(l3VocabularyModule, 'array');
873
+ if (!((vocabularyData === null || vocabularyData === void 0 ? void 0 : vocabularyData.words) == null || !Array.isArray(l3VocabularyData))) {
874
+ _context2.next = 3;
875
+ break;
876
+ }
877
+ throw new Error('[text-predictor] vocabulary JSON modules could not be unwrapped');
878
+ case 3:
879
+ terms = Object.entries(vocabularyData.words).map(function (_ref8) {
880
+ var _ref9 = _slicedToArray(_ref8, 2),
881
+ word = _ref9[0],
882
+ stats = _ref9[1];
883
+ return {
884
+ word: word,
885
+ freq: stats.freq,
886
+ docFreq: stats.doc_freq,
887
+ authorFreq: stats.author_freq
888
+ };
889
+ }); // Load L3 before L2: initVocabulary flips `isInitialized = true`, so it
890
+ // must run last — otherwise a throw in initL3Vocabulary would strand
891
+ // `isInitialized` true and the retry path could never reload L3.
892
+ initL3Vocabulary(l3VocabularyData);
893
+ initVocabulary({
894
+ terms: terms
895
+ });
896
+ succeedExp(EXPERIENCE_NAME.LOAD_VOCABULARY, 'singleton', {
897
+ l2WordCount: terms.length,
898
+ l3WordCount: l3VocabularyData.length
899
+ });
900
+ _context2.next = 5;
901
+ break;
902
+ case 4:
903
+ _context2.prev = 4;
904
+ _t3 = _context2["catch"](1);
905
+ failExp(EXPERIENCE_NAME.LOAD_VOCABULARY, 'singleton', {
906
+ errorType: 'parse_error'
907
+ });
908
+ // Allow a later call to retry the load rather than caching the failure.
909
+ vocabularyLoadPromise = undefined;
910
+ throw _t3;
911
+ case 5:
912
+ case "end":
913
+ return _context2.stop();
914
+ }
915
+ }, _callee2, null, [[1, 4]]);
916
+ }))();
917
+ return vocabularyLoadPromise;
834
918
  };
@@ -1,18 +1,27 @@
1
1
  /**
2
2
  * Local Slow Lane Client: On-device inference via @mlc-ai/web-llm.
3
3
  *
4
- * Drop-in replacement for the network-based slow-lane-client. Instead of
5
- * calling a backend API, this client uses MLC WebLLM to run a small language
6
- * model (SmolLM 135M) directly in the browser via WebGPU.
4
+ * Drop-in replacement for the network-based slow-lane-client. Instead of calling
5
+ * a backend API, this client runs two models in the browser via WebGPU, in a
6
+ * single MLCEngine, to reproduce the BE encoder's outputs on-device:
7
+ *
8
+ * - Causal LM (SmolLM2-135M-Instruct): one decode step per word boundary. A
9
+ * registered LogitProcessor captures the raw next-token logits, which
10
+ * `computeBePayload` turns into a whole-word `lm_logits` payload — a faithful
11
+ * port of the BE `CausalLMEncoder._get_top_k_probs` (masked softmax over the
12
+ * vocab's first-tokens, prefix expansion, L2 reservation, log-space pooling).
13
+ * - Semantic embedder (Snowflake Arctic Embed S): produces the real 384-d
14
+ * `semantic_vector`. Inputs are wrapped as passages (see `wrapForArctic`) so
15
+ * the runtime vector lands in the same space as the precomputed word bin.
7
16
  *
8
17
  * ── Why main thread (no Web Worker)? ─────────────────────────────────────
9
- * SmolLM 135M is small enough (~270 MB weights, 350-400 MB VRAM) that
10
- * WebGPU inference on the main thread is production-viable:
18
+ * The models are small enough (~640 MB combined VRAM) that WebGPU inference on
19
+ * the main thread is viable:
11
20
  *
12
21
  * - WebGPU GPU compute is inherently async (doesn't block the main thread)
13
- * - CPU overhead (tokenization + post-processing) is only 5-10 ms
14
- * - Single forward pass latency is 50-150 ms — well within autocomplete
15
- * expectations (~250 ms between word boundaries)
22
+ * - CPU overhead (BE-parity post-processing) is a few ms
23
+ * - Per-inference latency is well within autocomplete expectations
24
+ * (~250 ms between word boundaries)
16
25
  *
17
26
  * This avoids all the complexity of Web Workers:
18
27
  * - No CSP workarounds (blob URLs, inline scripts)
@@ -72,15 +81,93 @@ export interface LocalSlowLaneClient {
72
81
  setLmLogits: (logits: Record<string, number> | null) => void;
73
82
  updateContext: (text: string) => void;
74
83
  }
75
- export declare const LOCAL_MLC_MODEL_ID = "SmolLM2-135M-Instruct-q0f16-MLC";
76
- /** HF root for the default weights (includes `tensor-cache.json` for WebLLM 0.2+). */
77
- export declare const LOCAL_MLC_HF_MODEL_REPO = "https://huggingface.co/mlc-ai/SmolLM2-135M-Instruct-q0f16-MLC";
78
- export declare const LOCAL_MLC_MODEL_LIB_WASM_NAME = "SmolLM2-135M-Instruct-q0f16-ctx4k_cs1k-webgpu.wasm";
84
+ export declare const LOCAL_MLC_CAUSAL_MODEL_ID = "SmolLM2-135M-Instruct-q0f16-MLC";
85
+ /**
86
+ * MLC ID for the semantic embedder (Snowflake Arctic Embed S, batch=4 variant).
87
+ *
88
+ * The `-b4` suffix selects the prebuilt variant compiled for a max batch size of
89
+ * 4 (≈239 MB VRAM) rather than `-b32` (≈1023 MB VRAM). Autocomplete embeds one
90
+ * context at a time, so `-b4` is the right fit. This model IS in
91
+ * `prebuiltAppConfig.model_list` of web-llm 0.2.82 — no `customModelConfig` needed.
92
+ */
93
+ export declare const LOCAL_MLC_EMBEDDING_MODEL_ID = "snowflake-arctic-embed-s-q0f32-MLC-b4";
94
+ /**
95
+ * Wrap raw context text with BERT special tokens before embedding.
96
+ *
97
+ * web-llm's `EmbeddingPipeline` does NOT auto-prepend `[CLS]` / append `[SEP]`
98
+ * (the official MLC embeddings example wraps manually). The Python
99
+ * `sentence_transformers` side that generated the word-vector bin adds these
100
+ * inside `model.encode()`, so we must mirror it here for the runtime context
101
+ * vector to land in the same region of Arctic's embedding space as the bin.
102
+ *
103
+ * No query prefix is applied: the semantic step is sentence-to-sentence (`s2s`)
104
+ * similarity ("which words are conceptually similar to this context?"), not
105
+ * sentence-to-passage (`s2p`) retrieval. Arctic's query prefix would misframe
106
+ * the relationship. Encode both sides as passages. See implementation.md §4.3.
107
+ */
108
+ export declare const wrapForArctic: (text: string) => string;
109
+ /**
110
+ * BE-parity constants — must match `CausalLMEncoder` defaults in the Python
111
+ * sidecar (`cc-smarts/python-sidecar/src/causal_lm_encoder.py`) and
112
+ * `SlowLaneEngine` (`typeahead_context_encoding.py`) so local payloads behave
113
+ * identically to the server-client setup.
114
+ */
115
+ export declare const BE_PARITY: {
116
+ /** Final payload size cap (BE: `top_k_words`). */
117
+ readonly TOP_K_WORDS: 2000;
118
+ /** L2 (domain) words admitted unconditionally before pooling (BE: `reserved_l2_slots`). */
119
+ readonly RESERVED_L2_SLOTS: 500;
120
+ /** Log-space additive bias favouring L2 over L3 in the pool (BE: `l2_bias`). */
121
+ readonly L2_BIAS: 1;
122
+ /** Drop words below this probability from the final payload (BE: `> 0.00001`). */
123
+ readonly MIN_PROB: 0.00001;
124
+ /**
125
+ * Word-level approximation of the BE causal LM token limit.
126
+ *
127
+ * BE: `CausalLMEncoder.max_context_tokens = 100` (BPE tokens, left-truncated).
128
+ * FE: no tokenizer available, so we approximate with word count. English text
129
+ * averages ~1.3–1.5 BPE tokens/word, meaning 100 words ≈ 130–150 tokens.
130
+ * Using 100 words keeps the approximation simple and errs on the side of
131
+ * sending slightly more context than the BE sees — acceptable for a PoC.
132
+ */
133
+ readonly MAX_CONTEXT_TOKENS: 100;
134
+ /**
135
+ * Word-level rolling window for the semantic embedder.
136
+ *
137
+ * BE: `SlowLaneEngine.max_context_words = 100` (applied in
138
+ * `typeahead_context_encoding.py` before calling `SemanticEncoder.encode`).
139
+ * Truncated identically here so the runtime Arctic vector lands in the same
140
+ * region of the embedding space as the precomputed word-vector bin.
141
+ */
142
+ readonly MAX_CONTEXT_WORDS: 100;
143
+ };
144
+ /**
145
+ * Convert a raw next-token logit vector into a whole-word probability payload,
146
+ * faithfully porting the BE `CausalLMEncoder._get_top_k_probs`
147
+ * (`cc-smarts/python-sidecar/src/causal_lm_encoder.py`).
148
+ *
149
+ * Steps: (1) numerically-stable masked softmax over only the token ids present
150
+ * in the prefix-expansion map; (2) spread each token's probability to every
151
+ * whole word sharing that first token, taking the max; (3) reserve the top L2
152
+ * words unconditionally; (4) rank the remainder in a log-space pool with an
153
+ * additive L2 bias; (5) emit raw probabilities for the survivors, lowercased
154
+ * and trimmed at `MIN_PROB`.
155
+ *
156
+ * :params:
157
+ * rawLogits: Full-vocabulary logits from the LM's single decode step
158
+ * prefixMap: Map of first-token id to the words starting with that token
159
+ * domainWords: Set of L2 (domain) words, for tier-aware ranking
160
+ * :returns:
161
+ * A record of lowercase word to probability — the BE `lm_logits` payload
162
+ */
163
+ export declare const computeBePayload: (rawLogits: Float32Array, prefixMap: Map<number, string[]>, domainWords: Set<string>,
79
164
  /**
80
- * Original target repo (add-basics fine-tune). **Not compatible with WebLLM 0.2.x** (no `tensor-cache.json`).
81
- * @see module doc above
165
+ * Pre-derived token-ID array for the softmax mask. Defaults to the
166
+ * module-level `prefixMapTokenIds` (zero allocation in production). Pass
167
+ * `Array.from(prefixMap.keys())` in tests that supply a custom prefixMap so
168
+ * the softmax mask stays consistent with the iteration in Step 2.
82
169
  */
83
- export declare const HUGGINGFACE_TB_SMOLLM_ADD_BASICS_REPO = "https://huggingface.co/HuggingFaceTB/smollm-135M-instruct-add-basics-q0f16-MLC";
170
+ validTokenIds?: number[]) => Record<string, number>;
84
171
  /**
85
172
  * Create a local slow-lane client powered by MLC WebLLM.
86
173
  *
@@ -84,5 +84,5 @@ export declare const loadVectorsAsync: (options?: {
84
84
  getBinaryUrl?: () => Promise<string>;
85
85
  }) => Promise<void>;
86
86
  export declare const initVectors: (store: VectorStore) => void;
87
- export declare const loadDefaultVocabulary: () => void;
87
+ export declare const loadDefaultVocabulary: () => Promise<void>;
88
88
  export {};
@@ -1,18 +1,27 @@
1
1
  /**
2
2
  * Local Slow Lane Client: On-device inference via @mlc-ai/web-llm.
3
3
  *
4
- * Drop-in replacement for the network-based slow-lane-client. Instead of
5
- * calling a backend API, this client uses MLC WebLLM to run a small language
6
- * model (SmolLM 135M) directly in the browser via WebGPU.
4
+ * Drop-in replacement for the network-based slow-lane-client. Instead of calling
5
+ * a backend API, this client runs two models in the browser via WebGPU, in a
6
+ * single MLCEngine, to reproduce the BE encoder's outputs on-device:
7
+ *
8
+ * - Causal LM (SmolLM2-135M-Instruct): one decode step per word boundary. A
9
+ * registered LogitProcessor captures the raw next-token logits, which
10
+ * `computeBePayload` turns into a whole-word `lm_logits` payload — a faithful
11
+ * port of the BE `CausalLMEncoder._get_top_k_probs` (masked softmax over the
12
+ * vocab's first-tokens, prefix expansion, L2 reservation, log-space pooling).
13
+ * - Semantic embedder (Snowflake Arctic Embed S): produces the real 384-d
14
+ * `semantic_vector`. Inputs are wrapped as passages (see `wrapForArctic`) so
15
+ * the runtime vector lands in the same space as the precomputed word bin.
7
16
  *
8
17
  * ── Why main thread (no Web Worker)? ─────────────────────────────────────
9
- * SmolLM 135M is small enough (~270 MB weights, 350-400 MB VRAM) that
10
- * WebGPU inference on the main thread is production-viable:
18
+ * The models are small enough (~640 MB combined VRAM) that WebGPU inference on
19
+ * the main thread is viable:
11
20
  *
12
21
  * - WebGPU GPU compute is inherently async (doesn't block the main thread)
13
- * - CPU overhead (tokenization + post-processing) is only 5-10 ms
14
- * - Single forward pass latency is 50-150 ms — well within autocomplete
15
- * expectations (~250 ms between word boundaries)
22
+ * - CPU overhead (BE-parity post-processing) is a few ms
23
+ * - Per-inference latency is well within autocomplete expectations
24
+ * (~250 ms between word boundaries)
16
25
  *
17
26
  * This avoids all the complexity of Web Workers:
18
27
  * - No CSP workarounds (blob URLs, inline scripts)
@@ -72,15 +81,93 @@ export interface LocalSlowLaneClient {
72
81
  setLmLogits: (logits: Record<string, number> | null) => void;
73
82
  updateContext: (text: string) => void;
74
83
  }
75
- export declare const LOCAL_MLC_MODEL_ID = "SmolLM2-135M-Instruct-q0f16-MLC";
76
- /** HF root for the default weights (includes `tensor-cache.json` for WebLLM 0.2+). */
77
- export declare const LOCAL_MLC_HF_MODEL_REPO = "https://huggingface.co/mlc-ai/SmolLM2-135M-Instruct-q0f16-MLC";
78
- export declare const LOCAL_MLC_MODEL_LIB_WASM_NAME = "SmolLM2-135M-Instruct-q0f16-ctx4k_cs1k-webgpu.wasm";
84
+ export declare const LOCAL_MLC_CAUSAL_MODEL_ID = "SmolLM2-135M-Instruct-q0f16-MLC";
85
+ /**
86
+ * MLC ID for the semantic embedder (Snowflake Arctic Embed S, batch=4 variant).
87
+ *
88
+ * The `-b4` suffix selects the prebuilt variant compiled for a max batch size of
89
+ * 4 (≈239 MB VRAM) rather than `-b32` (≈1023 MB VRAM). Autocomplete embeds one
90
+ * context at a time, so `-b4` is the right fit. This model IS in
91
+ * `prebuiltAppConfig.model_list` of web-llm 0.2.82 — no `customModelConfig` needed.
92
+ */
93
+ export declare const LOCAL_MLC_EMBEDDING_MODEL_ID = "snowflake-arctic-embed-s-q0f32-MLC-b4";
94
+ /**
95
+ * Wrap raw context text with BERT special tokens before embedding.
96
+ *
97
+ * web-llm's `EmbeddingPipeline` does NOT auto-prepend `[CLS]` / append `[SEP]`
98
+ * (the official MLC embeddings example wraps manually). The Python
99
+ * `sentence_transformers` side that generated the word-vector bin adds these
100
+ * inside `model.encode()`, so we must mirror it here for the runtime context
101
+ * vector to land in the same region of Arctic's embedding space as the bin.
102
+ *
103
+ * No query prefix is applied: the semantic step is sentence-to-sentence (`s2s`)
104
+ * similarity ("which words are conceptually similar to this context?"), not
105
+ * sentence-to-passage (`s2p`) retrieval. Arctic's query prefix would misframe
106
+ * the relationship. Encode both sides as passages. See implementation.md §4.3.
107
+ */
108
+ export declare const wrapForArctic: (text: string) => string;
109
+ /**
110
+ * BE-parity constants — must match `CausalLMEncoder` defaults in the Python
111
+ * sidecar (`cc-smarts/python-sidecar/src/causal_lm_encoder.py`) and
112
+ * `SlowLaneEngine` (`typeahead_context_encoding.py`) so local payloads behave
113
+ * identically to the server-client setup.
114
+ */
115
+ export declare const BE_PARITY: {
116
+ /** Final payload size cap (BE: `top_k_words`). */
117
+ readonly TOP_K_WORDS: 2000;
118
+ /** L2 (domain) words admitted unconditionally before pooling (BE: `reserved_l2_slots`). */
119
+ readonly RESERVED_L2_SLOTS: 500;
120
+ /** Log-space additive bias favouring L2 over L3 in the pool (BE: `l2_bias`). */
121
+ readonly L2_BIAS: 1;
122
+ /** Drop words below this probability from the final payload (BE: `> 0.00001`). */
123
+ readonly MIN_PROB: 0.00001;
124
+ /**
125
+ * Word-level approximation of the BE causal LM token limit.
126
+ *
127
+ * BE: `CausalLMEncoder.max_context_tokens = 100` (BPE tokens, left-truncated).
128
+ * FE: no tokenizer available, so we approximate with word count. English text
129
+ * averages ~1.3–1.5 BPE tokens/word, meaning 100 words ≈ 130–150 tokens.
130
+ * Using 100 words keeps the approximation simple and errs on the side of
131
+ * sending slightly more context than the BE sees — acceptable for a PoC.
132
+ */
133
+ readonly MAX_CONTEXT_TOKENS: 100;
134
+ /**
135
+ * Word-level rolling window for the semantic embedder.
136
+ *
137
+ * BE: `SlowLaneEngine.max_context_words = 100` (applied in
138
+ * `typeahead_context_encoding.py` before calling `SemanticEncoder.encode`).
139
+ * Truncated identically here so the runtime Arctic vector lands in the same
140
+ * region of the embedding space as the precomputed word-vector bin.
141
+ */
142
+ readonly MAX_CONTEXT_WORDS: 100;
143
+ };
144
+ /**
145
+ * Convert a raw next-token logit vector into a whole-word probability payload,
146
+ * faithfully porting the BE `CausalLMEncoder._get_top_k_probs`
147
+ * (`cc-smarts/python-sidecar/src/causal_lm_encoder.py`).
148
+ *
149
+ * Steps: (1) numerically-stable masked softmax over only the token ids present
150
+ * in the prefix-expansion map; (2) spread each token's probability to every
151
+ * whole word sharing that first token, taking the max; (3) reserve the top L2
152
+ * words unconditionally; (4) rank the remainder in a log-space pool with an
153
+ * additive L2 bias; (5) emit raw probabilities for the survivors, lowercased
154
+ * and trimmed at `MIN_PROB`.
155
+ *
156
+ * :params:
157
+ * rawLogits: Full-vocabulary logits from the LM's single decode step
158
+ * prefixMap: Map of first-token id to the words starting with that token
159
+ * domainWords: Set of L2 (domain) words, for tier-aware ranking
160
+ * :returns:
161
+ * A record of lowercase word to probability — the BE `lm_logits` payload
162
+ */
163
+ export declare const computeBePayload: (rawLogits: Float32Array, prefixMap: Map<number, string[]>, domainWords: Set<string>,
79
164
  /**
80
- * Original target repo (add-basics fine-tune). **Not compatible with WebLLM 0.2.x** (no `tensor-cache.json`).
81
- * @see module doc above
165
+ * Pre-derived token-ID array for the softmax mask. Defaults to the
166
+ * module-level `prefixMapTokenIds` (zero allocation in production). Pass
167
+ * `Array.from(prefixMap.keys())` in tests that supply a custom prefixMap so
168
+ * the softmax mask stays consistent with the iteration in Step 2.
82
169
  */
83
- export declare const HUGGINGFACE_TB_SMOLLM_ADD_BASICS_REPO = "https://huggingface.co/HuggingFaceTB/smollm-135M-instruct-add-basics-q0f16-MLC";
170
+ validTokenIds?: number[]) => Record<string, number>;
84
171
  /**
85
172
  * Create a local slow-lane client powered by MLC WebLLM.
86
173
  *
@@ -84,5 +84,5 @@ export declare const loadVectorsAsync: (options?: {
84
84
  getBinaryUrl?: () => Promise<string>;
85
85
  }) => Promise<void>;
86
86
  export declare const initVectors: (store: VectorStore) => void;
87
- export declare const loadDefaultVocabulary: () => void;
87
+ export declare const loadDefaultVocabulary: () => Promise<void>;
88
88
  export {};
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@atlaskit/editor-plugin-autocomplete",
3
- "version": "3.0.0",
3
+ "version": "3.2.0",
4
4
  "description": "Client-side text autocomplete plugin for @atlaskit/editor-core",
5
5
  "author": "Atlassian Pty Ltd",
6
6
  "license": "Apache-2.0",
@@ -35,7 +35,7 @@
35
35
  "wink-nlp": "^2.4.0"
36
36
  },
37
37
  "peerDependencies": {
38
- "@atlaskit/editor-common": "^115.0.0",
38
+ "@atlaskit/editor-common": "^115.6.0",
39
39
  "@atlaskit/editor-plugin-analytics": "^11.0.0",
40
40
  "react": "^18.2.0"
41
41
  },