@atlaskit/editor-plugin-autocomplete 3.0.0 → 3.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -135,15 +135,16 @@ function applyGrammarFilter(candidates, previousWord) {
135
135
  dropped.push(entry.candidate.word);
136
136
  }
137
137
  }
138
- const finalFiltered = filtered.length > 0 ? filtered : candidates;
138
+
139
+ // Grammar is authoritative.
139
140
  return {
140
- filtered: finalFiltered,
141
+ filtered: filtered,
141
142
  grammarMeta: {
142
143
  prevWord: lowerPrev,
143
144
  prevTags,
144
145
  before: candidates.length,
145
- after: finalFiltered.length,
146
- dropped: filtered.length > 0 ? dropped : []
146
+ after: filtered.length,
147
+ dropped: dropped
147
148
  }
148
149
  };
149
150
  }
@@ -18,11 +18,9 @@ import _defineProperty from "@babel/runtime/helpers/defineProperty";
18
18
 
19
19
  import { EXPERIENCE_NAME, failExp, startExp, succeedExp } from '../analytics/ufo';
20
20
 
21
- // import bigramsData from './data/bigrams.json';
22
- import l3VocabularyData from './data/l3_vocabulary.json';
23
- import vocabularyData from './data/vocabulary_10k.json';
24
- import wordIndexData from './data/word_index_10k.json';
25
- // import { rankCandidates, isGrammarAllowed } from './scoring-pipeline';
21
+ // The vocabulary, L3 and word-index JSON payloads are dynamically imported in
22
+ // loadDefaultVocabulary / loadVectorsAsync so their (large) contents stay out of
23
+ // the editor's main chunk and only load when autocomplete is initialised.
26
24
  import { isAutocompleteDebugEnabled } from './debug-mode';
27
25
  import { rankCandidates, STAGE1_WEIGHT, STAGE2_WEIGHT, MIN_STAGE1_SCORE } from './scoring-pipeline';
28
26
  import { getStoredContextVector, getStoredLmLogits } from './slow-lane-client';
@@ -34,7 +32,7 @@ const PUNCTUATION_BOUNDARY_REGEX = /^[.,;:!?()\[\]{}"'`]+|[.,;:!?()\[\]{}"'`]+$/
34
32
  const MIN_PREFIX_LENGTH = 3;
35
33
  const MAX_CANDIDATES = 200;
36
34
  const CONTEXT_WORDS = 10;
37
- const MIN_SCORE_THRESHOLD = 0.2;
35
+ const MIN_SCORE_THRESHOLD = 0.35;
38
36
  const L3_BASELINE_FREQ = 0.001;
39
37
 
40
38
  // ─── Types ───────────────────────────────────────────────────────────────────
@@ -153,7 +151,6 @@ const wordTrie = new WeightedWordTrie();
153
151
  // L3 Trie (General English Fallback)
154
152
  const l3Trie = new WeightedWordTrie();
155
153
 
156
- // --- Initialization Function ---
157
154
  /**
158
155
  * Loads the General English vocabulary.
159
156
  * expects a simple array of strings: ["about", "above", "actually", ...]
@@ -249,14 +246,11 @@ const tokenize = text => {
249
246
  return tokens;
250
247
  };
251
248
  const extractPreviousWord = text => {
252
- // 1. Split the text by newlines or punctuation (. ? !)
249
+ // Only consider the current sentence/line the user is typing in.
253
250
  // eslint-disable-next-line require-unicode-regexp
254
251
  const sentences = text.split(/[\n.?!]+/);
255
-
256
- // 2. Only look at the current sentence/line the user is typing in
257
252
  const currentSentence = sentences[sentences.length - 1];
258
253
 
259
- // 3. Extract the previous word as normal
260
254
  // eslint-disable-next-line require-unicode-regexp
261
255
  const words = currentSentence.trimEnd().split(/\s+/);
262
256
  return words.length >= 2 ? words[words.length - 2] : '';
@@ -310,7 +304,6 @@ export const incrementSessionFreq = word => {
310
304
  * Pass `undefined` (or omit the argument) to skip priming — useful when the
311
305
  * calling context does not yet have a page value available.
312
306
  */
313
- // NOTE: We ingest full page context here
314
307
  export const ingestDocumentPage = pageContent => {
315
308
  if (!pageContent) {
316
309
  return;
@@ -334,7 +327,11 @@ export const ingestDocumentPage = pageContent => {
334
327
  };
335
328
  export const predict = textBefore => {
336
329
  if (!isInitialized) {
337
- loadDefaultVocabulary();
330
+ // Vocabulary JSON is code-split and loads asynchronously. Kick off the load
331
+ // and skip this keystroke; the plugin also primes it on focus, so the tries
332
+ // are usually ready before the user types.
333
+ void loadDefaultVocabulary().catch(() => {});
334
+ return null;
338
335
  }
339
336
  const t0 = performance.now();
340
337
 
@@ -386,19 +383,15 @@ export const predict = textBefore => {
386
383
  if (currentWord.length < MIN_PREFIX_LENGTH) {
387
384
  return null;
388
385
  }
389
-
390
- // 1. Primary Query: Ask the L2 Domain Trie
391
386
  const candidates = wordTrie.getCandidates(currentWord, MAX_CANDIDATES);
392
387
 
393
- // 2. Fallback Query: Gap-fill with the L3 General English Trie
388
+ // Gap-fill from the L3 general-English trie, requesting a full buffer so
389
+ // enough survive de-duplication against the L2 results.
394
390
  if (candidates.length < MAX_CANDIDATES) {
395
- // Ask L3 for MAX_CANDIDATES to guarantee we have enough buffer
396
- // to survive the deduplication process.
397
391
  const l3Candidates = l3Trie.getCandidates(currentWord, MAX_CANDIDATES);
398
392
  const existingWords = new Set(candidates.map(c => c.word));
399
393
  for (const l3c of l3Candidates) {
400
- if (candidates.length >= MAX_CANDIDATES) break; // Stop exactly at the limit
401
-
394
+ if (candidates.length >= MAX_CANDIDATES) break;
402
395
  if (!existingWords.has(l3c.word)) {
403
396
  candidates.push(l3c);
404
397
  }
@@ -558,6 +551,41 @@ export const predict = textBefore => {
558
551
 
559
552
  // ─── Data Loading ────────────────────────────────────────────────────────────
560
553
 
554
+ /**
555
+ * Unwrap a dynamically imported JSON module to its parsed value, handling both
556
+ * interop modes AFM's bundler chain emits: a `.default`-wrapped namespace
557
+ * (classic webpack) and a named-exports namespace (webpack 5 / atlaspack JSON
558
+ * modules, where `default` can be a misleading scalar). Named exports are
559
+ * preferred when present. The caller declares the JSON `shape` because a dense
560
+ * array and a sparse numeric-keyed object are emitted identically as named
561
+ * exports. Kept in lock-step with the matching helper in local-slow-lane-client.ts.
562
+ */
563
+ const unwrapJsonModule = (mod, shape) => {
564
+ if (mod == null || typeof mod !== 'object') {
565
+ return null;
566
+ }
567
+ const namespace = mod;
568
+ const ownKeys = Object.keys(namespace).filter(k => k !== 'default' && k !== '__esModule');
569
+ if (ownKeys.length > 0) {
570
+ if (shape === 'array') {
571
+ const len = ownKeys.length;
572
+ const arr = new Array(len);
573
+ for (let i = 0; i < len; i++) {
574
+ arr[i] = namespace[String(i)];
575
+ }
576
+ return arr;
577
+ }
578
+ const obj = {};
579
+ for (const k of ownKeys) {
580
+ obj[k] = namespace[k];
581
+ }
582
+ return obj;
583
+ }
584
+ if ('default' in namespace && namespace.default != null) {
585
+ return namespace.default;
586
+ }
587
+ return null;
588
+ };
561
589
  export const loadVectorsAsync = async options => {
562
590
  if (vectorStore || vectorsLoadStarted) {
563
591
  return;
@@ -582,6 +610,7 @@ export const loadVectorsAsync = async options => {
582
610
  return;
583
611
  }
584
612
  try {
613
+ var _wordIndexOuter$index;
585
614
  const res = await fetch(url);
586
615
  if (!res.ok) {
587
616
  vectorsLoadStarted = false;
@@ -595,8 +624,18 @@ export const loadVectorsAsync = async options => {
595
624
  }
596
625
  const buffer = await res.arrayBuffer();
597
626
  const float32 = new Float32Array(buffer);
598
- const wordIndex = wordIndexData;
627
+
628
+ // word_index_10k.json is wrapped as `{ "index": {…} }` so no real entry
629
+ // (e.g. the word "default") can shadow the synthetic ESM `default` export
630
+ // the bundler creates for dynamically-imported JSON.
631
+ const wordIndexModule = await import( /* webpackChunkName: "@atlaskit-internal_editor-plugin-autocomplete-word-index-10k" */'./data/word_index_10k.json');
632
+ const wordIndexOuter = unwrapJsonModule(wordIndexModule, 'object');
633
+ const wordIndex = (_wordIndexOuter$index = wordIndexOuter === null || wordIndexOuter === void 0 ? void 0 : wordIndexOuter.index) !== null && _wordIndexOuter$index !== void 0 ? _wordIndexOuter$index : {};
599
634
  const nWords = Object.keys(wordIndex).length;
635
+ if (nWords === 0) {
636
+ // eslint-disable-next-line no-console
637
+ console.warn('[text-predictor] word_index_10k.json missing its `index` wrapper — wordIndex is empty, semantic scoring will be a no-op.');
638
+ }
600
639
  const dim = float32.length / nWords;
601
640
  vectorStore = {
602
641
  float32,
@@ -628,35 +667,51 @@ export const loadVectorsAsync = async options => {
628
667
  export const initVectors = store => {
629
668
  vectorStore = store;
630
669
  };
670
+ let vocabularyLoadPromise;
631
671
  export const loadDefaultVocabulary = () => {
632
672
  if (isInitialized) {
633
- return;
634
- }
635
- startExp(EXPERIENCE_NAME.LOAD_VOCABULARY, 'singleton');
636
- try {
637
- // 1. Load the Atlassian Domain (L2)
638
- const data = vocabularyData;
639
- const terms = Object.entries(data.words).map(([word, stats]) => ({
640
- word,
641
- freq: stats.freq,
642
- docFreq: stats.doc_freq,
643
- authorFreq: stats.author_freq
644
- }));
645
- initVocabulary({
646
- terms
647
- });
648
-
649
- // 2. Load General English (L3)
650
- const l3Words = l3VocabularyData;
651
- initL3Vocabulary(l3Words);
652
- succeedExp(EXPERIENCE_NAME.LOAD_VOCABULARY, 'singleton', {
653
- l2WordCount: terms.length,
654
- l3WordCount: l3Words.length
655
- });
656
- } catch (e) {
657
- failExp(EXPERIENCE_NAME.LOAD_VOCABULARY, 'singleton', {
658
- errorType: 'parse_error'
659
- });
660
- throw e;
661
- }
673
+ return Promise.resolve();
674
+ }
675
+ if (vocabularyLoadPromise) {
676
+ return vocabularyLoadPromise;
677
+ }
678
+ vocabularyLoadPromise = (async () => {
679
+ startExp(EXPERIENCE_NAME.LOAD_VOCABULARY, 'singleton');
680
+ try {
681
+ // The L2 vocabulary and L3 word list are code-split into their own async
682
+ // chunks so they stay out of the editor's main bundle.
683
+ const [vocabularyModule, l3VocabularyModule] = await Promise.all([import( /* webpackChunkName: "@atlaskit-internal_editor-plugin-autocomplete-vocabulary-10k" */'./data/vocabulary_10k.json'), import( /* webpackChunkName: "@atlaskit-internal_editor-plugin-autocomplete-l3-vocabulary" */'./data/l3_vocabulary.json')]);
684
+ const vocabularyData = unwrapJsonModule(vocabularyModule, 'object');
685
+ const l3VocabularyData = unwrapJsonModule(l3VocabularyModule, 'array');
686
+ if ((vocabularyData === null || vocabularyData === void 0 ? void 0 : vocabularyData.words) == null || !Array.isArray(l3VocabularyData)) {
687
+ throw new Error('[text-predictor] vocabulary JSON modules could not be unwrapped');
688
+ }
689
+ const terms = Object.entries(vocabularyData.words).map(([word, stats]) => ({
690
+ word,
691
+ freq: stats.freq,
692
+ docFreq: stats.doc_freq,
693
+ authorFreq: stats.author_freq
694
+ }));
695
+
696
+ // Load L3 before L2: initVocabulary flips `isInitialized = true`, so it
697
+ // must run last — otherwise a throw in initL3Vocabulary would strand
698
+ // `isInitialized` true and the retry path could never reload L3.
699
+ initL3Vocabulary(l3VocabularyData);
700
+ initVocabulary({
701
+ terms
702
+ });
703
+ succeedExp(EXPERIENCE_NAME.LOAD_VOCABULARY, 'singleton', {
704
+ l2WordCount: terms.length,
705
+ l3WordCount: l3VocabularyData.length
706
+ });
707
+ } catch (e) {
708
+ failExp(EXPERIENCE_NAME.LOAD_VOCABULARY, 'singleton', {
709
+ errorType: 'parse_error'
710
+ });
711
+ // Allow a later call to retry the load rather than caching the failure.
712
+ vocabularyLoadPromise = undefined;
713
+ throw e;
714
+ }
715
+ })();
716
+ return vocabularyLoadPromise;
662
717
  };
@@ -31,8 +31,6 @@ var createInitialState = function createInitialState() {
31
31
  var getTextBeforeCursor = function getTextBeforeCursor(state) {
32
32
  var $from = state.selection.$from;
33
33
  var maxChars = 200;
34
-
35
- // 1. Get the perfectly flattened text of the current block up to the cursor
36
34
  var blockNode = $from.parent;
37
35
  var offsetInBlock = $from.parentOffset;
38
36
  var blockText = blockNode.textContent.slice(0, offsetInBlock);
@@ -41,7 +39,7 @@ var getTextBeforeCursor = function getTextBeforeCursor(state) {
41
39
  }
42
40
  var fullText = blockText;
43
41
 
44
- // 2. Walk backwards through previous blocks
42
+ // Walk backwards through previous blocks until we have enough context.
45
43
  var depth = $from.depth - 1;
46
44
  while (fullText.length < maxChars && depth >= 0) {
47
45
  var parentNode = $from.node(depth);
@@ -242,8 +240,6 @@ export var createAutocompletePlugin = function createAutocompletePlugin(options,
242
240
  try {
243
241
  var state = view.state;
244
242
  var selection = state.selection;
245
-
246
- // Only predict for cursor selections (not range selections)
247
243
  if (!selection.empty) {
248
244
  return;
249
245
  }
@@ -255,13 +251,9 @@ export var createAutocompletePlugin = function createAutocompletePlugin(options,
255
251
  return;
256
252
  }
257
253
  dismissedContext = null;
258
-
259
- // Don't predict if there's not enough context
260
254
  if (textBefore.trim().length < 3) {
261
255
  return;
262
256
  }
263
-
264
- // Tier 1 prediction is synchronous -- no async needed
265
257
  var prediction = predict(textBefore);
266
258
  if (prediction && prediction.length > 0) {
267
259
  var typedLength = getTypedLengthForPrediction(textBefore);
@@ -323,8 +315,7 @@ export var createAutocompletePlugin = function createAutocompletePlugin(options,
323
315
  return _objectSpread(_objectSpread({}, pluginState), meta);
324
316
  }
325
317
 
326
- // If the document changed, clear the ghost text
327
- // (new prediction will be scheduled from view.update)
318
+ // A new prediction is scheduled from view.update.
328
319
  if (tr.docChanged) {
329
320
  return _objectSpread(_objectSpread({}, pluginState), {}, {
330
321
  ghostText: '',
@@ -332,8 +323,6 @@ export var createAutocompletePlugin = function createAutocompletePlugin(options,
332
323
  decorationSet: DecorationSet.empty
333
324
  });
334
325
  }
335
-
336
- // If selection changed without doc change, clear ghost text
337
326
  if (tr.selectionSet && pluginState.ghostText) {
338
327
  return _objectSpread(_objectSpread({}, pluginState), {}, {
339
328
  ghostText: '',
@@ -394,13 +383,11 @@ export var createAutocompletePlugin = function createAutocompletePlugin(options,
394
383
  return false;
395
384
  },
396
385
  focus: function focus() {
397
- try {
398
- loadDefaultVocabulary();
399
- } catch (error) {
386
+ loadDefaultVocabulary().catch(function (error) {
400
387
  logException(error, {
401
388
  location: 'editor-plugin-autocomplete/loadDefaultVocabulary'
402
389
  });
403
- }
390
+ });
404
391
  loadVectorsAsync({
405
392
  getBinaryUrl: options === null || options === void 0 ? void 0 : options.getVectorsBinaryUrl
406
393
  }).catch(function (error) {
@@ -442,12 +429,9 @@ export var createAutocompletePlugin = function createAutocompletePlugin(options,
442
429
  if (justAccepted) {
443
430
  justAccepted = false;
444
431
 
445
- // ✨ THE FIX: Memorize the text state right after acceptance.
446
- // Any follow-up transactions will hit the 'dismissedContext'
447
- // block and abort until the user actually types a new character!
432
+ // Snapshot the post-acceptance text so follow-up transactions hit
433
+ // the dismissedContext guard and abort until the user types again.
448
434
  dismissedContext = getTextBeforeCursor(view.state);
449
-
450
- // Also clear any pending debounce timers from before the acceptance
451
435
  if (debounceTimer) {
452
436
  clearTimeout(debounceTimer);
453
437
  }