@atlaskit/editor-plugin-autocomplete 3.2.0 → 3.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -14,6 +14,7 @@ import type { EditorView } from '@atlaskit/editor-prosemirror/view';
14
14
 
15
15
  import type { AutocompletePlugin } from '../autocompletePluginType';
16
16
 
17
+ import { isAutocompleteDebugEnabled } from './debug-mode';
17
18
  import { createGhostTextDecorationSet } from './ghost-text-decoration';
18
19
  import { createLocalSlowLaneClient, type LocalSlowLaneClient } from './local-slow-lane-client';
19
20
  import { createSlowLaneClient, setDefaultSlowLaneClient, isWordBoundary } from './slow-lane-client';
@@ -28,6 +29,19 @@ import {
28
29
  export const autocompletePluginKey: PluginKey = new PluginKey('autocomplete');
29
30
 
30
31
  const DEBOUNCE_MS = 150;
32
+ const NETWORK_SLOW_LANE_DEBOUNCE_MS = 300;
33
+ const LOCAL_SLOW_LANE_DEBOUNCE_MS = 100;
34
+ const CONTEXT_REFRESH_THROTTLE_MS = 1000;
35
+ // Caps the dedup Set so long editing sessions don't retain every distinct
36
+ // version of the (potentially hundreds-of-KB) page content for the plugin's
37
+ // lifetime. Eviction is FIFO; a re-ingest of an evicted text is harmless.
38
+ const MAX_INGESTED_CONTEXT_TEXTS = 50;
39
+ // Caps how many times the word-boundary path will retry getContext() while the
40
+ // parent comment is still missing. Combined with the 1s throttle this gives a
41
+ // ~5s window to cover a still-loading comment thread, then stops permanently so
42
+ // non-comment editors (where parentCommentContent never arrives) don't refetch
43
+ // on every word boundary for the plugin's lifetime.
44
+ const MAX_CONTEXT_REFRESH_ATTEMPTS = 5;
31
45
 
32
46
  const hasDestroy = (
33
47
  client: ReturnType<typeof createSlowLaneClient> | LocalSlowLaneClient,
@@ -241,6 +255,12 @@ export const createAutocompletePlugin = (
241
255
  let debounceTimer: ReturnType<typeof setTimeout> | null = null;
242
256
  let hasIngestedPage = false;
243
257
  let resolvedContext: AutocompleteContext | undefined;
258
+ /**
259
+ * Kept in sync with the live EditorView so the async getContext() promise
260
+ * can re-trigger a slow-lane update the moment context arrives, even if the
261
+ * user has already typed several words before the promise resolved.
262
+ */
263
+ let currentView: EditorView | null = null;
244
264
  /**
245
265
  * Set after accepting a suggestion so the next doc-change update
246
266
  * skips scheduling a new prediction for the just-inserted text.
@@ -285,14 +305,139 @@ export const createAutocompletePlugin = (
285
305
 
286
306
  const slowLaneClient = options?.useLocalModel
287
307
  ? createLocalSlowLaneClient({
288
- debounceMs: 300,
308
+ debounceMs: LOCAL_SLOW_LANE_DEBOUNCE_MS,
289
309
  })
290
310
  : createSlowLaneClient({
291
311
  baseUrl: '',
292
- debounceMs: 300,
312
+ debounceMs: NETWORK_SLOW_LANE_DEBOUNCE_MS,
293
313
  });
294
314
  setDefaultSlowLaneClient(slowLaneClient);
295
315
 
316
+ let contextRequestInFlight = false;
317
+ let lastContextRefreshAt = 0;
318
+ // Bounds the word-boundary retry loop so it terminates even when the editor is
319
+ // not in a comment thread (parentCommentContent never resolves).
320
+ let wordBoundaryRefreshAttempts = 0;
321
+ // Set when the plugin is torn down so in-flight getContext() resolutions don't
322
+ // mutate the global text-predictor state after destruction.
323
+ let destroyed = false;
324
+ const ingestedContextTexts = new Set<string>();
325
+
326
+ const logContextResolved = (source: string, context?: AutocompleteContext): void => {
327
+ if (!isAutocompleteDebugEnabled()) {
328
+ return;
329
+ }
330
+
331
+ // eslint-disable-next-line no-console
332
+ console.log(
333
+ '%c[Autocomplete] %cgetContext resolved',
334
+ 'color: #00b8d9; font-weight: bold;',
335
+ 'color: inherit;',
336
+ {
337
+ source,
338
+ hasParentComment: !!context?.parentCommentContent,
339
+ parentCommentPreview: context?.parentCommentContent?.slice(0, 80),
340
+ siblingCount: context?.siblingCommentsContents?.length ?? 0,
341
+ hasFullPage: !!context?.fullPageContent,
342
+ },
343
+ );
344
+ };
345
+
346
+ const applyContext = (context: AutocompleteContext): void => {
347
+ // Merge rather than replace: the word-boundary retry may resolve only a
348
+ // late-arriving field (e.g. parentCommentContent) without re-sending
349
+ // fullPageContent, so replacing would drop previously resolved context.
350
+ // Strip undefined values first so a field explicitly set to undefined by
351
+ // getContext doesn't overwrite a previously resolved value.
352
+ const definedContext = Object.fromEntries(
353
+ Object.entries(context).filter(([, value]) => value !== undefined),
354
+ );
355
+ resolvedContext = { ...resolvedContext, ...definedContext };
356
+
357
+ const ingestContextText = (text?: string): void => {
358
+ if (!text || ingestedContextTexts.has(text)) {
359
+ return;
360
+ }
361
+
362
+ ingestedContextTexts.add(text);
363
+ // Evict oldest entries (Set preserves insertion order) to bound memory.
364
+ while (ingestedContextTexts.size > MAX_INGESTED_CONTEXT_TEXTS) {
365
+ const oldest = ingestedContextTexts.values().next().value;
366
+ if (oldest === undefined) {
367
+ break;
368
+ }
369
+ ingestedContextTexts.delete(oldest);
370
+ }
371
+ ingestDocumentPage(text);
372
+ };
373
+
374
+ ingestContextText(context.fullPageContent);
375
+ ingestContextText(context.parentCommentContent);
376
+ for (const siblingCommentContent of context.siblingCommentsContents ?? []) {
377
+ ingestContextText(siblingCommentContent);
378
+ }
379
+
380
+ // Context arrived after word boundaries may already have fired. Re-send
381
+ // slow-lane context immediately so the next inference includes the thread.
382
+ if (currentView) {
383
+ slowLaneClient.updateContext(
384
+ buildSlowLaneText(currentView.state.doc.textContent, resolvedContext),
385
+ );
386
+ }
387
+ };
388
+
389
+ /**
390
+ * Returns true when a fetch was actually started, false when it was skipped
391
+ * (no getContext, a request already in flight, or throttled). Callers that
392
+ * track a retry budget should only count attempts where this returns true.
393
+ */
394
+ const refreshContext = ({
395
+ source,
396
+ allowThrottle = true,
397
+ }: {
398
+ allowThrottle?: boolean;
399
+ source: string;
400
+ }): boolean => {
401
+ if (!options?.getContext || contextRequestInFlight) {
402
+ return false;
403
+ }
404
+
405
+ const now = Date.now();
406
+ if (allowThrottle && now - lastContextRefreshAt < CONTEXT_REFRESH_THROTTLE_MS) {
407
+ return false;
408
+ }
409
+
410
+ contextRequestInFlight = true;
411
+ lastContextRefreshAt = now;
412
+
413
+ options
414
+ .getContext()
415
+ .then((context) => {
416
+ // Bail if the plugin was destroyed while the fetch was in flight —
417
+ // applyContext mutates global text-predictor state we must not touch
418
+ // after teardown.
419
+ if (destroyed) {
420
+ return;
421
+ }
422
+ logContextResolved(source, context);
423
+ if (!context) {
424
+ return;
425
+ }
426
+
427
+ applyContext(context);
428
+ })
429
+ .catch((error) => {
430
+ logException(error as Error, {
431
+ location: 'editor-plugin-autocomplete/getContext',
432
+ });
433
+ })
434
+ .finally(() => {
435
+ contextRequestInFlight = false;
436
+ });
437
+
438
+ return true;
439
+ };
440
+
296
441
  /**
297
442
  * Schedule a prediction after a short debounce.
298
443
  * Tier 1 predictions are synchronous (<0.1ms) but we still debounce
@@ -490,29 +635,10 @@ export const createAutocompletePlugin = (
490
635
  });
491
636
  if (!hasIngestedPage) {
492
637
  hasIngestedPage = true;
493
- if (options?.getContext) {
494
- options
495
- .getContext()
496
- .then((context) => {
497
- if (!context) {
498
- return;
499
- }
500
- resolvedContext = context;
501
-
502
- if (context.fullPageContent) {
503
- ingestDocumentPage(context.fullPageContent);
504
- }
505
- if (context.parentCommentContent) {
506
- ingestDocumentPage(context.parentCommentContent);
507
- }
508
- context.siblingCommentsContents?.forEach(ingestDocumentPage);
509
- })
510
- .catch((error) => {
511
- logException(error as Error, {
512
- location: 'editor-plugin-autocomplete/getContext',
513
- });
514
- });
515
- }
638
+ refreshContext({
639
+ source: 'focus',
640
+ allowThrottle: false,
641
+ });
516
642
  }
517
643
  return false;
518
644
  },
@@ -521,6 +647,7 @@ export const createAutocompletePlugin = (
521
647
 
522
648
  view: () => ({
523
649
  update: (view: EditorView, prevState: EditorState) => {
650
+ currentView = view;
524
651
  if (!prevState.doc.eq(view.state.doc)) {
525
652
  if (justAccepted) {
526
653
  justAccepted = false;
@@ -542,18 +669,36 @@ export const createAutocompletePlugin = (
542
669
  slowLaneClient.updateContext(
543
670
  buildSlowLaneText(view.state.doc.textContent, resolvedContext),
544
671
  );
672
+
673
+ // Context may not have resolved on first focus (e.g. comment
674
+ // thread still loading). Retry on word boundaries until we have
675
+ // the parent comment, throttled so we don't refetch constantly
676
+ // and capped so non-comment editors stop retrying entirely.
677
+ if (
678
+ !resolvedContext?.parentCommentContent &&
679
+ wordBoundaryRefreshAttempts < MAX_CONTEXT_REFRESH_ATTEMPTS
680
+ ) {
681
+ // Only count the attempt when a fetch actually started, so an
682
+ // in-flight or throttled no-op doesn't burn the retry budget.
683
+ if (refreshContext({ source: 'word-boundary' })) {
684
+ wordBoundaryRefreshAttempts++;
685
+ }
686
+ }
545
687
  }
546
688
 
547
689
  schedulePrediction(view);
548
690
  }
549
691
  },
550
692
  destroy: () => {
693
+ destroyed = true;
694
+ currentView = null;
551
695
  if (debounceTimer) {
552
696
  clearTimeout(debounceTimer);
553
697
  }
554
698
  if (hasDestroy(slowLaneClient)) {
555
699
  slowLaneClient.destroy();
556
700
  }
701
+ ingestedContextTexts.clear();
557
702
  setDefaultSlowLaneClient(null);
558
703
  },
559
704
  }),
@@ -41,6 +41,7 @@ import { isAutocompleteDebugEnabled } from './debug-mode';
41
41
  import { isWordBoundary } from './slow-lane-client';
42
42
 
43
43
  type WebLlmModelRecord = NonNullable<AppConfig['model_list']>[number];
44
+ type EmbeddingApiResponse = { data?: Array<{ embedding?: unknown }> };
44
45
 
45
46
  // ─── Types ───────────────────────────────────────────────────────────────────
46
47
 
@@ -158,12 +159,42 @@ export const BE_PARITY = {
158
159
  MAX_CONTEXT_WORDS: 100,
159
160
  } as const;
160
161
 
162
+ const splitOnWhitespace = (text: string): string[] => {
163
+ const trimmed = text.trim();
164
+ if (trimmed === '') {
165
+ return [];
166
+ }
167
+
168
+ const words: string[] = [];
169
+ let wordStart = -1;
170
+
171
+ for (let i = 0; i < trimmed.length; i++) {
172
+ if (trimmed[i].trim() === '') {
173
+ if (wordStart !== -1) {
174
+ words.push(trimmed.slice(wordStart, i));
175
+ wordStart = -1;
176
+ }
177
+ continue;
178
+ }
179
+
180
+ if (wordStart === -1) {
181
+ wordStart = i;
182
+ }
183
+ }
184
+
185
+ if (wordStart !== -1) {
186
+ words.push(trimmed.slice(wordStart));
187
+ }
188
+
189
+ return words;
190
+ };
191
+
161
192
  /**
162
193
  * Return the last `n` whitespace-separated words of `text`, joined by spaces.
163
194
  * Mirrors the BE rolling-window truncation applied before both encoders.
164
195
  */
165
196
  const truncateToLastNWords = (text: string, n: number): string => {
166
- const words = text.trim().split(/\s+/u);
197
+ const words = splitOnWhitespace(text);
167
198
  return words.length <= n ? text : words.slice(-n).join(' ');
168
199
  };
169
200
 
@@ -564,6 +595,14 @@ export const createLocalSlowLaneClient = (
564
595
  let lastRequestedText = '';
565
596
  let requestCounter = 0;
566
597
  let latestRequestId = -1;
598
+ let inferenceInFlight = false;
599
+ let activeInferenceText: string | null = null;
600
+ // The requestId of the inference currently in flight. Tracked so the
601
+ // in-flight dedup path can restore `latestRequestId` to it — otherwise an
602
+ // intermediate keystroke that bumped `latestRequestId` would cause the
603
+ // in-flight (still-current) result to be discarded as stale.
604
+ let activeInferenceRequestId = -1;
605
+ let pendingInference: { requestId: number; text: string } | null = null;
567
606
  let ready = false;
568
607
  let destroyed = false;
569
608
  let initFailed = false;
@@ -755,17 +794,25 @@ export const createLocalSlowLaneClient = (
755
794
  const lmText = truncateToLastNWords(text, BE_PARITY.MAX_CONTEXT_TOKENS);
756
795
  const semanticText = truncateToLastNWords(text, BE_PARITY.MAX_CONTEXT_WORDS);
757
796
  const arcticInput = wrapForArctic(semanticText);
797
+ const captureCompletionTime = <T,>(
798
+ promise: Promise<T>,
799
+ onResolved: (resolvedAt: number) => void,
800
+ ): Promise<T> =>
801
+ promise.then((value: T) => {
802
+ onResolved(performance.now());
803
+ return value;
804
+ });
758
805
 
759
806
  if (isAutocompleteDebugEnabled()) {
760
807
  // eslint-disable-next-line no-console
761
808
  console.log(
762
- `%c[LocalSlowLane] %c🔢 Arctic input (${arcticInput.length} chars, ${semanticText.split(/\s+/u).length} words): "${arcticInput.length > 100 ? `${arcticInput.slice(0, 100)}…` : arcticInput}"`,
809
+ `%c[LocalSlowLane] %c🔢 Arctic input (${arcticInput.length} chars, ${splitOnWhitespace(semanticText).length} words): "${arcticInput.length > 100 ? `${arcticInput.slice(0, 100)}…` : arcticInput}"`,
763
810
  'color: #9c27b0; font-weight: bold;',
764
811
  'color: #009688;',
765
812
  );
766
813
  // eslint-disable-next-line no-console
767
814
  console.log(
768
- `%c[LocalSlowLane] %c🧠 LM input (${lmText.length} chars, ${lmText.split(/\s+/u).length} words): "${lmText.length > 100 ? `${lmText.slice(0, 100)}…` : lmText}"`,
815
+ `%c[LocalSlowLane] %c🧠 LM input (${lmText.length} chars, ${splitOnWhitespace(lmText).length} words): "${lmText.length > 100 ? `${lmText.slice(0, 100)}…` : lmText}"`,
769
816
  'color: #9c27b0; font-weight: bold;',
770
817
  'color: #2196f3;',
771
818
  );
@@ -777,27 +824,29 @@ export const createLocalSlowLaneClient = (
777
824
  let tEmbDone = 0;
778
825
 
779
826
  const [, embeddingResponse] = await Promise.all([
780
- engine.completions
781
- .create({
827
+ captureCompletionTime(
828
+ engine.completions.create({
782
829
  model: modelId,
783
830
  prompt: lmText,
784
831
  max_tokens: 1,
785
832
  temperature: 0,
786
833
  logprobs: false,
787
834
  })
788
- .then((r) => {
789
- tLmDone = performance.now();
790
- return r;
791
- }),
792
- engine.embeddings
793
- .create({
835
+ ,
836
+ (resolvedAt) => {
837
+ tLmDone = resolvedAt;
838
+ },
839
+ ),
840
+ captureCompletionTime(
841
+ engine.embeddings.create({
794
842
  model: LOCAL_MLC_EMBEDDING_MODEL_ID,
795
843
  input: arcticInput,
796
844
  })
797
- .then((r) => {
798
- tEmbDone = performance.now();
799
- return r;
800
- }),
845
+ ,
846
+ (resolvedAt) => {
847
+ tEmbDone = resolvedAt;
848
+ },
849
+ ),
801
850
  ]);
802
851
 
803
852
  if (isAutocompleteDebugEnabled()) {
@@ -827,7 +876,7 @@ export const createLocalSlowLaneClient = (
827
876
  // Guard against base64-encoded responses (encoding_format: 'base64' would
828
877
  // yield a string, and new Float32Array(string) silently produces an empty
829
878
  // array, corrupting downstream cosine-similarity scoring).
830
- const embedding = embeddingResponse.data?.[0]?.embedding;
879
+ const embedding = (embeddingResponse as EmbeddingApiResponse).data?.[0]?.embedding;
831
880
  storedContextVector =
832
881
  Array.isArray(embedding) && embedding.length > 0
833
882
  ? new Float32Array(embedding as number[])
@@ -879,8 +928,8 @@ export const createLocalSlowLaneClient = (
879
928
  hasLmLogits: storedLmLogits !== null,
880
929
  });
881
930
  } catch (err) {
882
- // Discard errors for stale requests
883
- if (requestId < latestRequestId) {
931
+ // Discard errors for stale requests or after teardown
932
+ if (requestId < latestRequestId || destroyed) {
884
933
  return;
885
934
  }
886
935
 
@@ -902,11 +951,55 @@ export const createLocalSlowLaneClient = (
902
951
 
903
952
  // ── Context update (debounced) ─────────────────────────────────────────
904
953
 
954
+ const startInference = (text: string, requestId: number): void => {
955
+ // Self-contained guard: never start a new inference cycle after teardown,
956
+ // regardless of caller discipline.
957
+ if (destroyed) {
958
+ return;
959
+ }
960
+
961
+ inferenceInFlight = true;
962
+ activeInferenceText = text;
963
+ activeInferenceRequestId = requestId;
964
+
965
+ void ensureEngineInitialized()
966
+ .then(() => runInference(text, requestId))
967
+ .catch(() => {})
968
+ .finally(() => {
969
+ inferenceInFlight = false;
970
+ activeInferenceText = null;
971
+ activeInferenceRequestId = -1;
972
+
973
+ const next = pendingInference;
974
+ pendingInference = null;
975
+ if (next && !destroyed) {
976
+ startInference(next.text, next.requestId);
977
+ }
978
+ });
979
+ };
980
+
905
981
  const doUpdateContext = (text: string): void => {
906
982
  if (destroyed || !text || text.trim().length === 0) {
907
983
  return;
908
984
  }
909
985
 
986
+ if (inferenceInFlight && text === activeInferenceText) {
987
+ // The latest desired text already matches the in-flight inference, so
988
+ // re-running it would be wasted work. But an intermediate keystroke may
989
+ // have bumped `latestRequestId` past the in-flight request (and then been
990
+ // coalesced away), which would cause runInference to discard the
991
+ // still-current result as stale. Pin `latestRequestId` back to the active
992
+ // request so its result is accepted, and drop any now-superseded pending
993
+ // request.
994
+ latestRequestId = activeInferenceRequestId;
995
+ pendingInference = null;
996
+ return;
997
+ }
998
+
999
+ if (inferenceInFlight && pendingInference?.text === text) {
1000
+ return;
1001
+ }
1002
+
910
1003
  const requestId = ++requestCounter;
911
1004
  latestRequestId = requestId;
912
1005
 
@@ -926,15 +1019,26 @@ export const createLocalSlowLaneClient = (
926
1019
  console.groupEnd();
927
1020
  }
928
1021
 
929
- void ensureEngineInitialized()
930
- .then(() => runInference(text, requestId))
931
- .catch(() => {});
1022
+ if (inferenceInFlight) {
1023
+ pendingInference = { text, requestId };
1024
+ return;
1025
+ }
1026
+
1027
+ startInference(text, requestId);
932
1028
  };
933
1029
 
934
1030
  const updateContextDebounced = (text: string): void => {
935
1031
  if (debounceTimer) {
936
1032
  clearTimeout(debounceTimer);
937
1033
  }
1034
+ if (inferenceInFlight) {
1035
+ pendingInference = null;
1036
+ if (text === activeInferenceText) {
1037
+ latestRequestId = activeInferenceRequestId;
1038
+ lastRequestedText = text;
1039
+ return;
1040
+ }
1041
+ }
938
1042
  lastRequestedText = text;
939
1043
  debounceTimer = setTimeout(() => {
940
1044
  debounceTimer = null;
@@ -967,6 +1071,10 @@ export const createLocalSlowLaneClient = (
967
1071
  unloadEngine(engineToUnload);
968
1072
  }
969
1073
  engineInitPromise = null;
1074
+ inferenceInFlight = false;
1075
+ activeInferenceText = null;
1076
+ activeInferenceRequestId = -1;
1077
+ pendingInference = null;
970
1078
  storedContextVector = null;
971
1079
  storedLmLogits = null;
972
1080
  },