@atlaskit/editor-plugin-autocomplete 3.2.0 → 3.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/dist/cjs/pm-plugins/autocomplete-plugin.js +162 -22
- package/dist/cjs/pm-plugins/local-slow-lane-client.js +108 -17
- package/dist/es2019/pm-plugins/autocomplete-plugin.js +148 -22
- package/dist/es2019/pm-plugins/local-slow-lane-client.js +105 -16
- package/dist/esm/pm-plugins/autocomplete-plugin.js +162 -22
- package/dist/esm/pm-plugins/local-slow-lane-client.js +108 -17
- package/package.json +2 -2
- package/src/pm-plugins/autocomplete-plugin.ts +170 -25
- package/src/pm-plugins/local-slow-lane-client.ts +129 -21
|
@@ -14,6 +14,7 @@ import type { EditorView } from '@atlaskit/editor-prosemirror/view';
|
|
|
14
14
|
|
|
15
15
|
import type { AutocompletePlugin } from '../autocompletePluginType';
|
|
16
16
|
|
|
17
|
+
import { isAutocompleteDebugEnabled } from './debug-mode';
|
|
17
18
|
import { createGhostTextDecorationSet } from './ghost-text-decoration';
|
|
18
19
|
import { createLocalSlowLaneClient, type LocalSlowLaneClient } from './local-slow-lane-client';
|
|
19
20
|
import { createSlowLaneClient, setDefaultSlowLaneClient, isWordBoundary } from './slow-lane-client';
|
|
@@ -28,6 +29,19 @@ import {
|
|
|
28
29
|
export const autocompletePluginKey: PluginKey = new PluginKey('autocomplete');
|
|
29
30
|
|
|
30
31
|
const DEBOUNCE_MS = 150;
|
|
32
|
+
const NETWORK_SLOW_LANE_DEBOUNCE_MS = 300;
|
|
33
|
+
const LOCAL_SLOW_LANE_DEBOUNCE_MS = 100;
|
|
34
|
+
const CONTEXT_REFRESH_THROTTLE_MS = 1000;
|
|
35
|
+
// Caps the dedup Set so long editing sessions don't retain every distinct
|
|
36
|
+
// version of the (potentially hundreds-of-KB) page content for the plugin's
|
|
37
|
+
// lifetime. Eviction is FIFO; a re-ingest of an evicted text is harmless.
|
|
38
|
+
const MAX_INGESTED_CONTEXT_TEXTS = 50;
|
|
39
|
+
// Caps how many times the word-boundary path will retry getContext() while the
|
|
40
|
+
// parent comment is still missing. Combined with the 1s throttle this gives a
|
|
41
|
+
// ~5s window to cover a still-loading comment thread, then stops permanently so
|
|
42
|
+
// non-comment editors (where parentCommentContent never arrives) don't refetch
|
|
43
|
+
// on every word boundary for the plugin's lifetime.
|
|
44
|
+
const MAX_CONTEXT_REFRESH_ATTEMPTS = 5;
|
|
31
45
|
|
|
32
46
|
const hasDestroy = (
|
|
33
47
|
client: ReturnType<typeof createSlowLaneClient> | LocalSlowLaneClient,
|
|
@@ -241,6 +255,12 @@ export const createAutocompletePlugin = (
|
|
|
241
255
|
let debounceTimer: ReturnType<typeof setTimeout> | null = null;
|
|
242
256
|
let hasIngestedPage = false;
|
|
243
257
|
let resolvedContext: AutocompleteContext | undefined;
|
|
258
|
+
/**
|
|
259
|
+
* Kept in sync with the live EditorView so the async getContext() promise
|
|
260
|
+
* can re-trigger a slow-lane update the moment context arrives, even if the
|
|
261
|
+
* user has already typed several words before the promise resolved.
|
|
262
|
+
*/
|
|
263
|
+
let currentView: EditorView | null = null;
|
|
244
264
|
/**
|
|
245
265
|
* Set after accepting a suggestion so the next doc-change update
|
|
246
266
|
* skips scheduling a new prediction for the just-inserted text.
|
|
@@ -285,14 +305,139 @@ export const createAutocompletePlugin = (
|
|
|
285
305
|
|
|
286
306
|
const slowLaneClient = options?.useLocalModel
|
|
287
307
|
? createLocalSlowLaneClient({
|
|
288
|
-
debounceMs:
|
|
308
|
+
debounceMs: LOCAL_SLOW_LANE_DEBOUNCE_MS,
|
|
289
309
|
})
|
|
290
310
|
: createSlowLaneClient({
|
|
291
311
|
baseUrl: '',
|
|
292
|
-
debounceMs:
|
|
312
|
+
debounceMs: NETWORK_SLOW_LANE_DEBOUNCE_MS,
|
|
293
313
|
});
|
|
294
314
|
setDefaultSlowLaneClient(slowLaneClient);
|
|
295
315
|
|
|
316
|
+
let contextRequestInFlight = false;
|
|
317
|
+
let lastContextRefreshAt = 0;
|
|
318
|
+
// Bounds the word-boundary retry loop so it terminates even when the editor is
|
|
319
|
+
// not in a comment thread (parentCommentContent never resolves).
|
|
320
|
+
let wordBoundaryRefreshAttempts = 0;
|
|
321
|
+
// Set when the plugin is torn down so in-flight getContext() resolutions don't
|
|
322
|
+
// mutate the global text-predictor state after destruction.
|
|
323
|
+
let destroyed = false;
|
|
324
|
+
const ingestedContextTexts = new Set<string>();
|
|
325
|
+
|
|
326
|
+
const logContextResolved = (source: string, context?: AutocompleteContext): void => {
|
|
327
|
+
if (!isAutocompleteDebugEnabled()) {
|
|
328
|
+
return;
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
// eslint-disable-next-line no-console
|
|
332
|
+
console.log(
|
|
333
|
+
'%c[Autocomplete] %cgetContext resolved',
|
|
334
|
+
'color: #00b8d9; font-weight: bold;',
|
|
335
|
+
'color: inherit;',
|
|
336
|
+
{
|
|
337
|
+
source,
|
|
338
|
+
hasParentComment: !!context?.parentCommentContent,
|
|
339
|
+
parentCommentPreview: context?.parentCommentContent?.slice(0, 80),
|
|
340
|
+
siblingCount: context?.siblingCommentsContents?.length ?? 0,
|
|
341
|
+
hasFullPage: !!context?.fullPageContent,
|
|
342
|
+
},
|
|
343
|
+
);
|
|
344
|
+
};
|
|
345
|
+
|
|
346
|
+
const applyContext = (context: AutocompleteContext): void => {
|
|
347
|
+
// Merge rather than replace: the word-boundary retry may resolve only a
|
|
348
|
+
// late-arriving field (e.g. parentCommentContent) without re-sending
|
|
349
|
+
// fullPageContent, so replacing would drop previously resolved context.
|
|
350
|
+
// Strip undefined values first so a field explicitly set to undefined by
|
|
351
|
+
// getContext doesn't overwrite a previously resolved value.
|
|
352
|
+
const definedContext = Object.fromEntries(
|
|
353
|
+
Object.entries(context).filter(([, value]) => value !== undefined),
|
|
354
|
+
);
|
|
355
|
+
resolvedContext = { ...resolvedContext, ...definedContext };
|
|
356
|
+
|
|
357
|
+
const ingestContextText = (text?: string): void => {
|
|
358
|
+
if (!text || ingestedContextTexts.has(text)) {
|
|
359
|
+
return;
|
|
360
|
+
}
|
|
361
|
+
|
|
362
|
+
ingestedContextTexts.add(text);
|
|
363
|
+
// Evict oldest entries (Set preserves insertion order) to bound memory.
|
|
364
|
+
while (ingestedContextTexts.size > MAX_INGESTED_CONTEXT_TEXTS) {
|
|
365
|
+
const oldest = ingestedContextTexts.values().next().value;
|
|
366
|
+
if (oldest === undefined) {
|
|
367
|
+
break;
|
|
368
|
+
}
|
|
369
|
+
ingestedContextTexts.delete(oldest);
|
|
370
|
+
}
|
|
371
|
+
ingestDocumentPage(text);
|
|
372
|
+
};
|
|
373
|
+
|
|
374
|
+
ingestContextText(context.fullPageContent);
|
|
375
|
+
ingestContextText(context.parentCommentContent);
|
|
376
|
+
for (const siblingCommentContent of context.siblingCommentsContents ?? []) {
|
|
377
|
+
ingestContextText(siblingCommentContent);
|
|
378
|
+
}
|
|
379
|
+
|
|
380
|
+
// Context arrived after word boundaries may already have fired. Re-send
|
|
381
|
+
// slow-lane context immediately so the next inference includes the thread.
|
|
382
|
+
if (currentView) {
|
|
383
|
+
slowLaneClient.updateContext(
|
|
384
|
+
buildSlowLaneText(currentView.state.doc.textContent, resolvedContext),
|
|
385
|
+
);
|
|
386
|
+
}
|
|
387
|
+
};
|
|
388
|
+
|
|
389
|
+
/**
|
|
390
|
+
* Returns true when a fetch was actually started, false when it was skipped
|
|
391
|
+
* (no getContext, a request already in flight, or throttled). Callers that
|
|
392
|
+
* track a retry budget should only count attempts where this returns true.
|
|
393
|
+
*/
|
|
394
|
+
const refreshContext = ({
|
|
395
|
+
source,
|
|
396
|
+
allowThrottle = true,
|
|
397
|
+
}: {
|
|
398
|
+
allowThrottle?: boolean;
|
|
399
|
+
source: string;
|
|
400
|
+
}): boolean => {
|
|
401
|
+
if (!options?.getContext || contextRequestInFlight) {
|
|
402
|
+
return false;
|
|
403
|
+
}
|
|
404
|
+
|
|
405
|
+
const now = Date.now();
|
|
406
|
+
if (allowThrottle && now - lastContextRefreshAt < CONTEXT_REFRESH_THROTTLE_MS) {
|
|
407
|
+
return false;
|
|
408
|
+
}
|
|
409
|
+
|
|
410
|
+
contextRequestInFlight = true;
|
|
411
|
+
lastContextRefreshAt = now;
|
|
412
|
+
|
|
413
|
+
options
|
|
414
|
+
.getContext()
|
|
415
|
+
.then((context) => {
|
|
416
|
+
// Bail if the plugin was destroyed while the fetch was in flight —
|
|
417
|
+
// applyContext mutates global text-predictor state we must not touch
|
|
418
|
+
// after teardown.
|
|
419
|
+
if (destroyed) {
|
|
420
|
+
return;
|
|
421
|
+
}
|
|
422
|
+
logContextResolved(source, context);
|
|
423
|
+
if (!context) {
|
|
424
|
+
return;
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
applyContext(context);
|
|
428
|
+
})
|
|
429
|
+
.catch((error) => {
|
|
430
|
+
logException(error as Error, {
|
|
431
|
+
location: 'editor-plugin-autocomplete/getContext',
|
|
432
|
+
});
|
|
433
|
+
})
|
|
434
|
+
.finally(() => {
|
|
435
|
+
contextRequestInFlight = false;
|
|
436
|
+
});
|
|
437
|
+
|
|
438
|
+
return true;
|
|
439
|
+
};
|
|
440
|
+
|
|
296
441
|
/**
|
|
297
442
|
* Schedule a prediction after a short debounce.
|
|
298
443
|
* Tier 1 predictions are synchronous (<0.1ms) but we still debounce
|
|
@@ -490,29 +635,10 @@ export const createAutocompletePlugin = (
|
|
|
490
635
|
});
|
|
491
636
|
if (!hasIngestedPage) {
|
|
492
637
|
hasIngestedPage = true;
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
if (!context) {
|
|
498
|
-
return;
|
|
499
|
-
}
|
|
500
|
-
resolvedContext = context;
|
|
501
|
-
|
|
502
|
-
if (context.fullPageContent) {
|
|
503
|
-
ingestDocumentPage(context.fullPageContent);
|
|
504
|
-
}
|
|
505
|
-
if (context.parentCommentContent) {
|
|
506
|
-
ingestDocumentPage(context.parentCommentContent);
|
|
507
|
-
}
|
|
508
|
-
context.siblingCommentsContents?.forEach(ingestDocumentPage);
|
|
509
|
-
})
|
|
510
|
-
.catch((error) => {
|
|
511
|
-
logException(error as Error, {
|
|
512
|
-
location: 'editor-plugin-autocomplete/getContext',
|
|
513
|
-
});
|
|
514
|
-
});
|
|
515
|
-
}
|
|
638
|
+
refreshContext({
|
|
639
|
+
source: 'focus',
|
|
640
|
+
allowThrottle: false,
|
|
641
|
+
});
|
|
516
642
|
}
|
|
517
643
|
return false;
|
|
518
644
|
},
|
|
@@ -521,6 +647,7 @@ export const createAutocompletePlugin = (
|
|
|
521
647
|
|
|
522
648
|
view: () => ({
|
|
523
649
|
update: (view: EditorView, prevState: EditorState) => {
|
|
650
|
+
currentView = view;
|
|
524
651
|
if (!prevState.doc.eq(view.state.doc)) {
|
|
525
652
|
if (justAccepted) {
|
|
526
653
|
justAccepted = false;
|
|
@@ -542,18 +669,36 @@ export const createAutocompletePlugin = (
|
|
|
542
669
|
slowLaneClient.updateContext(
|
|
543
670
|
buildSlowLaneText(view.state.doc.textContent, resolvedContext),
|
|
544
671
|
);
|
|
672
|
+
|
|
673
|
+
// Context may not have resolved on first focus (e.g. comment
|
|
674
|
+
// thread still loading). Retry on word boundaries until we have
|
|
675
|
+
// the parent comment, throttled so we don't refetch constantly
|
|
676
|
+
// and capped so non-comment editors stop retrying entirely.
|
|
677
|
+
if (
|
|
678
|
+
!resolvedContext?.parentCommentContent &&
|
|
679
|
+
wordBoundaryRefreshAttempts < MAX_CONTEXT_REFRESH_ATTEMPTS
|
|
680
|
+
) {
|
|
681
|
+
// Only count the attempt when a fetch actually started, so an
|
|
682
|
+
// in-flight or throttled no-op doesn't burn the retry budget.
|
|
683
|
+
if (refreshContext({ source: 'word-boundary' })) {
|
|
684
|
+
wordBoundaryRefreshAttempts++;
|
|
685
|
+
}
|
|
686
|
+
}
|
|
545
687
|
}
|
|
546
688
|
|
|
547
689
|
schedulePrediction(view);
|
|
548
690
|
}
|
|
549
691
|
},
|
|
550
692
|
destroy: () => {
|
|
693
|
+
destroyed = true;
|
|
694
|
+
currentView = null;
|
|
551
695
|
if (debounceTimer) {
|
|
552
696
|
clearTimeout(debounceTimer);
|
|
553
697
|
}
|
|
554
698
|
if (hasDestroy(slowLaneClient)) {
|
|
555
699
|
slowLaneClient.destroy();
|
|
556
700
|
}
|
|
701
|
+
ingestedContextTexts.clear();
|
|
557
702
|
setDefaultSlowLaneClient(null);
|
|
558
703
|
},
|
|
559
704
|
}),
|
|
@@ -41,6 +41,7 @@ import { isAutocompleteDebugEnabled } from './debug-mode';
|
|
|
41
41
|
import { isWordBoundary } from './slow-lane-client';
|
|
42
42
|
|
|
43
43
|
type WebLlmModelRecord = NonNullable<AppConfig['model_list']>[number];
|
|
44
|
+
type EmbeddingApiResponse = { data?: Array<{ embedding?: unknown }> };
|
|
44
45
|
|
|
45
46
|
// ─── Types ───────────────────────────────────────────────────────────────────
|
|
46
47
|
|
|
@@ -158,12 +159,42 @@ export const BE_PARITY = {
|
|
|
158
159
|
MAX_CONTEXT_WORDS: 100,
|
|
159
160
|
} as const;
|
|
160
161
|
|
|
162
|
+
const splitOnWhitespace = (text: string): string[] => {
|
|
163
|
+
const trimmed = text.trim();
|
|
164
|
+
if (trimmed === '') {
|
|
165
|
+
return [];
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
const words: string[] = [];
|
|
169
|
+
let wordStart = -1;
|
|
170
|
+
|
|
171
|
+
for (let i = 0; i < trimmed.length; i++) {
|
|
172
|
+
if (trimmed[i].trim() === '') {
|
|
173
|
+
if (wordStart !== -1) {
|
|
174
|
+
words.push(trimmed.slice(wordStart, i));
|
|
175
|
+
wordStart = -1;
|
|
176
|
+
}
|
|
177
|
+
continue;
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
if (wordStart === -1) {
|
|
181
|
+
wordStart = i;
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
if (wordStart !== -1) {
|
|
186
|
+
words.push(trimmed.slice(wordStart));
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
return words;
|
|
190
|
+
};
|
|
191
|
+
|
|
161
192
|
/**
|
|
162
193
|
* Return the last `n` whitespace-separated words of `text`, joined by spaces.
|
|
163
194
|
* Mirrors the BE rolling-window truncation applied before both encoders.
|
|
164
195
|
*/
|
|
165
196
|
const truncateToLastNWords = (text: string, n: number): string => {
|
|
166
|
-
const words = text
|
|
197
|
+
const words = splitOnWhitespace(text);
|
|
167
198
|
return words.length <= n ? text : words.slice(-n).join(' ');
|
|
168
199
|
};
|
|
169
200
|
|
|
@@ -564,6 +595,14 @@ export const createLocalSlowLaneClient = (
|
|
|
564
595
|
let lastRequestedText = '';
|
|
565
596
|
let requestCounter = 0;
|
|
566
597
|
let latestRequestId = -1;
|
|
598
|
+
let inferenceInFlight = false;
|
|
599
|
+
let activeInferenceText: string | null = null;
|
|
600
|
+
// The requestId of the inference currently in flight. Tracked so the
|
|
601
|
+
// in-flight dedup path can restore `latestRequestId` to it — otherwise an
|
|
602
|
+
// intermediate keystroke that bumped `latestRequestId` would cause the
|
|
603
|
+
// in-flight (still-current) result to be discarded as stale.
|
|
604
|
+
let activeInferenceRequestId = -1;
|
|
605
|
+
let pendingInference: { requestId: number; text: string } | null = null;
|
|
567
606
|
let ready = false;
|
|
568
607
|
let destroyed = false;
|
|
569
608
|
let initFailed = false;
|
|
@@ -755,17 +794,25 @@ export const createLocalSlowLaneClient = (
|
|
|
755
794
|
const lmText = truncateToLastNWords(text, BE_PARITY.MAX_CONTEXT_TOKENS);
|
|
756
795
|
const semanticText = truncateToLastNWords(text, BE_PARITY.MAX_CONTEXT_WORDS);
|
|
757
796
|
const arcticInput = wrapForArctic(semanticText);
|
|
797
|
+
const captureCompletionTime = <T,>(
|
|
798
|
+
promise: Promise<T>,
|
|
799
|
+
onResolved: (resolvedAt: number) => void,
|
|
800
|
+
): Promise<T> =>
|
|
801
|
+
promise.then((value: T) => {
|
|
802
|
+
onResolved(performance.now());
|
|
803
|
+
return value;
|
|
804
|
+
});
|
|
758
805
|
|
|
759
806
|
if (isAutocompleteDebugEnabled()) {
|
|
760
807
|
// eslint-disable-next-line no-console
|
|
761
808
|
console.log(
|
|
762
|
-
`%c[LocalSlowLane] %c🔢 Arctic input (${arcticInput.length} chars, ${semanticText
|
|
809
|
+
`%c[LocalSlowLane] %c🔢 Arctic input (${arcticInput.length} chars, ${splitOnWhitespace(semanticText).length} words): "${arcticInput.length > 100 ? `${arcticInput.slice(0, 100)}…` : arcticInput}"`,
|
|
763
810
|
'color: #9c27b0; font-weight: bold;',
|
|
764
811
|
'color: #009688;',
|
|
765
812
|
);
|
|
766
813
|
// eslint-disable-next-line no-console
|
|
767
814
|
console.log(
|
|
768
|
-
`%c[LocalSlowLane] %c🧠 LM input (${lmText.length} chars, ${lmText
|
|
815
|
+
`%c[LocalSlowLane] %c🧠 LM input (${lmText.length} chars, ${splitOnWhitespace(lmText).length} words): "${lmText.length > 100 ? `${lmText.slice(0, 100)}…` : lmText}"`,
|
|
769
816
|
'color: #9c27b0; font-weight: bold;',
|
|
770
817
|
'color: #2196f3;',
|
|
771
818
|
);
|
|
@@ -777,27 +824,29 @@ export const createLocalSlowLaneClient = (
|
|
|
777
824
|
let tEmbDone = 0;
|
|
778
825
|
|
|
779
826
|
const [, embeddingResponse] = await Promise.all([
|
|
780
|
-
|
|
781
|
-
.create({
|
|
827
|
+
captureCompletionTime(
|
|
828
|
+
engine.completions.create({
|
|
782
829
|
model: modelId,
|
|
783
830
|
prompt: lmText,
|
|
784
831
|
max_tokens: 1,
|
|
785
832
|
temperature: 0,
|
|
786
833
|
logprobs: false,
|
|
787
834
|
})
|
|
788
|
-
|
|
789
|
-
|
|
790
|
-
|
|
791
|
-
}
|
|
792
|
-
|
|
793
|
-
|
|
835
|
+
,
|
|
836
|
+
(resolvedAt) => {
|
|
837
|
+
tLmDone = resolvedAt;
|
|
838
|
+
},
|
|
839
|
+
),
|
|
840
|
+
captureCompletionTime(
|
|
841
|
+
engine.embeddings.create({
|
|
794
842
|
model: LOCAL_MLC_EMBEDDING_MODEL_ID,
|
|
795
843
|
input: arcticInput,
|
|
796
844
|
})
|
|
797
|
-
|
|
798
|
-
|
|
799
|
-
|
|
800
|
-
}
|
|
845
|
+
,
|
|
846
|
+
(resolvedAt) => {
|
|
847
|
+
tEmbDone = resolvedAt;
|
|
848
|
+
},
|
|
849
|
+
),
|
|
801
850
|
]);
|
|
802
851
|
|
|
803
852
|
if (isAutocompleteDebugEnabled()) {
|
|
@@ -827,7 +876,7 @@ export const createLocalSlowLaneClient = (
|
|
|
827
876
|
// Guard against base64-encoded responses (encoding_format: 'base64' would
|
|
828
877
|
// yield a string, and new Float32Array(string) silently produces an empty
|
|
829
878
|
// array, corrupting downstream cosine-similarity scoring).
|
|
830
|
-
const embedding = embeddingResponse.data?.[0]?.embedding;
|
|
879
|
+
const embedding = (embeddingResponse as EmbeddingApiResponse).data?.[0]?.embedding;
|
|
831
880
|
storedContextVector =
|
|
832
881
|
Array.isArray(embedding) && embedding.length > 0
|
|
833
882
|
? new Float32Array(embedding as number[])
|
|
@@ -879,8 +928,8 @@ export const createLocalSlowLaneClient = (
|
|
|
879
928
|
hasLmLogits: storedLmLogits !== null,
|
|
880
929
|
});
|
|
881
930
|
} catch (err) {
|
|
882
|
-
// Discard errors for stale requests
|
|
883
|
-
if (requestId < latestRequestId) {
|
|
931
|
+
// Discard errors for stale requests or after teardown
|
|
932
|
+
if (requestId < latestRequestId || destroyed) {
|
|
884
933
|
return;
|
|
885
934
|
}
|
|
886
935
|
|
|
@@ -902,11 +951,55 @@ export const createLocalSlowLaneClient = (
|
|
|
902
951
|
|
|
903
952
|
// ── Context update (debounced) ─────────────────────────────────────────
|
|
904
953
|
|
|
954
|
+
const startInference = (text: string, requestId: number): void => {
|
|
955
|
+
// Self-contained guard: never start a new inference cycle after teardown,
|
|
956
|
+
// regardless of caller discipline.
|
|
957
|
+
if (destroyed) {
|
|
958
|
+
return;
|
|
959
|
+
}
|
|
960
|
+
|
|
961
|
+
inferenceInFlight = true;
|
|
962
|
+
activeInferenceText = text;
|
|
963
|
+
activeInferenceRequestId = requestId;
|
|
964
|
+
|
|
965
|
+
void ensureEngineInitialized()
|
|
966
|
+
.then(() => runInference(text, requestId))
|
|
967
|
+
.catch(() => {})
|
|
968
|
+
.finally(() => {
|
|
969
|
+
inferenceInFlight = false;
|
|
970
|
+
activeInferenceText = null;
|
|
971
|
+
activeInferenceRequestId = -1;
|
|
972
|
+
|
|
973
|
+
const next = pendingInference;
|
|
974
|
+
pendingInference = null;
|
|
975
|
+
if (next && !destroyed) {
|
|
976
|
+
startInference(next.text, next.requestId);
|
|
977
|
+
}
|
|
978
|
+
});
|
|
979
|
+
};
|
|
980
|
+
|
|
905
981
|
const doUpdateContext = (text: string): void => {
|
|
906
982
|
if (destroyed || !text || text.trim().length === 0) {
|
|
907
983
|
return;
|
|
908
984
|
}
|
|
909
985
|
|
|
986
|
+
if (inferenceInFlight && text === activeInferenceText) {
|
|
987
|
+
// The latest desired text already matches the in-flight inference, so
|
|
988
|
+
// re-running it would be wasted work. But an intermediate keystroke may
|
|
989
|
+
// have bumped `latestRequestId` past the in-flight request (and then been
|
|
990
|
+
// coalesced away), which would cause runInference to discard the
|
|
991
|
+
// still-current result as stale. Pin `latestRequestId` back to the active
|
|
992
|
+
// request so its result is accepted, and drop any now-superseded pending
|
|
993
|
+
// request.
|
|
994
|
+
latestRequestId = activeInferenceRequestId;
|
|
995
|
+
pendingInference = null;
|
|
996
|
+
return;
|
|
997
|
+
}
|
|
998
|
+
|
|
999
|
+
if (inferenceInFlight && pendingInference?.text === text) {
|
|
1000
|
+
return;
|
|
1001
|
+
}
|
|
1002
|
+
|
|
910
1003
|
const requestId = ++requestCounter;
|
|
911
1004
|
latestRequestId = requestId;
|
|
912
1005
|
|
|
@@ -926,15 +1019,26 @@ export const createLocalSlowLaneClient = (
|
|
|
926
1019
|
console.groupEnd();
|
|
927
1020
|
}
|
|
928
1021
|
|
|
929
|
-
|
|
930
|
-
|
|
931
|
-
|
|
1022
|
+
if (inferenceInFlight) {
|
|
1023
|
+
pendingInference = { text, requestId };
|
|
1024
|
+
return;
|
|
1025
|
+
}
|
|
1026
|
+
|
|
1027
|
+
startInference(text, requestId);
|
|
932
1028
|
};
|
|
933
1029
|
|
|
934
1030
|
const updateContextDebounced = (text: string): void => {
|
|
935
1031
|
if (debounceTimer) {
|
|
936
1032
|
clearTimeout(debounceTimer);
|
|
937
1033
|
}
|
|
1034
|
+
if (inferenceInFlight) {
|
|
1035
|
+
pendingInference = null;
|
|
1036
|
+
if (text === activeInferenceText) {
|
|
1037
|
+
latestRequestId = activeInferenceRequestId;
|
|
1038
|
+
lastRequestedText = text;
|
|
1039
|
+
return;
|
|
1040
|
+
}
|
|
1041
|
+
}
|
|
938
1042
|
lastRequestedText = text;
|
|
939
1043
|
debounceTimer = setTimeout(() => {
|
|
940
1044
|
debounceTimer = null;
|
|
@@ -967,6 +1071,10 @@ export const createLocalSlowLaneClient = (
|
|
|
967
1071
|
unloadEngine(engineToUnload);
|
|
968
1072
|
}
|
|
969
1073
|
engineInitPromise = null;
|
|
1074
|
+
inferenceInFlight = false;
|
|
1075
|
+
activeInferenceText = null;
|
|
1076
|
+
activeInferenceRequestId = -1;
|
|
1077
|
+
pendingInference = null;
|
|
970
1078
|
storedContextVector = null;
|
|
971
1079
|
storedLmLogits = null;
|
|
972
1080
|
},
|