@atlaskit/editor-plugin-autocomplete 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/CHANGELOG.md +8 -0
  2. package/afm-cc/tsconfig.json +5 -1
  3. package/afm-jira/tsconfig.json +5 -1
  4. package/afm-products/tsconfig.json +5 -1
  5. package/build/tsconfig.json +24 -0
  6. package/dist/cjs/autocompletePlugin.js +18 -5
  7. package/dist/cjs/autocompletePluginType.js +5 -1
  8. package/dist/cjs/pm-plugins/autocomplete-plugin.js +361 -0
  9. package/dist/cjs/pm-plugins/ghost-text-decoration.js +39 -0
  10. package/dist/cjs/pm-plugins/scoring-pipeline.js +258 -0
  11. package/dist/cjs/pm-plugins/slow-lane-client.js +197 -0
  12. package/dist/cjs/pm-plugins/text-predictor.js +786 -0
  13. package/dist/es2019/autocompletePlugin.js +20 -6
  14. package/dist/es2019/autocompletePluginType.js +1 -0
  15. package/dist/es2019/pm-plugins/autocomplete-plugin.js +360 -0
  16. package/dist/es2019/pm-plugins/ghost-text-decoration.js +33 -0
  17. package/dist/es2019/pm-plugins/scoring-pipeline.js +224 -0
  18. package/dist/es2019/pm-plugins/slow-lane-client.js +154 -0
  19. package/dist/es2019/pm-plugins/text-predictor.js +624 -0
  20. package/dist/esm/autocompletePlugin.js +18 -5
  21. package/dist/esm/autocompletePluginType.js +1 -0
  22. package/dist/esm/pm-plugins/autocomplete-plugin.js +354 -0
  23. package/dist/esm/pm-plugins/ghost-text-decoration.js +33 -0
  24. package/dist/esm/pm-plugins/scoring-pipeline.js +255 -0
  25. package/dist/esm/pm-plugins/slow-lane-client.js +190 -0
  26. package/dist/esm/pm-plugins/text-predictor.js +783 -0
  27. package/dist/types/autocompletePluginType.d.ts +6 -3
  28. package/dist/types/pm-plugins/autocomplete-plugin.d.ts +36 -0
  29. package/dist/types/pm-plugins/ghost-text-decoration.d.ts +7 -0
  30. package/dist/types/pm-plugins/scoring-pipeline.d.ts +33 -0
  31. package/dist/types/pm-plugins/slow-lane-client.d.ts +46 -0
  32. package/dist/types/pm-plugins/text-predictor.d.ts +90 -0
  33. package/dist/types-ts4.5/autocompletePluginType.d.ts +6 -3
  34. package/dist/types-ts4.5/pm-plugins/autocomplete-plugin.d.ts +36 -0
  35. package/dist/types-ts4.5/pm-plugins/ghost-text-decoration.d.ts +7 -0
  36. package/dist/types-ts4.5/pm-plugins/scoring-pipeline.d.ts +33 -0
  37. package/dist/types-ts4.5/pm-plugins/slow-lane-client.d.ts +46 -0
  38. package/dist/types-ts4.5/pm-plugins/text-predictor.d.ts +90 -0
  39. package/package.json +2 -2
  40. package/src/autocompletePlugin.tsx +25 -5
  41. package/src/autocompletePluginType.ts +14 -3
  42. package/src/pm-plugins/autocomplete-plugin/package.json +15 -0
  43. package/src/pm-plugins/autocomplete-plugin.ts +429 -0
  44. package/src/pm-plugins/ghost-text-decoration.ts +44 -0
  45. package/src/pm-plugins/scoring-pipeline.ts +297 -0
  46. package/src/pm-plugins/slow-lane-client/package.json +15 -0
  47. package/src/pm-plugins/slow-lane-client.ts +220 -0
  48. package/src/pm-plugins/text-predictor/package.json +15 -0
  49. package/src/pm-plugins/text-predictor.ts +771 -0
  50. package/src/pm-plugins/typings.d.ts +12 -0
  51. package/tsconfig.app.json +10 -2
  52. package/tsconfig.json +3 -1
@@ -0,0 +1,771 @@
1
+ /**
2
+ * Fast Lane Predictor: Local autocomplete using weighted trie + frequency + semantic scoring.
3
+ *
4
+ * Two prediction modes:
5
+ * 1. Word boundary → bigram-based next-word suggestion (grammar-filtered)
6
+ * 2. Mid-word (≥3 chars) → trie prefix search → scoring pipeline → top result
7
+ *
8
+ * Scoring is delegated to scoring-pipeline.ts which handles:
9
+ * Stage 1 (semantic + frequency), grammar filter, Stage 2 (optional LM re-ranking).
10
+ *
11
+ * Context vector: average of word vectors from text before cursor (last N words).
12
+ * Falls back to cold mode (freq-only) when vectors not yet loaded.
13
+ *
14
+ * Session personalization (L1): words the user types are incrementally boosted
15
+ * via incrementSessionFreq(), called on word boundaries from the plugin.
16
+ */
17
+
18
+ // import bigramsData from './data/bigrams.json';
19
+ import l3VocabularyData from './data/l3_vocabulary.json';
20
+ import vocabularyData from './data/vocabulary_10k.json';
21
+ import wordIndexData from './data/word_index_10k.json';
22
+ // import { rankCandidates, isGrammarAllowed } from './scoring-pipeline';
23
+ import { rankCandidates } from './scoring-pipeline';
24
+ import type { ScoringCandidate } from './scoring-pipeline';
25
+ import { getStoredContextVector, getStoredLmLogits } from './slow-lane-client';
26
+
27
+ // ─── Constants ───────────────────────────────────────────────────────────────
28
+
29
+ const PUNCTUATION_BOUNDARY_REGEX = /^[.,;:!?()\[\]{}"'`]+|[.,;:!?()\[\]{}"'`]+$/gu;
30
+
31
+ const MIN_PREFIX_LENGTH = 3;
32
+ const MAX_CANDIDATES = 200;
33
+ const CONTEXT_WORDS = 10;
34
+ const MIN_SCORE_THRESHOLD = 0.2;
35
+ const L3_BASELINE_FREQ = 0.001;
36
+
37
+ // ─── Types ───────────────────────────────────────────────────────────────────
38
+
39
+ interface Candidate {
40
+ node: TrieNode;
41
+ word: string;
42
+ }
43
+
44
+ export interface WeightedTerm {
45
+ authorFreq: number;
46
+ docFreq: number;
47
+ freq: number;
48
+ word: string;
49
+ }
50
+
51
+ export interface TenantVocabulary {
52
+ terms: WeightedTerm[];
53
+ }
54
+
55
+ interface VectorStore {
56
+ dim: number;
57
+ float32: Float32Array;
58
+ wordIndex: Record<string, number>;
59
+ }
60
+
61
+ class TrieNode {
62
+ children: Map<string, TrieNode> = new Map();
63
+ word: string | null = null;
64
+ tenantFreq: number = 0;
65
+ docFreq: number = 0;
66
+ authorFreq: number = 0;
67
+ sessionFreq: number = 0;
68
+ }
69
+
70
+ class WeightedWordTrie {
71
+ private root = new TrieNode();
72
+ /** Highest tenantFreq seen — used to normalize freq scores at query time */
73
+ maxTenantFreq: number = 1;
74
+
75
+ insert(word: string, tenantFreq: number, docFreq: number, authorFreq: number): void {
76
+ let node = this.root;
77
+ for (const char of word.toLowerCase()) {
78
+ let next = node.children.get(char);
79
+ if (!next) {
80
+ next = new TrieNode();
81
+ node.children.set(char, next);
82
+ }
83
+ node = next;
84
+ }
85
+ node.word = word;
86
+ node.tenantFreq = tenantFreq;
87
+ node.docFreq = docFreq;
88
+ node.authorFreq = authorFreq;
89
+ if (tenantFreq > this.maxTenantFreq) {
90
+ this.maxTenantFreq = tenantFreq;
91
+ }
92
+ }
93
+
94
+ /**
95
+ * Return all words matching this prefix, up to maxResults.
96
+ * O(prefix_length + results) — traverses to the prefix node then collects subtree.
97
+ */
98
+ getCandidates(prefix: string, maxResults: number = MAX_CANDIDATES): Candidate[] {
99
+ let node = this.root;
100
+ for (const char of prefix.toLowerCase()) {
101
+ const next = node.children.get(char);
102
+ if (!next) {
103
+ return [];
104
+ }
105
+ node = next;
106
+ }
107
+
108
+ const candidates: Candidate[] = [];
109
+ const stack: TrieNode[] = [node];
110
+
111
+ while (stack.length > 0 && candidates.length < maxResults) {
112
+ const current = stack.pop();
113
+ if (!current) {
114
+ continue;
115
+ }
116
+
117
+ // Only add candidates that are longer than the prefix
118
+ if (current.word && current.word.length > prefix.length) {
119
+ candidates.push({ word: current.word, node: current });
120
+ }
121
+ for (const child of current.children.values()) {
122
+ stack.push(child);
123
+ }
124
+ }
125
+
126
+ return candidates;
127
+ }
128
+
129
+ private findNode(word: string): TrieNode | null {
130
+ let node = this.root;
131
+ for (const char of word.toLowerCase()) {
132
+ const next = node.children.get(char);
133
+ if (!next) {
134
+ return null;
135
+ }
136
+ node = next;
137
+ }
138
+ return node.word !== null ? node : null;
139
+ }
140
+
141
+ /**
142
+ * Set the session frequency for a word.
143
+ * Returns true if the word exists in the trie.
144
+ */
145
+ updateSessionFreq(word: string, count: number): boolean {
146
+ const node = this.findNode(word);
147
+ if (!node) {
148
+ return false;
149
+ }
150
+ node.sessionFreq = count;
151
+ return true;
152
+ }
153
+
154
+ /**
155
+ * Increment the session frequency for a word by 1.
156
+ * Returns true if the word exists in the trie.
157
+ */
158
+ incrementSessionFreq(word: string): boolean {
159
+ const node = this.findNode(word);
160
+ if (!node) {
161
+ return false;
162
+ }
163
+ node.sessionFreq += 1;
164
+ return true;
165
+ }
166
+ }
167
+
168
+ // L1/L2 Trie (Session + Atlassian Domain)
169
+ const wordTrie = new WeightedWordTrie();
170
+
171
+ // L3 Trie (General English Fallback)
172
+ const l3Trie = new WeightedWordTrie();
173
+
174
+ // --- Initialization Function ---
175
+ /**
176
+ * Loads the General English vocabulary.
177
+ * expects a simple array of strings: ["about", "above", "actually", ...]
178
+ */
179
+ export const initL3Vocabulary = (l3Words: string[]): void => {
180
+ for (const word of l3Words) {
181
+ // Insert with a tiny baseline frequency so it mathematically
182
+ // loses to any domain word in Stage 1, but still scores above 0.
183
+ l3Trie.insert(word, L3_BASELINE_FREQ, 0, 0);
184
+ }
185
+ if (debugMode) {
186
+ // eslint-disable-next-line no-console
187
+ console.log(`[text-predictor] L3 General English loaded: ${l3Words.length} words`);
188
+ }
189
+ };
190
+
191
+ // const bigramMap: Map<string, Record<string, number>> = new Map(
192
+ // Object.entries(bigramsData as Record<string, Record<string, number>>),
193
+ // );
194
+
195
+ let isInitialized = false;
196
+
197
+ let vectorStore: VectorStore | null = null;
198
+
199
+ let vectorsLoadStarted = false;
200
+
201
+ let debugMode = true;
202
+
203
+ let lastPredictionDebug: {
204
+ contextWords: string[];
205
+ currentWord: string;
206
+ mode: 'cold' | 'warm';
207
+ suggestion: string | null;
208
+ textBefore: string;
209
+ topCandidates: Array<{
210
+ finalScore: number;
211
+ freqScore: number;
212
+ lmScore: number;
213
+ semanticScore: number;
214
+ word: string;
215
+ }>;
216
+ } | null = null;
217
+
218
+ let hasLoggedSemanticActive = false;
219
+
220
+ /** Get vector for a word from the store. */
221
+ const getWordVector = (word: string): Float32Array | null => {
222
+ if (!vectorStore) {
223
+ return null;
224
+ }
225
+ const idx = vectorStore.wordIndex[word.toLowerCase()];
226
+ if (idx === undefined) {
227
+ return null;
228
+ }
229
+ const start = idx * vectorStore.dim;
230
+ return vectorStore.float32.subarray(start, start + vectorStore.dim);
231
+ };
232
+
233
+ /**
234
+ * Compute context vector by averaging vectors of last N words in text.
235
+ * Falls back to null (cold mode) if no words have vectors.
236
+ */
237
+ const computeContextVectorLocal = (textBefore: string): Float32Array | null => {
238
+ if (!vectorStore) {
239
+ return null;
240
+ }
241
+ const tokens = tokenize(textBefore);
242
+ const words = tokens.slice(-CONTEXT_WORDS);
243
+ const vectors: Float32Array[] = [];
244
+ for (const word of words) {
245
+ const v = getWordVector(word);
246
+ if (v) {
247
+ vectors.push(v);
248
+ }
249
+ }
250
+ if (vectors.length === 0) {
251
+ return null;
252
+ }
253
+
254
+ const dim = vectorStore.dim;
255
+ const avg = new Float32Array(dim);
256
+ for (const v of vectors) {
257
+ for (let i = 0; i < dim; i++) {
258
+ avg[i] += v[i];
259
+ }
260
+ }
261
+ for (let i = 0; i < dim; i++) {
262
+ avg[i] /= vectors.length;
263
+ }
264
+ return avg;
265
+ };
266
+
267
+ /**
268
+ * Get context vector for scoring. Prefers Slow Lane (BE) context when available
269
+ * and dimension matches; otherwise falls back to local averaging.
270
+ */
271
+ const getContextVectorForScoring = (textBefore: string): Float32Array | null => {
272
+ const slowLaneVector = getStoredContextVector();
273
+ if (slowLaneVector && vectorStore && slowLaneVector.length === vectorStore.dim) {
274
+ return slowLaneVector;
275
+ }
276
+ return computeContextVectorLocal(textBefore);
277
+ };
278
+
279
+ const tokenize = (text: string): string[] => {
280
+ const tokens: string[] = [];
281
+ // eslint-disable-next-line @atlassian/perf-linting/no-expensive-split-replace
282
+ for (const raw of text.toLowerCase().split(/\s+/u)) {
283
+ // eslint-disable-next-line @atlassian/perf-linting/no-expensive-split-replace
284
+ const clean = raw.replace(PUNCTUATION_BOUNDARY_REGEX, '');
285
+ if (clean.length >= 2) {
286
+ tokens.push(clean);
287
+ }
288
+ }
289
+ return tokens;
290
+ };
291
+
292
+ const extractPreviousWord = (text: string): string => {
293
+ // 1. Split the text by newlines or punctuation (. ? !)
294
+ const sentences = text.split(/[\n.?!]+/u);
295
+
296
+ // 2. Only look at the current sentence/line the user is typing in
297
+ const currentSentence = sentences[sentences.length - 1];
298
+
299
+ // 3. Extract the previous word as normal
300
+ const words = currentSentence.trimEnd().split(/\s+/u);
301
+ return words.length >= 2 ? words[words.length - 2] : '';
302
+ };
303
+
304
+ // ─── Debug Helpers ───────────────────────────────────────────────────────────
305
+
306
+ /** Enable or disable debug logging. Also checks localStorage key `autocomplete-debug`. */
307
+ export const setDebugMode = (enabled: boolean): void => {
308
+ debugMode = enabled;
309
+ };
310
+
311
+ /** Check localStorage for autocomplete-debug on first access. */
312
+ const ensureDebugModeFromStorage = (): void => {
313
+ if (typeof localStorage !== 'undefined' && localStorage.getItem('autocomplete-debug') === '1') {
314
+ debugMode = true;
315
+ }
316
+ };
317
+
318
+ /**
319
+ * Get predictor status for debugging.
320
+ * vectorsLoaded: true when semantic scoring is active
321
+ * wordCount: number of words in vector store (0 if not loaded)
322
+ */
323
+ export const getPredictorStatus = (): {
324
+ isInitialized: boolean;
325
+ vectorsLoaded: boolean;
326
+ vectorsLoadStarted: boolean;
327
+ wordCount: number;
328
+ } => {
329
+ ensureDebugModeFromStorage();
330
+ return {
331
+ vectorsLoaded: vectorStore !== null,
332
+ wordCount: vectorStore ? Object.keys(vectorStore.wordIndex).length : 0,
333
+ vectorsLoadStarted,
334
+ isInitialized,
335
+ };
336
+ };
337
+
338
+ /**
339
+ * Get details of the last prediction (for debugging).
340
+ * Returns null if no prediction has run yet or debug was off.
341
+ */
342
+ export const getLastPredictionDebug = (): {
343
+ contextWords: string[];
344
+ currentWord: string;
345
+ mode: 'cold' | 'warm';
346
+ suggestion: string | null;
347
+ textBefore: string;
348
+ topCandidates: Array<{
349
+ finalScore: number;
350
+ freqScore: number;
351
+ lmScore: number;
352
+ semanticScore: number;
353
+ word: string;
354
+ }>;
355
+ } | null => {
356
+ ensureDebugModeFromStorage();
357
+ return lastPredictionDebug;
358
+ };
359
+
360
+ export const initVocabulary = (vocabulary: TenantVocabulary): void => {
361
+ for (const term of vocabulary.terms) {
362
+ wordTrie.insert(term.word, term.freq, term.docFreq, term.authorFreq);
363
+ }
364
+ isInitialized = true;
365
+ };
366
+
367
+ /**
368
+ * Increment L1 session frequency for a single word.
369
+ * Called from the plugin on word boundaries for efficient incremental boosting.
370
+ */
371
+ export const incrementSessionFreq = (word: string): void => {
372
+ wordTrie.incrementSessionFreq(word);
373
+ };
374
+
375
+ /**
376
+ * Prime session frequencies from a document page string.
377
+ *
378
+ * Iterates through every token in `pageContent` and increments its session
379
+ * frequency so that words already present on the page receive an L1 boost
380
+ * before the user starts typing.
381
+ *
382
+ * Pass `undefined` (or omit the argument) to skip priming — useful when the
383
+ * calling context does not yet have a page value available.
384
+ */
385
+ // NOTE: We ingest full page context here
386
+ export const ingestDocumentPage = (pageContent: string | undefined): void => {
387
+ if (!pageContent) {
388
+ return;
389
+ }
390
+
391
+ const words = tokenize(pageContent);
392
+ const validBoostedWords = new Set<string>();
393
+
394
+ for (const word of words) {
395
+ const didBoost = wordTrie.incrementSessionFreq(word);
396
+ if (didBoost) {
397
+ validBoostedWords.add(word);
398
+ }
399
+ }
400
+
401
+ if (debugMode && validBoostedWords.size > 0) {
402
+ // eslint-disable-next-line no-console
403
+ console.groupCollapsed(
404
+ `%c[L1 Session] %cPrimed ${validBoostedWords.size} valid dictionary words from page`,
405
+ 'color: #00b8d9; font-weight: bold;',
406
+ 'color: inherit; font-style: italic;',
407
+ );
408
+ // eslint-disable-next-line no-console
409
+ console.dir(Array.from(validBoostedWords).sort());
410
+ // eslint-disable-next-line no-console
411
+ console.groupEnd();
412
+ }
413
+ };
414
+
415
+ export const predict = (textBefore: string): string | null => {
416
+ ensureDebugModeFromStorage();
417
+
418
+ if (!isInitialized) {
419
+ loadDefaultVocabulary();
420
+ }
421
+
422
+ const t0 = performance.now();
423
+
424
+ // ── Step 1: Bigram-based next-word suggestion at word boundary ───────────
425
+ // if (textBefore.length > 0 && /\s$/u.test(textBefore)) {
426
+ // const words = textBefore.toLowerCase().trimEnd().split(/\s+/u);
427
+ // const prevWord = words[words.length - 1];
428
+ // const nextWords = bigramMap.get(prevWord);
429
+ // if (nextWords) {
430
+ // const sorted = Object.entries(nextWords).sort((a, b) => b[1] - a[1]);
431
+ // let bestWord = '';
432
+ // for (const [word] of sorted) {
433
+ // if (!isGrammarAllowed(prevWord, word)) {
434
+ // continue;
435
+ // }
436
+ // bestWord = word;
437
+ // break;
438
+ // }
439
+ // if (bestWord) {
440
+ // if (debugMode) {
441
+ // const latencyMs = performance.now() - t0;
442
+ // console.log(
443
+ // '%c[autocomplete] BIGRAM',
444
+ // 'color:cyan',
445
+ // '| "' + prevWord + '" -> "' + bestWord + '" | ' + latencyMs.toFixed(1) + 'ms',
446
+ // );
447
+ // }
448
+ // return bestWord;
449
+ // }
450
+ // }
451
+ // if (debugMode) {
452
+ // console.log(
453
+ // '%c[autocomplete] BIGRAM-MISS',
454
+ // 'color:gray',
455
+ // '| no bigram for "' + words[words.length - 1] + '", skipping prefix completion',
456
+ // );
457
+ // }
458
+ // return null;
459
+ // }
460
+
461
+ // ── Step 2: Prefix completion (≥3 chars typed) ──────────────────────────
462
+ if (textBefore.length > 0 && /\s$/u.test(textBefore)) {
463
+ return null;
464
+ }
465
+
466
+ const trimmed = textBefore.trimEnd();
467
+ const lastSpaceIdx = trimmed.lastIndexOf(' ');
468
+ const currentWord = lastSpaceIdx === -1 ? trimmed : trimmed.slice(lastSpaceIdx + 1);
469
+
470
+ if (currentWord.length < MIN_PREFIX_LENGTH) {
471
+ return null;
472
+ }
473
+
474
+ // 1. Primary Query: Ask the L2 Domain Trie
475
+ const candidates = wordTrie.getCandidates(currentWord, MAX_CANDIDATES);
476
+
477
+ // 2. Fallback Query: Gap-fill with the L3 General English Trie
478
+ if (candidates.length < MAX_CANDIDATES) {
479
+ // Ask L3 for MAX_CANDIDATES to guarantee we have enough buffer
480
+ // to survive the deduplication process.
481
+ const l3Candidates = l3Trie.getCandidates(currentWord, MAX_CANDIDATES);
482
+
483
+ const existingWords = new Set(candidates.map((c) => c.word));
484
+
485
+ for (const l3c of l3Candidates) {
486
+ if (candidates.length >= MAX_CANDIDATES) break; // Stop exactly at the limit
487
+
488
+ if (!existingWords.has(l3c.word)) {
489
+ candidates.push(l3c);
490
+ }
491
+ }
492
+ }
493
+
494
+ // If both Tries are completely empty for this prefix
495
+ if (candidates.length === 0) {
496
+ return null;
497
+ }
498
+
499
+ const previousWord = extractPreviousWord(trimmed);
500
+
501
+ const contextVector = getContextVectorForScoring(trimmed);
502
+ const lmLogits = getStoredLmLogits();
503
+
504
+ // Raw LM Output Logger
505
+ if (debugMode && lmLogits && Object.keys(lmLogits).length > 0) {
506
+ const rawLmTop = Object.entries(lmLogits)
507
+ .sort((a, b) => b[1] - a[1])
508
+ .slice(0, 5)
509
+ .map(([word, score]) => ({
510
+ Word: word,
511
+ Prob: Number(score.toFixed(5)),
512
+ }));
513
+ // eslint-disable-next-line no-console
514
+ console.log('%c[Raw LM Prediction]🧠', 'color: #e83e8c; font-weight: bold;', rawLmTop);
515
+ }
516
+
517
+ const mode: 'cold' | 'warm' = contextVector ? 'warm' : 'cold';
518
+
519
+ if (debugMode && contextVector && !hasLoggedSemanticActive) {
520
+ hasLoggedSemanticActive = true;
521
+ }
522
+
523
+ // Build ScoringCandidate array from TrieNodes
524
+ const scoringCandidates: ScoringCandidate[] = candidates.map(({ word, node }) => ({
525
+ word,
526
+ tenantFreq: node.tenantFreq,
527
+ docFreq: node.docFreq,
528
+ authorFreq: node.authorFreq,
529
+ sessionFreq: node.sessionFreq,
530
+ }));
531
+
532
+ const { candidates: ranked, grammarMeta } = rankCandidates(
533
+ scoringCandidates,
534
+ contextVector,
535
+ (w: string) => getWordVector(w),
536
+ lmLogits,
537
+ wordTrie.maxTenantFreq,
538
+ previousWord,
539
+ );
540
+
541
+ const best = ranked[0];
542
+ const suggestion =
543
+ best && best.finalScore >= MIN_SCORE_THRESHOLD ? best.word.slice(currentWord.length) : null;
544
+
545
+ if (debugMode) {
546
+ const latencyMs = performance.now() - t0;
547
+ const tokens = tokenize(trimmed);
548
+ const contextWords = tokens.slice(-CONTEXT_WORDS);
549
+ const belowThreshold = best && best.finalScore < MIN_SCORE_THRESHOLD;
550
+
551
+ const suggestionLabel = belowThreshold
552
+ ? '🚫 (below threshold)'
553
+ : suggestion && suggestion.length > 0
554
+ ? `✨ "${suggestion}"`
555
+ : '🚫 (no match)';
556
+
557
+ lastPredictionDebug = {
558
+ textBefore: trimmed,
559
+ currentWord,
560
+ mode,
561
+ contextWords,
562
+ topCandidates: ranked.slice(0, 5).map((r) => ({
563
+ word: r.word,
564
+ finalScore: r.finalScore,
565
+ semanticScore: r.semanticScore,
566
+ freqScore: r.freqScore,
567
+ lmScore: r.lmScore,
568
+ })),
569
+ suggestion: suggestion && suggestion.length > 0 ? suggestion : null,
570
+ };
571
+
572
+ // ── Mode label: COLD / WARM(local) / WARM(BE) ───────────────────────
573
+ const slowLaneVec = getStoredContextVector();
574
+ const isUsingSlowLaneVector =
575
+ slowLaneVec !== null && vectorStore !== null && slowLaneVec.length === vectorStore.dim;
576
+ const modeLabel = !contextVector ? 'COLD' : isUsingSlowLaneVector ? 'WARM(BE)' : 'WARM(local)';
577
+ const modeColor =
578
+ modeLabel === 'WARM(BE)'
579
+ ? 'color: #ff9800; font-weight: bold;'
580
+ : modeLabel === 'WARM(local)'
581
+ ? 'color: #4caf50; font-weight: bold;'
582
+ : 'color: #9e9e9e; font-weight: bold;';
583
+
584
+ // 1. Collapsible group header
585
+ // eslint-disable-next-line no-console
586
+ console.groupCollapsed(
587
+ `%c[Autocomplete] %c${modeLabel} %c| "${currentWord}" ➔ ${suggestionLabel} | ⏱ ${latencyMs.toFixed(1)}ms`,
588
+ 'color: #00b8d9; font-weight: bold;',
589
+ modeColor,
590
+ 'color: inherit; font-weight: normal;',
591
+ );
592
+
593
+ // 2. Context window (what local vector averaging sees)
594
+ // eslint-disable-next-line no-console
595
+ console.log(
596
+ '%cContext Window:',
597
+ 'color: #888; font-style: italic;',
598
+ contextWords.length ? contextWords.join(' ') : '(none)',
599
+ );
600
+
601
+ // 3. Slow-lane status
602
+ const vectorStatus = isUsingSlowLaneVector
603
+ ? `✅ BE semantic vector (dim=${slowLaneVec?.length})`
604
+ : vectorStore
605
+ ? '⚠️ Local vector average (slow-lane not yet returned)'
606
+ : '❌ No vectors (cold)';
607
+ const logitsStatus =
608
+ lmLogits && Object.keys(lmLogits).length > 0
609
+ ? `✅ LM logits active (${Object.keys(lmLogits).length} tokens)`
610
+ : '⏳ No LM logits (slow-lane pending or failed)';
611
+ // eslint-disable-next-line no-console
612
+ console.log(
613
+ '%cSlow Lane:',
614
+ 'color: #888; font-style: italic;',
615
+ vectorStatus,
616
+ '|',
617
+ logitsStatus,
618
+ );
619
+
620
+ // 4. Scoring formula active this prediction
621
+ const formulaLabel =
622
+ lmLogits && Object.keys(lmLogits).length > 0
623
+ ? 'Stage1(×0.6) + LM(×0.4)'
624
+ : 'Stage1 only (no LM logits)';
625
+ // eslint-disable-next-line no-console
626
+ console.log('%cFormula:', 'color: #888; font-style: italic;', formulaLabel);
627
+
628
+ // 5. Grammar filter result (collected inside rankCandidates, logged here)
629
+ if (grammarMeta) {
630
+ // eslint-disable-next-line no-console
631
+ console.log(
632
+ `%c[Grammar] "${grammarMeta.prevWord}" [${grammarMeta.prevTags.join('|')}] → ${grammarMeta.before} candidates → ${grammarMeta.after} after filter`,
633
+ 'color: #4caf50; font-weight: bold;',
634
+ );
635
+ if (grammarMeta.dropped.length > 0) {
636
+ // eslint-disable-next-line no-console
637
+ console.log(
638
+ `%c🚫 Dropped: ${grammarMeta.dropped.join(', ')}`,
639
+ 'color: #f44336; font-style: italic;',
640
+ );
641
+ }
642
+ }
643
+
644
+ // 6. Candidate table
645
+ if (ranked.length > 0) {
646
+ const lmCoverage = ranked.slice(0, 10).filter((r) => r.lmScore > 0.05).length;
647
+ // eslint-disable-next-line no-console
648
+ console.log(
649
+ `%cLM coverage: ${lmCoverage}/${Math.min(ranked.length, 10)} candidates had real logit scores`,
650
+ 'color: #888; font-style: italic;',
651
+ );
652
+
653
+ const tableData = ranked.slice(0, 10).map((r) => {
654
+ let rawLogit: string | number = 'Not in Payload';
655
+ if (lmLogits) {
656
+ const val = lmLogits[r.word.toLowerCase()];
657
+ if (val !== undefined) {
658
+ rawLogit = Number(val.toFixed(5));
659
+ }
660
+ }
661
+
662
+ const original = scoringCandidates.find((sc) => sc.word === r.word);
663
+ let source = 'Unknown';
664
+ if (original) {
665
+ if (original.docFreq === 0 && original.tenantFreq === L3_BASELINE_FREQ) {
666
+ source = '🌍 L3 (Generic)';
667
+ } else if (original.sessionFreq > 0 && original.tenantFreq === 0) {
668
+ source = '👤 L1 (Session Only)';
669
+ } else {
670
+ source = '🏢 L2 (Domain)';
671
+ }
672
+ }
673
+
674
+ return {
675
+ Candidate: r.word,
676
+ Source: source,
677
+ 'Final Score': Number(r.finalScore.toFixed(4)),
678
+ Semantics: Number(r.semanticScore.toFixed(4)),
679
+ Freq: Number(r.freqScore.toFixed(4)),
680
+ 'LM Score': Number(r.lmScore.toFixed(4)),
681
+ 'Raw Logit': rawLogit,
682
+ 'Session Freq': original?.sessionFreq || 0,
683
+ };
684
+ });
685
+
686
+ // eslint-disable-next-line no-console
687
+ console.table(tableData);
688
+ } else {
689
+ // eslint-disable-next-line no-console
690
+ console.log('No candidates found.');
691
+ }
692
+
693
+ // eslint-disable-next-line no-console
694
+ console.groupEnd();
695
+ }
696
+
697
+ return suggestion && suggestion.length > 0 ? suggestion : null;
698
+ };
699
+
700
+ // ─── Data Loading ────────────────────────────────────────────────────────────
701
+
702
+ interface VocabularyJson {
703
+ words: Record<
704
+ string,
705
+ {
706
+ author_freq: number;
707
+ doc_freq: number;
708
+ freq: number;
709
+ }
710
+ >;
711
+ }
712
+
713
+ const DEFAULT_VECTORS_URL = '/data/word-vectors_10k.bin';
714
+
715
+ export const loadVectorsAsync = async (options?: { vectorsUrl?: string }): Promise<void> => {
716
+ if (vectorStore || vectorsLoadStarted) {
717
+ return;
718
+ }
719
+ vectorsLoadStarted = true;
720
+
721
+ const url = options?.vectorsUrl ?? DEFAULT_VECTORS_URL;
722
+ try {
723
+ const res = await fetch(url);
724
+ if (!res.ok) {
725
+ vectorsLoadStarted = false;
726
+ // eslint-disable-next-line no-console
727
+ console.warn(`[text-predictor] Failed to load vectors: ${res.status}`);
728
+ return;
729
+ }
730
+ const buffer = await res.arrayBuffer();
731
+ const float32 = new Float32Array(buffer);
732
+ const wordIndex = wordIndexData as Record<string, number>;
733
+ const nWords = Object.keys(wordIndex).length;
734
+ const dim = float32.length / nWords;
735
+
736
+ vectorStore = { float32, wordIndex, dim };
737
+ ensureDebugModeFromStorage();
738
+ if (debugMode) {
739
+ // eslint-disable-next-line no-console
740
+ console.log('[text-predictor] Vectors loaded:', {
741
+ wordCount: nWords,
742
+ dim,
743
+ sizeBytes: float32.byteLength,
744
+ });
745
+ }
746
+ } catch (e) {
747
+ vectorsLoadStarted = false;
748
+ // eslint-disable-next-line no-console
749
+ console.warn('[text-predictor] Failed to load vectors:', e);
750
+ }
751
+ };
752
+
753
+ export const initVectors = (store: VectorStore): void => {
754
+ vectorStore = store;
755
+ };
756
+
757
+ export const loadDefaultVocabulary = (): void => {
758
+ // 1. Load the Atlassian Domain (L2)
759
+ const data = vocabularyData as VocabularyJson;
760
+ const terms = Object.entries(data.words).map(([word, stats]) => ({
761
+ word,
762
+ freq: stats.freq,
763
+ docFreq: stats.doc_freq,
764
+ authorFreq: stats.author_freq,
765
+ }));
766
+ initVocabulary({ terms });
767
+
768
+ // 2. Load General English (L3)
769
+ const l3Words = l3VocabularyData as string[];
770
+ initL3Vocabulary(l3Words);
771
+ };