@atlaskit/editor-plugin-autocomplete 0.1.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/CHANGELOG.md +16 -0
  2. package/afm-cc/tsconfig.json +2 -1
  3. package/afm-jira/tsconfig.json +2 -1
  4. package/afm-products/tsconfig.json +2 -1
  5. package/build/tsconfig.json +20 -0
  6. package/build/url-module.d.ts +8 -0
  7. package/dist/cjs/autocompletePlugin.js +18 -5
  8. package/dist/cjs/autocompletePluginType.js +5 -1
  9. package/dist/cjs/pm-plugins/autocomplete-plugin.js +368 -0
  10. package/dist/cjs/pm-plugins/ghost-text-decoration.js +39 -0
  11. package/dist/cjs/pm-plugins/scoring-pipeline.js +256 -0
  12. package/dist/cjs/pm-plugins/slow-lane-client.js +199 -0
  13. package/dist/cjs/pm-plugins/text-predictor.js +796 -0
  14. package/dist/es2019/autocompletePlugin.js +20 -6
  15. package/dist/es2019/autocompletePluginType.js +1 -0
  16. package/dist/es2019/pm-plugins/autocomplete-plugin.js +368 -0
  17. package/dist/es2019/pm-plugins/ghost-text-decoration.js +33 -0
  18. package/dist/es2019/pm-plugins/scoring-pipeline.js +221 -0
  19. package/dist/es2019/pm-plugins/slow-lane-client.js +157 -0
  20. package/dist/es2019/pm-plugins/text-predictor.js +631 -0
  21. package/dist/esm/autocompletePlugin.js +18 -5
  22. package/dist/esm/autocompletePluginType.js +1 -0
  23. package/dist/esm/pm-plugins/autocomplete-plugin.js +362 -0
  24. package/dist/esm/pm-plugins/ghost-text-decoration.js +33 -0
  25. package/dist/esm/pm-plugins/scoring-pipeline.js +252 -0
  26. package/dist/esm/pm-plugins/slow-lane-client.js +192 -0
  27. package/dist/esm/pm-plugins/text-predictor.js +793 -0
  28. package/dist/types/autocompletePluginType.d.ts +6 -3
  29. package/dist/types/pm-plugins/autocomplete-plugin.d.ts +36 -0
  30. package/dist/types/pm-plugins/ghost-text-decoration.d.ts +7 -0
  31. package/dist/types/pm-plugins/scoring-pipeline.d.ts +33 -0
  32. package/dist/types/pm-plugins/slow-lane-client.d.ts +46 -0
  33. package/dist/types/pm-plugins/text-predictor.d.ts +90 -0
  34. package/dist/types-ts4.5/autocompletePluginType.d.ts +6 -3
  35. package/dist/types-ts4.5/pm-plugins/autocomplete-plugin.d.ts +36 -0
  36. package/dist/types-ts4.5/pm-plugins/ghost-text-decoration.d.ts +7 -0
  37. package/dist/types-ts4.5/pm-plugins/scoring-pipeline.d.ts +33 -0
  38. package/dist/types-ts4.5/pm-plugins/slow-lane-client.d.ts +46 -0
  39. package/dist/types-ts4.5/pm-plugins/text-predictor.d.ts +90 -0
  40. package/package.json +2 -2
  41. package/src/autocompletePlugin.tsx +25 -5
  42. package/src/autocompletePluginType.ts +14 -3
  43. package/src/pm-plugins/autocomplete-plugin/package.json +15 -0
  44. package/src/pm-plugins/autocomplete-plugin.ts +443 -0
  45. package/src/pm-plugins/data/combined_l2_l3_pos_tags.json +73571 -3
  46. package/src/pm-plugins/data/ghost_pos_tags.json +43 -3
  47. package/src/pm-plugins/data/grammar_transitions_10k.json +46 -3
  48. package/src/pm-plugins/data/l3_vocabulary.json +20002 -3
  49. package/src/pm-plugins/data/vocabulary_10k.json +38794 -3
  50. package/src/pm-plugins/data/word_index_10k.json +7760 -3
  51. package/src/pm-plugins/ghost-text-decoration.ts +44 -0
  52. package/src/pm-plugins/scoring-pipeline.ts +294 -0
  53. package/src/pm-plugins/slow-lane-client/package.json +15 -0
  54. package/src/pm-plugins/slow-lane-client.ts +222 -0
  55. package/src/pm-plugins/text-predictor/package.json +15 -0
  56. package/src/pm-plugins/text-predictor.ts +780 -0
  57. package/tsconfig.app.json +12 -3
  58. package/tsconfig.json +4 -1
@@ -1,6 +1,20 @@
1
- // This file will contain the autocomplete plugin implementation.
2
- // The full implementation will be added in a follow-up PR.
3
-
4
- export const autocompletePlugin = () => ({
5
- name: 'autocomplete'
6
- });
1
+ import { autocompletePluginKey, createAutocompletePlugin } from './pm-plugins/autocomplete-plugin';
2
+ export const autocompletePlugin = ({
3
+ config: options
4
+ }) => {
5
+ return {
6
+ name: 'autocomplete',
7
+ getSharedState(editorState) {
8
+ if (!editorState) {
9
+ return undefined;
10
+ }
11
+ return autocompletePluginKey.getState(editorState);
12
+ },
13
+ pmPlugins() {
14
+ return [{
15
+ name: 'autocomplete',
16
+ plugin: () => createAutocompletePlugin(options)
17
+ }];
18
+ }
19
+ };
20
+ };
@@ -0,0 +1 @@
1
+ export {};
@@ -0,0 +1,368 @@
1
+ // url: prefix is an Atlaspack/Parcel directive that resolves this file as an emitted
2
+ // asset URL (content-hashed). For rspack, this is handled via staticAssetsLoader.
3
+ // eslint-disable-next-line @repo/internal/import/no-unresolved
4
+ import wordVectorsUrl from 'url:./data/word-vectors_10k.bin';
5
+ import { SafePlugin } from '@atlaskit/editor-common/safe-plugin';
6
+ import { keydownHandler } from '@atlaskit/editor-prosemirror/keymap';
7
+ import { PluginKey } from '@atlaskit/editor-prosemirror/state';
8
+ import { DecorationSet } from '@atlaskit/editor-prosemirror/view';
9
+ import { createGhostTextDecorationSet } from './ghost-text-decoration';
10
+ import { createSlowLaneClient, setDefaultSlowLaneClient, isWordBoundary } from './slow-lane-client';
11
+ import { predict, loadDefaultVocabulary, loadVectorsAsync, incrementSessionFreq, ingestDocumentPage } from './text-predictor';
12
+ const SLOW_LANE_ENDPOINT = '/gateway/api/v1/autocomplete/typeahead-encodings';
13
+ export const autocompletePluginKey = new PluginKey('autocomplete');
14
+ const DEBOUNCE_MS = 150;
15
+ const createInitialState = () => ({
16
+ ghostText: '',
17
+ ghostPosition: -1,
18
+ decorationSet: DecorationSet.empty
19
+ });
20
+
21
+ /**
22
+ * Extract text content before the cursor from the current document.
23
+ * Returns the last ~200 characters for context.
24
+ */
25
+ const getTextBeforeCursor = state => {
26
+ const {
27
+ $from
28
+ } = state.selection;
29
+ const maxChars = 200;
30
+
31
+ // 1. Get the perfectly flattened text of the current block up to the cursor
32
+ const blockNode = $from.parent;
33
+ const offsetInBlock = $from.parentOffset;
34
+ const blockText = blockNode.textContent.slice(0, offsetInBlock);
35
+ if (blockText.length >= maxChars) {
36
+ return blockText.slice(-maxChars);
37
+ }
38
+ let fullText = blockText;
39
+
40
+ // 2. Walk backwards through previous blocks
41
+ let depth = $from.depth - 1;
42
+ while (fullText.length < maxChars && depth >= 0) {
43
+ const parentNode = $from.node(depth);
44
+ const indexInParent = $from.index(depth);
45
+ for (let i = indexInParent - 1; i >= 0 && fullText.length < maxChars; i--) {
46
+ const sibling = parentNode.child(i);
47
+ const siblingText = sibling.textContent;
48
+ fullText = siblingText + '\n' + fullText;
49
+ }
50
+ depth--;
51
+ }
52
+ return fullText.slice(-maxChars);
53
+ };
54
+
55
+ /**
56
+ * Set the autocomplete state via a transaction metadata.
57
+ */
58
+ const setAutocompleteMeta = (tr, meta) => {
59
+ return tr.setMeta(autocompletePluginKey, meta);
60
+ };
61
+
62
+ /**
63
+ * Apply a ghost text suggestion to the editor state.
64
+ */
65
+ const showGhostText = (view, text, position) => {
66
+ const {
67
+ state,
68
+ dispatch
69
+ } = view;
70
+ const decorationSet = createGhostTextDecorationSet(state, position, text);
71
+ const tr = setAutocompleteMeta(state.tr, {
72
+ ghostText: text,
73
+ ghostPosition: position,
74
+ decorationSet
75
+ });
76
+ dispatch(tr);
77
+ };
78
+
79
+ /**
80
+ * Clear the current ghost text from the editor.
81
+ */
82
+ const clearGhostText = (state, dispatch) => {
83
+ const pluginState = autocompletePluginKey.getState(state);
84
+ if (!pluginState || !pluginState.ghostText) {
85
+ return false;
86
+ }
87
+ if (dispatch) {
88
+ const tr = setAutocompleteMeta(state.tr, {
89
+ ghostText: '',
90
+ ghostPosition: -1,
91
+ decorationSet: DecorationSet.empty
92
+ });
93
+ dispatch(tr);
94
+ }
95
+ return true;
96
+ };
97
+
98
+ /**
99
+ * Accept the current ghost text suggestion and insert it into the document.
100
+ */
101
+ const acceptGhostText = (state, dispatch) => {
102
+ const pluginState = autocompletePluginKey.getState(state);
103
+ if (!pluginState || !pluginState.ghostText) {
104
+ return false;
105
+ }
106
+ if (dispatch) {
107
+ const {
108
+ ghostText,
109
+ ghostPosition
110
+ } = pluginState;
111
+ let tr = state.tr.insertText(ghostText, ghostPosition);
112
+ tr = setAutocompleteMeta(tr, {
113
+ ghostText: '',
114
+ ghostPosition: -1,
115
+ decorationSet: DecorationSet.empty
116
+ });
117
+ dispatch(tr);
118
+ }
119
+ return true;
120
+ };
121
+
122
+ /**
123
+ * Context provided to the autocomplete plugin on first editor focus.
124
+ * Text fields are selectively ingested to boost word-frequency scoring for
125
+ * predictions, giving words already present in the document/thread an L1
126
+ * priority boost.
127
+ */
128
+
129
+ /**
130
+ * Build the text payload for the slow-lane request by prepending any available
131
+ * comment context ahead of the live document text. This gives the backend
132
+ * model richer context about the thread the user is writing in.
133
+ */
134
+ const buildSlowLaneText = (docText, context) => {
135
+ var _context$siblingComme, _context$siblingComme2, _context$siblingComme3;
136
+ const lines = [];
137
+ if (context !== null && context !== void 0 && context.parentCommentContent) {
138
+ lines.push(`comment: ${context.parentCommentContent}`);
139
+ }
140
+ context === null || context === void 0 ? void 0 : (_context$siblingComme = context.siblingCommentsContents) === null || _context$siblingComme === void 0 ? void 0 : _context$siblingComme.forEach((sibling, index) => {
141
+ lines.push(`reply ${index + 1}: ${sibling}`);
142
+ });
143
+ const nextReplyNumber = ((_context$siblingComme2 = context === null || context === void 0 ? void 0 : (_context$siblingComme3 = context.siblingCommentsContents) === null || _context$siblingComme3 === void 0 ? void 0 : _context$siblingComme3.length) !== null && _context$siblingComme2 !== void 0 ? _context$siblingComme2 : 0) + 1;
144
+ lines.push(`reply ${nextReplyNumber}: ${docText}`);
145
+ return lines.join('\n');
146
+ };
147
+ export const createAutocompletePlugin = options => {
148
+ let debounceTimer = null;
149
+ let hasIngestedPage = false;
150
+ let resolvedContext;
151
+ /**
152
+ * Set after accepting a suggestion so the next doc-change update
153
+ * skips scheduling a new prediction for the just-inserted text.
154
+ * Scoped to the factory so multiple editor instances don't share state.
155
+ */
156
+ let justAccepted = false;
157
+
158
+ /**
159
+ * Stores the text-before-cursor snapshot at the moment the user dismissed
160
+ * a suggestion via Escape. While the context remains identical, we suppress
161
+ * re-showing the same suggestion. Resets to null as soon as the text changes.
162
+ */
163
+ let dismissedContext = null;
164
+ const slowLaneClient = createSlowLaneClient({
165
+ baseUrl: '',
166
+ endpoint: SLOW_LANE_ENDPOINT
167
+ });
168
+ setDefaultSlowLaneClient(slowLaneClient);
169
+
170
+ /**
171
+ * Schedule a prediction after a short debounce.
172
+ * Tier 1 predictions are synchronous (<0.1ms) but we still debounce
173
+ * to avoid unnecessary work on rapid keystrokes.
174
+ */
175
+ const schedulePrediction = view => {
176
+ if (debounceTimer) {
177
+ clearTimeout(debounceTimer);
178
+ }
179
+ debounceTimer = setTimeout(() => {
180
+ const {
181
+ state
182
+ } = view;
183
+ const {
184
+ selection
185
+ } = state;
186
+
187
+ // Only predict for cursor selections (not range selections)
188
+ if (!selection.empty) {
189
+ return;
190
+ }
191
+ const textBefore = getTextBeforeCursor(state);
192
+
193
+ // Suppress re-showing the same suggestion the user just dismissed.
194
+ // Once the text context changes (user types or deletes), this clears automatically.
195
+ if (textBefore === dismissedContext) {
196
+ return;
197
+ }
198
+ dismissedContext = null;
199
+
200
+ // Don't predict if there's not enough context
201
+ if (textBefore.trim().length < 3) {
202
+ return;
203
+ }
204
+
205
+ // Tier 1 prediction is synchronous -- no async needed
206
+ const prediction = predict(textBefore);
207
+ if (prediction && prediction.length > 0) {
208
+ showGhostText(view, prediction, selection.from);
209
+ }
210
+ }, DEBOUNCE_MS);
211
+ };
212
+ const maybeUpdateSessionFrequency = (view, prevState) => {
213
+ const newText = getTextBeforeCursor(view.state);
214
+ const prevText = getTextBeforeCursor(prevState);
215
+ if (newText.length <= prevText.length) {
216
+ return;
217
+ }
218
+ const lastChar = newText[newText.length - 1];
219
+ // eslint-disable-next-line require-unicode-regexp
220
+ if (!/[\s.,;:!?]/.test(lastChar)) {
221
+ return;
222
+ }
223
+
224
+ // Only fire if the previous state did not already end on a boundary,
225
+ // so we don't double-count when multiple boundary chars are inserted.
226
+ const prevLastChar = prevText[prevText.length - 1];
227
+ // eslint-disable-next-line require-unicode-regexp
228
+ if (prevLastChar && /[\s.,;:!?]/.test(prevLastChar)) {
229
+ return;
230
+ }
231
+ const beforeBoundary = newText.slice(0, -1).trimEnd();
232
+ const lastSpaceIdx = beforeBoundary.lastIndexOf(' ');
233
+ const completedWord = beforeBoundary.slice(lastSpaceIdx + 1).toLowerCase();
234
+ if (completedWord.length >= 2) {
235
+ incrementSessionFreq(completedWord);
236
+ }
237
+ };
238
+ return new SafePlugin({
239
+ key: autocompletePluginKey,
240
+ state: {
241
+ init: () => createInitialState(),
242
+ apply: (tr, pluginState) => {
243
+ const meta = tr.getMeta(autocompletePluginKey);
244
+ if (meta) {
245
+ return {
246
+ ...pluginState,
247
+ ...meta
248
+ };
249
+ }
250
+
251
+ // If the document changed, clear the ghost text
252
+ // (new prediction will be scheduled from view.update)
253
+ if (tr.docChanged) {
254
+ return {
255
+ ...pluginState,
256
+ ghostText: '',
257
+ ghostPosition: -1,
258
+ decorationSet: DecorationSet.empty
259
+ };
260
+ }
261
+
262
+ // If selection changed without doc change, clear ghost text
263
+ if (tr.selectionSet && pluginState.ghostText) {
264
+ return {
265
+ ...pluginState,
266
+ ghostText: '',
267
+ ghostPosition: -1,
268
+ decorationSet: DecorationSet.empty
269
+ };
270
+ }
271
+ return pluginState;
272
+ }
273
+ },
274
+ props: {
275
+ decorations: state => {
276
+ var _pluginState$decorati;
277
+ const pluginState = autocompletePluginKey.getState(state);
278
+ return (_pluginState$decorati = pluginState === null || pluginState === void 0 ? void 0 : pluginState.decorationSet) !== null && _pluginState$decorati !== void 0 ? _pluginState$decorati : DecorationSet.empty;
279
+ },
280
+ handleKeyDown: keydownHandler({
281
+ Tab: (state, dispatch) => {
282
+ const accepted = acceptGhostText(state, dispatch);
283
+ if (accepted) justAccepted = true;
284
+ return accepted;
285
+ },
286
+ ArrowRight: (state, dispatch) => {
287
+ const accepted = acceptGhostText(state, dispatch);
288
+ if (accepted) justAccepted = true;
289
+ return accepted;
290
+ },
291
+ Escape: (state, dispatch) => {
292
+ const didClear = clearGhostText(state, dispatch);
293
+ if (didClear) {
294
+ dismissedContext = getTextBeforeCursor(state);
295
+ }
296
+ return didClear;
297
+ }
298
+ }),
299
+ handleDOMEvents: {
300
+ blur: view => {
301
+ const pluginState = autocompletePluginKey.getState(view.state);
302
+ if (pluginState !== null && pluginState !== void 0 && pluginState.ghostText) {
303
+ clearGhostText(view.state, view.dispatch);
304
+ }
305
+ return false;
306
+ },
307
+ focus: () => {
308
+ loadDefaultVocabulary();
309
+ loadVectorsAsync({
310
+ vectorsUrl: wordVectorsUrl
311
+ }).catch(() => {});
312
+ if (!hasIngestedPage) {
313
+ hasIngestedPage = true;
314
+ if (options !== null && options !== void 0 && options.getContext) {
315
+ options.getContext().then(context => {
316
+ var _context$siblingComme4;
317
+ if (!context) {
318
+ return;
319
+ }
320
+ resolvedContext = context;
321
+ if (context.fullPageContent) {
322
+ ingestDocumentPage(context.fullPageContent);
323
+ }
324
+ if (context.parentCommentContent) {
325
+ ingestDocumentPage(context.parentCommentContent);
326
+ }
327
+ (_context$siblingComme4 = context.siblingCommentsContents) === null || _context$siblingComme4 === void 0 ? void 0 : _context$siblingComme4.forEach(ingestDocumentPage);
328
+ }).catch(() => {});
329
+ }
330
+ }
331
+ return false;
332
+ }
333
+ }
334
+ },
335
+ view: () => ({
336
+ update: (view, prevState) => {
337
+ if (!prevState.doc.eq(view.state.doc)) {
338
+ if (justAccepted) {
339
+ justAccepted = false;
340
+
341
+ // ✨ THE FIX: Memorize the text state right after acceptance.
342
+ // Any follow-up transactions will hit the 'dismissedContext'
343
+ // block and abort until the user actually types a new character!
344
+ dismissedContext = getTextBeforeCursor(view.state);
345
+
346
+ // Also clear any pending debounce timers from before the acceptance
347
+ if (debounceTimer) {
348
+ clearTimeout(debounceTimer);
349
+ }
350
+ return;
351
+ }
352
+ maybeUpdateSessionFrequency(view, prevState);
353
+ const textBefore = getTextBeforeCursor(view.state);
354
+ if (isWordBoundary(textBefore)) {
355
+ slowLaneClient.updateContext(buildSlowLaneText(view.state.doc.textContent, resolvedContext));
356
+ }
357
+ schedulePrediction(view);
358
+ }
359
+ },
360
+ destroy: () => {
361
+ if (debounceTimer) {
362
+ clearTimeout(debounceTimer);
363
+ }
364
+ setDefaultSlowLaneClient(null);
365
+ }
366
+ })
367
+ });
368
+ };
@@ -0,0 +1,33 @@
1
+ import { Decoration, DecorationSet } from '@atlaskit/editor-prosemirror/view';
2
+ const GHOST_TEXT_CLASS = 'autocomplete-ghost-text';
3
+
4
+ /**
5
+ * Creates a DecorationSet containing a ghost text widget at the given position.
6
+ * The ghost text is rendered as a styled <span> that appears after the cursor.
7
+ */
8
+ export const createGhostTextDecorationSet = (state, position, text) => {
9
+ if (!text) {
10
+ return DecorationSet.empty;
11
+ }
12
+ const decoration = Decoration.widget(position, () => {
13
+ const container = document.createElement('span');
14
+ container.className = GHOST_TEXT_CLASS;
15
+ container.setAttribute('data-autocomplete-ghost', 'true');
16
+ container.style.color = '#999';
17
+ container.style.opacity = '0.6';
18
+ container.style.pointerEvents = 'none';
19
+ container.style.userSelect = 'none';
20
+ container.style.fontStyle = 'italic';
21
+ // U+200B (Zero Width Space) gives the browser a line-break opportunity
22
+ // immediately before the ghost text. This ensures the typed text before
23
+ // the span is never pushed to the next line by the ghost text's width —
24
+ // only the ghost text itself will wrap if it doesn't fit.
25
+ container.textContent = '\u200b' + text;
26
+ return container;
27
+ }, {
28
+ side: 1,
29
+ // Render after content at this position
30
+ key: 'autocomplete-ghost-text'
31
+ });
32
+ return DecorationSet.create(state.doc, [decoration]);
33
+ };
@@ -0,0 +1,221 @@
1
+ /**
2
+ * Scoring Pipeline: Stage 1 (Semantic + Frequency), Grammar Filter, Stage 2 (LM Re-ranking).
3
+ *
4
+ * Operates synchronously on pre-loaded data. Each stage gracefully degrades
5
+ * when its required data isn't available (cold → warm → full warm).
6
+ */
7
+
8
+ import posTagsData from './data/combined_l2_l3_pos_tags.json';
9
+ import ghostPosTagsData from './data/ghost_pos_tags.json';
10
+ import grammarTransitionsData from './data/grammar_transitions_10k.json';
11
+
12
+ // ─── Types ──────────────────────────────────────────────────
13
+
14
+ /** Metadata returned by the grammar filter for debug logging in the caller. */
15
+
16
+ // ─── Scoring Constants ──────────────────────────────────────
17
+
18
+ const ALPHA = 0.5;
19
+ const BETA = 0.5;
20
+ const NEUTRAL_SCORE = 0.5;
21
+ const STAGE1_WEIGHT = 0.6;
22
+ const STAGE2_WEIGHT = 0.4;
23
+ const MIN_STAGE1_SCORE = 0.35;
24
+ const L1_SESSION_CAP = 1.2;
25
+
26
+ // ─── Grammar Data (loaded once on import) ───────────────────
27
+
28
+ const posTags = new Map(Object.entries(posTagsData));
29
+ const grammarTransitions = grammarTransitionsData;
30
+
31
+ /**
32
+ * Precomputed map from each POS tag to the set of allowed next POS tags.
33
+ * Built once at module load from grammarTransitions so applyGrammarFilter
34
+ * never re-iterates the transition rules per call.
35
+ */
36
+ const precomputedAllowedByPos = new Map(Object.entries(grammarTransitions.transitions).map(([pos, rule]) => [pos, new Set(rule.allowed)]));
37
+
38
+ // ─── Math ───────────────────────────────────────────────────
39
+
40
+ function cosineSimilarity(a, b) {
41
+ let dot = 0;
42
+ let normA = 0;
43
+ let normB = 0;
44
+ for (let i = 0; i < a.length; i++) {
45
+ dot += a[i] * b[i];
46
+ normA += a[i] * a[i];
47
+ normB += b[i] * b[i];
48
+ }
49
+ const dNormA = Math.sqrt(normA);
50
+ const dNormB = Math.sqrt(normB);
51
+ if (dNormA === 0 || dNormB === 0) {
52
+ return NEUTRAL_SCORE;
53
+ }
54
+ return (1 + dot / (dNormA * dNormB)) / 2;
55
+ }
56
+
57
+ // ─── Stage 1: Semantic + Frequency ──────────────────────────
58
+
59
+ function scoreStage1(candidate, contextVector, getWordVector, maxTenantFreq) {
60
+ // 1. Calculate Base Global Score (Normalized Log)
61
+ const maxPossibleLog = Math.log10(maxTenantFreq + 1);
62
+
63
+ // Diversity Adjustment
64
+ const diversityRaw = (Math.log10(candidate.tenantFreq + 1) * 0.50 + Math.log10(candidate.docFreq + 1) * 0.25 + Math.log10(candidate.authorFreq + 1) * 0.25) / maxPossibleLog;
65
+ const sessionMultiplier = candidate.sessionFreq > 0 ? 1 + Math.log10(candidate.sessionFreq + 1) * 2.5 : 1;
66
+
67
+ // Apply multiplier; capped at L1_SESSION_CAP (default 1.2) to prevent excessive over-indexing
68
+ const freqScore = Math.min(diversityRaw * sessionMultiplier, L1_SESSION_CAP);
69
+
70
+ // 3. Semantic Scoring
71
+ let semanticScore = NEUTRAL_SCORE;
72
+ if (contextVector) {
73
+ const wordVec = getWordVector(candidate.word);
74
+ semanticScore = wordVec ? cosineSimilarity(contextVector, wordVec) : NEUTRAL_SCORE;
75
+ }
76
+ return {
77
+ semanticScore,
78
+ freqScore,
79
+ stage1Score: ALPHA * semanticScore + BETA * freqScore
80
+ };
81
+ }
82
+
83
+ // ─── Grammar Filter ─────────────────────────────────────────
84
+
85
+ /**
86
+ * GHOST POS DICTIONARY
87
+ * A hardcoded mapping of common structural English words that were stripped
88
+ * from the main domain vocabulary. This allows the grammar filter to understand
89
+ * context without suggesting these words to the user.
90
+ */
91
+ const ghostPosTags = ghostPosTagsData;
92
+ function applyGrammarFilter(candidates, previousWord) {
93
+ if (!previousWord) return {
94
+ filtered: candidates,
95
+ grammarMeta: null
96
+ };
97
+ const lowerPrev = previousWord.toLowerCase();
98
+ const prevTags = ghostPosTags[lowerPrev] || posTags.get(lowerPrev);
99
+ if (!prevTags || prevTags.length === 0) {
100
+ return {
101
+ filtered: candidates,
102
+ grammarMeta: null
103
+ };
104
+ }
105
+ let allowedNextTags;
106
+ if (prevTags.length === 1) {
107
+ var _precomputedAllowedBy;
108
+ // Common case: single POS tag — reuse the precomputed Set directly (no allocation)
109
+ allowedNextTags = (_precomputedAllowedBy = precomputedAllowedByPos.get(prevTags[0])) !== null && _precomputedAllowedBy !== void 0 ? _precomputedAllowedBy : new Set();
110
+ } else {
111
+ allowedNextTags = new Set();
112
+ for (const pt of prevTags) {
113
+ const allowed = precomputedAllowedByPos.get(pt);
114
+ if (allowed) allowed.forEach(tag => allowedNextTags.add(tag));
115
+ }
116
+ }
117
+ const filtered = [];
118
+ const dropped = [];
119
+ for (const entry of candidates) {
120
+ const candidateTags = posTags.get(entry.candidate.word.toLowerCase());
121
+
122
+ // If candidate has no tags (unknown word), let it pass to be safe
123
+ if (!candidateTags || candidateTags.length === 0) {
124
+ filtered.push(entry);
125
+ continue;
126
+ }
127
+ if (candidateTags.some(ct => allowedNextTags.has(ct))) {
128
+ filtered.push(entry);
129
+ } else {
130
+ dropped.push(entry.candidate.word);
131
+ }
132
+ }
133
+ const finalFiltered = filtered.length > 0 ? filtered : candidates;
134
+ return {
135
+ filtered: finalFiltered,
136
+ grammarMeta: {
137
+ prevWord: lowerPrev,
138
+ prevTags,
139
+ before: candidates.length,
140
+ after: finalFiltered.length,
141
+ dropped: filtered.length > 0 ? dropped : []
142
+ }
143
+ };
144
+ }
145
+
146
+ // ─── Stage 2: LM Re-ranking ────────────────────────────────
147
+ function getLmScore(word, lmLogits) {
148
+ if (!lmLogits) return 0;
149
+
150
+ // Look up the word directly! No more tokens.
151
+ const val = lmLogits[word.toLowerCase()];
152
+ if (typeof val === 'number') {
153
+ return val;
154
+ }
155
+ return 0;
156
+ }
157
+
158
+ // ─── Public API ─────────────────────────────────────────────
159
+
160
+ export function rankCandidates(candidates, contextVector, getWordVector, lmLogits, maxTenantFreq, previousWord) {
161
+ // Stage 1
162
+ const stage1Results = candidates.map(candidate => {
163
+ const {
164
+ semanticScore,
165
+ freqScore,
166
+ stage1Score
167
+ } = scoreStage1(candidate, contextVector, getWordVector, maxTenantFreq);
168
+ return {
169
+ candidate,
170
+ semanticScore,
171
+ freqScore,
172
+ stage1Score
173
+ };
174
+ });
175
+ const stage1Survivors = stage1Results.filter(entry => entry.stage1Score >= MIN_STAGE1_SCORE);
176
+
177
+ // Grammar Filter
178
+ const {
179
+ filtered,
180
+ grammarMeta
181
+ } = applyGrammarFilter(stage1Survivors, previousWord);
182
+
183
+ // Stage 2 + final assembly
184
+ let lmMax = 0;
185
+ if (lmLogits && Object.keys(lmLogits).length > 0) {
186
+ const values = Object.values(lmLogits);
187
+ lmMax = Math.max(...values);
188
+ }
189
+ const scored = filtered.map(entry => {
190
+ let lmScore = 0;
191
+ let finalScore = entry.stage1Score;
192
+ if (lmLogits && Object.keys(lmLogits).length > 0) {
193
+ const rawLm = getLmScore(entry.candidate.word, lmLogits);
194
+ if (rawLm !== 0) {
195
+ // The word was in the top_k! Score it normally.
196
+ const logitDiff = Math.log(rawLm) - Math.log(lmMax);
197
+ lmScore = Math.exp(logitDiff);
198
+ } else {
199
+ lmScore = 0.05;
200
+ }
201
+ finalScore = STAGE1_WEIGHT * entry.stage1Score + STAGE2_WEIGHT * lmScore;
202
+ }
203
+ return {
204
+ word: entry.candidate.word,
205
+ freqScore: entry.freqScore,
206
+ semanticScore: entry.semanticScore,
207
+ lmScore,
208
+ finalScore
209
+ };
210
+ });
211
+ scored.sort((a, b) => {
212
+ if (b.finalScore !== a.finalScore) {
213
+ return b.finalScore - a.finalScore;
214
+ }
215
+ return a.word.length - b.word.length;
216
+ });
217
+ return {
218
+ candidates: scored,
219
+ grammarMeta
220
+ };
221
+ }