@atlaskit/editor-plugin-autocomplete 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/CHANGELOG.md +8 -0
  2. package/afm-cc/tsconfig.json +5 -1
  3. package/afm-jira/tsconfig.json +5 -1
  4. package/afm-products/tsconfig.json +5 -1
  5. package/build/tsconfig.json +24 -0
  6. package/dist/cjs/autocompletePlugin.js +18 -5
  7. package/dist/cjs/autocompletePluginType.js +5 -1
  8. package/dist/cjs/pm-plugins/autocomplete-plugin.js +361 -0
  9. package/dist/cjs/pm-plugins/ghost-text-decoration.js +39 -0
  10. package/dist/cjs/pm-plugins/scoring-pipeline.js +258 -0
  11. package/dist/cjs/pm-plugins/slow-lane-client.js +197 -0
  12. package/dist/cjs/pm-plugins/text-predictor.js +786 -0
  13. package/dist/es2019/autocompletePlugin.js +20 -6
  14. package/dist/es2019/autocompletePluginType.js +1 -0
  15. package/dist/es2019/pm-plugins/autocomplete-plugin.js +360 -0
  16. package/dist/es2019/pm-plugins/ghost-text-decoration.js +33 -0
  17. package/dist/es2019/pm-plugins/scoring-pipeline.js +224 -0
  18. package/dist/es2019/pm-plugins/slow-lane-client.js +154 -0
  19. package/dist/es2019/pm-plugins/text-predictor.js +624 -0
  20. package/dist/esm/autocompletePlugin.js +18 -5
  21. package/dist/esm/autocompletePluginType.js +1 -0
  22. package/dist/esm/pm-plugins/autocomplete-plugin.js +354 -0
  23. package/dist/esm/pm-plugins/ghost-text-decoration.js +33 -0
  24. package/dist/esm/pm-plugins/scoring-pipeline.js +255 -0
  25. package/dist/esm/pm-plugins/slow-lane-client.js +190 -0
  26. package/dist/esm/pm-plugins/text-predictor.js +783 -0
  27. package/dist/types/autocompletePluginType.d.ts +6 -3
  28. package/dist/types/pm-plugins/autocomplete-plugin.d.ts +36 -0
  29. package/dist/types/pm-plugins/ghost-text-decoration.d.ts +7 -0
  30. package/dist/types/pm-plugins/scoring-pipeline.d.ts +33 -0
  31. package/dist/types/pm-plugins/slow-lane-client.d.ts +46 -0
  32. package/dist/types/pm-plugins/text-predictor.d.ts +90 -0
  33. package/dist/types-ts4.5/autocompletePluginType.d.ts +6 -3
  34. package/dist/types-ts4.5/pm-plugins/autocomplete-plugin.d.ts +36 -0
  35. package/dist/types-ts4.5/pm-plugins/ghost-text-decoration.d.ts +7 -0
  36. package/dist/types-ts4.5/pm-plugins/scoring-pipeline.d.ts +33 -0
  37. package/dist/types-ts4.5/pm-plugins/slow-lane-client.d.ts +46 -0
  38. package/dist/types-ts4.5/pm-plugins/text-predictor.d.ts +90 -0
  39. package/package.json +2 -2
  40. package/src/autocompletePlugin.tsx +25 -5
  41. package/src/autocompletePluginType.ts +14 -3
  42. package/src/pm-plugins/autocomplete-plugin/package.json +15 -0
  43. package/src/pm-plugins/autocomplete-plugin.ts +429 -0
  44. package/src/pm-plugins/ghost-text-decoration.ts +44 -0
  45. package/src/pm-plugins/scoring-pipeline.ts +297 -0
  46. package/src/pm-plugins/slow-lane-client/package.json +15 -0
  47. package/src/pm-plugins/slow-lane-client.ts +220 -0
  48. package/src/pm-plugins/text-predictor/package.json +15 -0
  49. package/src/pm-plugins/text-predictor.ts +771 -0
  50. package/src/pm-plugins/typings.d.ts +12 -0
  51. package/tsconfig.app.json +10 -2
  52. package/tsconfig.json +3 -1
@@ -1,6 +1,20 @@
1
- // This file will contain the autocomplete plugin implementation.
2
- // The full implementation will be added in a follow-up PR.
3
-
4
- export const autocompletePlugin = () => ({
5
- name: 'autocomplete'
6
- });
1
+ import { autocompletePluginKey, createAutocompletePlugin } from './pm-plugins/autocomplete-plugin';
2
+ export const autocompletePlugin = ({
3
+ config: options
4
+ }) => {
5
+ return {
6
+ name: 'autocomplete',
7
+ getSharedState(editorState) {
8
+ if (!editorState) {
9
+ return undefined;
10
+ }
11
+ return autocompletePluginKey.getState(editorState);
12
+ },
13
+ pmPlugins() {
14
+ return [{
15
+ name: 'autocomplete',
16
+ plugin: () => createAutocompletePlugin(options)
17
+ }];
18
+ }
19
+ };
20
+ };
@@ -0,0 +1 @@
1
+ export {};
@@ -0,0 +1,360 @@
1
+ import { SafePlugin } from '@atlaskit/editor-common/safe-plugin';
2
+ import { keydownHandler } from '@atlaskit/editor-prosemirror/keymap';
3
+ import { PluginKey } from '@atlaskit/editor-prosemirror/state';
4
+ import { DecorationSet } from '@atlaskit/editor-prosemirror/view';
5
+ import { createGhostTextDecorationSet } from './ghost-text-decoration';
6
+ import { createSlowLaneClient, setDefaultSlowLaneClient, isWordBoundary } from './slow-lane-client';
7
+ import { predict, loadDefaultVocabulary, loadVectorsAsync, incrementSessionFreq, ingestDocumentPage } from './text-predictor';
8
+ const SLOW_LANE_ENDPOINT = '/gateway/api/v1/autocomplete/typeahead-encodings';
9
+ export const autocompletePluginKey = new PluginKey('autocomplete');
10
+ const DEBOUNCE_MS = 150;
11
+ const createInitialState = () => ({
12
+ ghostText: '',
13
+ ghostPosition: -1,
14
+ decorationSet: DecorationSet.empty
15
+ });
16
+
17
+ /**
18
+ * Extract text content before the cursor from the current document.
19
+ * Returns the last ~200 characters for context.
20
+ */
21
+ const getTextBeforeCursor = state => {
22
+ const {
23
+ $from
24
+ } = state.selection;
25
+ const maxChars = 200;
26
+
27
+ // 1. Get the perfectly flattened text of the current block up to the cursor
28
+ const blockNode = $from.parent;
29
+ const offsetInBlock = $from.parentOffset;
30
+ const blockText = blockNode.textContent.slice(0, offsetInBlock);
31
+ if (blockText.length >= maxChars) {
32
+ return blockText.slice(-maxChars);
33
+ }
34
+ let fullText = blockText;
35
+
36
+ // 2. Walk backwards through previous blocks
37
+ let depth = $from.depth - 1;
38
+ while (fullText.length < maxChars && depth >= 0) {
39
+ const parentNode = $from.node(depth);
40
+ const indexInParent = $from.index(depth);
41
+ for (let i = indexInParent - 1; i >= 0 && fullText.length < maxChars; i--) {
42
+ const sibling = parentNode.child(i);
43
+ const siblingText = sibling.textContent;
44
+ fullText = siblingText + '\n' + fullText;
45
+ }
46
+ depth--;
47
+ }
48
+ return fullText.slice(-maxChars);
49
+ };
50
+
51
+ /**
52
+ * Set the autocomplete state via a transaction metadata.
53
+ */
54
+ const setAutocompleteMeta = (tr, meta) => {
55
+ return tr.setMeta(autocompletePluginKey, meta);
56
+ };
57
+
58
+ /**
59
+ * Apply a ghost text suggestion to the editor state.
60
+ */
61
+ const showGhostText = (view, text, position) => {
62
+ const {
63
+ state,
64
+ dispatch
65
+ } = view;
66
+ const decorationSet = createGhostTextDecorationSet(state, position, text);
67
+ const tr = setAutocompleteMeta(state.tr, {
68
+ ghostText: text,
69
+ ghostPosition: position,
70
+ decorationSet
71
+ });
72
+ dispatch(tr);
73
+ };
74
+
75
+ /**
76
+ * Clear the current ghost text from the editor.
77
+ */
78
+ const clearGhostText = (state, dispatch) => {
79
+ const pluginState = autocompletePluginKey.getState(state);
80
+ if (!pluginState || !pluginState.ghostText) {
81
+ return false;
82
+ }
83
+ if (dispatch) {
84
+ const tr = setAutocompleteMeta(state.tr, {
85
+ ghostText: '',
86
+ ghostPosition: -1,
87
+ decorationSet: DecorationSet.empty
88
+ });
89
+ dispatch(tr);
90
+ }
91
+ return true;
92
+ };
93
+
94
+ /**
95
+ * Accept the current ghost text suggestion and insert it into the document.
96
+ */
97
+ const acceptGhostText = (state, dispatch) => {
98
+ const pluginState = autocompletePluginKey.getState(state);
99
+ if (!pluginState || !pluginState.ghostText) {
100
+ return false;
101
+ }
102
+ if (dispatch) {
103
+ const {
104
+ ghostText,
105
+ ghostPosition
106
+ } = pluginState;
107
+ let tr = state.tr.insertText(ghostText, ghostPosition);
108
+ tr = setAutocompleteMeta(tr, {
109
+ ghostText: '',
110
+ ghostPosition: -1,
111
+ decorationSet: DecorationSet.empty
112
+ });
113
+ dispatch(tr);
114
+ }
115
+ return true;
116
+ };
117
+
118
+ /**
119
+ * Context provided to the autocomplete plugin on first editor focus.
120
+ * Text fields are selectively ingested to boost word-frequency scoring for
121
+ * predictions, giving words already present in the document/thread an L1
122
+ * priority boost.
123
+ */
124
+
125
+ /**
126
+ * Build the text payload for the slow-lane request by prepending any available
127
+ * comment context ahead of the live document text. This gives the backend
128
+ * model richer context about the thread the user is writing in.
129
+ */
130
+ const buildSlowLaneText = (docText, context) => {
131
+ var _context$siblingComme, _context$siblingComme2, _context$siblingComme3;
132
+ const lines = [];
133
+ if (context !== null && context !== void 0 && context.parentCommentContent) {
134
+ lines.push(`comment: ${context.parentCommentContent}`);
135
+ }
136
+ context === null || context === void 0 ? void 0 : (_context$siblingComme = context.siblingCommentsContents) === null || _context$siblingComme === void 0 ? void 0 : _context$siblingComme.forEach((sibling, index) => {
137
+ lines.push(`reply ${index + 1}: ${sibling}`);
138
+ });
139
+ const nextReplyNumber = ((_context$siblingComme2 = context === null || context === void 0 ? void 0 : (_context$siblingComme3 = context.siblingCommentsContents) === null || _context$siblingComme3 === void 0 ? void 0 : _context$siblingComme3.length) !== null && _context$siblingComme2 !== void 0 ? _context$siblingComme2 : 0) + 1;
140
+ lines.push(`reply ${nextReplyNumber}: ${docText}`);
141
+ return lines.join('\n');
142
+ };
143
+ export const createAutocompletePlugin = options => {
144
+ let debounceTimer = null;
145
+ let hasIngestedPage = false;
146
+ let resolvedContext;
147
+ /**
148
+ * Set after accepting a suggestion so the next doc-change update
149
+ * skips scheduling a new prediction for the just-inserted text.
150
+ * Scoped to the factory so multiple editor instances don't share state.
151
+ */
152
+ let justAccepted = false;
153
+
154
+ /**
155
+ * Stores the text-before-cursor snapshot at the moment the user dismissed
156
+ * a suggestion via Escape. While the context remains identical, we suppress
157
+ * re-showing the same suggestion. Resets to null as soon as the text changes.
158
+ */
159
+ let dismissedContext = null;
160
+ const slowLaneClient = createSlowLaneClient({
161
+ baseUrl: '',
162
+ endpoint: SLOW_LANE_ENDPOINT
163
+ });
164
+ setDefaultSlowLaneClient(slowLaneClient);
165
+
166
+ /**
167
+ * Schedule a prediction after a short debounce.
168
+ * Tier 1 predictions are synchronous (<0.1ms) but we still debounce
169
+ * to avoid unnecessary work on rapid keystrokes.
170
+ */
171
+ const schedulePrediction = view => {
172
+ if (debounceTimer) {
173
+ clearTimeout(debounceTimer);
174
+ }
175
+ debounceTimer = setTimeout(() => {
176
+ const {
177
+ state
178
+ } = view;
179
+ const {
180
+ selection
181
+ } = state;
182
+
183
+ // Only predict for cursor selections (not range selections)
184
+ if (!selection.empty) {
185
+ return;
186
+ }
187
+ const textBefore = getTextBeforeCursor(state);
188
+
189
+ // Suppress re-showing the same suggestion the user just dismissed.
190
+ // Once the text context changes (user types or deletes), this clears automatically.
191
+ if (textBefore === dismissedContext) {
192
+ return;
193
+ }
194
+ dismissedContext = null;
195
+
196
+ // Don't predict if there's not enough context
197
+ if (textBefore.trim().length < 3) {
198
+ return;
199
+ }
200
+
201
+ // Tier 1 prediction is synchronous -- no async needed
202
+ const prediction = predict(textBefore);
203
+ if (prediction && prediction.length > 0) {
204
+ showGhostText(view, prediction, selection.from);
205
+ }
206
+ }, DEBOUNCE_MS);
207
+ };
208
+ const maybeUpdateSessionFrequency = (view, prevState) => {
209
+ const newText = getTextBeforeCursor(view.state);
210
+ const prevText = getTextBeforeCursor(prevState);
211
+ if (newText.length <= prevText.length) {
212
+ return;
213
+ }
214
+ const lastChar = newText[newText.length - 1];
215
+ if (!/[\s.,;:!?]/u.test(lastChar)) {
216
+ return;
217
+ }
218
+
219
+ // Only fire if the previous state did not already end on a boundary,
220
+ // so we don't double-count when multiple boundary chars are inserted.
221
+ const prevLastChar = prevText[prevText.length - 1];
222
+ if (prevLastChar && /[\s.,;:!?]/u.test(prevLastChar)) {
223
+ return;
224
+ }
225
+ const beforeBoundary = newText.slice(0, -1).trimEnd();
226
+ const lastSpaceIdx = beforeBoundary.lastIndexOf(' ');
227
+ const completedWord = beforeBoundary.slice(lastSpaceIdx + 1).toLowerCase();
228
+ if (completedWord.length >= 2) {
229
+ incrementSessionFreq(completedWord);
230
+ }
231
+ };
232
+ return new SafePlugin({
233
+ key: autocompletePluginKey,
234
+ state: {
235
+ init: () => createInitialState(),
236
+ apply: (tr, pluginState) => {
237
+ const meta = tr.getMeta(autocompletePluginKey);
238
+ if (meta) {
239
+ return {
240
+ ...pluginState,
241
+ ...meta
242
+ };
243
+ }
244
+
245
+ // If the document changed, clear the ghost text
246
+ // (new prediction will be scheduled from view.update)
247
+ if (tr.docChanged) {
248
+ return {
249
+ ...pluginState,
250
+ ghostText: '',
251
+ ghostPosition: -1,
252
+ decorationSet: DecorationSet.empty
253
+ };
254
+ }
255
+
256
+ // If selection changed without doc change, clear ghost text
257
+ if (tr.selectionSet && pluginState.ghostText) {
258
+ return {
259
+ ...pluginState,
260
+ ghostText: '',
261
+ ghostPosition: -1,
262
+ decorationSet: DecorationSet.empty
263
+ };
264
+ }
265
+ return pluginState;
266
+ }
267
+ },
268
+ props: {
269
+ decorations: state => {
270
+ var _pluginState$decorati;
271
+ const pluginState = autocompletePluginKey.getState(state);
272
+ return (_pluginState$decorati = pluginState === null || pluginState === void 0 ? void 0 : pluginState.decorationSet) !== null && _pluginState$decorati !== void 0 ? _pluginState$decorati : DecorationSet.empty;
273
+ },
274
+ handleKeyDown: keydownHandler({
275
+ Tab: (state, dispatch) => {
276
+ const accepted = acceptGhostText(state, dispatch);
277
+ if (accepted) justAccepted = true;
278
+ return accepted;
279
+ },
280
+ ArrowRight: (state, dispatch) => {
281
+ const accepted = acceptGhostText(state, dispatch);
282
+ if (accepted) justAccepted = true;
283
+ return accepted;
284
+ },
285
+ Escape: (state, dispatch) => {
286
+ const didClear = clearGhostText(state, dispatch);
287
+ if (didClear) {
288
+ dismissedContext = getTextBeforeCursor(state);
289
+ }
290
+ return didClear;
291
+ }
292
+ }),
293
+ handleDOMEvents: {
294
+ blur: view => {
295
+ const pluginState = autocompletePluginKey.getState(view.state);
296
+ if (pluginState !== null && pluginState !== void 0 && pluginState.ghostText) {
297
+ clearGhostText(view.state, view.dispatch);
298
+ }
299
+ return false;
300
+ },
301
+ focus: () => {
302
+ loadDefaultVocabulary();
303
+ loadVectorsAsync().catch(() => {});
304
+ if (!hasIngestedPage) {
305
+ hasIngestedPage = true;
306
+ if (options !== null && options !== void 0 && options.getContext) {
307
+ options.getContext().then(context => {
308
+ var _context$siblingComme4;
309
+ if (!context) {
310
+ return;
311
+ }
312
+ resolvedContext = context;
313
+ if (context.fullPageContent) {
314
+ ingestDocumentPage(context.fullPageContent);
315
+ }
316
+ if (context.parentCommentContent) {
317
+ ingestDocumentPage(context.parentCommentContent);
318
+ }
319
+ (_context$siblingComme4 = context.siblingCommentsContents) === null || _context$siblingComme4 === void 0 ? void 0 : _context$siblingComme4.forEach(ingestDocumentPage);
320
+ }).catch(() => {});
321
+ }
322
+ }
323
+ return false;
324
+ }
325
+ }
326
+ },
327
+ view: () => ({
328
+ update: (view, prevState) => {
329
+ if (!prevState.doc.eq(view.state.doc)) {
330
+ if (justAccepted) {
331
+ justAccepted = false;
332
+
333
+ // ✨ THE FIX: Memorize the text state right after acceptance.
334
+ // Any follow-up transactions will hit the 'dismissedContext'
335
+ // block and abort until the user actually types a new character!
336
+ dismissedContext = getTextBeforeCursor(view.state);
337
+
338
+ // Also clear any pending debounce timers from before the acceptance
339
+ if (debounceTimer) {
340
+ clearTimeout(debounceTimer);
341
+ }
342
+ return;
343
+ }
344
+ maybeUpdateSessionFrequency(view, prevState);
345
+ const textBefore = getTextBeforeCursor(view.state);
346
+ if (isWordBoundary(textBefore)) {
347
+ slowLaneClient.updateContext(buildSlowLaneText(view.state.doc.textContent, resolvedContext));
348
+ }
349
+ schedulePrediction(view);
350
+ }
351
+ },
352
+ destroy: () => {
353
+ if (debounceTimer) {
354
+ clearTimeout(debounceTimer);
355
+ }
356
+ setDefaultSlowLaneClient(null);
357
+ }
358
+ })
359
+ });
360
+ };
@@ -0,0 +1,33 @@
1
+ import { Decoration, DecorationSet } from '@atlaskit/editor-prosemirror/view';
2
+ const GHOST_TEXT_CLASS = 'autocomplete-ghost-text';
3
+
4
+ /**
5
+ * Creates a DecorationSet containing a ghost text widget at the given position.
6
+ * The ghost text is rendered as a styled <span> that appears after the cursor.
7
+ */
8
+ export const createGhostTextDecorationSet = (state, position, text) => {
9
+ if (!text) {
10
+ return DecorationSet.empty;
11
+ }
12
+ const decoration = Decoration.widget(position, () => {
13
+ const container = document.createElement('span');
14
+ container.className = GHOST_TEXT_CLASS;
15
+ container.setAttribute('data-autocomplete-ghost', 'true');
16
+ container.style.color = '#999';
17
+ container.style.opacity = '0.6';
18
+ container.style.pointerEvents = 'none';
19
+ container.style.userSelect = 'none';
20
+ container.style.fontStyle = 'italic';
21
+ // U+200B (Zero Width Space) gives the browser a line-break opportunity
22
+ // immediately before the ghost text. This ensures the typed text before
23
+ // the span is never pushed to the next line by the ghost text's width —
24
+ // only the ghost text itself will wrap if it doesn't fit.
25
+ container.textContent = '\u200b' + text;
26
+ return container;
27
+ }, {
28
+ side: 1,
29
+ // Render after content at this position
30
+ key: 'autocomplete-ghost-text'
31
+ });
32
+ return DecorationSet.create(state.doc, [decoration]);
33
+ };
@@ -0,0 +1,224 @@
1
+ /**
2
+ * Scoring Pipeline: Stage 1 (Semantic + Frequency), Grammar Filter, Stage 2 (LM Re-ranking).
3
+ *
4
+ * Operates synchronously on pre-loaded data. Each stage gracefully degrades
5
+ * when its required data isn't available (cold → warm → full warm).
6
+ */
7
+
8
+ // resolveJsonModule is disabled for this package (see tsconfig.json) to prevent
9
+ // TypeScript from parsing LFS pointer files during CI typecheck. JSON imports
10
+ // are typed via the '*.json' declaration in typings.d.ts.
11
+ import posTagsData from './data/combined_l2_l3_pos_tags.json';
12
+ import ghostPosTagsData from './data/ghost_pos_tags.json';
13
+ import grammarTransitionsData from './data/grammar_transitions_10k.json';
14
+
15
+ // ─── Types ──────────────────────────────────────────────────
16
+
17
+ /** Metadata returned by the grammar filter for debug logging in the caller. */
18
+
19
+ // ─── Scoring Constants ──────────────────────────────────────
20
+
21
+ const ALPHA = 0.5;
22
+ const BETA = 0.5;
23
+ const NEUTRAL_SCORE = 0.5;
24
+ const STAGE1_WEIGHT = 0.6;
25
+ const STAGE2_WEIGHT = 0.4;
26
+ const MIN_STAGE1_SCORE = 0.35;
27
+ const L1_SESSION_CAP = 1.2;
28
+
29
+ // ─── Grammar Data (loaded once on import) ───────────────────
30
+
31
+ const posTags = new Map(Object.entries(posTagsData));
32
+ const grammarTransitions = grammarTransitionsData;
33
+
34
+ /**
35
+ * Precomputed map from each POS tag to the set of allowed next POS tags.
36
+ * Built once at module load from grammarTransitions so applyGrammarFilter
37
+ * never re-iterates the transition rules per call.
38
+ */
39
+ const precomputedAllowedByPos = new Map(Object.entries(grammarTransitions.transitions).map(([pos, rule]) => [pos, new Set(rule.allowed)]));
40
+
41
+ // ─── Math ───────────────────────────────────────────────────
42
+
43
+ function cosineSimilarity(a, b) {
44
+ let dot = 0;
45
+ let normA = 0;
46
+ let normB = 0;
47
+ for (let i = 0; i < a.length; i++) {
48
+ dot += a[i] * b[i];
49
+ normA += a[i] * a[i];
50
+ normB += b[i] * b[i];
51
+ }
52
+ const dNormA = Math.sqrt(normA);
53
+ const dNormB = Math.sqrt(normB);
54
+ if (dNormA === 0 || dNormB === 0) {
55
+ return NEUTRAL_SCORE;
56
+ }
57
+ return (1 + dot / (dNormA * dNormB)) / 2;
58
+ }
59
+
60
+ // ─── Stage 1: Semantic + Frequency ──────────────────────────
61
+
62
+ function scoreStage1(candidate, contextVector, getWordVector, maxTenantFreq) {
63
+ // 1. Calculate Base Global Score (Normalized Log)
64
+ const maxPossibleLog = Math.log10(maxTenantFreq + 1);
65
+
66
+ // Diversity Adjustment
67
+ const diversityRaw = (Math.log10(candidate.tenantFreq + 1) * 0.50 + Math.log10(candidate.docFreq + 1) * 0.25 + Math.log10(candidate.authorFreq + 1) * 0.25) / maxPossibleLog;
68
+ const sessionMultiplier = candidate.sessionFreq > 0 ? 1 + Math.log10(candidate.sessionFreq + 1) * 2.5 : 1;
69
+
70
+ // Apply multiplier; capped at L1_SESSION_CAP (default 1.2) to prevent excessive over-indexing
71
+ const freqScore = Math.min(diversityRaw * sessionMultiplier, L1_SESSION_CAP);
72
+
73
+ // 3. Semantic Scoring
74
+ let semanticScore = NEUTRAL_SCORE;
75
+ if (contextVector) {
76
+ const wordVec = getWordVector(candidate.word);
77
+ semanticScore = wordVec ? cosineSimilarity(contextVector, wordVec) : NEUTRAL_SCORE;
78
+ }
79
+ return {
80
+ semanticScore,
81
+ freqScore,
82
+ stage1Score: ALPHA * semanticScore + BETA * freqScore
83
+ };
84
+ }
85
+
86
+ // ─── Grammar Filter ─────────────────────────────────────────
87
+
88
+ /**
89
+ * GHOST POS DICTIONARY
90
+ * A hardcoded mapping of common structural English words that were stripped
91
+ * from the main domain vocabulary. This allows the grammar filter to understand
92
+ * context without suggesting these words to the user.
93
+ */
94
+ const ghostPosTags = ghostPosTagsData;
95
+ function applyGrammarFilter(candidates, previousWord) {
96
+ if (!previousWord) return {
97
+ filtered: candidates,
98
+ grammarMeta: null
99
+ };
100
+ const lowerPrev = previousWord.toLowerCase();
101
+ const prevTags = ghostPosTags[lowerPrev] || posTags.get(lowerPrev);
102
+ if (!prevTags || prevTags.length === 0) {
103
+ return {
104
+ filtered: candidates,
105
+ grammarMeta: null
106
+ };
107
+ }
108
+ let allowedNextTags;
109
+ if (prevTags.length === 1) {
110
+ var _precomputedAllowedBy;
111
+ // Common case: single POS tag — reuse the precomputed Set directly (no allocation)
112
+ allowedNextTags = (_precomputedAllowedBy = precomputedAllowedByPos.get(prevTags[0])) !== null && _precomputedAllowedBy !== void 0 ? _precomputedAllowedBy : new Set();
113
+ } else {
114
+ allowedNextTags = new Set();
115
+ for (const pt of prevTags) {
116
+ const allowed = precomputedAllowedByPos.get(pt);
117
+ if (allowed) allowed.forEach(tag => allowedNextTags.add(tag));
118
+ }
119
+ }
120
+ const filtered = [];
121
+ const dropped = [];
122
+ for (const entry of candidates) {
123
+ const candidateTags = posTags.get(entry.candidate.word.toLowerCase());
124
+
125
+ // If candidate has no tags (unknown word), let it pass to be safe
126
+ if (!candidateTags || candidateTags.length === 0) {
127
+ filtered.push(entry);
128
+ continue;
129
+ }
130
+ if (candidateTags.some(ct => allowedNextTags.has(ct))) {
131
+ filtered.push(entry);
132
+ } else {
133
+ dropped.push(entry.candidate.word);
134
+ }
135
+ }
136
+ const finalFiltered = filtered.length > 0 ? filtered : candidates;
137
+ return {
138
+ filtered: finalFiltered,
139
+ grammarMeta: {
140
+ prevWord: lowerPrev,
141
+ prevTags,
142
+ before: candidates.length,
143
+ after: finalFiltered.length,
144
+ dropped: filtered.length > 0 ? dropped : []
145
+ }
146
+ };
147
+ }
148
+
149
+ // ─── Stage 2: LM Re-ranking ────────────────────────────────
150
+ function getLmScore(word, lmLogits) {
151
+ if (!lmLogits) return 0;
152
+
153
+ // Look up the word directly! No more tokens.
154
+ const val = lmLogits[word.toLowerCase()];
155
+ if (typeof val === 'number') {
156
+ return val;
157
+ }
158
+ return 0;
159
+ }
160
+
161
+ // ─── Public API ─────────────────────────────────────────────
162
+
163
+ export function rankCandidates(candidates, contextVector, getWordVector, lmLogits, maxTenantFreq, previousWord) {
164
+ // Stage 1
165
+ const stage1Results = candidates.map(candidate => {
166
+ const {
167
+ semanticScore,
168
+ freqScore,
169
+ stage1Score
170
+ } = scoreStage1(candidate, contextVector, getWordVector, maxTenantFreq);
171
+ return {
172
+ candidate,
173
+ semanticScore,
174
+ freqScore,
175
+ stage1Score
176
+ };
177
+ });
178
+ const stage1Survivors = stage1Results.filter(entry => entry.stage1Score >= MIN_STAGE1_SCORE);
179
+
180
+ // Grammar Filter
181
+ const {
182
+ filtered,
183
+ grammarMeta
184
+ } = applyGrammarFilter(stage1Survivors, previousWord);
185
+
186
+ // Stage 2 + final assembly
187
+ let lmMax = 0;
188
+ if (lmLogits && Object.keys(lmLogits).length > 0) {
189
+ const values = Object.values(lmLogits);
190
+ lmMax = Math.max(...values);
191
+ }
192
+ const scored = filtered.map(entry => {
193
+ let lmScore = 0;
194
+ let finalScore = entry.stage1Score;
195
+ if (lmLogits && Object.keys(lmLogits).length > 0) {
196
+ const rawLm = getLmScore(entry.candidate.word, lmLogits);
197
+ if (rawLm !== 0) {
198
+ // The word was in the top_k! Score it normally.
199
+ const logitDiff = Math.log(rawLm) - Math.log(lmMax);
200
+ lmScore = Math.exp(logitDiff);
201
+ } else {
202
+ lmScore = 0.05;
203
+ }
204
+ finalScore = STAGE1_WEIGHT * entry.stage1Score + STAGE2_WEIGHT * lmScore;
205
+ }
206
+ return {
207
+ word: entry.candidate.word,
208
+ freqScore: entry.freqScore,
209
+ semanticScore: entry.semanticScore,
210
+ lmScore,
211
+ finalScore
212
+ };
213
+ });
214
+ scored.sort((a, b) => {
215
+ if (b.finalScore !== a.finalScore) {
216
+ return b.finalScore - a.finalScore;
217
+ }
218
+ return a.word.length - b.word.length;
219
+ });
220
+ return {
221
+ candidates: scored,
222
+ grammarMeta
223
+ };
224
+ }