@atlaskit/editor-plugin-autocomplete 0.3.0 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,22 @@
1
1
  # @atlaskit/editor-plugin-autocomplete
2
2
 
3
+ ## 1.0.0
4
+
5
+ ### Patch Changes
6
+
7
+ - Updated dependencies
8
+
9
+ ## 0.4.0
10
+
11
+ ### Minor Changes
12
+
13
+ - [`603cd44e7b8c3`](https://bitbucket.org/atlassian/atlassian-frontend-monorepo/commits/603cd44e7b8c3) -
14
+ Fetch vectors through media client
15
+
16
+ ### Patch Changes
17
+
18
+ - Updated dependencies
19
+
3
20
  ## 0.3.0
4
21
 
5
22
  ### Minor Changes
@@ -4,7 +4,6 @@
4
4
  "paths": {},
5
5
  "resolveJsonModule": true
6
6
  },
7
- "files": ["./url-module.d.ts"],
8
7
  "include": ["../src/**/*.ts", "../src/**/*.tsx"],
9
8
  "exclude": [
10
9
  "../src/**/__tests__/*",
@@ -6,7 +6,6 @@ Object.defineProperty(exports, "__esModule", {
6
6
  });
7
7
  exports.createAutocompletePlugin = exports.autocompletePluginKey = void 0;
8
8
  var _defineProperty2 = _interopRequireDefault(require("@babel/runtime/helpers/defineProperty"));
9
- var _wordVectors_10k = _interopRequireDefault(require("url:./data/word-vectors_10k.bin"));
10
9
  var _safePlugin = require("@atlaskit/editor-common/safe-plugin");
11
10
  var _keymap = require("@atlaskit/editor-prosemirror/keymap");
12
11
  var _state = require("@atlaskit/editor-prosemirror/state");
@@ -15,9 +14,7 @@ var _ghostTextDecoration = require("./ghost-text-decoration");
15
14
  var _slowLaneClient = require("./slow-lane-client");
16
15
  var _textPredictor = require("./text-predictor");
17
16
  function ownKeys(e, r) { var t = Object.keys(e); if (Object.getOwnPropertySymbols) { var o = Object.getOwnPropertySymbols(e); r && (o = o.filter(function (r) { return Object.getOwnPropertyDescriptor(e, r).enumerable; })), t.push.apply(t, o); } return t; }
18
- function _objectSpread(e) { for (var r = 1; r < arguments.length; r++) { var t = null != arguments[r] ? arguments[r] : {}; r % 2 ? ownKeys(Object(t), !0).forEach(function (r) { (0, _defineProperty2.default)(e, r, t[r]); }) : Object.getOwnPropertyDescriptors ? Object.defineProperties(e, Object.getOwnPropertyDescriptors(t)) : ownKeys(Object(t)).forEach(function (r) { Object.defineProperty(e, r, Object.getOwnPropertyDescriptor(t, r)); }); } return e; } // url: prefix is an Atlaspack/Parcel directive that resolves this file as an emitted
19
- // asset URL (content-hashed). For rspack, this is handled via staticAssetsLoader.
20
- // eslint-disable-next-line @repo/internal/import/no-unresolved
17
+ function _objectSpread(e) { for (var r = 1; r < arguments.length; r++) { var t = null != arguments[r] ? arguments[r] : {}; r % 2 ? ownKeys(Object(t), !0).forEach(function (r) { (0, _defineProperty2.default)(e, r, t[r]); }) : Object.getOwnPropertyDescriptors ? Object.defineProperties(e, Object.getOwnPropertyDescriptors(t)) : ownKeys(Object(t)).forEach(function (r) { Object.defineProperty(e, r, Object.getOwnPropertyDescriptor(t, r)); }); } return e; }
21
18
  var SLOW_LANE_ENDPOINT = '/gateway/api/v1/autocomplete/typeahead-encodings';
22
19
  var autocompletePluginKey = exports.autocompletePluginKey = new _state.PluginKey('autocomplete');
23
20
  var DEBOUNCE_MS = 150;
@@ -305,7 +302,7 @@ var createAutocompletePlugin = exports.createAutocompletePlugin = function creat
305
302
  focus: function focus() {
306
303
  (0, _textPredictor.loadDefaultVocabulary)();
307
304
  (0, _textPredictor.loadVectorsAsync)({
308
- vectorsUrl: _wordVectors_10k.default
305
+ getBinaryUrl: options === null || options === void 0 ? void 0 : options.getVectorsBinaryUrl
309
306
  }).catch(function () {});
310
307
  if (!hasIngestedPage) {
311
308
  hasIngestedPage = true;
@@ -74,7 +74,7 @@ function scoreStage1(candidate, contextVector, getWordVector, maxTenantFreq) {
74
74
  var maxPossibleLog = Math.log10(maxTenantFreq + 1);
75
75
 
76
76
  // Diversity Adjustment
77
- var diversityRaw = (Math.log10(candidate.tenantFreq + 1) * 0.50 + Math.log10(candidate.docFreq + 1) * 0.25 + Math.log10(candidate.authorFreq + 1) * 0.25) / maxPossibleLog;
77
+ var diversityRaw = (Math.log10(candidate.tenantFreq + 1) * 0.5 + Math.log10(candidate.docFreq + 1) * 0.25 + Math.log10(candidate.authorFreq + 1) * 0.25) / maxPossibleLog;
78
78
  var sessionMultiplier = candidate.sessionFreq > 0 ? 1 + Math.log10(candidate.sessionFreq + 1) * 2.5 : 1;
79
79
 
80
80
  // Apply multiplier; capped at L1_SESSION_CAP (default 1.2) to prevent excessive over-indexing
@@ -97,8 +97,8 @@ function scoreStage1(candidate, contextVector, getWordVector, maxTenantFreq) {
97
97
 
98
98
  /**
99
99
  * GHOST POS DICTIONARY
100
- * A hardcoded mapping of common structural English words that were stripped
101
- * from the main domain vocabulary. This allows the grammar filter to understand
100
+ * A hardcoded mapping of common structural English words that were stripped
101
+ * from the main domain vocabulary. This allows the grammar filter to understand
102
102
  * context without suggesting these words to the user.
103
103
  */
104
104
  var ghostPosTags = _ghost_pos_tags.default;
@@ -705,33 +705,47 @@ var loadVectorsAsync = exports.loadVectorsAsync = /*#__PURE__*/function () {
705
705
  }
706
706
  return _context.abrupt("return");
707
707
  case 2:
708
- if (options !== null && options !== void 0 && options.vectorsUrl) {
708
+ if (options !== null && options !== void 0 && options.getBinaryUrl) {
709
709
  _context.next = 5;
710
710
  break;
711
711
  }
712
712
  // eslint-disable-next-line no-console
713
- console.warn('[text-predictor] loadVectorsAsync called without a vectorsUrl — vectors will not load. Pass vectorsUrl via plugin options.');
713
+ console.warn('[text-predictor] loadVectorsAsync called without a getBinaryUrl — vectors will not load. Pass getVectorsBinaryUrl via plugin options.');
714
714
  return _context.abrupt("return");
715
715
  case 5:
716
716
  vectorsLoadStarted = true;
717
- url = options.vectorsUrl;
718
- _context.prev = 7;
719
- _context.next = 10;
717
+ _context.prev = 6;
718
+ _context.next = 9;
719
+ return options.getBinaryUrl();
720
+ case 9:
721
+ url = _context.sent;
722
+ _context.next = 17;
723
+ break;
724
+ case 12:
725
+ _context.prev = 12;
726
+ _context.t0 = _context["catch"](6);
727
+ vectorsLoadStarted = false;
728
+ // eslint-disable-next-line no-console
729
+ console.warn('[text-predictor] Failed to resolve vectors URL:', _context.t0);
730
+ return _context.abrupt("return");
731
+ case 17:
732
+ _context.prev = 17;
733
+ _context.next = 20;
720
734
  return fetch(url);
721
- case 10:
735
+ case 20:
722
736
  res = _context.sent;
723
737
  if (res.ok) {
724
- _context.next = 15;
738
+ _context.next = 25;
725
739
  break;
726
740
  }
727
741
  vectorsLoadStarted = false;
728
742
  // eslint-disable-next-line no-console
729
743
  console.warn("[text-predictor] Failed to load vectors: ".concat(res.status));
730
744
  return _context.abrupt("return");
731
- case 15:
732
- _context.next = 17;
745
+ case 25:
746
+ _context.next = 27;
733
747
  return res.arrayBuffer();
734
- case 17:
748
+ case 27:
735
749
  buffer = _context.sent;
736
750
  float32 = new Float32Array(buffer);
737
751
  wordIndex = _word_index_10k.default;
@@ -751,19 +765,19 @@ var loadVectorsAsync = exports.loadVectorsAsync = /*#__PURE__*/function () {
751
765
  sizeBytes: float32.byteLength
752
766
  });
753
767
  }
754
- _context.next = 31;
768
+ _context.next = 41;
755
769
  break;
756
- case 27:
757
- _context.prev = 27;
758
- _context.t0 = _context["catch"](7);
770
+ case 37:
771
+ _context.prev = 37;
772
+ _context.t1 = _context["catch"](17);
759
773
  vectorsLoadStarted = false;
760
774
  // eslint-disable-next-line no-console
761
- console.warn('[text-predictor] Failed to load vectors:', _context.t0);
762
- case 31:
775
+ console.warn('[text-predictor] Failed to load vectors:', _context.t1);
776
+ case 41:
763
777
  case "end":
764
778
  return _context.stop();
765
779
  }
766
- }, _callee, null, [[7, 27]]);
780
+ }, _callee, null, [[6, 12], [17, 37]]);
767
781
  }));
768
782
  return function loadVectorsAsync(_x) {
769
783
  return _ref4.apply(this, arguments);
@@ -1,7 +1,3 @@
1
- // url: prefix is an Atlaspack/Parcel directive that resolves this file as an emitted
2
- // asset URL (content-hashed). For rspack, this is handled via staticAssetsLoader.
3
- // eslint-disable-next-line @repo/internal/import/no-unresolved
4
- import wordVectorsUrl from 'url:./data/word-vectors_10k.bin';
5
1
  import { SafePlugin } from '@atlaskit/editor-common/safe-plugin';
6
2
  import { keydownHandler } from '@atlaskit/editor-prosemirror/keymap';
7
3
  import { PluginKey } from '@atlaskit/editor-prosemirror/state';
@@ -307,7 +303,7 @@ export const createAutocompletePlugin = options => {
307
303
  focus: () => {
308
304
  loadDefaultVocabulary();
309
305
  loadVectorsAsync({
310
- vectorsUrl: wordVectorsUrl
306
+ getBinaryUrl: options === null || options === void 0 ? void 0 : options.getVectorsBinaryUrl
311
307
  }).catch(() => {});
312
308
  if (!hasIngestedPage) {
313
309
  hasIngestedPage = true;
@@ -61,7 +61,7 @@ function scoreStage1(candidate, contextVector, getWordVector, maxTenantFreq) {
61
61
  const maxPossibleLog = Math.log10(maxTenantFreq + 1);
62
62
 
63
63
  // Diversity Adjustment
64
- const diversityRaw = (Math.log10(candidate.tenantFreq + 1) * 0.50 + Math.log10(candidate.docFreq + 1) * 0.25 + Math.log10(candidate.authorFreq + 1) * 0.25) / maxPossibleLog;
64
+ const diversityRaw = (Math.log10(candidate.tenantFreq + 1) * 0.5 + Math.log10(candidate.docFreq + 1) * 0.25 + Math.log10(candidate.authorFreq + 1) * 0.25) / maxPossibleLog;
65
65
  const sessionMultiplier = candidate.sessionFreq > 0 ? 1 + Math.log10(candidate.sessionFreq + 1) * 2.5 : 1;
66
66
 
67
67
  // Apply multiplier; capped at L1_SESSION_CAP (default 1.2) to prevent excessive over-indexing
@@ -84,8 +84,8 @@ function scoreStage1(candidate, contextVector, getWordVector, maxTenantFreq) {
84
84
 
85
85
  /**
86
86
  * GHOST POS DICTIONARY
87
- * A hardcoded mapping of common structural English words that were stripped
88
- * from the main domain vocabulary. This allows the grammar filter to understand
87
+ * A hardcoded mapping of common structural English words that were stripped
88
+ * from the main domain vocabulary. This allows the grammar filter to understand
89
89
  * context without suggesting these words to the user.
90
90
  */
91
91
  const ghostPosTags = ghostPosTagsData;
@@ -569,13 +569,21 @@ export const loadVectorsAsync = async options => {
569
569
  if (vectorStore || vectorsLoadStarted) {
570
570
  return;
571
571
  }
572
- if (!(options !== null && options !== void 0 && options.vectorsUrl)) {
572
+ if (!(options !== null && options !== void 0 && options.getBinaryUrl)) {
573
573
  // eslint-disable-next-line no-console
574
- console.warn('[text-predictor] loadVectorsAsync called without a vectorsUrl — vectors will not load. Pass vectorsUrl via plugin options.');
574
+ console.warn('[text-predictor] loadVectorsAsync called without a getBinaryUrl — vectors will not load. Pass getVectorsBinaryUrl via plugin options.');
575
575
  return;
576
576
  }
577
577
  vectorsLoadStarted = true;
578
- const url = options.vectorsUrl;
578
+ let url;
579
+ try {
580
+ url = await options.getBinaryUrl();
581
+ } catch (e) {
582
+ vectorsLoadStarted = false;
583
+ // eslint-disable-next-line no-console
584
+ console.warn('[text-predictor] Failed to resolve vectors URL:', e);
585
+ return;
586
+ }
579
587
  try {
580
588
  const res = await fetch(url);
581
589
  if (!res.ok) {
@@ -1,10 +1,6 @@
1
1
  import _defineProperty from "@babel/runtime/helpers/defineProperty";
2
2
  function ownKeys(e, r) { var t = Object.keys(e); if (Object.getOwnPropertySymbols) { var o = Object.getOwnPropertySymbols(e); r && (o = o.filter(function (r) { return Object.getOwnPropertyDescriptor(e, r).enumerable; })), t.push.apply(t, o); } return t; }
3
3
  function _objectSpread(e) { for (var r = 1; r < arguments.length; r++) { var t = null != arguments[r] ? arguments[r] : {}; r % 2 ? ownKeys(Object(t), !0).forEach(function (r) { _defineProperty(e, r, t[r]); }) : Object.getOwnPropertyDescriptors ? Object.defineProperties(e, Object.getOwnPropertyDescriptors(t)) : ownKeys(Object(t)).forEach(function (r) { Object.defineProperty(e, r, Object.getOwnPropertyDescriptor(t, r)); }); } return e; }
4
- // url: prefix is an Atlaspack/Parcel directive that resolves this file as an emitted
5
- // asset URL (content-hashed). For rspack, this is handled via staticAssetsLoader.
6
- // eslint-disable-next-line @repo/internal/import/no-unresolved
7
- import wordVectorsUrl from 'url:./data/word-vectors_10k.bin';
8
4
  import { SafePlugin } from '@atlaskit/editor-common/safe-plugin';
9
5
  import { keydownHandler } from '@atlaskit/editor-prosemirror/keymap';
10
6
  import { PluginKey } from '@atlaskit/editor-prosemirror/state';
@@ -299,7 +295,7 @@ export var createAutocompletePlugin = function createAutocompletePlugin(options)
299
295
  focus: function focus() {
300
296
  loadDefaultVocabulary();
301
297
  loadVectorsAsync({
302
- vectorsUrl: wordVectorsUrl
298
+ getBinaryUrl: options === null || options === void 0 ? void 0 : options.getVectorsBinaryUrl
303
299
  }).catch(function () {});
304
300
  if (!hasIngestedPage) {
305
301
  hasIngestedPage = true;
@@ -70,7 +70,7 @@ function scoreStage1(candidate, contextVector, getWordVector, maxTenantFreq) {
70
70
  var maxPossibleLog = Math.log10(maxTenantFreq + 1);
71
71
 
72
72
  // Diversity Adjustment
73
- var diversityRaw = (Math.log10(candidate.tenantFreq + 1) * 0.50 + Math.log10(candidate.docFreq + 1) * 0.25 + Math.log10(candidate.authorFreq + 1) * 0.25) / maxPossibleLog;
73
+ var diversityRaw = (Math.log10(candidate.tenantFreq + 1) * 0.5 + Math.log10(candidate.docFreq + 1) * 0.25 + Math.log10(candidate.authorFreq + 1) * 0.25) / maxPossibleLog;
74
74
  var sessionMultiplier = candidate.sessionFreq > 0 ? 1 + Math.log10(candidate.sessionFreq + 1) * 2.5 : 1;
75
75
 
76
76
  // Apply multiplier; capped at L1_SESSION_CAP (default 1.2) to prevent excessive over-indexing
@@ -93,8 +93,8 @@ function scoreStage1(candidate, contextVector, getWordVector, maxTenantFreq) {
93
93
 
94
94
  /**
95
95
  * GHOST POS DICTIONARY
96
- * A hardcoded mapping of common structural English words that were stripped
97
- * from the main domain vocabulary. This allows the grammar filter to understand
96
+ * A hardcoded mapping of common structural English words that were stripped
97
+ * from the main domain vocabulary. This allows the grammar filter to understand
98
98
  * context without suggesting these words to the user.
99
99
  */
100
100
  var ghostPosTags = ghostPosTagsData;
@@ -702,33 +702,47 @@ export var loadVectorsAsync = /*#__PURE__*/function () {
702
702
  }
703
703
  return _context.abrupt("return");
704
704
  case 2:
705
- if (options !== null && options !== void 0 && options.vectorsUrl) {
705
+ if (options !== null && options !== void 0 && options.getBinaryUrl) {
706
706
  _context.next = 5;
707
707
  break;
708
708
  }
709
709
  // eslint-disable-next-line no-console
710
- console.warn('[text-predictor] loadVectorsAsync called without a vectorsUrl — vectors will not load. Pass vectorsUrl via plugin options.');
710
+ console.warn('[text-predictor] loadVectorsAsync called without a getBinaryUrl — vectors will not load. Pass getVectorsBinaryUrl via plugin options.');
711
711
  return _context.abrupt("return");
712
712
  case 5:
713
713
  vectorsLoadStarted = true;
714
- url = options.vectorsUrl;
715
- _context.prev = 7;
716
- _context.next = 10;
714
+ _context.prev = 6;
715
+ _context.next = 9;
716
+ return options.getBinaryUrl();
717
+ case 9:
718
+ url = _context.sent;
719
+ _context.next = 17;
720
+ break;
721
+ case 12:
722
+ _context.prev = 12;
723
+ _context.t0 = _context["catch"](6);
724
+ vectorsLoadStarted = false;
725
+ // eslint-disable-next-line no-console
726
+ console.warn('[text-predictor] Failed to resolve vectors URL:', _context.t0);
727
+ return _context.abrupt("return");
728
+ case 17:
729
+ _context.prev = 17;
730
+ _context.next = 20;
717
731
  return fetch(url);
718
- case 10:
732
+ case 20:
719
733
  res = _context.sent;
720
734
  if (res.ok) {
721
- _context.next = 15;
735
+ _context.next = 25;
722
736
  break;
723
737
  }
724
738
  vectorsLoadStarted = false;
725
739
  // eslint-disable-next-line no-console
726
740
  console.warn("[text-predictor] Failed to load vectors: ".concat(res.status));
727
741
  return _context.abrupt("return");
728
- case 15:
729
- _context.next = 17;
742
+ case 25:
743
+ _context.next = 27;
730
744
  return res.arrayBuffer();
731
- case 17:
745
+ case 27:
732
746
  buffer = _context.sent;
733
747
  float32 = new Float32Array(buffer);
734
748
  wordIndex = wordIndexData;
@@ -748,19 +762,19 @@ export var loadVectorsAsync = /*#__PURE__*/function () {
748
762
  sizeBytes: float32.byteLength
749
763
  });
750
764
  }
751
- _context.next = 31;
765
+ _context.next = 41;
752
766
  break;
753
- case 27:
754
- _context.prev = 27;
755
- _context.t0 = _context["catch"](7);
767
+ case 37:
768
+ _context.prev = 37;
769
+ _context.t1 = _context["catch"](17);
756
770
  vectorsLoadStarted = false;
757
771
  // eslint-disable-next-line no-console
758
- console.warn('[text-predictor] Failed to load vectors:', _context.t0);
759
- case 31:
772
+ console.warn('[text-predictor] Failed to load vectors:', _context.t1);
773
+ case 41:
760
774
  case "end":
761
775
  return _context.stop();
762
776
  }
763
- }, _callee, null, [[7, 27]]);
777
+ }, _callee, null, [[6, 12], [17, 37]]);
764
778
  }));
765
779
  return function loadVectorsAsync(_x) {
766
780
  return _ref4.apply(this, arguments);
@@ -32,5 +32,11 @@ export interface AutocompletePluginOptions {
32
32
  * word-frequency boosting. Called lazily so the preset can remain synchronous.
33
33
  */
34
34
  getContext?: () => Promise<AutocompleteContext | undefined>;
35
+ /**
36
+ * Async function that resolves to a URL for the word vectors binary file.
37
+ * When provided, this takes precedence over the bundled asset URL.
38
+ * Use this to serve vectors from a CDN or media service in production.
39
+ */
40
+ getVectorsBinaryUrl?: () => Promise<string>;
35
41
  }
36
- export declare const createAutocompletePlugin: (options?: AutocompletePluginOptions) => SafePlugin<any>;
42
+ export declare const createAutocompletePlugin: (options?: AutocompletePluginOptions) => SafePlugin<AutocompletePluginState>;
@@ -83,7 +83,7 @@ export declare const incrementSessionFreq: (word: string) => void;
83
83
  export declare const ingestDocumentPage: (pageContent: string | undefined) => void;
84
84
  export declare const predict: (textBefore: string) => string | null;
85
85
  export declare const loadVectorsAsync: (options?: {
86
- vectorsUrl?: string;
86
+ getBinaryUrl?: () => Promise<string>;
87
87
  }) => Promise<void>;
88
88
  export declare const initVectors: (store: VectorStore) => void;
89
89
  export declare const loadDefaultVocabulary: () => void;
@@ -32,5 +32,11 @@ export interface AutocompletePluginOptions {
32
32
  * word-frequency boosting. Called lazily so the preset can remain synchronous.
33
33
  */
34
34
  getContext?: () => Promise<AutocompleteContext | undefined>;
35
+ /**
36
+ * Async function that resolves to a URL for the word vectors binary file.
37
+ * When provided, this takes precedence over the bundled asset URL.
38
+ * Use this to serve vectors from a CDN or media service in production.
39
+ */
40
+ getVectorsBinaryUrl?: () => Promise<string>;
35
41
  }
36
- export declare const createAutocompletePlugin: (options?: AutocompletePluginOptions) => SafePlugin<any>;
42
+ export declare const createAutocompletePlugin: (options?: AutocompletePluginOptions) => SafePlugin<AutocompletePluginState>;
@@ -83,7 +83,7 @@ export declare const incrementSessionFreq: (word: string) => void;
83
83
  export declare const ingestDocumentPage: (pageContent: string | undefined) => void;
84
84
  export declare const predict: (textBefore: string) => string | null;
85
85
  export declare const loadVectorsAsync: (options?: {
86
- vectorsUrl?: string;
86
+ getBinaryUrl?: () => Promise<string>;
87
87
  }) => Promise<void>;
88
88
  export declare const initVectors: (store: VectorStore) => void;
89
89
  export declare const loadDefaultVocabulary: () => void;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@atlaskit/editor-plugin-autocomplete",
3
- "version": "0.3.0",
3
+ "version": "1.0.0",
4
4
  "description": "Client-side text autocomplete plugin for @atlaskit/editor-core",
5
5
  "author": "Atlassian Pty Ltd",
6
6
  "license": "Apache-2.0",
@@ -33,7 +33,7 @@
33
33
  "wink-nlp": "^2.4.0"
34
34
  },
35
35
  "peerDependencies": {
36
- "@atlaskit/editor-common": "^112.16.0",
36
+ "@atlaskit/editor-common": "^113.0.0",
37
37
  "react": "^18.2.0"
38
38
  },
39
39
  "techstack": {
@@ -1,8 +1,5 @@
1
1
  import type { AutocompletePlugin } from './autocompletePluginType';
2
- import {
3
- autocompletePluginKey,
4
- createAutocompletePlugin,
5
- } from './pm-plugins/autocomplete-plugin';
2
+ import { autocompletePluginKey, createAutocompletePlugin } from './pm-plugins/autocomplete-plugin';
6
3
  import type { AutocompletePluginState } from './pm-plugins/autocomplete-plugin';
7
4
 
8
5
  export const autocompletePlugin: AutocompletePlugin = ({ config: options }) => {
@@ -1,8 +1,3 @@
1
- // url: prefix is an Atlaspack/Parcel directive that resolves this file as an emitted
2
- // asset URL (content-hashed). For rspack, this is handled via staticAssetsLoader.
3
- // eslint-disable-next-line @repo/internal/import/no-unresolved
4
- import wordVectorsUrl from 'url:./data/word-vectors_10k.bin';
5
-
6
1
  import { SafePlugin } from '@atlaskit/editor-common/safe-plugin';
7
2
  import { keydownHandler } from '@atlaskit/editor-prosemirror/keymap';
8
3
  import { PluginKey } from '@atlaskit/editor-prosemirror/state';
@@ -171,6 +166,12 @@ export interface AutocompletePluginOptions {
171
166
  * word-frequency boosting. Called lazily so the preset can remain synchronous.
172
167
  */
173
168
  getContext?: () => Promise<AutocompleteContext | undefined>;
169
+ /**
170
+ * Async function that resolves to a URL for the word vectors binary file.
171
+ * When provided, this takes precedence over the bundled asset URL.
172
+ * Use this to serve vectors from a CDN or media service in production.
173
+ */
174
+ getVectorsBinaryUrl?: () => Promise<string>;
174
175
  }
175
176
 
176
177
  /**
@@ -374,7 +375,7 @@ export const createAutocompletePlugin = (options?: AutocompletePluginOptions) =>
374
375
  },
375
376
  focus: () => {
376
377
  loadDefaultVocabulary();
377
- loadVectorsAsync({ vectorsUrl: wordVectorsUrl }).catch(() => {});
378
+ loadVectorsAsync({ getBinaryUrl: options?.getVectorsBinaryUrl }).catch(() => {});
378
379
  if (!hasIngestedPage) {
379
380
  hasIngestedPage = true;
380
381
  if (options?.getContext) {
@@ -12,36 +12,36 @@ import grammarTransitionsData from './data/grammar_transitions_10k.json';
12
12
  // ─── Types ──────────────────────────────────────────────────
13
13
 
14
14
  export interface ScoringCandidate {
15
- authorFreq: number;
16
- docFreq: number;
17
- sessionFreq: number;
18
- tenantFreq: number;
19
- word: string;
15
+ authorFreq: number;
16
+ docFreq: number;
17
+ sessionFreq: number;
18
+ tenantFreq: number;
19
+ word: string;
20
20
  }
21
21
 
22
22
  export interface ScoredCandidate {
23
- finalScore: number;
24
- freqScore: number;
25
- lmScore: number;
26
- semanticScore: number;
27
- word: string;
23
+ finalScore: number;
24
+ freqScore: number;
25
+ lmScore: number;
26
+ semanticScore: number;
27
+ word: string;
28
28
  }
29
29
 
30
30
  /** Metadata returned by the grammar filter for debug logging in the caller. */
31
31
  export interface GrammarFilterMeta {
32
- after: number;
33
- before: number;
34
- dropped: string[];
35
- prevTags: string[];
36
- prevWord: string;
32
+ after: number;
33
+ before: number;
34
+ dropped: string[];
35
+ prevTags: string[];
36
+ prevWord: string;
37
37
  }
38
38
 
39
39
  interface PosTransitionRule {
40
- allowed: string[];
40
+ allowed: string[];
41
41
  }
42
42
 
43
43
  interface GrammarTransitions {
44
- transitions: Record<string, PosTransitionRule>;
44
+ transitions: Record<string, PosTransitionRule>;
45
45
  }
46
46
 
47
47
  // ─── Scoring Constants ──────────────────────────────────────
@@ -57,7 +57,7 @@ const L1_SESSION_CAP = 1.2;
57
57
  // ─── Grammar Data (loaded once on import) ───────────────────
58
58
 
59
59
  const posTags: Map<string, string[]> = new Map(
60
- Object.entries(posTagsData as Record<string, string[]>),
60
+ Object.entries(posTagsData as Record<string, string[]>),
61
61
  );
62
62
 
63
63
  const grammarTransitions = grammarTransitionsData as GrammarTransitions;
@@ -68,227 +68,224 @@ const grammarTransitions = grammarTransitionsData as GrammarTransitions;
68
68
  * never re-iterates the transition rules per call.
69
69
  */
70
70
  const precomputedAllowedByPos: Map<string, Set<string>> = new Map(
71
- Object.entries(grammarTransitions.transitions).map(([pos, rule]) => [
72
- pos,
73
- new Set(rule.allowed),
74
- ]),
71
+ Object.entries(grammarTransitions.transitions).map(([pos, rule]) => [pos, new Set(rule.allowed)]),
75
72
  );
76
73
 
77
74
  // ─── Math ───────────────────────────────────────────────────
78
75
 
79
76
  function cosineSimilarity(a: Float32Array, b: Float32Array): number {
80
- let dot = 0;
81
- let normA = 0;
82
- let normB = 0;
83
- for (let i = 0; i < a.length; i++) {
84
- dot += a[i] * b[i];
85
- normA += a[i] * a[i];
86
- normB += b[i] * b[i];
87
- }
88
- const dNormA = Math.sqrt(normA);
89
- const dNormB = Math.sqrt(normB);
90
- if (dNormA === 0 || dNormB === 0) {
91
- return NEUTRAL_SCORE;
92
- }
93
- return (1 + dot / (dNormA * dNormB)) / 2;
77
+ let dot = 0;
78
+ let normA = 0;
79
+ let normB = 0;
80
+ for (let i = 0; i < a.length; i++) {
81
+ dot += a[i] * b[i];
82
+ normA += a[i] * a[i];
83
+ normB += b[i] * b[i];
84
+ }
85
+ const dNormA = Math.sqrt(normA);
86
+ const dNormB = Math.sqrt(normB);
87
+ if (dNormA === 0 || dNormB === 0) {
88
+ return NEUTRAL_SCORE;
89
+ }
90
+ return (1 + dot / (dNormA * dNormB)) / 2;
94
91
  }
95
92
 
96
93
  // ─── Stage 1: Semantic + Frequency ──────────────────────────
97
94
 
98
95
  function scoreStage1(
99
- candidate: ScoringCandidate,
100
- contextVector: Float32Array | null,
101
- getWordVector: (word: string) => Float32Array | null,
102
- maxTenantFreq: number,
96
+ candidate: ScoringCandidate,
97
+ contextVector: Float32Array | null,
98
+ getWordVector: (word: string) => Float32Array | null,
99
+ maxTenantFreq: number,
103
100
  ): { freqScore: number; semanticScore: number; stage1Score: number } {
104
-
105
101
  // 1. Calculate Base Global Score (Normalized Log)
106
- const maxPossibleLog = Math.log10(maxTenantFreq + 1);
107
-
108
- // Diversity Adjustment
109
- const diversityRaw = (
110
- Math.log10(candidate.tenantFreq + 1) * 0.50 +
111
- Math.log10(candidate.docFreq + 1) * 0.25 +
112
- Math.log10(candidate.authorFreq + 1) * 0.25
113
- ) / maxPossibleLog;
114
-
115
- const sessionMultiplier = candidate.sessionFreq > 0
116
- ? 1 + (Math.log10(candidate.sessionFreq + 1) * 2.5)
117
- : 1;
118
-
119
- // Apply multiplier; capped at L1_SESSION_CAP (default 1.2) to prevent excessive over-indexing
120
- const freqScore = Math.min(diversityRaw * sessionMultiplier, L1_SESSION_CAP);
121
-
122
- // 3. Semantic Scoring
123
- let semanticScore = NEUTRAL_SCORE;
124
- if (contextVector) {
125
- const wordVec = getWordVector(candidate.word);
126
- semanticScore = wordVec ? cosineSimilarity(contextVector, wordVec) : NEUTRAL_SCORE;
127
- }
128
-
129
- return {
130
- semanticScore,
131
- freqScore,
132
- stage1Score: (ALPHA * semanticScore) + (BETA * freqScore),
133
- };
102
+ const maxPossibleLog = Math.log10(maxTenantFreq + 1);
103
+
104
+ // Diversity Adjustment
105
+ const diversityRaw =
106
+ (Math.log10(candidate.tenantFreq + 1) * 0.5 +
107
+ Math.log10(candidate.docFreq + 1) * 0.25 +
108
+ Math.log10(candidate.authorFreq + 1) * 0.25) /
109
+ maxPossibleLog;
110
+
111
+ const sessionMultiplier =
112
+ candidate.sessionFreq > 0 ? 1 + Math.log10(candidate.sessionFreq + 1) * 2.5 : 1;
113
+
114
+ // Apply multiplier; capped at L1_SESSION_CAP (default 1.2) to prevent excessive over-indexing
115
+ const freqScore = Math.min(diversityRaw * sessionMultiplier, L1_SESSION_CAP);
116
+
117
+ // 3. Semantic Scoring
118
+ let semanticScore = NEUTRAL_SCORE;
119
+ if (contextVector) {
120
+ const wordVec = getWordVector(candidate.word);
121
+ semanticScore = wordVec ? cosineSimilarity(contextVector, wordVec) : NEUTRAL_SCORE;
122
+ }
123
+
124
+ return {
125
+ semanticScore,
126
+ freqScore,
127
+ stage1Score: ALPHA * semanticScore + BETA * freqScore,
128
+ };
134
129
  }
135
130
 
136
131
  // ─── Grammar Filter ─────────────────────────────────────────
137
132
 
138
133
  /**
139
134
  * GHOST POS DICTIONARY
140
- * A hardcoded mapping of common structural English words that were stripped
141
- * from the main domain vocabulary. This allows the grammar filter to understand
135
+ * A hardcoded mapping of common structural English words that were stripped
136
+ * from the main domain vocabulary. This allows the grammar filter to understand
142
137
  * context without suggesting these words to the user.
143
138
  */
144
139
  const ghostPosTags: Record<string, string[]> = ghostPosTagsData as Record<string, string[]>;
145
140
 
146
- type FilterEntry = { candidate: ScoringCandidate; freqScore: number; semanticScore: number; stage1Score: number };
141
+ type FilterEntry = {
142
+ candidate: ScoringCandidate;
143
+ freqScore: number;
144
+ semanticScore: number;
145
+ stage1Score: number;
146
+ };
147
147
 
148
148
  function applyGrammarFilter(
149
- candidates: FilterEntry[],
150
- previousWord: string,
149
+ candidates: FilterEntry[],
150
+ previousWord: string,
151
151
  ): { filtered: FilterEntry[]; grammarMeta: GrammarFilterMeta | null } {
152
- if (!previousWord) return { filtered: candidates, grammarMeta: null };
153
-
154
- const lowerPrev = previousWord.toLowerCase();
155
- const prevTags = ghostPosTags[lowerPrev] || posTags.get(lowerPrev);
156
-
157
- if (!prevTags || prevTags.length === 0) {
158
- return { filtered: candidates, grammarMeta: null };
159
- }
160
-
161
- let allowedNextTags: Set<string>;
162
- if (prevTags.length === 1) {
163
- // Common case: single POS tag — reuse the precomputed Set directly (no allocation)
164
- allowedNextTags = precomputedAllowedByPos.get(prevTags[0]) ?? new Set();
165
- } else {
166
- allowedNextTags = new Set<string>();
167
- for (const pt of prevTags) {
168
- const allowed = precomputedAllowedByPos.get(pt);
169
- if (allowed) allowed.forEach(tag => allowedNextTags.add(tag));
170
- }
171
- }
172
-
173
- const filtered: FilterEntry[] = [];
174
- const dropped: string[] = [];
175
-
176
- for (const entry of candidates) {
177
- const candidateTags = posTags.get(entry.candidate.word.toLowerCase());
178
-
179
- // If candidate has no tags (unknown word), let it pass to be safe
180
- if (!candidateTags || candidateTags.length === 0) {
181
- filtered.push(entry);
182
- continue;
183
- }
184
-
185
- if (candidateTags.some((ct) => allowedNextTags.has(ct))) {
186
- filtered.push(entry);
187
- } else {
188
- dropped.push(entry.candidate.word);
189
- }
190
- }
191
-
192
- const finalFiltered = filtered.length > 0 ? filtered : candidates;
193
-
194
- return {
195
- filtered: finalFiltered,
196
- grammarMeta: {
197
- prevWord: lowerPrev,
198
- prevTags,
199
- before: candidates.length,
200
- after: finalFiltered.length,
201
- dropped: filtered.length > 0 ? dropped : [],
202
- },
203
- };
152
+ if (!previousWord) return { filtered: candidates, grammarMeta: null };
153
+
154
+ const lowerPrev = previousWord.toLowerCase();
155
+ const prevTags = ghostPosTags[lowerPrev] || posTags.get(lowerPrev);
156
+
157
+ if (!prevTags || prevTags.length === 0) {
158
+ return { filtered: candidates, grammarMeta: null };
159
+ }
160
+
161
+ let allowedNextTags: Set<string>;
162
+ if (prevTags.length === 1) {
163
+ // Common case: single POS tag — reuse the precomputed Set directly (no allocation)
164
+ allowedNextTags = precomputedAllowedByPos.get(prevTags[0]) ?? new Set();
165
+ } else {
166
+ allowedNextTags = new Set<string>();
167
+ for (const pt of prevTags) {
168
+ const allowed = precomputedAllowedByPos.get(pt);
169
+ if (allowed) allowed.forEach((tag) => allowedNextTags.add(tag));
170
+ }
171
+ }
172
+
173
+ const filtered: FilterEntry[] = [];
174
+ const dropped: string[] = [];
175
+
176
+ for (const entry of candidates) {
177
+ const candidateTags = posTags.get(entry.candidate.word.toLowerCase());
178
+
179
+ // If candidate has no tags (unknown word), let it pass to be safe
180
+ if (!candidateTags || candidateTags.length === 0) {
181
+ filtered.push(entry);
182
+ continue;
183
+ }
184
+
185
+ if (candidateTags.some((ct) => allowedNextTags.has(ct))) {
186
+ filtered.push(entry);
187
+ } else {
188
+ dropped.push(entry.candidate.word);
189
+ }
190
+ }
191
+
192
+ const finalFiltered = filtered.length > 0 ? filtered : candidates;
193
+
194
+ return {
195
+ filtered: finalFiltered,
196
+ grammarMeta: {
197
+ prevWord: lowerPrev,
198
+ prevTags,
199
+ before: candidates.length,
200
+ after: finalFiltered.length,
201
+ dropped: filtered.length > 0 ? dropped : [],
202
+ },
203
+ };
204
204
  }
205
205
 
206
206
  // ─── Stage 2: LM Re-ranking ────────────────────────────────
207
- function getLmScore(
208
- word: string,
209
- lmLogits: Record<string, number> | null
210
- ): number {
211
- if (!lmLogits) return 0;
212
-
213
- // Look up the word directly! No more tokens.
214
- const val = lmLogits[word.toLowerCase()];
215
- if (typeof val === 'number') {
216
- return val;
217
- }
218
- return 0;
207
+ function getLmScore(word: string, lmLogits: Record<string, number> | null): number {
208
+ if (!lmLogits) return 0;
209
+
210
+ // Look up the word directly! No more tokens.
211
+ const val = lmLogits[word.toLowerCase()];
212
+ if (typeof val === 'number') {
213
+ return val;
214
+ }
215
+ return 0;
219
216
  }
220
217
 
221
218
  // ─── Public API ─────────────────────────────────────────────
222
219
 
223
220
  export interface RankCandidatesResult {
224
- candidates: ScoredCandidate[];
225
- grammarMeta: GrammarFilterMeta | null;
221
+ candidates: ScoredCandidate[];
222
+ grammarMeta: GrammarFilterMeta | null;
226
223
  }
227
224
 
228
225
  export function rankCandidates(
229
- candidates: ScoringCandidate[],
230
- contextVector: Float32Array | null,
231
- getWordVector: (word: string) => Float32Array | null,
232
- lmLogits: Record<string, number> | null,
233
- maxTenantFreq: number,
234
- previousWord: string,
226
+ candidates: ScoringCandidate[],
227
+ contextVector: Float32Array | null,
228
+ getWordVector: (word: string) => Float32Array | null,
229
+ lmLogits: Record<string, number> | null,
230
+ maxTenantFreq: number,
231
+ previousWord: string,
235
232
  ): RankCandidatesResult {
236
- // Stage 1
237
- const stage1Results = candidates.map((candidate) => {
238
- const { semanticScore, freqScore, stage1Score } = scoreStage1(
239
- candidate,
240
- contextVector,
241
- getWordVector,
242
- maxTenantFreq,
243
- );
244
- return { candidate, semanticScore, freqScore, stage1Score };
245
- });
246
-
247
- const stage1Survivors = stage1Results.filter(entry => entry.stage1Score >= MIN_STAGE1_SCORE);
248
-
249
- // Grammar Filter
250
- const { filtered, grammarMeta } = applyGrammarFilter(stage1Survivors, previousWord);
251
-
252
- // Stage 2 + final assembly
253
- let lmMax = 0;
254
- if (lmLogits && Object.keys(lmLogits).length > 0) {
255
- const values = Object.values(lmLogits);
256
- lmMax = Math.max(...values);
257
- }
258
-
259
- const scored: ScoredCandidate[] = filtered.map((entry) => {
260
- let lmScore = 0;
261
- let finalScore = entry.stage1Score;
262
-
263
- if (lmLogits && Object.keys(lmLogits).length > 0) {
264
- const rawLm = getLmScore(entry.candidate.word, lmLogits);
265
-
266
- if (rawLm !== 0) {
267
- // The word was in the top_k! Score it normally.
268
- const logitDiff = Math.log(rawLm) - Math.log(lmMax);
269
- lmScore = Math.exp(logitDiff);
270
- } else {
271
- lmScore = 0.05;
272
- }
273
-
274
- finalScore = STAGE1_WEIGHT * entry.stage1Score + STAGE2_WEIGHT * lmScore;
275
- }
276
-
277
- return {
278
- word: entry.candidate.word,
279
- freqScore: entry.freqScore,
280
- semanticScore: entry.semanticScore,
281
- lmScore,
282
- finalScore,
283
- };
284
- });
285
-
233
+ // Stage 1
234
+ const stage1Results = candidates.map((candidate) => {
235
+ const { semanticScore, freqScore, stage1Score } = scoreStage1(
236
+ candidate,
237
+ contextVector,
238
+ getWordVector,
239
+ maxTenantFreq,
240
+ );
241
+ return { candidate, semanticScore, freqScore, stage1Score };
242
+ });
243
+
244
+ const stage1Survivors = stage1Results.filter((entry) => entry.stage1Score >= MIN_STAGE1_SCORE);
245
+
246
+ // Grammar Filter
247
+ const { filtered, grammarMeta } = applyGrammarFilter(stage1Survivors, previousWord);
248
+
249
+ // Stage 2 + final assembly
250
+ let lmMax = 0;
251
+ if (lmLogits && Object.keys(lmLogits).length > 0) {
252
+ const values = Object.values(lmLogits);
253
+ lmMax = Math.max(...values);
254
+ }
255
+
256
+ const scored: ScoredCandidate[] = filtered.map((entry) => {
257
+ let lmScore = 0;
258
+ let finalScore = entry.stage1Score;
259
+
260
+ if (lmLogits && Object.keys(lmLogits).length > 0) {
261
+ const rawLm = getLmScore(entry.candidate.word, lmLogits);
262
+
263
+ if (rawLm !== 0) {
264
+ // The word was in the top_k! Score it normally.
265
+ const logitDiff = Math.log(rawLm) - Math.log(lmMax);
266
+ lmScore = Math.exp(logitDiff);
267
+ } else {
268
+ lmScore = 0.05;
269
+ }
270
+
271
+ finalScore = STAGE1_WEIGHT * entry.stage1Score + STAGE2_WEIGHT * lmScore;
272
+ }
273
+
274
+ return {
275
+ word: entry.candidate.word,
276
+ freqScore: entry.freqScore,
277
+ semanticScore: entry.semanticScore,
278
+ lmScore,
279
+ finalScore,
280
+ };
281
+ });
282
+
286
283
  scored.sort((a, b) => {
287
- if (b.finalScore !== a.finalScore) {
288
- return b.finalScore - a.finalScore;
289
- }
290
- return a.word.length - b.word.length;
291
- });
292
-
293
- return { candidates: scored, grammarMeta };
294
- }
284
+ if (b.finalScore !== a.finalScore) {
285
+ return b.finalScore - a.finalScore;
286
+ }
287
+ return a.word.length - b.word.length;
288
+ });
289
+
290
+ return { candidates: scored, grammarMeta };
291
+ }
@@ -714,20 +714,31 @@ interface VocabularyJson {
714
714
  >;
715
715
  }
716
716
 
717
- export const loadVectorsAsync = async (options?: { vectorsUrl?: string }): Promise<void> => {
717
+ export const loadVectorsAsync = async (options?: {
718
+ getBinaryUrl?: () => Promise<string>;
719
+ }): Promise<void> => {
718
720
  if (vectorStore || vectorsLoadStarted) {
719
721
  return;
720
722
  }
721
- if (!options?.vectorsUrl) {
723
+ if (!options?.getBinaryUrl) {
722
724
  // eslint-disable-next-line no-console
723
725
  console.warn(
724
- '[text-predictor] loadVectorsAsync called without a vectorsUrl — vectors will not load. Pass vectorsUrl via plugin options.',
726
+ '[text-predictor] loadVectorsAsync called without a getBinaryUrl — vectors will not load. Pass getVectorsBinaryUrl via plugin options.',
725
727
  );
726
728
  return;
727
729
  }
728
730
  vectorsLoadStarted = true;
729
731
 
730
- const url = options.vectorsUrl;
732
+ let url: string;
733
+ try {
734
+ url = await options.getBinaryUrl();
735
+ } catch (e) {
736
+ vectorsLoadStarted = false;
737
+ // eslint-disable-next-line no-console
738
+ console.warn('[text-predictor] Failed to resolve vectors URL:', e);
739
+ return;
740
+ }
741
+
731
742
  try {
732
743
  const res = await fetch(url);
733
744
  if (!res.ok) {
package/tsconfig.app.json CHANGED
@@ -1,6 +1,5 @@
1
1
  {
2
2
  "extends": "../../../tsconfig.base.json",
3
- "files": ["./build/url-module.d.ts"],
4
3
  "include": ["./src/**/*.ts", "./src/**/*.tsx", "./src/**/*.json"],
5
4
  "exclude": [
6
5
  "**/docs/**/*",
package/tsconfig.json CHANGED
@@ -1,6 +1,5 @@
1
1
  {
2
2
  "extends": "../../../tsconfig.json",
3
- "files": ["./build/url-module.d.ts"],
4
3
  "include": [
5
4
  "src/**/*.ts",
6
5
  "src/**/*.tsx"
@@ -1,8 +0,0 @@
1
- // Ambient declaration for Atlaspack/Parcel url: prefix imports.
2
- // The url: prefix instructs the bundler to emit the referenced file as a
3
- // content-hashed asset and return its public URL as a string.
4
- // For rspack, this is handled via staticAssetsLoader config.
5
- declare module 'url:*' {
6
- const url: string;
7
- export default url;
8
- }