@atlaskit/editor-plugin-autocomplete 3.0.0 → 3.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,170 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ Offline build script: generate `first_token_to_words.json`.
4
+
5
+ This is a ONE-TIME / build-time tool. It is never imported by the plugin and
6
+ never runs in CI. It reproduces the prefix-expansion map that the BE sidecar
7
+ builds in-memory at runtime (`cc-smarts/python-sidecar/src/causal_lm_encoder.py`
8
+ `CausalLMEncoder._ensure_loaded`), so the local (client-only) slow-lane client
9
+ can ship it as a static artifact instead of running a tokenizer in the browser.
10
+
11
+ The output maps each SmolLM2 first-token id to every L2/L3 vocabulary word whose
12
+ space-prefixed encoding starts with that token. The local client loads it and,
13
+ per inference, spreads the next-token logit mass over whole words (masked softmax
14
+ + prefix expansion) to match the BE's whole-word `lm_logits` payload.
15
+
16
+ Three details MUST match the BE exactly, or the map is silently wrong:
17
+ 1. Tokenizer = HuggingFaceTB/SmolLM2-135M (base; vocab identical to Instruct).
18
+ 2. Leading space: encode(" " + word) — BPE tokenizes " word" != "word".
19
+ 3. add_special_tokens=False — no BOS/EOS, we want the word's own first token.
20
+
21
+ Usage:
22
+ pip install transformers # torch NOT required (SmolLM2 uses a fast tokenizer)
23
+ python scripts/gen_first_token_to_words.py
24
+
25
+ Run it from anywhere — paths are resolved relative to this file's location.
26
+ """
27
+
28
+ import os
29
+ import json
30
+ import collections
31
+
32
+ from transformers import AutoTokenizer
33
+
34
+ # Ground truth: must match the BE tokenizer (causal_lm_encoder.py line 30).
35
+ TOKENIZER_NAME = "HuggingFaceTB/SmolLM2-135M"
36
+
37
+ # Resolve data paths relative to this script, so cwd does not matter.
38
+ _SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
39
+ _DATA_DIR = os.path.join(_SCRIPT_DIR, "..", "src", "pm-plugins", "data")
40
+ L2_PATH = os.path.join(_DATA_DIR, "vocabulary_10k.json")
41
+ L3_PATH = os.path.join(_DATA_DIR, "l3_vocabulary.json")
42
+ OUT_PATH = os.path.join(_DATA_DIR, "first_token_to_words.json")
43
+
44
+ # Probe words for the post-build sanity check.
45
+ # Must be words that actually appear in vocabulary_10k.json (L2) or l3_vocabulary.json (L3).
46
+ # Common stop words like "the" / "a" are NOT in either vocabulary by design.
47
+ # L2 confirmed: "atlassian", "service", "product", "customer" (vocabulary_10k.json lines 3-18)
48
+ # L3 confirmed: "about", "search", "information", "business" (l3_vocabulary.json lines 2-15)
49
+ _PROBE_WORDS = ["atlassian", "service", "product", "about", "search"]
50
+
51
+
52
+ def load_l2_words(path):
53
+ """
54
+ Load the L2 (Atlassian-domain) vocabulary as a set of words.
55
+
56
+ The file shape is {"words": {"<word>": {freq, ...}}}, matching the BE's
57
+ `vocab_data.get("words", {})`. Only the keys are needed.
58
+
59
+ :params:
60
+ path: Absolute path to vocabulary_10k.json
61
+ :returns:
62
+ A set of L2 word strings
63
+ """
64
+ with open(path, "r") as f:
65
+ data = json.load(f)
66
+ return set(data["words"].keys())
67
+
68
+
69
+ def load_l3_words(path):
70
+ """
71
+ Load the L3 (general English) vocabulary as a list of words.
72
+
73
+ The file shape is a flat JSON array of strings, matching the BE's L3 list.
74
+
75
+ :params:
76
+ path: Absolute path to l3_vocabulary.json
77
+ :returns:
78
+ A list of L3 word strings
79
+ """
80
+ with open(path, "r") as f:
81
+ return json.load(f)
82
+
83
+
84
+ def build_first_token_map(tokenizer, words):
85
+ """
86
+ Build the first-token-id -> [words] prefix-expansion map.
87
+
88
+ Mirrors the BE loop exactly: each word is encoded with a leading space and
89
+ no special tokens, and the word is bucketed under its first token id. A set
90
+ of words is expected so each word is processed once (L2/L3 overlap removed).
91
+
92
+ :params:
93
+ tokenizer: A HuggingFace tokenizer for SmolLM2
94
+ words: An iterable of unique words (L2 union L3)
95
+ :returns:
96
+ A dict mapping int first-token-id to a list of word strings
97
+ """
98
+ table = collections.defaultdict(list)
99
+ for word in words:
100
+ ids = tokenizer.encode(" " + word, add_special_tokens=False)
101
+ if ids:
102
+ table[ids[0]].append(word)
103
+ return table
104
+
105
+
106
+ def verify_map(tokenizer, table, probe_words):
107
+ """
108
+ Sanity-check the generated map by confirming probe words land in the right
109
+ first-token bucket.
110
+
111
+ :params:
112
+ tokenizer: The same SmolLM2 tokenizer used to build the map
113
+ table: The dict mapping int first-token-id to a list of words
114
+ probe_words: A list of words expected to be present in the map
115
+ :returns:
116
+ None. Raises AssertionError if any probe word is missing or misplaced.
117
+ """
118
+ for word in probe_words:
119
+ ids = tokenizer.encode(" " + word, add_special_tokens=False)
120
+ assert ids, f"Probe word '{word}' produced no tokens"
121
+ first_token_id = ids[0]
122
+ bucket = table.get(first_token_id, [])
123
+ assert word in bucket, (
124
+ f"Probe word '{word}' missing under token {first_token_id} "
125
+ f"(bucket head: {bucket[:5]})"
126
+ )
127
+ print(f" OK: '{word}' -> token {first_token_id} -> {bucket[:5]}...")
128
+
129
+
130
+ def main():
131
+ """
132
+ Generate first_token_to_words.json from the L2 and L3 vocabularies.
133
+
134
+ :params:
135
+ None
136
+ :returns:
137
+ None. Writes the JSON artifact to OUT_PATH and prints a summary.
138
+ """
139
+ print(f"[gen] Loading tokenizer: {TOKENIZER_NAME} ...")
140
+ tokenizer = AutoTokenizer.from_pretrained(TOKENIZER_NAME)
141
+
142
+ print(f"[gen] Loading vocabularies ...")
143
+ l2_words = load_l2_words(L2_PATH)
144
+ l3_words = load_l3_words(L3_PATH)
145
+ all_words = l2_words.union(l3_words)
146
+
147
+ print(f"[gen] Building prefix-expansion map for {len(all_words)} words ...")
148
+ table = build_first_token_map(tokenizer, all_words)
149
+
150
+ # JSON object keys must be strings; the FE parses them back with Number(key).
151
+ out = {str(token_id): words for token_id, words in table.items()}
152
+ with open(OUT_PATH, "w") as f:
153
+ json.dump(out, f)
154
+
155
+ total_words = sum(len(v) for v in table.values())
156
+ print(
157
+ f"[gen] Mapped {len(all_words)} words ({len(l2_words)} L2) "
158
+ f"-> {len(table)} unique first-tokens ({total_words} word entries)"
159
+ )
160
+
161
+ print(f"[gen] Verifying probe words ...")
162
+ verify_map(tokenizer, table, _PROBE_WORDS)
163
+
164
+ out_abs = os.path.abspath(OUT_PATH)
165
+ size_kb = os.path.getsize(OUT_PATH) / 1024
166
+ print(f"[gen] Wrote {out_abs} ({size_kb:.0f} KB)")
167
+
168
+
169
+ if __name__ == "__main__":
170
+ main()
@@ -56,7 +56,6 @@ const getTextBeforeCursor = (state: EditorState): string => {
56
56
  const { $from } = state.selection;
57
57
  const maxChars = 200;
58
58
 
59
- // 1. Get the perfectly flattened text of the current block up to the cursor
60
59
  const blockNode = $from.parent;
61
60
  const offsetInBlock = $from.parentOffset;
62
61
  const blockText = blockNode.textContent.slice(0, offsetInBlock);
@@ -67,7 +66,7 @@ const getTextBeforeCursor = (state: EditorState): string => {
67
66
 
68
67
  let fullText = blockText;
69
68
 
70
- // 2. Walk backwards through previous blocks
69
+ // Walk backwards through previous blocks until we have enough context.
71
70
  let depth = $from.depth - 1;
72
71
 
73
72
  while (fullText.length < maxChars && depth >= 0) {
@@ -309,7 +308,6 @@ export const createAutocompletePlugin = (
309
308
  const { state } = view;
310
309
  const { selection } = state;
311
310
 
312
- // Only predict for cursor selections (not range selections)
313
311
  if (!selection.empty) {
314
312
  return;
315
313
  }
@@ -323,12 +321,10 @@ export const createAutocompletePlugin = (
323
321
  }
324
322
  dismissedContext = null;
325
323
 
326
- // Don't predict if there's not enough context
327
324
  if (textBefore.trim().length < 3) {
328
325
  return;
329
326
  }
330
327
 
331
- // Tier 1 prediction is synchronous -- no async needed
332
328
  const prediction = predict(textBefore);
333
329
 
334
330
  if (prediction && prediction.length > 0) {
@@ -402,8 +398,7 @@ export const createAutocompletePlugin = (
402
398
  return { ...pluginState, ...meta };
403
399
  }
404
400
 
405
- // If the document changed, clear the ghost text
406
- // (new prediction will be scheduled from view.update)
401
+ // A new prediction is scheduled from view.update.
407
402
  if (tr.docChanged) {
408
403
  return {
409
404
  ...pluginState,
@@ -413,7 +408,6 @@ export const createAutocompletePlugin = (
413
408
  };
414
409
  }
415
410
 
416
- // If selection changed without doc change, clear ghost text
417
411
  if (tr.selectionSet && pluginState.ghostText) {
418
412
  return {
419
413
  ...pluginState,
@@ -484,13 +478,11 @@ export const createAutocompletePlugin = (
484
478
  return false;
485
479
  },
486
480
  focus: () => {
487
- try {
488
- loadDefaultVocabulary();
489
- } catch (error) {
481
+ loadDefaultVocabulary().catch((error) => {
490
482
  logException(error as Error, {
491
483
  location: 'editor-plugin-autocomplete/loadDefaultVocabulary',
492
484
  });
493
- }
485
+ });
494
486
  loadVectorsAsync({ getBinaryUrl: options?.getVectorsBinaryUrl }).catch((error) => {
495
487
  logException(error as Error, {
496
488
  location: 'editor-plugin-autocomplete/loadVectorsAsync',
@@ -533,12 +525,10 @@ export const createAutocompletePlugin = (
533
525
  if (justAccepted) {
534
526
  justAccepted = false;
535
527
 
536
- // ✨ THE FIX: Memorize the text state right after acceptance.
537
- // Any follow-up transactions will hit the 'dismissedContext'
538
- // block and abort until the user actually types a new character!
528
+ // Snapshot the post-acceptance text so follow-up transactions hit
529
+ // the dismissedContext guard and abort until the user types again.
539
530
  dismissedContext = getTextBeforeCursor(view.state);
540
531
 
541
- // Also clear any pending debounce timers from before the acceptance
542
532
  if (debounceTimer) {
543
533
  clearTimeout(debounceTimer);
544
534
  }