@atlaskit/editor-plugin-autocomplete 3.0.0 → 3.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +16 -0
- package/dist/cjs/pm-plugins/autocomplete-plugin.js +6 -22
- package/dist/cjs/pm-plugins/local-slow-lane-client.js +676 -148
- package/dist/cjs/pm-plugins/scoring-pipeline.js +5 -4
- package/dist/cjs/pm-plugins/text-predictor.js +152 -60
- package/dist/es2019/pm-plugins/autocomplete-plugin.js +6 -22
- package/dist/es2019/pm-plugins/local-slow-lane-client.js +493 -85
- package/dist/es2019/pm-plugins/scoring-pipeline.js +5 -4
- package/dist/es2019/pm-plugins/text-predictor.js +105 -50
- package/dist/esm/pm-plugins/autocomplete-plugin.js +6 -22
- package/dist/esm/pm-plugins/local-slow-lane-client.js +668 -144
- package/dist/esm/pm-plugins/scoring-pipeline.js +5 -4
- package/dist/esm/pm-plugins/text-predictor.js +144 -60
- package/dist/types/pm-plugins/local-slow-lane-client.d.ts +102 -15
- package/dist/types/pm-plugins/text-predictor.d.ts +1 -1
- package/dist/types-ts4.5/pm-plugins/local-slow-lane-client.d.ts +102 -15
- package/dist/types-ts4.5/pm-plugins/text-predictor.d.ts +1 -1
- package/package.json +2 -2
- package/scripts/gen_first_token_to_words.py +170 -0
- package/src/pm-plugins/autocomplete-plugin.ts +6 -16
- package/src/pm-plugins/data/first_token_to_words.json +1 -0
- package/src/pm-plugins/data/word_index_10k.json +7761 -7759
- package/src/pm-plugins/local-slow-lane-client.ts +587 -95
- package/src/pm-plugins/scoring-pipeline.ts +4 -5
- package/src/pm-plugins/text-predictor.ts +124 -45
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
Offline build script: generate `first_token_to_words.json`.
|
|
4
|
+
|
|
5
|
+
This is a ONE-TIME / build-time tool. It is never imported by the plugin and
|
|
6
|
+
never runs in CI. It reproduces the prefix-expansion map that the BE sidecar
|
|
7
|
+
builds in-memory at runtime (`cc-smarts/python-sidecar/src/causal_lm_encoder.py`
|
|
8
|
+
`CausalLMEncoder._ensure_loaded`), so the local (client-only) slow-lane client
|
|
9
|
+
can ship it as a static artifact instead of running a tokenizer in the browser.
|
|
10
|
+
|
|
11
|
+
The output maps each SmolLM2 first-token id to every L2/L3 vocabulary word whose
|
|
12
|
+
space-prefixed encoding starts with that token. The local client loads it and,
|
|
13
|
+
per inference, spreads the next-token logit mass over whole words (masked softmax
|
|
14
|
+
+ prefix expansion) to match the BE's whole-word `lm_logits` payload.
|
|
15
|
+
|
|
16
|
+
Three details MUST match the BE exactly, or the map is silently wrong:
|
|
17
|
+
1. Tokenizer = HuggingFaceTB/SmolLM2-135M (base; vocab identical to Instruct).
|
|
18
|
+
2. Leading space: encode(" " + word) — BPE tokenizes " word" != "word".
|
|
19
|
+
3. add_special_tokens=False — no BOS/EOS, we want the word's own first token.
|
|
20
|
+
|
|
21
|
+
Usage:
|
|
22
|
+
pip install transformers # torch NOT required (SmolLM2 uses a fast tokenizer)
|
|
23
|
+
python scripts/gen_first_token_to_words.py
|
|
24
|
+
|
|
25
|
+
Run it from anywhere — paths are resolved relative to this file's location.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
import os
|
|
29
|
+
import json
|
|
30
|
+
import collections
|
|
31
|
+
|
|
32
|
+
from transformers import AutoTokenizer
|
|
33
|
+
|
|
34
|
+
# Ground truth: must match the BE tokenizer (causal_lm_encoder.py line 30).
|
|
35
|
+
TOKENIZER_NAME = "HuggingFaceTB/SmolLM2-135M"
|
|
36
|
+
|
|
37
|
+
# Resolve data paths relative to this script, so cwd does not matter.
|
|
38
|
+
_SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
|
|
39
|
+
_DATA_DIR = os.path.join(_SCRIPT_DIR, "..", "src", "pm-plugins", "data")
|
|
40
|
+
L2_PATH = os.path.join(_DATA_DIR, "vocabulary_10k.json")
|
|
41
|
+
L3_PATH = os.path.join(_DATA_DIR, "l3_vocabulary.json")
|
|
42
|
+
OUT_PATH = os.path.join(_DATA_DIR, "first_token_to_words.json")
|
|
43
|
+
|
|
44
|
+
# Probe words for the post-build sanity check.
|
|
45
|
+
# Must be words that actually appear in vocabulary_10k.json (L2) or l3_vocabulary.json (L3).
|
|
46
|
+
# Common stop words like "the" / "a" are NOT in either vocabulary by design.
|
|
47
|
+
# L2 confirmed: "atlassian", "service", "product", "customer" (vocabulary_10k.json lines 3-18)
|
|
48
|
+
# L3 confirmed: "about", "search", "information", "business" (l3_vocabulary.json lines 2-15)
|
|
49
|
+
_PROBE_WORDS = ["atlassian", "service", "product", "about", "search"]
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def load_l2_words(path):
|
|
53
|
+
"""
|
|
54
|
+
Load the L2 (Atlassian-domain) vocabulary as a set of words.
|
|
55
|
+
|
|
56
|
+
The file shape is {"words": {"<word>": {freq, ...}}}, matching the BE's
|
|
57
|
+
`vocab_data.get("words", {})`. Only the keys are needed.
|
|
58
|
+
|
|
59
|
+
:params:
|
|
60
|
+
path: Absolute path to vocabulary_10k.json
|
|
61
|
+
:returns:
|
|
62
|
+
A set of L2 word strings
|
|
63
|
+
"""
|
|
64
|
+
with open(path, "r") as f:
|
|
65
|
+
data = json.load(f)
|
|
66
|
+
return set(data["words"].keys())
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def load_l3_words(path):
|
|
70
|
+
"""
|
|
71
|
+
Load the L3 (general English) vocabulary as a list of words.
|
|
72
|
+
|
|
73
|
+
The file shape is a flat JSON array of strings, matching the BE's L3 list.
|
|
74
|
+
|
|
75
|
+
:params:
|
|
76
|
+
path: Absolute path to l3_vocabulary.json
|
|
77
|
+
:returns:
|
|
78
|
+
A list of L3 word strings
|
|
79
|
+
"""
|
|
80
|
+
with open(path, "r") as f:
|
|
81
|
+
return json.load(f)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def build_first_token_map(tokenizer, words):
|
|
85
|
+
"""
|
|
86
|
+
Build the first-token-id -> [words] prefix-expansion map.
|
|
87
|
+
|
|
88
|
+
Mirrors the BE loop exactly: each word is encoded with a leading space and
|
|
89
|
+
no special tokens, and the word is bucketed under its first token id. A set
|
|
90
|
+
of words is expected so each word is processed once (L2/L3 overlap removed).
|
|
91
|
+
|
|
92
|
+
:params:
|
|
93
|
+
tokenizer: A HuggingFace tokenizer for SmolLM2
|
|
94
|
+
words: An iterable of unique words (L2 union L3)
|
|
95
|
+
:returns:
|
|
96
|
+
A dict mapping int first-token-id to a list of word strings
|
|
97
|
+
"""
|
|
98
|
+
table = collections.defaultdict(list)
|
|
99
|
+
for word in words:
|
|
100
|
+
ids = tokenizer.encode(" " + word, add_special_tokens=False)
|
|
101
|
+
if ids:
|
|
102
|
+
table[ids[0]].append(word)
|
|
103
|
+
return table
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def verify_map(tokenizer, table, probe_words):
|
|
107
|
+
"""
|
|
108
|
+
Sanity-check the generated map by confirming probe words land in the right
|
|
109
|
+
first-token bucket.
|
|
110
|
+
|
|
111
|
+
:params:
|
|
112
|
+
tokenizer: The same SmolLM2 tokenizer used to build the map
|
|
113
|
+
table: The dict mapping int first-token-id to a list of words
|
|
114
|
+
probe_words: A list of words expected to be present in the map
|
|
115
|
+
:returns:
|
|
116
|
+
None. Raises AssertionError if any probe word is missing or misplaced.
|
|
117
|
+
"""
|
|
118
|
+
for word in probe_words:
|
|
119
|
+
ids = tokenizer.encode(" " + word, add_special_tokens=False)
|
|
120
|
+
assert ids, f"Probe word '{word}' produced no tokens"
|
|
121
|
+
first_token_id = ids[0]
|
|
122
|
+
bucket = table.get(first_token_id, [])
|
|
123
|
+
assert word in bucket, (
|
|
124
|
+
f"Probe word '{word}' missing under token {first_token_id} "
|
|
125
|
+
f"(bucket head: {bucket[:5]})"
|
|
126
|
+
)
|
|
127
|
+
print(f" OK: '{word}' -> token {first_token_id} -> {bucket[:5]}...")
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def main():
|
|
131
|
+
"""
|
|
132
|
+
Generate first_token_to_words.json from the L2 and L3 vocabularies.
|
|
133
|
+
|
|
134
|
+
:params:
|
|
135
|
+
None
|
|
136
|
+
:returns:
|
|
137
|
+
None. Writes the JSON artifact to OUT_PATH and prints a summary.
|
|
138
|
+
"""
|
|
139
|
+
print(f"[gen] Loading tokenizer: {TOKENIZER_NAME} ...")
|
|
140
|
+
tokenizer = AutoTokenizer.from_pretrained(TOKENIZER_NAME)
|
|
141
|
+
|
|
142
|
+
print(f"[gen] Loading vocabularies ...")
|
|
143
|
+
l2_words = load_l2_words(L2_PATH)
|
|
144
|
+
l3_words = load_l3_words(L3_PATH)
|
|
145
|
+
all_words = l2_words.union(l3_words)
|
|
146
|
+
|
|
147
|
+
print(f"[gen] Building prefix-expansion map for {len(all_words)} words ...")
|
|
148
|
+
table = build_first_token_map(tokenizer, all_words)
|
|
149
|
+
|
|
150
|
+
# JSON object keys must be strings; the FE parses them back with Number(key).
|
|
151
|
+
out = {str(token_id): words for token_id, words in table.items()}
|
|
152
|
+
with open(OUT_PATH, "w") as f:
|
|
153
|
+
json.dump(out, f)
|
|
154
|
+
|
|
155
|
+
total_words = sum(len(v) for v in table.values())
|
|
156
|
+
print(
|
|
157
|
+
f"[gen] Mapped {len(all_words)} words ({len(l2_words)} L2) "
|
|
158
|
+
f"-> {len(table)} unique first-tokens ({total_words} word entries)"
|
|
159
|
+
)
|
|
160
|
+
|
|
161
|
+
print(f"[gen] Verifying probe words ...")
|
|
162
|
+
verify_map(tokenizer, table, _PROBE_WORDS)
|
|
163
|
+
|
|
164
|
+
out_abs = os.path.abspath(OUT_PATH)
|
|
165
|
+
size_kb = os.path.getsize(OUT_PATH) / 1024
|
|
166
|
+
print(f"[gen] Wrote {out_abs} ({size_kb:.0f} KB)")
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
if __name__ == "__main__":
|
|
170
|
+
main()
|
|
@@ -56,7 +56,6 @@ const getTextBeforeCursor = (state: EditorState): string => {
|
|
|
56
56
|
const { $from } = state.selection;
|
|
57
57
|
const maxChars = 200;
|
|
58
58
|
|
|
59
|
-
// 1. Get the perfectly flattened text of the current block up to the cursor
|
|
60
59
|
const blockNode = $from.parent;
|
|
61
60
|
const offsetInBlock = $from.parentOffset;
|
|
62
61
|
const blockText = blockNode.textContent.slice(0, offsetInBlock);
|
|
@@ -67,7 +66,7 @@ const getTextBeforeCursor = (state: EditorState): string => {
|
|
|
67
66
|
|
|
68
67
|
let fullText = blockText;
|
|
69
68
|
|
|
70
|
-
//
|
|
69
|
+
// Walk backwards through previous blocks until we have enough context.
|
|
71
70
|
let depth = $from.depth - 1;
|
|
72
71
|
|
|
73
72
|
while (fullText.length < maxChars && depth >= 0) {
|
|
@@ -309,7 +308,6 @@ export const createAutocompletePlugin = (
|
|
|
309
308
|
const { state } = view;
|
|
310
309
|
const { selection } = state;
|
|
311
310
|
|
|
312
|
-
// Only predict for cursor selections (not range selections)
|
|
313
311
|
if (!selection.empty) {
|
|
314
312
|
return;
|
|
315
313
|
}
|
|
@@ -323,12 +321,10 @@ export const createAutocompletePlugin = (
|
|
|
323
321
|
}
|
|
324
322
|
dismissedContext = null;
|
|
325
323
|
|
|
326
|
-
// Don't predict if there's not enough context
|
|
327
324
|
if (textBefore.trim().length < 3) {
|
|
328
325
|
return;
|
|
329
326
|
}
|
|
330
327
|
|
|
331
|
-
// Tier 1 prediction is synchronous -- no async needed
|
|
332
328
|
const prediction = predict(textBefore);
|
|
333
329
|
|
|
334
330
|
if (prediction && prediction.length > 0) {
|
|
@@ -402,8 +398,7 @@ export const createAutocompletePlugin = (
|
|
|
402
398
|
return { ...pluginState, ...meta };
|
|
403
399
|
}
|
|
404
400
|
|
|
405
|
-
//
|
|
406
|
-
// (new prediction will be scheduled from view.update)
|
|
401
|
+
// A new prediction is scheduled from view.update.
|
|
407
402
|
if (tr.docChanged) {
|
|
408
403
|
return {
|
|
409
404
|
...pluginState,
|
|
@@ -413,7 +408,6 @@ export const createAutocompletePlugin = (
|
|
|
413
408
|
};
|
|
414
409
|
}
|
|
415
410
|
|
|
416
|
-
// If selection changed without doc change, clear ghost text
|
|
417
411
|
if (tr.selectionSet && pluginState.ghostText) {
|
|
418
412
|
return {
|
|
419
413
|
...pluginState,
|
|
@@ -484,13 +478,11 @@ export const createAutocompletePlugin = (
|
|
|
484
478
|
return false;
|
|
485
479
|
},
|
|
486
480
|
focus: () => {
|
|
487
|
-
|
|
488
|
-
loadDefaultVocabulary();
|
|
489
|
-
} catch (error) {
|
|
481
|
+
loadDefaultVocabulary().catch((error) => {
|
|
490
482
|
logException(error as Error, {
|
|
491
483
|
location: 'editor-plugin-autocomplete/loadDefaultVocabulary',
|
|
492
484
|
});
|
|
493
|
-
}
|
|
485
|
+
});
|
|
494
486
|
loadVectorsAsync({ getBinaryUrl: options?.getVectorsBinaryUrl }).catch((error) => {
|
|
495
487
|
logException(error as Error, {
|
|
496
488
|
location: 'editor-plugin-autocomplete/loadVectorsAsync',
|
|
@@ -533,12 +525,10 @@ export const createAutocompletePlugin = (
|
|
|
533
525
|
if (justAccepted) {
|
|
534
526
|
justAccepted = false;
|
|
535
527
|
|
|
536
|
-
//
|
|
537
|
-
//
|
|
538
|
-
// block and abort until the user actually types a new character!
|
|
528
|
+
// Snapshot the post-acceptance text so follow-up transactions hit
|
|
529
|
+
// the dismissedContext guard and abort until the user types again.
|
|
539
530
|
dismissedContext = getTextBeforeCursor(view.state);
|
|
540
531
|
|
|
541
|
-
// Also clear any pending debounce timers from before the acceptance
|
|
542
532
|
if (debounceTimer) {
|
|
543
533
|
clearTimeout(debounceTimer);
|
|
544
534
|
}
|