dsh-plugin-term-dictionary 0.0.0-stage → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,398 @@
1
+ "use strict";
2
+
3
+ /**
4
+ * The term detector: decides which words in a message are terminology and how
5
+ * sure it is.
6
+ *
7
+ * It combines four evidence sources, strongest first:
8
+ * 1. the user's own dictionary (an explicit entry is never questioned);
9
+ * 2. the built-in glossary, which also supplies a ready-made explanation, so
10
+ * the plugin works with no model call and no network;
11
+ * 3. the spelling of the word itself (`camelCase`, `ALL_CAPS`, `snake_case`),
12
+ * which is how a term nobody has ever catalogued still gets noticed;
13
+ * 4. technical morphology (`-ology`, `-ization`, `meta-`, `poly-`).
14
+ *
15
+ * Dictionary and glossary matching runs over raw text rather than tokens, which
16
+ * is what makes multi-word English phrases (`event sourcing`) and spacer-less
17
+ * Chinese terms (`幂等`) work through one code path. Everything is synchronous
18
+ * and pure, so it can run on every rendered message without a network round
19
+ * trip.
20
+ */
21
+
22
+ const core = require("./core.js");
23
+
24
+ const {
25
+ normalizeTerm,
26
+ termId,
27
+ contextAround,
28
+ isTermShaped,
29
+ looksLikeIdentifier,
30
+ isAcronym,
31
+ hasTechnicalMorphology,
32
+ tokenize
33
+ } = core;
34
+
35
+ /** Evidence weights, also the confidence reported per occurrence. */
36
+ const CONFIDENCE = {
37
+ /** The user curated this term. */
38
+ dictionary: 1,
39
+ /** The built-in glossary knows this term. */
40
+ glossary: 0.8,
41
+ /** An acronym or initialism: `API`, `SLO`. */
42
+ acronym: 0.55,
43
+ /** Identifier spelling: `eventSourcing`, `snake_case`. */
44
+ identifier: 0.5,
45
+ /** A long English word with technical morphology. */
46
+ morphology: 0.45
47
+ };
48
+
49
+ /** Longest known term, in characters, the substring matcher will look for. */
50
+ const MAX_TERM_LENGTH = 48;
51
+
52
+ /**
53
+ * Words that are common enough to refuse even when they arrive capitalized,
54
+ * which is how they appear at the start of most sentences. Kept small on
55
+ * purpose: the stopword list already covers the real prose, and this only has to
56
+ * catch the sentence-initial cases the lowercase test cannot see.
57
+ */
58
+ const CAPITALIZED_STOPWORDS = new Set([
59
+ "a", "an", "and", "as", "at", "be", "but", "by", "can", "do", "for", "from",
60
+ "he", "her", "here", "his", "how", "i", "if", "in", "is", "it", "its", "let",
61
+ "my", "no", "not", "of", "on", "or", "our", "out", "she", "so", "that", "the",
62
+ "their", "them", "then", "there", "these", "they", "this", "those", "to", "up",
63
+ "us", "was", "we", "were", "what", "when", "where", "which", "who", "why",
64
+ "will", "with", "you", "your"
65
+ ]);
66
+
67
+ /**
68
+ * The detector. Construct one per (glossary, dictionary) generation and reuse it
69
+ * for every message; {@link TermDetector#scan} is stateless.
70
+ */
71
+ class TermDetector {
72
+ /**
73
+ * Build the term index the detector matches against.
74
+ *
75
+ * @param options - `lexicon` is the built-in map (key -> `{ zh, domain, gloss }`),
76
+ * `entries` is the user's dictionary, and `stopwords` holds the two word
77
+ * lists the detector must ignore.
78
+ */
79
+ constructor(options) {
80
+ const config = options ?? {};
81
+ const lexicon = config.lexicon ?? {};
82
+ const stopwords = config.stopwords ?? {};
83
+ const entries = Array.isArray(config.entries) ? config.entries : [];
84
+
85
+ /** Known glossary records, keyed by normalized term. */
86
+ const glossary = new Map();
87
+ /** Match keys ordered longest first, so a phrase wins over the words in it. */
88
+ const keys = [];
89
+ for (const [term, value] of Object.entries(lexicon)) {
90
+ const key = normalizeTerm(term);
91
+ if (key === "" || key.length > MAX_TERM_LENGTH) continue;
92
+ glossary.set(key, {
93
+ key,
94
+ zh: typeof value?.zh === "string" ? value.zh : "",
95
+ domain: typeof value?.domain === "string" ? value.domain : "",
96
+ gloss: typeof value?.gloss === "string" ? value.gloss : ""
97
+ });
98
+ keys.push(key);
99
+ }
100
+
101
+ /** User dictionary projected for highlighting, keyed by term and alias. */
102
+ const known = new Map();
103
+ for (const entry of entries) {
104
+ if (entry === null || typeof entry !== "object") continue;
105
+ const key = normalizeTerm(entry.term);
106
+ if (key === "") continue;
107
+ const record = {
108
+ term: typeof entry.term === "string" && entry.term.trim() !== "" ? entry.term.trim() : key,
109
+ id: typeof entry.id === "string" && entry.id !== "" ? entry.id : termId(key),
110
+ key,
111
+ entry
112
+ };
113
+ known.set(key, record);
114
+ for (const alias of Array.isArray(entry.aliases) ? entry.aliases : []) {
115
+ const aliasKey = normalizeTerm(alias);
116
+ if (aliasKey !== "" && !known.has(aliasKey)) known.set(aliasKey, record);
117
+ }
118
+ }
119
+
120
+ // The user's own terms must be MATCHED, not merely recorded.
121
+ //
122
+ // `scan` finds occurrences by searching the text for each key in `keys`, and the
123
+ // loop above only ever filled `keys` from the lexicon. Every collected term that is
124
+ // not also a built-in glossary word was therefore invisible to the matcher: a
125
+ // dictionary entry for `WriteAheadLog` came back as an unknown `identifier`
126
+ // candidate (`known: false`) instead of a dictionary hit. Three user-visible
127
+ // consequences, all reported from the running plugin:
128
+ //
129
+ // - the term was never marked in the transcript ("没有变成词条块");
130
+ // - hovering it offered no explanation;
131
+ // - clicking it offered to CREATE an entry that already existed.
132
+ //
133
+ // Only entries that are also lexicon words (like `quorum`) worked, which is why
134
+ // this survived every test: the tests happened to use glossary terms.
135
+ for (const key of known.keys()) {
136
+ if (key.length > MAX_TERM_LENGTH) continue;
137
+ if (!keys.includes(key)) keys.push(key);
138
+ }
139
+ keys.sort((left, right) => right.length - left.length);
140
+
141
+ this.glossary = glossary;
142
+ this.keys = keys;
143
+ this.known = known;
144
+ this.stopEn = new Set((stopwords.EN ?? []).map((word) => String(word).toLowerCase()));
145
+ this.stopZh = new Set(stopwords.ZH ?? []);
146
+ }
147
+
148
+ /**
149
+ * Glossary knowledge for one normalized term, if the built-in lexicon has it.
150
+ * @param key - normalized term text.
151
+ * @returns the glossary record, or undefined.
152
+ */
153
+ glossaryOf(key) {
154
+ return this.glossary.get(key);
155
+ }
156
+
157
+ /**
158
+ * The user's own entry covering this term or alias.
159
+ * @param key - normalized term text.
160
+ * @returns the known-term record, or undefined.
161
+ */
162
+ knownOf(key) {
163
+ return this.known.get(key);
164
+ }
165
+
166
+ /**
167
+ * Whether the word is common enough that it must never count as jargon on its
168
+ * own. A dictionary entry or glossary hit outranks this.
169
+ * @param token - a tokenized word.
170
+ * @returns true when the word carries no terminology signal.
171
+ */
172
+ isCommon(token) {
173
+ if (core.hasCjk(token.raw)) return this.stopZh.has(token.raw);
174
+ return token.raw === token.lower && this.stopEn.has(token.lower);
175
+ }
176
+
177
+ /**
178
+ * Classify one token that is not covered by the dictionary or the glossary.
179
+ * @param token - a tokenized word.
180
+ * @returns the confidence and evidence name, or null for ordinary prose.
181
+ */
182
+ classify(token) {
183
+ const { raw } = token;
184
+ if (!isTermShaped(raw)) return null;
185
+ if (this.isCommon(token)) return null;
186
+ if (core.hasCjk(raw)) return null;
187
+ if (looksLikeIdentifier(raw)) return { confidence: CONFIDENCE.identifier, source: "identifier" };
188
+ if (isAcronym(raw)) return { confidence: CONFIDENCE.acronym, source: "acronym" };
189
+ if (raw === token.lower && hasTechnicalMorphology(token.lower)) return { confidence: CONFIDENCE.morphology, source: "morphology" };
190
+ return null;
191
+ }
192
+
193
+ /**
194
+ * Detect every term occurrence in one message.
195
+ *
196
+ * Matches never overlap: a longer dictionary or glossary term wins over the
197
+ * words inside it, and later candidates that intersect an accepted range are
198
+ * dropped. The result is ordered by position, which is what the renderer
199
+ * walks.
200
+ *
201
+ * @param text - the message's plain text.
202
+ * @param options - `includeCandidates` (default true) controls whether weak
203
+ * evidence (spelling and morphology) is reported alongside real hits.
204
+ * @returns occurrences sorted by start offset.
205
+ */
206
+ scan(text, options) {
207
+ if (typeof text !== "string" || text === "") return [];
208
+ const includeCandidates = options?.includeCandidates !== false;
209
+ const accepted = [];
210
+ const claimed = [];
211
+ const overlaps = (start, end) => claimed.some((range) => start < range.end && end > range.start);
212
+ const accept = (occurrence) => {
213
+ accepted.push(occurrence);
214
+ claimed.push({ start: occurrence.start, end: occurrence.end });
215
+ };
216
+
217
+ // 1. Known terms, longest first, so `event sourcing` is one hit and the
218
+ // words inside it are never claimed separately.
219
+ const lowerText = text.toLowerCase();
220
+ for (const key of this.keys) {
221
+ if (key.length > text.length) continue;
222
+ let from = 0;
223
+ for (;;) {
224
+ const at = lowerText.indexOf(key, from);
225
+ if (at < 0) break;
226
+ from = at + 1;
227
+ const end = at + key.length;
228
+ if (!boundaryOk(text, at, end, hasCjkKey(key))) continue;
229
+ if (overlaps(at, end)) continue;
230
+ const occurrence = this.occurrenceFor(at, end, key, text);
231
+ if (occurrence !== null) accept(occurrence);
232
+ }
233
+ }
234
+
235
+ // 2. Token-level evidence for everything no known term claimed.
236
+ for (const token of tokenize(text)) {
237
+ if (overlaps(token.start, token.end)) continue;
238
+ const classified = includeCandidates ? this.classify(token) : null;
239
+ if (classified === null) continue;
240
+ accept({
241
+ term: token.raw,
242
+ key: token.lower,
243
+ id: termId(token.raw),
244
+ start: token.start,
245
+ end: token.end,
246
+ confidence: classified.confidence,
247
+ source: classified.source,
248
+ known: false
249
+ });
250
+ }
251
+
252
+ accepted.sort((left, right) => left.start - right.start || right.end - left.end);
253
+ return accepted;
254
+ }
255
+
256
+ /**
257
+ * Build one occurrence for a known-term match, preferring the user's own
258
+ * dictionary over the built-in glossary.
259
+ * @param start - match start offset.
260
+ * @param end - match end offset.
261
+ * @param key - the normalized term that matched.
262
+ * @param text - the containing message text.
263
+ * @returns the occurrence, or null when neither source covers the key.
264
+ */
265
+ occurrenceFor(start, end, key, text) {
266
+ const known = this.knownOf(key);
267
+ if (known !== undefined) {
268
+ return {
269
+ term: known.term,
270
+ key: known.key,
271
+ id: known.id,
272
+ start,
273
+ end,
274
+ confidence: CONFIDENCE.dictionary,
275
+ source: "dictionary",
276
+ known: true
277
+ };
278
+ }
279
+ const glossary = this.glossaryOf(key);
280
+ if (glossary === undefined) return null;
281
+ return {
282
+ term: text.slice(start, end),
283
+ key,
284
+ id: termId(key),
285
+ start,
286
+ end,
287
+ confidence: CONFIDENCE.glossary,
288
+ source: "glossary",
289
+ known: true
290
+ };
291
+ }
292
+
293
+ /**
294
+ * Whether a bare word is a plausible *new* dictionary term, used when a click
295
+ * lands on text the scan did not flag.
296
+ *
297
+ * Without this gate, clicking any ordinary word would offer to create an entry
298
+ * for it. English is judged by the same shape and morphology rules the scan
299
+ * uses, so a plain word like "folder" is refused while `runInTransaction` and
300
+ * `idempotency` are accepted. Chinese has no such signal — a two-character word
301
+ * is already a plausible term — so any in-range Chinese word is accepted, and
302
+ * the built-in glossary decides what is *known* rather than what is offered.
303
+ *
304
+ * @param raw - the word as it appeared.
305
+ * @returns true when the word may become a new entry.
306
+ */
307
+ isCandidateWord(raw) {
308
+ if (typeof raw !== "string") return false;
309
+ if (core.hasCjk(raw)) return raw.length >= 2 && raw.length <= 8;
310
+ if (raw.length < 3 || raw.length > 48) return false;
311
+ if (!/[A-Za-z]/.test(raw)) return false;
312
+ const lower = raw.toLowerCase();
313
+ if (this.glossary.has(lower) || this.known.has(lower)) return true;
314
+ if (looksLikeIdentifier(raw) || isAcronym(raw)) return true;
315
+ if (raw === lower && hasTechnicalMorphology(lower)) return true;
316
+ if (this.stopEn.has(lower)) return false;
317
+ // Capitalization alone is not evidence: it is also how every sentence starts,
318
+ // so accepting it would offer to define "The" and "We". What remains is a word
319
+ // with no morphological signal and no stopword entry — rare enough to be worth
320
+ // offering, though never certain.
321
+ return raw === lower ? false : !CAPITALIZED_STOPWORDS.has(lower);
322
+ }
323
+
324
+ /**
325
+ * The strongest evidence in a message that the dictionary does not have yet,
326
+ * used when the plugin offers a new entry.
327
+ * @param text - the message text.
328
+ * @returns the best unknown occurrence with its context, or null.
329
+ */
330
+ bestCandidate(text) {
331
+ const unknown = this.scan(text, { includeCandidates: true }).filter((occurrence) => !occurrence.known);
332
+ let best = null;
333
+ for (const occurrence of unknown) {
334
+ if (best === null || occurrence.confidence > best.occurrence.confidence) {
335
+ best = { occurrence, context: contextAround(text, occurrence.start, occurrence.end) };
336
+ }
337
+ }
338
+ return best;
339
+ }
340
+
341
+ /**
342
+ * Every occurrence of one exact term in a message, regardless of evidence.
343
+ * @param text - the message text.
344
+ * @param term - the term to locate.
345
+ * @returns occurrences in source order.
346
+ */
347
+ findAll(text, term) {
348
+ const key = normalizeTerm(term);
349
+ if (key === "" || typeof text !== "string") return [];
350
+ return this.scan(text, { includeCandidates: true }).filter((occurrence) => occurrence.key === key);
351
+ }
352
+ }
353
+
354
+ /** Memoized "does this key contain ideographs" test for the match loop. */
355
+ const CJK_KEY_CACHE = new Map();
356
+
357
+ /**
358
+ * Whether a normalized match key contains ideographs.
359
+ * @param key - normalized term text.
360
+ * @returns true for Chinese terms.
361
+ */
362
+ function hasCjkKey(key) {
363
+ let cached = CJK_KEY_CACHE.get(key);
364
+ if (cached === undefined) {
365
+ cached = core.hasCjk(key);
366
+ CJK_KEY_CACHE.set(key, cached);
367
+ }
368
+ return cached;
369
+ }
370
+
371
+ /**
372
+ * Whether a raw substring match is a real occurrence of the term.
373
+ *
374
+ * Latin terms must sit on token boundaries, so `act` inside `interaction` is
375
+ * not a hit. A Chinese term is exempt from the boundary test against ideograph
376
+ * neighbours: Chinese writes no spaces, so `幂等` is a genuine occurrence inside
377
+ * `幂等键` and `幂等性`, and a stricter rule would make a one-word Chinese entry
378
+ * almost unmatchable.
379
+ *
380
+ * @param text - the containing text.
381
+ * @param start - match start offset.
382
+ * @param end - match end offset.
383
+ * @param cjk - whether the matched term itself contains ideographs.
384
+ * @returns true when the match is a real occurrence.
385
+ */
386
+ function boundaryOk(text, start, end, cjk) {
387
+ if (cjk) return true;
388
+ if (start > 0 && core.isWordCharacter(text[start - 1])) return false;
389
+ if (end < text.length && core.isWordCharacter(text[end])) return false;
390
+ return true;
391
+ }
392
+
393
+ module.exports = {
394
+ CONFIDENCE,
395
+ MAX_TERM_LENGTH,
396
+ TermDetector,
397
+ boundaryOk
398
+ };