dsh-plugin-term-dictionary 0.0.0-stage → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,309 @@
1
+ "use strict";
2
+
3
+ /**
4
+ * Environment-free text primitives shared by the lexicon engine, the store, and
5
+ * the browser bundle.
6
+ *
7
+ * Everything here is pure: no DOM, no `node:` imports, no globals beyond the
8
+ * language built-ins. Both the browser bundle (through the module loader) and
9
+ * the Node-side tests load this same file, so a rule proved here holds in the
10
+ * page.
11
+ */
12
+
13
+ //#region character classes
14
+
15
+ /** Code point of the first CJK Unified Ideograph. */
16
+ const CJK_MIN = 0x3400;
17
+ /** Code point just past the last common CJK Unified Ideograph. */
18
+ const CJK_MAX = 0x9fff;
19
+
20
+ /**
21
+ * Whether one code point is a CJK ideograph.
22
+ * @param code - a UTF-16 code unit.
23
+ * @returns true for ideographs a Chinese dictionary could hold.
24
+ */
25
+ function isCjkCode(code) {
26
+ return code >= CJK_MIN && code <= CJK_MAX;
27
+ }
28
+
29
+ /** CJK punctuation and full-width symbols: sentence boundaries for Chinese text. */
30
+ const CJK_PUNCTUATION = ",。!?;:、()《》〈〉「」『』【】〔〕—…·~ ";
31
+
32
+ /** ASCII punctuation that ends an English sentence. */
33
+ const ASCII_SENTENCE_END = ".!?\n";
34
+
35
+ /**
36
+ * Whether one character can sit inside a term (letters, digits, connectors and
37
+ * ideographs). A term boundary is any position where this is false.
38
+ * @param character - a single code unit.
39
+ * @returns true when the character may belong to a word.
40
+ */
41
+ function isWordCharacter(character) {
42
+ if (character === "-" || character === "_") return true;
43
+ const code = character.charCodeAt(0);
44
+ if (isCjkCode(code)) return true;
45
+ return (
46
+ (code >= 0x30 && code <= 0x39) ||
47
+ (code >= 0x41 && code <= 0x5a) ||
48
+ (code >= 0x61 && code <= 0x7a)
49
+ );
50
+ }
51
+
52
+ /**
53
+ * Whether a string contains at least one CJK ideograph.
54
+ * @param text - candidate text.
55
+ * @returns true for Chinese text.
56
+ */
57
+ function hasCjk(text) {
58
+ for (const character of text) {
59
+ if (isCjkCode(character.codePointAt(0))) return true;
60
+ }
61
+ return false;
62
+ }
63
+
64
+ //#endregion
65
+
66
+ //#region tokenization
67
+
68
+ /**
69
+ * Split text into candidate tokens with their offsets.
70
+ *
71
+ * A token is a maximal run of letters, digits, `-` and `_`; every CJK ideograph
72
+ * is its own token because Chinese has no spaces and dictionary lookup is
73
+ * per-word. Offsets index the original string, so a caller can map a token back
74
+ * to a DOM text position.
75
+ *
76
+ * @param text - text to scan.
77
+ * @returns tokens in source order; each carries its raw slice and range.
78
+ * @throws {TypeError} when `text` is not a string.
79
+ */
80
+ function tokenize(text) {
81
+ if (typeof text !== "string") throw new TypeError("tokenize expects a string");
82
+ const tokens = [];
83
+ let start = -1;
84
+ let end = -1;
85
+
86
+ /** Emit the pending run, if any. */
87
+ const flush = () => {
88
+ if (start < 0) return;
89
+ const raw = text.slice(start, end);
90
+ tokens.push({ raw, lower: raw.toLowerCase(), start, end });
91
+ start = -1;
92
+ end = -1;
93
+ };
94
+
95
+ for (let index = 0; index < text.length; index++) {
96
+ const character = text[index];
97
+ if (!isWordCharacter(character)) {
98
+ flush();
99
+ continue;
100
+ }
101
+ // One ideograph is one token: Chinese compounds have no delimiter, so a
102
+ // second ideograph closes the previous run and opens its own.
103
+ if (isCjkCode(character.charCodeAt(0))) {
104
+ flush();
105
+ start = index;
106
+ end = index + 1;
107
+ flush();
108
+ continue;
109
+ }
110
+ if (start < 0) start = index;
111
+ end = index + 1;
112
+ }
113
+ flush();
114
+ return tokens;
115
+ }
116
+
117
+ //#endregion
118
+
119
+ //#region shape tests
120
+
121
+ /** Suffixes that reliably mark an English loanword as a technical noun. */
122
+ const TECHNICAL_SUFFIXES = [
123
+ "tion", "sion", "ism", "ity", "ance", "ence", "ology", "ography",
124
+ "ization", "isation", "ability", "ibility", "escence", "ification"
125
+ ];
126
+
127
+ /** Prefixes that reliably mark an English loanword as a technical coinage. */
128
+ const TECHNICAL_PREFIXES = [
129
+ "meta", "hyper", "multi", "poly", "mono", "micro", "macro", "pseudo",
130
+ "auto", "hetero", "homo", "isomorphic", "ortho", "tele", "cyber", "neuro"
131
+ ];
132
+
133
+ /**
134
+ * Whether a word is written as a code identifier rather than as prose:
135
+ * `camelCase`, `PascalCase`, `ALL_CAPS`, `snake_case` or `kebab-case`.
136
+ * @param raw - the token exactly as it appeared.
137
+ * @returns true when the spelling itself signals a technical term.
138
+ */
139
+ function looksLikeIdentifier(raw) {
140
+ if (raw.includes("_")) return /[A-Za-z]/.test(raw);
141
+ if (raw.includes("-")) return raw.length >= 5 && raw.split("-").every((part) => part.length >= 2);
142
+ if (raw.length >= 3 && /^[A-Z][a-z]*[A-Z]/.test(raw)) return true;
143
+ if (raw.length >= 4 && /[a-z][A-Z]/.test(raw)) return true;
144
+ return false;
145
+ }
146
+
147
+ /**
148
+ * Whether a token is an acronym or initialism (`API`, `TCP`, `ISO8601`).
149
+ * @param raw - the token exactly as it appeared.
150
+ * @returns true for an all-caps token of two or more letters.
151
+ */
152
+ function isAcronym(raw) {
153
+ return /^[A-Z][A-Z0-9]{1,7}$/.test(raw);
154
+ }
155
+
156
+ /**
157
+ * Whether a lowercase English word carries technical morphology, which is how a
158
+ * term missing from the built-in lexicon still gets noticed.
159
+ * @param lower - the lowercased token.
160
+ * @returns true when a prefix or suffix marks it as jargon.
161
+ */
162
+ function hasTechnicalMorphology(lower) {
163
+ if (lower.length < 8) return false;
164
+ if (TECHNICAL_SUFFIXES.some((suffix) => lower.endsWith(suffix) && lower.length > suffix.length + 3)) return true;
165
+ if (lower.length >= 10 && TECHNICAL_PREFIXES.some((prefix) => lower.startsWith(prefix) && lower.length > prefix.length + 4)) return true;
166
+ return false;
167
+ }
168
+
169
+ /** Mixed-script or full-width latin is data, not a term. */
170
+ const DIGITS_ONLY = /^[0-9]+$/;
171
+
172
+ /**
173
+ * Whether a token is worth registering as a dictionary entry on its own.
174
+ * @param raw - the token exactly as it appeared.
175
+ * @returns true when the token is a plausible standalone term.
176
+ */
177
+ function isTermShaped(raw) {
178
+ if (raw.length < 2 || DIGITS_ONLY.test(raw)) return false;
179
+ if (hasCjk(raw)) return raw.length >= 2 && raw.length <= 8;
180
+ if (!/[A-Za-z]/.test(raw)) return false;
181
+ if (raw.length > 48) return false;
182
+ return true;
183
+ }
184
+
185
+ //#endregion
186
+
187
+ //#region keys
188
+
189
+ /**
190
+ * Normalize a term to its dictionary key: trimmed, lowercased, and with runs of
191
+ * internal whitespace collapsed so `Event Sourcing` and `event sourcing` are
192
+ * one entry.
193
+ * @param term - user- or detector-supplied term text.
194
+ * @returns the canonical key, or an empty string when nothing survives.
195
+ */
196
+ function normalizeTerm(term) {
197
+ if (typeof term !== "string") return "";
198
+ return term.replace(/\s+/g, " ").trim().toLowerCase();
199
+ }
200
+
201
+ /**
202
+ * Stable identifier for a term, so the same term never becomes two entries.
203
+ * FNV-1a over the normalized key, rendered in base 36.
204
+ * @param term - term text in any casing.
205
+ * @returns a short, stable, filesystem-safe id.
206
+ */
207
+ function termId(term) {
208
+ const key = normalizeTerm(term);
209
+ let hash = 0x811c9dc5;
210
+ for (let index = 0; index < key.length; index++) {
211
+ hash ^= key.charCodeAt(index);
212
+ hash = Math.imul(hash, 0x01000193) >>> 0;
213
+ }
214
+ return `t${hash.toString(36)}`;
215
+ }
216
+
217
+ /**
218
+ * Human-readable title case for a key that lost its original casing.
219
+ * @param term - term text in any casing.
220
+ * @returns the term with each word capitalized.
221
+ */
222
+ function titleCase(term) {
223
+ const trimmed = typeof term === "string" ? term.trim() : "";
224
+ if (trimmed === "") return "";
225
+ return trimmed
226
+ .split(" ")
227
+ .map((word) => (word.length === 0 ? word : word[0].toUpperCase() + word.slice(1)))
228
+ .join(" ");
229
+ }
230
+
231
+ //#endregion
232
+
233
+ //#region context
234
+
235
+ /** Sentence terminators for either writing system. */
236
+ const SENTENCE_ENDS = "。!?;.!?;\n";
237
+
238
+ /**
239
+ * Extract the surrounding text of an occurrence, trimmed to whole sentences
240
+ * where possible and hard-clamped so one entry never stores a whole document.
241
+ * @param text - the full message text.
242
+ * @param start - occurrence start offset.
243
+ * @param end - occurrence end offset.
244
+ * @param limit - maximum characters to keep on the window.
245
+ * @returns the context sentence fragment, single-lined.
246
+ */
247
+ function contextAround(text, start, end, limit = 240) {
248
+ if (typeof text !== "string" || text.length === 0) return "";
249
+ const safeStart = Math.max(0, Math.min(start, text.length));
250
+ const safeEnd = Math.max(safeStart, Math.min(end, text.length));
251
+ let from = safeStart;
252
+ while (from > 0 && !SENTENCE_ENDS.includes(text[from - 1]) && safeStart - from < limit) from--;
253
+ let to = safeEnd;
254
+ while (to < text.length && !SENTENCE_ENDS.includes(text[to]) && to - safeEnd < limit) to++;
255
+ let window = text.slice(from, to);
256
+ if (window.length > limit) {
257
+ const cut = Math.floor((limit - (safeEnd - safeStart)) / 2);
258
+ window = text.slice(Math.max(from, safeStart - cut), Math.min(to, safeEnd + cut));
259
+ }
260
+ return window.replace(/\s+/g, " ").trim();
261
+ }
262
+
263
+ /** Characters that make a run of text a URL rather than prose. */
264
+ function isUrlLike(raw) {
265
+ return /^[a-z][a-z0-9+.-]*:\/\//i.test(raw) || /^www\./i.test(raw);
266
+ }
267
+
268
+ //#endregion
269
+
270
+ //#region formatting
271
+
272
+ /** Escape text for safe placement inside a regular expression. */
273
+ function escapeRegExp(text) {
274
+ return text.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
275
+ }
276
+
277
+ /**
278
+ * Short local timestamp for a dictionary entry.
279
+ * @param at - epoch milliseconds.
280
+ * @returns `YYYY-MM-DD HH:mm`, or an empty string without a usable value.
281
+ */
282
+ function formatStamp(at) {
283
+ if (typeof at !== "number" || !Number.isFinite(at)) return "";
284
+ const date = new Date(at);
285
+ const pad = (value) => String(value).padStart(2, "0");
286
+ return `${date.getFullYear()}-${pad(date.getMonth() + 1)}-${pad(date.getDate())} ${pad(date.getHours())}:${pad(date.getMinutes())}`;
287
+ }
288
+
289
+ //#endregion
290
+
291
+ module.exports = {
292
+ isCjkCode,
293
+ hasCjk,
294
+ isWordCharacter,
295
+ tokenize,
296
+ looksLikeIdentifier,
297
+ isAcronym,
298
+ hasTechnicalMorphology,
299
+ isTermShaped,
300
+ normalizeTerm,
301
+ termId,
302
+ titleCase,
303
+ contextAround,
304
+ isUrlLike,
305
+ escapeRegExp,
306
+ formatStamp,
307
+ CJK_PUNCTUATION,
308
+ ASCII_SENTENCE_END
309
+ };