@singapore-editor/spellcheck 0.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +106 -0
- package/THIRD_PARTY_NOTICES +611 -0
- package/dist/assets/spellcheck.worker-Dx-kt277.js +8971 -0
- package/dist/assets/spellcheck.worker-Dx-kt277.js.map +1 -0
- package/dist/controller.d.ts +51 -0
- package/dist/controller.d.ts.map +1 -0
- package/dist/controller.js +307 -0
- package/dist/controller.js.map +1 -0
- package/dist/dictionaryAssets.d.ts +4 -0
- package/dist/dictionaryAssets.d.ts.map +1 -0
- package/dist/dictionaryData.d.ts +13 -0
- package/dist/dictionaryData.d.ts.map +1 -0
- package/dist/engine.d.ts +18 -0
- package/dist/engine.d.ts.map +1 -0
- package/dist/feature.d.ts +19 -0
- package/dist/feature.d.ts.map +1 -0
- package/dist/feature.js +8 -0
- package/dist/feature.js.map +1 -0
- package/dist/index.d.ts +7 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +5 -0
- package/dist/plugin.d.ts +12 -0
- package/dist/plugin.d.ts.map +1 -0
- package/dist/plugin.js +27 -0
- package/dist/plugin.js.map +1 -0
- package/dist/proseRanges.d.ts +25 -0
- package/dist/proseRanges.d.ts.map +1 -0
- package/dist/proseRanges.js +75 -0
- package/dist/proseRanges.js.map +1 -0
- package/dist/protocol.d.ts +27 -0
- package/dist/protocol.d.ts.map +1 -0
- package/dist/service.d.ts +38 -0
- package/dist/service.d.ts.map +1 -0
- package/dist/service.js +128 -0
- package/dist/service.js.map +1 -0
- package/dist/spellcheck.worker.d.ts +2 -0
- package/dist/spellcheck.worker.d.ts.map +1 -0
- package/dist/styles.d.ts +7 -0
- package/dist/styles.d.ts.map +1 -0
- package/dist/styles.js +13 -0
- package/dist/styles.js.map +1 -0
- package/dist/tokenizer.d.ts +23 -0
- package/dist/tokenizer.d.ts.map +1 -0
- package/dist/tokenizer.js +140 -0
- package/dist/tokenizer.js.map +1 -0
- package/package.json +52 -0
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
//#region src/proseRanges.ts
|
|
2
|
+
var PLAIN_LANGUAGES = /* @__PURE__ */ new Set(["plaintext", "text"]);
|
|
3
|
+
var MARKDOWN_LANGUAGES = /* @__PURE__ */ new Set(["markdown", "mdx"]);
|
|
4
|
+
var MARKDOWN_SKIPPED = /* @__PURE__ */ new Set([
|
|
5
|
+
"text.literal",
|
|
6
|
+
"text.uri",
|
|
7
|
+
"text.reference",
|
|
8
|
+
"none"
|
|
9
|
+
]);
|
|
10
|
+
var NO_REGIONS = {
|
|
11
|
+
prose: [],
|
|
12
|
+
code: [],
|
|
13
|
+
excluded: []
|
|
14
|
+
};
|
|
15
|
+
/** Which parts of the window are checked, or null while the syntax that decides it is loading. */
|
|
16
|
+
function spellcheckRegions(input) {
|
|
17
|
+
const { languageId, window } = input;
|
|
18
|
+
if (languageId === null || PLAIN_LANGUAGES.has(languageId)) return {
|
|
19
|
+
prose: [window],
|
|
20
|
+
code: [],
|
|
21
|
+
excluded: []
|
|
22
|
+
};
|
|
23
|
+
const isMarkdown = MARKDOWN_LANGUAGES.has(languageId);
|
|
24
|
+
if (!isMarkdown && input.scope !== "proseAndCode") return NO_REGIONS;
|
|
25
|
+
const captures = input.captures;
|
|
26
|
+
if (!captures) return null;
|
|
27
|
+
if (isMarkdown) {
|
|
28
|
+
const excluded = capturesInWindow(captures, window, (name) => MARKDOWN_SKIPPED.has(name));
|
|
29
|
+
return {
|
|
30
|
+
prose: [window],
|
|
31
|
+
code: [],
|
|
32
|
+
excluded
|
|
33
|
+
};
|
|
34
|
+
}
|
|
35
|
+
return {
|
|
36
|
+
prose: [],
|
|
37
|
+
code: mergeRanges(capturesInWindow(captures, window, isCommentOrString)),
|
|
38
|
+
excluded: []
|
|
39
|
+
};
|
|
40
|
+
}
|
|
41
|
+
function isCommentOrString(name) {
|
|
42
|
+
return name.startsWith("comment") || name.startsWith("string");
|
|
43
|
+
}
|
|
44
|
+
function capturesInWindow(captures, window, keep) {
|
|
45
|
+
const ranges = [];
|
|
46
|
+
for (const capture of captures) {
|
|
47
|
+
if (!keep(capture.captureName)) continue;
|
|
48
|
+
const start = Math.max(capture.startIndex, window.start);
|
|
49
|
+
const end = Math.min(capture.endIndex, window.end);
|
|
50
|
+
if (end > start) ranges.push({
|
|
51
|
+
start,
|
|
52
|
+
end
|
|
53
|
+
});
|
|
54
|
+
}
|
|
55
|
+
return ranges;
|
|
56
|
+
}
|
|
57
|
+
function mergeRanges(ranges) {
|
|
58
|
+
const merged = [];
|
|
59
|
+
for (const range of ranges.toSorted((a, b) => a.start - b.start)) {
|
|
60
|
+
const last = merged.at(-1);
|
|
61
|
+
if (last && range.start <= last.end) {
|
|
62
|
+
merged[merged.length - 1] = {
|
|
63
|
+
start: last.start,
|
|
64
|
+
end: Math.max(last.end, range.end)
|
|
65
|
+
};
|
|
66
|
+
continue;
|
|
67
|
+
}
|
|
68
|
+
merged.push(range);
|
|
69
|
+
}
|
|
70
|
+
return merged;
|
|
71
|
+
}
|
|
72
|
+
//#endregion
|
|
73
|
+
export { spellcheckRegions };
|
|
74
|
+
|
|
75
|
+
//# sourceMappingURL=proseRanges.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"proseRanges.js","names":[],"sources":["../src/proseRanges.ts"],"sourcesContent":["import type { SpellTextRange } from './tokenizer'\n\n/** `prose` checks plain text and Markdown only; `proseAndCode` also checks comments and strings. */\nexport type SpellcheckScope = 'prose' | 'proseAndCode'\n\nexport type SpellcheckCapture = {\n readonly startIndex: number\n readonly endIndex: number\n readonly captureName: string\n}\n\nexport type SpellcheckRegionInput = {\n readonly languageId: string | null\n /** Null while the parse that decides which parts are prose has not landed. */\n readonly captures: readonly SpellcheckCapture[] | null\n readonly window: SpellTextRange\n readonly scope: SpellcheckScope\n}\n\nexport type SpellcheckRegions = {\n /** Checked word by word as written. */\n readonly prose: readonly SpellTextRange[]\n /** Comments and strings: camelCase and snake_case are split. */\n readonly code: readonly SpellTextRange[]\n readonly excluded: readonly SpellTextRange[]\n}\n\nconst PLAIN_LANGUAGES: ReadonlySet<string> = new Set(['plaintext', 'text'])\nconst MARKDOWN_LANGUAGES: ReadonlySet<string> = new Set(['markdown', 'mdx'])\n// Inline and indented code, link targets and labels, and fenced code (`none`).\nconst MARKDOWN_SKIPPED: ReadonlySet<string> = new Set([\n 'text.literal',\n 'text.uri',\n 'text.reference',\n 'none',\n])\nconst NO_REGIONS: SpellcheckRegions = { prose: [], code: [], excluded: [] }\n\n/** Which parts of the window are checked, or null while the syntax that decides it is loading. */\nexport function spellcheckRegions(input: SpellcheckRegionInput): SpellcheckRegions | null {\n const { languageId, window } = input\n if (languageId === null || PLAIN_LANGUAGES.has(languageId)) {\n return { prose: [window], code: [], excluded: [] }\n }\n const isMarkdown = MARKDOWN_LANGUAGES.has(languageId)\n if (!isMarkdown && input.scope !== 'proseAndCode') return NO_REGIONS\n // Checking before the parse lands would mark words inside code for a moment.\n const captures = input.captures\n if (!captures) return null\n if (isMarkdown) {\n const excluded = capturesInWindow(captures, window, (name) => MARKDOWN_SKIPPED.has(name))\n return { prose: [window], code: [], excluded }\n }\n\n const code = capturesInWindow(captures, window, isCommentOrString)\n return { prose: [], code: mergeRanges(code), excluded: [] }\n}\n\nfunction isCommentOrString(name: string): boolean {\n return name.startsWith('comment') || name.startsWith('string')\n}\n\nfunction capturesInWindow(\n captures: readonly SpellcheckCapture[],\n window: SpellTextRange,\n keep: (name: string) => boolean,\n): readonly SpellTextRange[] {\n const ranges: SpellTextRange[] = []\n for (const capture of captures) {\n if (!keep(capture.captureName)) continue\n const start = Math.max(capture.startIndex, window.start)\n const end = Math.min(capture.endIndex, window.end)\n if (end > start) ranges.push({ start, end })\n }\n return ranges\n}\n\nfunction mergeRanges(ranges: readonly SpellTextRange[]): readonly SpellTextRange[] {\n const merged: SpellTextRange[] = []\n for (const range of ranges.toSorted((a, b) => a.start - b.start)) {\n const last = merged.at(-1)\n if (last && range.start <= last.end) {\n merged[merged.length - 1] = { start: last.start, end: Math.max(last.end, range.end) }\n continue\n }\n merged.push(range)\n }\n return merged\n}\n"],"mappings":";AA2BA,IAAM,kCAAuC,IAAI,IAAI,CAAC,aAAa,MAAM,CAAC;AAC1E,IAAM,qCAA0C,IAAI,IAAI,CAAC,YAAY,KAAK,CAAC;AAE3E,IAAM,mCAAwC,IAAI,IAAI;CACpD;CACA;CACA;CACA;AACF,CAAC;AACD,IAAM,aAAgC;CAAE,OAAO,CAAC;CAAG,MAAM,CAAC;CAAG,UAAU,CAAC;AAAE;;AAG1E,SAAgB,kBAAkB,OAAwD;CACxF,MAAM,EAAE,YAAY,WAAW;CAC/B,IAAI,eAAe,QAAQ,gBAAgB,IAAI,UAAU,GACvD,OAAO;EAAE,OAAO,CAAC,MAAM;EAAG,MAAM,CAAC;EAAG,UAAU,CAAC;CAAE;CAEnD,MAAM,aAAa,mBAAmB,IAAI,UAAU;CACpD,IAAI,CAAC,cAAc,MAAM,UAAU,gBAAgB,OAAO;CAE1D,MAAM,WAAW,MAAM;CACvB,IAAI,CAAC,UAAU,OAAO;CACtB,IAAI,YAAY;EACd,MAAM,WAAW,iBAAiB,UAAU,SAAS,SAAS,iBAAiB,IAAI,IAAI,CAAC;EACxF,OAAO;GAAE,OAAO,CAAC,MAAM;GAAG,MAAM,CAAC;GAAG;EAAS;CAC/C;CAGA,OAAO;EAAE,OAAO,CAAC;EAAG,MAAM,YADb,iBAAiB,UAAU,QAAQ,iBACV,CAAI;EAAG,UAAU,CAAC;CAAE;AAC5D;AAEA,SAAS,kBAAkB,MAAuB;CAChD,OAAO,KAAK,WAAW,SAAS,KAAK,KAAK,WAAW,QAAQ;AAC/D;AAEA,SAAS,iBACP,UACA,QACA,MAC2B;CAC3B,MAAM,SAA2B,CAAC;CAClC,KAAK,MAAM,WAAW,UAAU;EAC9B,IAAI,CAAC,KAAK,QAAQ,WAAW,GAAG;EAChC,MAAM,QAAQ,KAAK,IAAI,QAAQ,YAAY,OAAO,KAAK;EACvD,MAAM,MAAM,KAAK,IAAI,QAAQ,UAAU,OAAO,GAAG;EACjD,IAAI,MAAM,OAAO,OAAO,KAAK;GAAE;GAAO;EAAI,CAAC;CAC7C;CACA,OAAO;AACT;AAEA,SAAS,YAAY,QAA8D;CACjF,MAAM,SAA2B,CAAC;CAClC,KAAK,MAAM,SAAS,OAAO,UAAU,GAAG,MAAM,EAAE,QAAQ,EAAE,KAAK,GAAG;EAChE,MAAM,OAAO,OAAO,GAAG,EAAE;EACzB,IAAI,QAAQ,MAAM,SAAS,KAAK,KAAK;GACnC,OAAO,OAAO,SAAS,KAAK;IAAE,OAAO,KAAK;IAAO,KAAK,KAAK,IAAI,KAAK,KAAK,MAAM,GAAG;GAAE;GACpF;EACF;EACA,OAAO,KAAK,KAAK;CACnB;CACA,OAAO;AACT"}
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
export type SpellcheckWorkerRequest = {
|
|
2
|
+
readonly type: 'check';
|
|
3
|
+
readonly id: number;
|
|
4
|
+
readonly words: readonly string[];
|
|
5
|
+
} | {
|
|
6
|
+
readonly type: 'suggest';
|
|
7
|
+
readonly id: number;
|
|
8
|
+
readonly word: string;
|
|
9
|
+
readonly limit: number;
|
|
10
|
+
} | {
|
|
11
|
+
readonly type: 'setAcceptedWords';
|
|
12
|
+
readonly words: readonly string[];
|
|
13
|
+
};
|
|
14
|
+
export type SpellcheckWorkerResponse = {
|
|
15
|
+
readonly type: 'check';
|
|
16
|
+
readonly id: number;
|
|
17
|
+
readonly misspelled: readonly string[];
|
|
18
|
+
} | {
|
|
19
|
+
readonly type: 'suggest';
|
|
20
|
+
readonly id: number;
|
|
21
|
+
readonly suggestions: readonly string[];
|
|
22
|
+
} | {
|
|
23
|
+
readonly type: 'error';
|
|
24
|
+
readonly id: number | null;
|
|
25
|
+
readonly message: string;
|
|
26
|
+
};
|
|
27
|
+
//# sourceMappingURL=protocol.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"protocol.d.ts","sourceRoot":"","sources":["../src/protocol.ts"],"names":[],"mappings":"AAAA,MAAM,MAAM,uBAAuB,GAC/B;IAAE,QAAQ,CAAC,IAAI,EAAE,OAAO,CAAC;IAAC,QAAQ,CAAC,EAAE,EAAE,MAAM,CAAC;IAAC,QAAQ,CAAC,KAAK,EAAE,SAAS,MAAM,EAAE,CAAA;CAAE,GAClF;IACE,QAAQ,CAAC,IAAI,EAAE,SAAS,CAAA;IACxB,QAAQ,CAAC,EAAE,EAAE,MAAM,CAAA;IACnB,QAAQ,CAAC,IAAI,EAAE,MAAM,CAAA;IACrB,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAA;CACvB,GACD;IAAE,QAAQ,CAAC,IAAI,EAAE,kBAAkB,CAAC;IAAC,QAAQ,CAAC,KAAK,EAAE,SAAS,MAAM,EAAE,CAAA;CAAE,CAAA;AAE5E,MAAM,MAAM,wBAAwB,GAChC;IAAE,QAAQ,CAAC,IAAI,EAAE,OAAO,CAAC;IAAC,QAAQ,CAAC,EAAE,EAAE,MAAM,CAAC;IAAC,QAAQ,CAAC,UAAU,EAAE,SAAS,MAAM,EAAE,CAAA;CAAE,GACvF;IAAE,QAAQ,CAAC,IAAI,EAAE,SAAS,CAAC;IAAC,QAAQ,CAAC,EAAE,EAAE,MAAM,CAAC;IAAC,QAAQ,CAAC,WAAW,EAAE,SAAS,MAAM,EAAE,CAAA;CAAE,GAC1F;IAAE,QAAQ,CAAC,IAAI,EAAE,OAAO,CAAC;IAAC,QAAQ,CAAC,EAAE,EAAE,MAAM,GAAG,IAAI,CAAC;IAAC,QAAQ,CAAC,OAAO,EAAE,MAAM,CAAA;CAAE,CAAA"}
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
export type SpellcheckServiceOptions = {
|
|
2
|
+
/** Replaces the bundled worker, for tests and hosts that serve workers themselves. */
|
|
3
|
+
readonly workerFactory?: () => Worker;
|
|
4
|
+
};
|
|
5
|
+
/**
|
|
6
|
+
* One dictionary worker, shared by every editor on the page. The host creates it once and hands it
|
|
7
|
+
* to each editor's spellcheck; the worker starts on the first request.
|
|
8
|
+
*/
|
|
9
|
+
export declare class SpellcheckService {
|
|
10
|
+
private readonly options;
|
|
11
|
+
private worker;
|
|
12
|
+
private disposed;
|
|
13
|
+
private nextId;
|
|
14
|
+
private acceptedWords;
|
|
15
|
+
private accepted;
|
|
16
|
+
private readonly acceptedListeners;
|
|
17
|
+
private readonly pending;
|
|
18
|
+
constructor(options?: SpellcheckServiceOptions);
|
|
19
|
+
/** The words among `words` that are misspelled. */
|
|
20
|
+
check(words: readonly string[]): Promise<readonly string[]>;
|
|
21
|
+
suggest(word: string, limit?: number): Promise<readonly string[]>;
|
|
22
|
+
/** Replaces the accepted list for every editor sharing the service; it survives a worker restart. */
|
|
23
|
+
setAcceptedWords(words: readonly string[]): void;
|
|
24
|
+
/** Answered here, so a verdict the worker gave before the word was accepted never needs asking again. */
|
|
25
|
+
isAccepted(word: string): boolean;
|
|
26
|
+
onDidChangeAcceptedWords(listener: () => void): {
|
|
27
|
+
dispose(): void;
|
|
28
|
+
};
|
|
29
|
+
dispose(): void;
|
|
30
|
+
private request;
|
|
31
|
+
private ensureWorker;
|
|
32
|
+
private receive;
|
|
33
|
+
private settleError;
|
|
34
|
+
/** A crashed worker is dropped; the next request starts a fresh one. */
|
|
35
|
+
private crash;
|
|
36
|
+
private stop;
|
|
37
|
+
}
|
|
38
|
+
//# sourceMappingURL=service.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"service.d.ts","sourceRoot":"","sources":["../src/service.ts"],"names":[],"mappings":"AAEA,MAAM,MAAM,wBAAwB,GAAG;IACrC,sFAAsF;IACtF,QAAQ,CAAC,aAAa,CAAC,EAAE,MAAM,MAAM,CAAA;CACtC,CAAA;AASD;;;GAGG;AACH,qBAAa,iBAAiB;IAST,OAAO,CAAC,QAAQ,CAAC,OAAO;IAR3C,OAAO,CAAC,MAAM,CAAsB;IACpC,OAAO,CAAC,QAAQ,CAAQ;IACxB,OAAO,CAAC,MAAM,CAAI;IAClB,OAAO,CAAC,aAAa,CAAwB;IAC7C,OAAO,CAAC,QAAQ,CAAoB;IACpC,OAAO,CAAC,QAAQ,CAAC,iBAAiB,CAAwB;IAC1D,OAAO,CAAC,QAAQ,CAAC,OAAO,CAA6B;gBAEjB,OAAO,GAAE,wBAA6B;IAE1E,mDAAmD;IAC5C,KAAK,CAAC,KAAK,EAAE,SAAS,MAAM,EAAE,GAAG,OAAO,CAAC,SAAS,MAAM,EAAE,CAAC;IAK3D,OAAO,CAAC,IAAI,EAAE,MAAM,EAAE,KAAK,SAAsB,GAAG,OAAO,CAAC,SAAS,MAAM,EAAE,CAAC;IAIrF,qGAAqG;IAC9F,gBAAgB,CAAC,KAAK,EAAE,SAAS,MAAM,EAAE,GAAG,IAAI;IAavD,yGAAyG;IAClG,UAAU,CAAC,IAAI,EAAE,MAAM,GAAG,OAAO;IAIjC,wBAAwB,CAAC,QAAQ,EAAE,MAAM,IAAI,GAAG;QAAE,OAAO,IAAI,IAAI,CAAA;KAAE;IAKnE,OAAO,IAAI,IAAI;IAOtB,OAAO,CAAC,OAAO;IAcf,OAAO,CAAC,YAAY;IAgBpB,OAAO,CAAC,OAAO;IAYf,OAAO,CAAC,WAAW;IAQnB,wEAAwE;IACxE,OAAO,CAAC,KAAK;IAIb,OAAO,CAAC,IAAI;CAWb"}
|
package/dist/service.js
ADDED
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
//#region src/service.ts
|
|
2
|
+
var DEFAULT_SUGGESTIONS = 5;
|
|
3
|
+
/**
|
|
4
|
+
* One dictionary worker, shared by every editor on the page. The host creates it once and hands it
|
|
5
|
+
* to each editor's spellcheck; the worker starts on the first request.
|
|
6
|
+
*/
|
|
7
|
+
var SpellcheckService = class {
|
|
8
|
+
options;
|
|
9
|
+
worker = null;
|
|
10
|
+
disposed = false;
|
|
11
|
+
nextId = 1;
|
|
12
|
+
acceptedWords = [];
|
|
13
|
+
accepted = /* @__PURE__ */ new Set();
|
|
14
|
+
acceptedListeners = /* @__PURE__ */ new Set();
|
|
15
|
+
pending = /* @__PURE__ */ new Map();
|
|
16
|
+
constructor(options = {}) {
|
|
17
|
+
this.options = options;
|
|
18
|
+
}
|
|
19
|
+
/** The words among `words` that are misspelled. */
|
|
20
|
+
check(words) {
|
|
21
|
+
if (words.length === 0) return Promise.resolve([]);
|
|
22
|
+
return this.request((id) => ({
|
|
23
|
+
type: "check",
|
|
24
|
+
id,
|
|
25
|
+
words
|
|
26
|
+
}));
|
|
27
|
+
}
|
|
28
|
+
suggest(word, limit = DEFAULT_SUGGESTIONS) {
|
|
29
|
+
return this.request((id) => ({
|
|
30
|
+
type: "suggest",
|
|
31
|
+
id,
|
|
32
|
+
word,
|
|
33
|
+
limit
|
|
34
|
+
}));
|
|
35
|
+
}
|
|
36
|
+
/** Replaces the accepted list for every editor sharing the service; it survives a worker restart. */
|
|
37
|
+
setAcceptedWords(words) {
|
|
38
|
+
this.acceptedWords = words;
|
|
39
|
+
this.accepted = new Set(words);
|
|
40
|
+
if (this.worker) try {
|
|
41
|
+
this.worker.postMessage({
|
|
42
|
+
type: "setAcceptedWords",
|
|
43
|
+
words
|
|
44
|
+
});
|
|
45
|
+
} catch (error) {
|
|
46
|
+
this.stop(error);
|
|
47
|
+
}
|
|
48
|
+
for (const listener of this.acceptedListeners) listener();
|
|
49
|
+
}
|
|
50
|
+
/** Answered here, so a verdict the worker gave before the word was accepted never needs asking again. */
|
|
51
|
+
isAccepted(word) {
|
|
52
|
+
return this.accepted.has(word) || this.accepted.has(word.toLowerCase());
|
|
53
|
+
}
|
|
54
|
+
onDidChangeAcceptedWords(listener) {
|
|
55
|
+
this.acceptedListeners.add(listener);
|
|
56
|
+
return { dispose: () => this.acceptedListeners.delete(listener) };
|
|
57
|
+
}
|
|
58
|
+
dispose() {
|
|
59
|
+
if (this.disposed) return;
|
|
60
|
+
this.disposed = true;
|
|
61
|
+
this.acceptedListeners.clear();
|
|
62
|
+
this.stop(/* @__PURE__ */ new Error("The spellcheck service was disposed"));
|
|
63
|
+
}
|
|
64
|
+
request(build) {
|
|
65
|
+
if (this.disposed) return Promise.reject(/* @__PURE__ */ new Error("The spellcheck service was disposed"));
|
|
66
|
+
const id = this.nextId++;
|
|
67
|
+
return new Promise((resolve, reject) => {
|
|
68
|
+
this.pending.set(id, {
|
|
69
|
+
resolve,
|
|
70
|
+
reject
|
|
71
|
+
});
|
|
72
|
+
try {
|
|
73
|
+
this.ensureWorker().postMessage(build(id));
|
|
74
|
+
} catch (error) {
|
|
75
|
+
this.stop(error);
|
|
76
|
+
}
|
|
77
|
+
});
|
|
78
|
+
}
|
|
79
|
+
ensureWorker() {
|
|
80
|
+
if (this.worker) return this.worker;
|
|
81
|
+
const worker = this.options.workerFactory?.() ?? new Worker(new URL("./assets/spellcheck.worker-Dx-kt277.js", import.meta.url), { type: "module" });
|
|
82
|
+
worker.onmessage = (event) => this.receive(event.data);
|
|
83
|
+
worker.onerror = (event) => this.crash(event.message);
|
|
84
|
+
worker.onmessageerror = () => this.crash("The worker response could not be decoded");
|
|
85
|
+
this.worker = worker;
|
|
86
|
+
if (this.acceptedWords.length > 0) worker.postMessage({
|
|
87
|
+
type: "setAcceptedWords",
|
|
88
|
+
words: this.acceptedWords
|
|
89
|
+
});
|
|
90
|
+
return worker;
|
|
91
|
+
}
|
|
92
|
+
receive(response) {
|
|
93
|
+
if (response.type === "error") {
|
|
94
|
+
this.settleError(response.id, response.message);
|
|
95
|
+
return;
|
|
96
|
+
}
|
|
97
|
+
const pending = this.pending.get(response.id);
|
|
98
|
+
if (!pending) return;
|
|
99
|
+
this.pending.delete(response.id);
|
|
100
|
+
pending.resolve(response.type === "check" ? response.misspelled : response.suggestions);
|
|
101
|
+
}
|
|
102
|
+
settleError(id, message) {
|
|
103
|
+
if (id === null) return this.crash(message);
|
|
104
|
+
const pending = this.pending.get(id);
|
|
105
|
+
if (!pending) return;
|
|
106
|
+
this.pending.delete(id);
|
|
107
|
+
pending.reject(/* @__PURE__ */ new Error(`Spellcheck failed: ${message}`));
|
|
108
|
+
}
|
|
109
|
+
/** A crashed worker is dropped; the next request starts a fresh one. */
|
|
110
|
+
crash(message) {
|
|
111
|
+
this.stop(/* @__PURE__ */ new Error(`The spellcheck worker stopped: ${message}`));
|
|
112
|
+
}
|
|
113
|
+
stop(error) {
|
|
114
|
+
if (this.worker) {
|
|
115
|
+
this.worker.onmessage = null;
|
|
116
|
+
this.worker.onerror = null;
|
|
117
|
+
this.worker.onmessageerror = null;
|
|
118
|
+
this.worker.terminate();
|
|
119
|
+
}
|
|
120
|
+
this.worker = null;
|
|
121
|
+
for (const pending of this.pending.values()) pending.reject(error);
|
|
122
|
+
this.pending.clear();
|
|
123
|
+
}
|
|
124
|
+
};
|
|
125
|
+
//#endregion
|
|
126
|
+
export { SpellcheckService };
|
|
127
|
+
|
|
128
|
+
//# sourceMappingURL=service.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"service.js","names":["import.meta.ROLLDOWN_FILE_URL_i9WMc1PFfvwp6TtOmeUZmw"],"sources":["../src/service.ts"],"sourcesContent":["import type { SpellcheckWorkerRequest, SpellcheckWorkerResponse } from './protocol'\n\nexport type SpellcheckServiceOptions = {\n /** Replaces the bundled worker, for tests and hosts that serve workers themselves. */\n readonly workerFactory?: () => Worker\n}\n\ntype Pending = {\n readonly resolve: (value: readonly string[]) => void\n readonly reject: (error: unknown) => void\n}\n\nconst DEFAULT_SUGGESTIONS = 5\n\n/**\n * One dictionary worker, shared by every editor on the page. The host creates it once and hands it\n * to each editor's spellcheck; the worker starts on the first request.\n */\nexport class SpellcheckService {\n private worker: Worker | null = null\n private disposed = false\n private nextId = 1\n private acceptedWords: readonly string[] = []\n private accepted = new Set<string>()\n private readonly acceptedListeners = new Set<() => void>()\n private readonly pending = new Map<number, Pending>()\n\n public constructor(private readonly options: SpellcheckServiceOptions = {}) {}\n\n /** The words among `words` that are misspelled. */\n public check(words: readonly string[]): Promise<readonly string[]> {\n if (words.length === 0) return Promise.resolve([])\n return this.request((id) => ({ type: 'check', id, words }))\n }\n\n public suggest(word: string, limit = DEFAULT_SUGGESTIONS): Promise<readonly string[]> {\n return this.request((id) => ({ type: 'suggest', id, word, limit }))\n }\n\n /** Replaces the accepted list for every editor sharing the service; it survives a worker restart. */\n public setAcceptedWords(words: readonly string[]): void {\n this.acceptedWords = words\n this.accepted = new Set(words)\n if (this.worker) {\n try {\n this.worker.postMessage({ type: 'setAcceptedWords', words })\n } catch (error) {\n this.stop(error)\n }\n }\n for (const listener of this.acceptedListeners) listener()\n }\n\n /** Answered here, so a verdict the worker gave before the word was accepted never needs asking again. */\n public isAccepted(word: string): boolean {\n return this.accepted.has(word) || this.accepted.has(word.toLowerCase())\n }\n\n public onDidChangeAcceptedWords(listener: () => void): { dispose(): void } {\n this.acceptedListeners.add(listener)\n return { dispose: () => this.acceptedListeners.delete(listener) }\n }\n\n public dispose(): void {\n if (this.disposed) return\n this.disposed = true\n this.acceptedListeners.clear()\n this.stop(new Error('The spellcheck service was disposed'))\n }\n\n private request(build: (id: number) => SpellcheckWorkerRequest): Promise<readonly string[]> {\n if (this.disposed) return Promise.reject(new Error('The spellcheck service was disposed'))\n\n const id = this.nextId++\n return new Promise((resolve, reject) => {\n this.pending.set(id, { resolve, reject })\n try {\n this.ensureWorker().postMessage(build(id))\n } catch (error) {\n this.stop(error)\n }\n })\n }\n\n private ensureWorker(): Worker {\n if (this.worker) return this.worker\n\n const worker =\n this.options.workerFactory?.() ??\n new Worker(new URL('./spellcheck.worker.ts', import.meta.url), { type: 'module' })\n worker.onmessage = (event: MessageEvent<SpellcheckWorkerResponse>) => this.receive(event.data)\n worker.onerror = (event: ErrorEvent) => this.crash(event.message)\n worker.onmessageerror = () => this.crash('The worker response could not be decoded')\n this.worker = worker\n if (this.acceptedWords.length > 0) {\n worker.postMessage({ type: 'setAcceptedWords', words: this.acceptedWords })\n }\n return worker\n }\n\n private receive(response: SpellcheckWorkerResponse): void {\n if (response.type === 'error') {\n this.settleError(response.id, response.message)\n return\n }\n\n const pending = this.pending.get(response.id)\n if (!pending) return\n this.pending.delete(response.id)\n pending.resolve(response.type === 'check' ? response.misspelled : response.suggestions)\n }\n\n private settleError(id: number | null, message: string): void {\n if (id === null) return this.crash(message)\n const pending = this.pending.get(id)\n if (!pending) return\n this.pending.delete(id)\n pending.reject(new Error(`Spellcheck failed: ${message}`))\n }\n\n /** A crashed worker is dropped; the next request starts a fresh one. */\n private crash(message: string): void {\n this.stop(new Error(`The spellcheck worker stopped: ${message}`))\n }\n\n private stop(error: unknown): void {\n if (this.worker) {\n this.worker.onmessage = null\n this.worker.onerror = null\n this.worker.onmessageerror = null\n this.worker.terminate()\n }\n this.worker = null\n for (const pending of this.pending.values()) pending.reject(error)\n this.pending.clear()\n }\n}\n"],"mappings":";AAYA,IAAM,sBAAsB;;;;;AAM5B,IAAa,oBAAb,MAA+B;CASO;CARpC,SAAgC;CAChC,WAAmB;CACnB,SAAiB;CACjB,gBAA2C,CAAC;CAC5C,2BAAmB,IAAI,IAAY;CACnC,oCAAqC,IAAI,IAAgB;CACzD,0BAA2B,IAAI,IAAqB;CAEpD,YAAmB,UAAqD,CAAC,GAAG;EAAxC,KAAA,UAAA;CAAyC;;CAG7E,MAAa,OAAsD;EACjE,IAAI,MAAM,WAAW,GAAG,OAAO,QAAQ,QAAQ,CAAC,CAAC;EACjD,OAAO,KAAK,SAAS,QAAQ;GAAE,MAAM;GAAS;GAAI;EAAM,EAAE;CAC5D;CAEA,QAAe,MAAc,QAAQ,qBAAiD;EACpF,OAAO,KAAK,SAAS,QAAQ;GAAE,MAAM;GAAW;GAAI;GAAM;EAAM,EAAE;CACpE;;CAGA,iBAAwB,OAAgC;EACtD,KAAK,gBAAgB;EACrB,KAAK,WAAW,IAAI,IAAI,KAAK;EAC7B,IAAI,KAAK,QACP,IAAI;GACF,KAAK,OAAO,YAAY;IAAE,MAAM;IAAoB;GAAM,CAAC;EAC7D,SAAS,OAAO;GACd,KAAK,KAAK,KAAK;EACjB;EAEF,KAAK,MAAM,YAAY,KAAK,mBAAmB,SAAS;CAC1D;;CAGA,WAAkB,MAAuB;EACvC,OAAO,KAAK,SAAS,IAAI,IAAI,KAAK,KAAK,SAAS,IAAI,KAAK,YAAY,CAAC;CACxE;CAEA,yBAAgC,UAA2C;EACzE,KAAK,kBAAkB,IAAI,QAAQ;EACnC,OAAO,EAAE,eAAe,KAAK,kBAAkB,OAAO,QAAQ,EAAE;CAClE;CAEA,UAAuB;EACrB,IAAI,KAAK,UAAU;EACnB,KAAK,WAAW;EAChB,KAAK,kBAAkB,MAAM;EAC7B,KAAK,qBAAK,IAAI,MAAM,qCAAqC,CAAC;CAC5D;CAEA,QAAgB,OAA4E;EAC1F,IAAI,KAAK,UAAU,OAAO,QAAQ,uBAAO,IAAI,MAAM,qCAAqC,CAAC;EAEzF,MAAM,KAAK,KAAK;EAChB,OAAO,IAAI,SAAS,SAAS,WAAW;GACtC,KAAK,QAAQ,IAAI,IAAI;IAAE;IAAS;GAAO,CAAC;GACxC,IAAI;IACF,KAAK,aAAa,CAAC,CAAC,YAAY,MAAM,EAAE,CAAC;GAC3C,SAAS,OAAO;IACd,KAAK,KAAK,KAAK;GACjB;EACF,CAAC;CACH;CAEA,eAA+B;EAC7B,IAAI,KAAK,QAAQ,OAAO,KAAK;EAE7B,MAAM,SACJ,KAAK,QAAQ,gBAAgB,KAC7B,IAAI,OAAO,IAAA;;GAAA,IAAA,IAAA,wCAAA,YAAA,GAAA,CAAA,CAAAA;GAAA,KAAA,YAAA;EAAA,GAAoD,EAAE,MAAM,SAAS,CAAC;EACnF,OAAO,aAAa,UAAkD,KAAK,QAAQ,MAAM,IAAI;EAC7F,OAAO,WAAW,UAAsB,KAAK,MAAM,MAAM,OAAO;EAChE,OAAO,uBAAuB,KAAK,MAAM,0CAA0C;EACnF,KAAK,SAAS;EACd,IAAI,KAAK,cAAc,SAAS,GAC9B,OAAO,YAAY;GAAE,MAAM;GAAoB,OAAO,KAAK;EAAc,CAAC;EAE5E,OAAO;CACT;CAEA,QAAgB,UAA0C;EACxD,IAAI,SAAS,SAAS,SAAS;GAC7B,KAAK,YAAY,SAAS,IAAI,SAAS,OAAO;GAC9C;EACF;EAEA,MAAM,UAAU,KAAK,QAAQ,IAAI,SAAS,EAAE;EAC5C,IAAI,CAAC,SAAS;EACd,KAAK,QAAQ,OAAO,SAAS,EAAE;EAC/B,QAAQ,QAAQ,SAAS,SAAS,UAAU,SAAS,aAAa,SAAS,WAAW;CACxF;CAEA,YAAoB,IAAmB,SAAuB;EAC5D,IAAI,OAAO,MAAM,OAAO,KAAK,MAAM,OAAO;EAC1C,MAAM,UAAU,KAAK,QAAQ,IAAI,EAAE;EACnC,IAAI,CAAC,SAAS;EACd,KAAK,QAAQ,OAAO,EAAE;EACtB,QAAQ,uBAAO,IAAI,MAAM,sBAAsB,SAAS,CAAC;CAC3D;;CAGA,MAAc,SAAuB;EACnC,KAAK,qBAAK,IAAI,MAAM,kCAAkC,SAAS,CAAC;CAClE;CAEA,KAAa,OAAsB;EACjC,IAAI,KAAK,QAAQ;GACf,KAAK,OAAO,YAAY;GACxB,KAAK,OAAO,UAAU;GACtB,KAAK,OAAO,iBAAiB;GAC7B,KAAK,OAAO,UAAU;EACxB;EACA,KAAK,SAAS;EACd,KAAK,MAAM,WAAW,KAAK,QAAQ,OAAO,GAAG,QAAQ,OAAO,KAAK;EACjE,KAAK,QAAQ,MAAM;CACrB;AACF"}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"spellcheck.worker.d.ts","sourceRoot":"","sources":["../src/spellcheck.worker.ts"],"names":[],"mappings":""}
|
package/dist/styles.d.ts
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
import { type VirtualizedTextHighlightStyle } from '@singapore-editor/core/rendering';
|
|
2
|
+
/**
|
|
3
|
+
* An overlay, because WebKit draws a highlight's wavy line only when the highlight also sets a text
|
|
4
|
+
* colour, and the overlay paints each token's own colour back with it.
|
|
5
|
+
*/
|
|
6
|
+
export declare const SPELLING_STYLE: VirtualizedTextHighlightStyle;
|
|
7
|
+
//# sourceMappingURL=styles.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"styles.d.ts","sourceRoot":"","sources":["../src/styles.ts"],"names":[],"mappings":"AAAA,OAAO,EAEL,KAAK,6BAA6B,EACnC,MAAM,kCAAkC,CAAA;AAQzC;;;GAGG;AACH,eAAO,MAAM,cAAc,EAAE,6BAE5B,CAAA"}
|
package/dist/styles.js
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
import { registerEditorColor } from "@singapore-editor/core/rendering";
|
|
2
|
+
/**
|
|
3
|
+
* An overlay, because WebKit draws a highlight's wavy line only when the highlight also sets a text
|
|
4
|
+
* colour, and the overlay paints each token's own colour back with it.
|
|
5
|
+
*/
|
|
6
|
+
var SPELLING_STYLE = { overlay: { textDecoration: `underline wavy ${registerEditorColor("spellcheck.misspelled", {
|
|
7
|
+
dark: "#38bdf8",
|
|
8
|
+
light: "#0284c7"
|
|
9
|
+
})}` } };
|
|
10
|
+
//#endregion
|
|
11
|
+
export { SPELLING_STYLE };
|
|
12
|
+
|
|
13
|
+
//# sourceMappingURL=styles.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"styles.js","names":[],"sources":["../src/styles.ts"],"sourcesContent":["import {\n registerEditorColor,\n type VirtualizedTextHighlightStyle,\n} from '@singapore-editor/core/rendering'\n\n// Blue keeps a spelling mark apart from a red error on the same word.\nconst MISSPELLED = registerEditorColor('spellcheck.misspelled', {\n dark: '#38bdf8',\n light: '#0284c7',\n})\n\n/**\n * An overlay, because WebKit draws a highlight's wavy line only when the highlight also sets a text\n * colour, and the overlay paints each token's own colour back with it.\n */\nexport const SPELLING_STYLE: VirtualizedTextHighlightStyle = {\n overlay: { textDecoration: `underline wavy ${MISSPELLED}` },\n}\n"],"mappings":";;;;;AAeA,IAAa,iBAAgD,EAC3D,SAAS,EAAE,gBAAgB,kBAVV,oBAAoB,yBAAyB;CAC9D,MAAM;CACN,OAAO;AACT,CAO+C,IAAa,EAC5D"}
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
export type SpellTextRange = {
|
|
2
|
+
readonly start: number;
|
|
3
|
+
readonly end: number;
|
|
4
|
+
};
|
|
5
|
+
export type SpellWord = SpellTextRange & {
|
|
6
|
+
/** The word as the dictionary is asked about it: typographic apostrophes folded to `'`. */
|
|
7
|
+
readonly word: string;
|
|
8
|
+
};
|
|
9
|
+
/**
|
|
10
|
+
* `prose` checks words as written and skips anything shaped like an identifier. `code` is for
|
|
11
|
+
* comments and strings: it splits camelCase and snake_case and checks each part.
|
|
12
|
+
*/
|
|
13
|
+
export type SpellTokenizeMode = 'prose' | 'code';
|
|
14
|
+
export type SpellTokenizeOptions = {
|
|
15
|
+
readonly mode?: SpellTokenizeMode;
|
|
16
|
+
/** Ranges never checked, such as inline replacements. Offsets are relative to the text. */
|
|
17
|
+
readonly excluded?: readonly SpellTextRange[];
|
|
18
|
+
};
|
|
19
|
+
export declare const MAX_SPELLCHECK_LINE_LENGTH = 16384;
|
|
20
|
+
/** Finds the words to check in `text`, with offsets relative to it. */
|
|
21
|
+
export declare function tokenizeSpellWords(text: string, options?: SpellTokenizeOptions): readonly SpellWord[];
|
|
22
|
+
export declare function foldApostrophes(word: string): string;
|
|
23
|
+
//# sourceMappingURL=tokenizer.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"tokenizer.d.ts","sourceRoot":"","sources":["../src/tokenizer.ts"],"names":[],"mappings":"AAAA,MAAM,MAAM,cAAc,GAAG;IAC3B,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAA;IACtB,QAAQ,CAAC,GAAG,EAAE,MAAM,CAAA;CACrB,CAAA;AAED,MAAM,MAAM,SAAS,GAAG,cAAc,GAAG;IACvC,2FAA2F;IAC3F,QAAQ,CAAC,IAAI,EAAE,MAAM,CAAA;CACtB,CAAA;AAED;;;GAGG;AACH,MAAM,MAAM,iBAAiB,GAAG,OAAO,GAAG,MAAM,CAAA;AAEhD,MAAM,MAAM,oBAAoB,GAAG;IACjC,QAAQ,CAAC,IAAI,CAAC,EAAE,iBAAiB,CAAA;IACjC,2FAA2F;IAC3F,QAAQ,CAAC,QAAQ,CAAC,EAAE,SAAS,cAAc,EAAE,CAAA;CAC9C,CAAA;AAOD,eAAO,MAAM,0BAA0B,QAAS,CAAA;AAWhD,uEAAuE;AACvE,wBAAgB,kBAAkB,CAChC,IAAI,EAAE,MAAM,EACZ,OAAO,GAAE,oBAAyB,GACjC,SAAS,SAAS,EAAE,CAatB;AA2ED,wBAAgB,eAAe,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,CAEpD"}
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
//#region src/tokenizer.ts
|
|
2
|
+
var MIN_PROSE_LENGTH = 2;
|
|
3
|
+
var MIN_CODE_PART_LENGTH = 4;
|
|
4
|
+
var MAX_STRUCTURED_LENGTH = 256;
|
|
5
|
+
var MAX_SPELLCHECK_LINE_LENGTH = 16384;
|
|
6
|
+
var URL_PATTERN = /\b(?:[a-z][a-z0-9+.-]*:\/\/|www\.)[^\s<>()"'`]+/gi;
|
|
7
|
+
var EMAIL_PATTERN = /[\w.+-]+@[\w-]+(?:\.[\w-]+)+/g;
|
|
8
|
+
var PATH_PATTERN = /(?:[\w.~@-]*[/\\])+[\w.@-]*/g;
|
|
9
|
+
var DOTTED_PATTERN = /[\p{L}\p{N}_]+(?:\.[\p{L}\p{N}_]+)+/gu;
|
|
10
|
+
var RUN_PATTERN = /[\p{L}\p{M}\p{N}_'’]+/gu;
|
|
11
|
+
var EDGE_APOSTROPHES = /^['’]+|['’]+$/g;
|
|
12
|
+
var ENGLISH_WORD = /^[a-zA-Z]+(?:['’][a-zA-Z]+)*$/;
|
|
13
|
+
var CAMEL_PART = /[A-Z]{2,}(?![a-z])|[A-Z]?[a-z]+(?:['’][a-z]+)*/g;
|
|
14
|
+
/** Finds the words to check in `text`, with offsets relative to it. */
|
|
15
|
+
function tokenizeSpellWords(text, options = {}) {
|
|
16
|
+
const words = [];
|
|
17
|
+
const excluded = sortedRanges(options.excluded ?? []);
|
|
18
|
+
let cursor = 0;
|
|
19
|
+
for (const chunk of text.matchAll(/\S+/g)) {
|
|
20
|
+
if (chunk[0].length > MAX_STRUCTURED_LENGTH) continue;
|
|
21
|
+
cursor = advancePast(excluded, cursor, chunk.index);
|
|
22
|
+
const local = chunkExclusions(excluded, cursor, chunk.index, chunk[0].length);
|
|
23
|
+
for (const word of tokenizeChunk(chunk[0], {
|
|
24
|
+
...options,
|
|
25
|
+
excluded: local
|
|
26
|
+
})) words.push({
|
|
27
|
+
...word,
|
|
28
|
+
start: word.start + chunk.index,
|
|
29
|
+
end: word.end + chunk.index
|
|
30
|
+
});
|
|
31
|
+
}
|
|
32
|
+
return words;
|
|
33
|
+
}
|
|
34
|
+
function chunkExclusions(ranges, cursor, start, length) {
|
|
35
|
+
const local = [];
|
|
36
|
+
for (let index = cursor; index < ranges.length; index++) {
|
|
37
|
+
const range = ranges[index];
|
|
38
|
+
if (range.start >= start + length) break;
|
|
39
|
+
local.push({
|
|
40
|
+
start: range.start - start,
|
|
41
|
+
end: range.end - start
|
|
42
|
+
});
|
|
43
|
+
}
|
|
44
|
+
return local;
|
|
45
|
+
}
|
|
46
|
+
function tokenizeChunk(text, options) {
|
|
47
|
+
const skipped = sortedRanges([...options.excluded ?? [], ...structuredRanges(text)]);
|
|
48
|
+
const mode = options.mode ?? "prose";
|
|
49
|
+
const words = [];
|
|
50
|
+
let cursor = 0;
|
|
51
|
+
for (const match of text.matchAll(RUN_PATTERN)) {
|
|
52
|
+
const run = trimmedRun(match[0], match.index);
|
|
53
|
+
if (!run) continue;
|
|
54
|
+
cursor = advancePast(skipped, cursor, run.start);
|
|
55
|
+
if (overlapsFrom(skipped, cursor, run)) continue;
|
|
56
|
+
if (mode === "code") words.push(...codeWords(run));
|
|
57
|
+
else pushProseWord(words, run);
|
|
58
|
+
}
|
|
59
|
+
return words;
|
|
60
|
+
}
|
|
61
|
+
function trimmedRun(raw, index) {
|
|
62
|
+
const leading = raw.length - raw.replace(/^['’]+/, "").length;
|
|
63
|
+
const text = raw.replace(EDGE_APOSTROPHES, "");
|
|
64
|
+
if (text.length === 0) return null;
|
|
65
|
+
const start = index + leading;
|
|
66
|
+
return {
|
|
67
|
+
start,
|
|
68
|
+
end: start + text.length,
|
|
69
|
+
text
|
|
70
|
+
};
|
|
71
|
+
}
|
|
72
|
+
function pushProseWord(words, run) {
|
|
73
|
+
if (run.text.length < MIN_PROSE_LENGTH) return;
|
|
74
|
+
if (!ENGLISH_WORD.test(run.text)) return;
|
|
75
|
+
if (isAcronym(run.text) || isCamelCase(run.text)) return;
|
|
76
|
+
words.push({
|
|
77
|
+
start: run.start,
|
|
78
|
+
end: run.end,
|
|
79
|
+
word: foldApostrophes(run.text)
|
|
80
|
+
});
|
|
81
|
+
}
|
|
82
|
+
function codeWords(run) {
|
|
83
|
+
if (/[^\p{N}_'’a-zA-Z]/u.test(run.text)) return [];
|
|
84
|
+
const words = [];
|
|
85
|
+
for (const part of run.text.matchAll(CAMEL_PART)) {
|
|
86
|
+
const text = part[0];
|
|
87
|
+
if (text.length < MIN_CODE_PART_LENGTH || isAcronym(text)) continue;
|
|
88
|
+
const start = run.start + part.index;
|
|
89
|
+
words.push({
|
|
90
|
+
start,
|
|
91
|
+
end: start + text.length,
|
|
92
|
+
word: foldApostrophes(text)
|
|
93
|
+
});
|
|
94
|
+
}
|
|
95
|
+
return words;
|
|
96
|
+
}
|
|
97
|
+
function isAcronym(word) {
|
|
98
|
+
return !/[a-z]/.test(word);
|
|
99
|
+
}
|
|
100
|
+
/** An upper-case letter after the first one: `useState`, `iPhone`, `McDonald`. */
|
|
101
|
+
function isCamelCase(word) {
|
|
102
|
+
return /[A-Z]/.test(word.slice(1));
|
|
103
|
+
}
|
|
104
|
+
function foldApostrophes(word) {
|
|
105
|
+
return word.replaceAll("’", "'");
|
|
106
|
+
}
|
|
107
|
+
function structuredRanges(text) {
|
|
108
|
+
const ranges = [];
|
|
109
|
+
for (const pattern of [
|
|
110
|
+
URL_PATTERN,
|
|
111
|
+
EMAIL_PATTERN,
|
|
112
|
+
PATH_PATTERN,
|
|
113
|
+
DOTTED_PATTERN
|
|
114
|
+
]) for (const match of text.matchAll(pattern)) ranges.push({
|
|
115
|
+
start: match.index,
|
|
116
|
+
end: match.index + match[0].length
|
|
117
|
+
});
|
|
118
|
+
return ranges;
|
|
119
|
+
}
|
|
120
|
+
function sortedRanges(ranges) {
|
|
121
|
+
return ranges.filter((range) => range.end > range.start).toSorted((a, b) => a.start - b.start);
|
|
122
|
+
}
|
|
123
|
+
/** Runs arrive in order, so ranges that end before one can never overlap a later one. */
|
|
124
|
+
function advancePast(ranges, cursor, offset) {
|
|
125
|
+
let next = cursor;
|
|
126
|
+
while (next < ranges.length && (ranges[next]?.end ?? 0) <= offset) next++;
|
|
127
|
+
return next;
|
|
128
|
+
}
|
|
129
|
+
function overlapsFrom(ranges, cursor, run) {
|
|
130
|
+
for (let index = cursor; index < ranges.length; index++) {
|
|
131
|
+
const range = ranges[index];
|
|
132
|
+
if (!range || range.start >= run.end) return false;
|
|
133
|
+
if (range.end > run.start) return true;
|
|
134
|
+
}
|
|
135
|
+
return false;
|
|
136
|
+
}
|
|
137
|
+
//#endregion
|
|
138
|
+
export { MAX_SPELLCHECK_LINE_LENGTH, foldApostrophes, tokenizeSpellWords };
|
|
139
|
+
|
|
140
|
+
//# sourceMappingURL=tokenizer.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"tokenizer.js","names":[],"sources":["../src/tokenizer.ts"],"sourcesContent":["export type SpellTextRange = {\n readonly start: number\n readonly end: number\n}\n\nexport type SpellWord = SpellTextRange & {\n /** The word as the dictionary is asked about it: typographic apostrophes folded to `'`. */\n readonly word: string\n}\n\n/**\n * `prose` checks words as written and skips anything shaped like an identifier. `code` is for\n * comments and strings: it splits camelCase and snake_case and checks each part.\n */\nexport type SpellTokenizeMode = 'prose' | 'code'\n\nexport type SpellTokenizeOptions = {\n readonly mode?: SpellTokenizeMode\n /** Ranges never checked, such as inline replacements. Offsets are relative to the text. */\n readonly excluded?: readonly SpellTextRange[]\n}\n\nconst MIN_PROSE_LENGTH = 2\n// cSpell's default for code: shorter camelCase parts are mostly abbreviations.\nconst MIN_CODE_PART_LENGTH = 4\nconst MAX_STRUCTURED_LENGTH = 256\n// Skip generated/minified lines before reading or caching their text on each keystroke.\nexport const MAX_SPELLCHECK_LINE_LENGTH = 16_384\n\nconst URL_PATTERN = /\\b(?:[a-z][a-z0-9+.-]*:\\/\\/|www\\.)[^\\s<>()\"'`]+/gi\nconst EMAIL_PATTERN = /[\\w.+-]+@[\\w-]+(?:\\.[\\w-]+)+/g\nconst PATH_PATTERN = /(?:[\\w.~@-]*[/\\\\])+[\\w.@-]*/g\nconst DOTTED_PATTERN = /[\\p{L}\\p{N}_]+(?:\\.[\\p{L}\\p{N}_]+)+/gu\nconst RUN_PATTERN = /[\\p{L}\\p{M}\\p{N}_'’]+/gu\nconst EDGE_APOSTROPHES = /^['’]+|['’]+$/g\nconst ENGLISH_WORD = /^[a-zA-Z]+(?:['’][a-zA-Z]+)*$/\nconst CAMEL_PART = /[A-Z]{2,}(?![a-z])|[A-Z]?[a-z]+(?:['’][a-z]+)*/g\n\n/** Finds the words to check in `text`, with offsets relative to it. */\nexport function tokenizeSpellWords(\n text: string,\n options: SpellTokenizeOptions = {},\n): readonly SpellWord[] {\n const words: SpellWord[] = []\n const excluded = sortedRanges(options.excluded ?? [])\n let cursor = 0\n for (const chunk of text.matchAll(/\\S+/g)) {\n if (chunk[0].length > MAX_STRUCTURED_LENGTH) continue\n cursor = advancePast(excluded, cursor, chunk.index)\n const local = chunkExclusions(excluded, cursor, chunk.index, chunk[0].length)\n for (const word of tokenizeChunk(chunk[0], { ...options, excluded: local })) {\n words.push({ ...word, start: word.start + chunk.index, end: word.end + chunk.index })\n }\n }\n return words\n}\n\nfunction chunkExclusions(\n ranges: readonly SpellTextRange[],\n cursor: number,\n start: number,\n length: number,\n): readonly SpellTextRange[] {\n const local: SpellTextRange[] = []\n for (let index = cursor; index < ranges.length; index++) {\n const range = ranges[index]!\n if (range.start >= start + length) break\n local.push({ start: range.start - start, end: range.end - start })\n }\n return local\n}\n\nfunction tokenizeChunk(text: string, options: SpellTokenizeOptions): readonly SpellWord[] {\n const skipped = sortedRanges([...(options.excluded ?? []), ...structuredRanges(text)])\n const mode = options.mode ?? 'prose'\n const words: SpellWord[] = []\n let cursor = 0\n\n for (const match of text.matchAll(RUN_PATTERN)) {\n const run = trimmedRun(match[0], match.index)\n if (!run) continue\n cursor = advancePast(skipped, cursor, run.start)\n if (overlapsFrom(skipped, cursor, run)) continue\n if (mode === 'code') words.push(...codeWords(run))\n else pushProseWord(words, run)\n }\n\n return words\n}\n\ntype Run = SpellTextRange & { readonly text: string }\n\nfunction trimmedRun(raw: string, index: number): Run | null {\n const leading = raw.length - raw.replace(/^['’]+/, '').length\n const text = raw.replace(EDGE_APOSTROPHES, '')\n if (text.length === 0) return null\n const start = index + leading\n return { start, end: start + text.length, text }\n}\n\nfunction pushProseWord(words: SpellWord[], run: Run): void {\n if (run.text.length < MIN_PROSE_LENGTH) return\n if (!ENGLISH_WORD.test(run.text)) return\n if (isAcronym(run.text) || isCamelCase(run.text)) return\n words.push({ start: run.start, end: run.end, word: foldApostrophes(run.text) })\n}\n\nfunction codeWords(run: Run): readonly SpellWord[] {\n // The bundled English policy only submits ASCII letters and supported apostrophes.\n if (/[^\\p{N}_'’a-zA-Z]/u.test(run.text)) return []\n\n const words: SpellWord[] = []\n for (const part of run.text.matchAll(CAMEL_PART)) {\n const text = part[0]\n if (text.length < MIN_CODE_PART_LENGTH || isAcronym(text)) continue\n const start = run.start + part.index\n words.push({ start, end: start + text.length, word: foldApostrophes(text) })\n }\n return words\n}\n\nfunction isAcronym(word: string): boolean {\n return !/[a-z]/.test(word)\n}\n\n/** An upper-case letter after the first one: `useState`, `iPhone`, `McDonald`. */\nfunction isCamelCase(word: string): boolean {\n return /[A-Z]/.test(word.slice(1))\n}\n\nexport function foldApostrophes(word: string): string {\n return word.replaceAll('’', \"'\")\n}\n\nfunction structuredRanges(text: string): readonly SpellTextRange[] {\n const ranges: SpellTextRange[] = []\n for (const pattern of [URL_PATTERN, EMAIL_PATTERN, PATH_PATTERN, DOTTED_PATTERN]) {\n for (const match of text.matchAll(pattern)) {\n ranges.push({ start: match.index, end: match.index + match[0].length })\n }\n }\n return ranges\n}\n\nfunction sortedRanges(ranges: readonly SpellTextRange[]): readonly SpellTextRange[] {\n return ranges.filter((range) => range.end > range.start).toSorted((a, b) => a.start - b.start)\n}\n\n/** Runs arrive in order, so ranges that end before one can never overlap a later one. */\nfunction advancePast(ranges: readonly SpellTextRange[], cursor: number, offset: number): number {\n let next = cursor\n while (next < ranges.length && (ranges[next]?.end ?? 0) <= offset) next++\n return next\n}\n\nfunction overlapsFrom(\n ranges: readonly SpellTextRange[],\n cursor: number,\n run: SpellTextRange,\n): boolean {\n for (let index = cursor; index < ranges.length; index++) {\n const range = ranges[index]\n if (!range || range.start >= run.end) return false\n if (range.end > run.start) return true\n }\n return false\n}\n"],"mappings":";AAsBA,IAAM,mBAAmB;AAEzB,IAAM,uBAAuB;AAC7B,IAAM,wBAAwB;AAE9B,IAAa,6BAA6B;AAE1C,IAAM,cAAc;AACpB,IAAM,gBAAgB;AACtB,IAAM,eAAe;AACrB,IAAM,iBAAiB;AACvB,IAAM,cAAc;AACpB,IAAM,mBAAmB;AACzB,IAAM,eAAe;AACrB,IAAM,aAAa;;AAGnB,SAAgB,mBACd,MACA,UAAgC,CAAC,GACX;CACtB,MAAM,QAAqB,CAAC;CAC5B,MAAM,WAAW,aAAa,QAAQ,YAAY,CAAC,CAAC;CACpD,IAAI,SAAS;CACb,KAAK,MAAM,SAAS,KAAK,SAAS,MAAM,GAAG;EACzC,IAAI,MAAM,EAAE,CAAC,SAAS,uBAAuB;EAC7C,SAAS,YAAY,UAAU,QAAQ,MAAM,KAAK;EAClD,MAAM,QAAQ,gBAAgB,UAAU,QAAQ,MAAM,OAAO,MAAM,EAAE,CAAC,MAAM;EAC5E,KAAK,MAAM,QAAQ,cAAc,MAAM,IAAI;GAAE,GAAG;GAAS,UAAU;EAAM,CAAC,GACxE,MAAM,KAAK;GAAE,GAAG;GAAM,OAAO,KAAK,QAAQ,MAAM;GAAO,KAAK,KAAK,MAAM,MAAM;EAAM,CAAC;CAExF;CACA,OAAO;AACT;AAEA,SAAS,gBACP,QACA,QACA,OACA,QAC2B;CAC3B,MAAM,QAA0B,CAAC;CACjC,KAAK,IAAI,QAAQ,QAAQ,QAAQ,OAAO,QAAQ,SAAS;EACvD,MAAM,QAAQ,OAAO;EACrB,IAAI,MAAM,SAAS,QAAQ,QAAQ;EACnC,MAAM,KAAK;GAAE,OAAO,MAAM,QAAQ;GAAO,KAAK,MAAM,MAAM;EAAM,CAAC;CACnE;CACA,OAAO;AACT;AAEA,SAAS,cAAc,MAAc,SAAqD;CACxF,MAAM,UAAU,aAAa,CAAC,GAAI,QAAQ,YAAY,CAAC,GAAI,GAAG,iBAAiB,IAAI,CAAC,CAAC;CACrF,MAAM,OAAO,QAAQ,QAAQ;CAC7B,MAAM,QAAqB,CAAC;CAC5B,IAAI,SAAS;CAEb,KAAK,MAAM,SAAS,KAAK,SAAS,WAAW,GAAG;EAC9C,MAAM,MAAM,WAAW,MAAM,IAAI,MAAM,KAAK;EAC5C,IAAI,CAAC,KAAK;EACV,SAAS,YAAY,SAAS,QAAQ,IAAI,KAAK;EAC/C,IAAI,aAAa,SAAS,QAAQ,GAAG,GAAG;EACxC,IAAI,SAAS,QAAQ,MAAM,KAAK,GAAG,UAAU,GAAG,CAAC;OAC5C,cAAc,OAAO,GAAG;CAC/B;CAEA,OAAO;AACT;AAIA,SAAS,WAAW,KAAa,OAA2B;CAC1D,MAAM,UAAU,IAAI,SAAS,IAAI,QAAQ,UAAU,EAAE,CAAC,CAAC;CACvD,MAAM,OAAO,IAAI,QAAQ,kBAAkB,EAAE;CAC7C,IAAI,KAAK,WAAW,GAAG,OAAO;CAC9B,MAAM,QAAQ,QAAQ;CACtB,OAAO;EAAE;EAAO,KAAK,QAAQ,KAAK;EAAQ;CAAK;AACjD;AAEA,SAAS,cAAc,OAAoB,KAAgB;CACzD,IAAI,IAAI,KAAK,SAAS,kBAAkB;CACxC,IAAI,CAAC,aAAa,KAAK,IAAI,IAAI,GAAG;CAClC,IAAI,UAAU,IAAI,IAAI,KAAK,YAAY,IAAI,IAAI,GAAG;CAClD,MAAM,KAAK;EAAE,OAAO,IAAI;EAAO,KAAK,IAAI;EAAK,MAAM,gBAAgB,IAAI,IAAI;CAAE,CAAC;AAChF;AAEA,SAAS,UAAU,KAAgC;CAEjD,IAAI,qBAAqB,KAAK,IAAI,IAAI,GAAG,OAAO,CAAC;CAEjD,MAAM,QAAqB,CAAC;CAC5B,KAAK,MAAM,QAAQ,IAAI,KAAK,SAAS,UAAU,GAAG;EAChD,MAAM,OAAO,KAAK;EAClB,IAAI,KAAK,SAAS,wBAAwB,UAAU,IAAI,GAAG;EAC3D,MAAM,QAAQ,IAAI,QAAQ,KAAK;EAC/B,MAAM,KAAK;GAAE;GAAO,KAAK,QAAQ,KAAK;GAAQ,MAAM,gBAAgB,IAAI;EAAE,CAAC;CAC7E;CACA,OAAO;AACT;AAEA,SAAS,UAAU,MAAuB;CACxC,OAAO,CAAC,QAAQ,KAAK,IAAI;AAC3B;;AAGA,SAAS,YAAY,MAAuB;CAC1C,OAAO,QAAQ,KAAK,KAAK,MAAM,CAAC,CAAC;AACnC;AAEA,SAAgB,gBAAgB,MAAsB;CACpD,OAAO,KAAK,WAAW,KAAK,GAAG;AACjC;AAEA,SAAS,iBAAiB,MAAyC;CACjE,MAAM,SAA2B,CAAC;CAClC,KAAK,MAAM,WAAW;EAAC;EAAa;EAAe;EAAc;CAAc,GAC7E,KAAK,MAAM,SAAS,KAAK,SAAS,OAAO,GACvC,OAAO,KAAK;EAAE,OAAO,MAAM;EAAO,KAAK,MAAM,QAAQ,MAAM,EAAE,CAAC;CAAO,CAAC;CAG1E,OAAO;AACT;AAEA,SAAS,aAAa,QAA8D;CAClF,OAAO,OAAO,QAAQ,UAAU,MAAM,MAAM,MAAM,KAAK,CAAC,CAAC,UAAU,GAAG,MAAM,EAAE,QAAQ,EAAE,KAAK;AAC/F;;AAGA,SAAS,YAAY,QAAmC,QAAgB,QAAwB;CAC9F,IAAI,OAAO;CACX,OAAO,OAAO,OAAO,WAAW,OAAO,KAAK,EAAE,OAAO,MAAM,QAAQ;CACnE,OAAO;AACT;AAEA,SAAS,aACP,QACA,QACA,KACS;CACT,KAAK,IAAI,QAAQ,QAAQ,QAAQ,OAAO,QAAQ,SAAS;EACvD,MAAM,QAAQ,OAAO;EACrB,IAAI,CAAC,SAAS,MAAM,SAAS,IAAI,KAAK,OAAO;EAC7C,IAAI,MAAM,MAAM,IAAI,OAAO,OAAO;CACpC;CACA,OAAO;AACT"}
|
package/package.json
ADDED
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "@singapore-editor/spellcheck",
|
|
3
|
+
"version": "0.1.1",
|
|
4
|
+
"files": [
|
|
5
|
+
"README.md",
|
|
6
|
+
"THIRD_PARTY_NOTICES",
|
|
7
|
+
"dist"
|
|
8
|
+
],
|
|
9
|
+
"type": "module",
|
|
10
|
+
"sideEffects": false,
|
|
11
|
+
"exports": {
|
|
12
|
+
".": {
|
|
13
|
+
"types": "./dist/index.d.ts",
|
|
14
|
+
"import": "./dist/index.js",
|
|
15
|
+
"default": "./dist/index.js"
|
|
16
|
+
}
|
|
17
|
+
},
|
|
18
|
+
"publishConfig": {
|
|
19
|
+
"access": "public",
|
|
20
|
+
"registry": "https://registry.npmjs.org/"
|
|
21
|
+
},
|
|
22
|
+
"scripts": {
|
|
23
|
+
"build": "bun ../../scripts/build-package.ts",
|
|
24
|
+
"build:dictionaries": "bun scripts/build-dictionaries.ts",
|
|
25
|
+
"bench:engine": "bun bench/engine.ts",
|
|
26
|
+
"test": "vitest run --project node --project dom --project browser --project engines",
|
|
27
|
+
"typecheck": "tsgo --noEmit",
|
|
28
|
+
"lint": "oxlint .",
|
|
29
|
+
"format": "oxfmt --write .",
|
|
30
|
+
"format:check": "oxfmt --check .",
|
|
31
|
+
"bench:typing": "vitest run --project typing-cost --silent=false"
|
|
32
|
+
},
|
|
33
|
+
"dependencies": {
|
|
34
|
+
"cspell-trie-lib": "10.3.4"
|
|
35
|
+
},
|
|
36
|
+
"devDependencies": {
|
|
37
|
+
"@playwright/test": "^1.63.0",
|
|
38
|
+
"@singapore-editor/core": "0.1.2",
|
|
39
|
+
"@singapore-editor/tree-sitter-languages": "0.1.1",
|
|
40
|
+
"@types/node": "^26.6.3",
|
|
41
|
+
"@typescript/native-preview": "7.0.0-dev.20260707.2",
|
|
42
|
+
"@vitest/browser-playwright": "^5.0.2",
|
|
43
|
+
"happy-dom": "^20.14.5",
|
|
44
|
+
"oxfmt": "0.70.0",
|
|
45
|
+
"oxlint": "1.85.0",
|
|
46
|
+
"typescript": "~6.0.3",
|
|
47
|
+
"vitest": "^5.0.2"
|
|
48
|
+
},
|
|
49
|
+
"peerDependencies": {
|
|
50
|
+
"@singapore-editor/core": "0.1.2"
|
|
51
|
+
}
|
|
52
|
+
}
|