@inshapardaz/likhari-react 0.1.21-dev.201 → 0.1.21-dev.203
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/EditorRoot.d.ts +6 -0
- package/dist/EditorRoot.d.ts.map +1 -1
- package/dist/EditorRoot.js +2 -2
- package/dist/autocorrect/AutoCorrectPlugin.d.ts +7 -1
- package/dist/autocorrect/AutoCorrectPlugin.d.ts.map +1 -1
- package/dist/autocorrect/AutoCorrectPlugin.js +22 -7
- package/dist/autocorrect/autoCorrectActions.d.ts +14 -2
- package/dist/autocorrect/autoCorrectActions.d.ts.map +1 -1
- package/dist/autocorrect/autoCorrectActions.js +86 -8
- package/dist/autocorrect/punctuationRules.d.ts +22 -0
- package/dist/autocorrect/punctuationRules.d.ts.map +1 -0
- package/dist/autocorrect/punctuationRules.js +45 -0
- package/dist/index.d.ts +3 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -0
- package/dist/normalization/urduNormalize.d.ts +24 -0
- package/dist/normalization/urduNormalize.d.ts.map +1 -0
- package/dist/normalization/urduNormalize.js +116 -0
- package/package.json +1 -1
package/dist/EditorRoot.d.ts
CHANGED
|
@@ -5,6 +5,8 @@ import { type EditorFeatureConfig, type FeatureConfigPresetName } from '@inshapa
|
|
|
5
5
|
import { type FontOption } from './fonts/index.js';
|
|
6
6
|
import { type DraftRestoreMode } from './components/DraftRestore.js';
|
|
7
7
|
import { type AutoCorrectStore } from './autocorrect/autoCorrectStores.js';
|
|
8
|
+
import type { UrduNormalizationOptions } from './normalization/urduNormalize.js';
|
|
9
|
+
import type { PunctuationOptions } from './autocorrect/punctuationRules.js';
|
|
8
10
|
import { type Locale } from './i18n/index.js';
|
|
9
11
|
/** What happens when the user is about to leave with unsaved changes. */
|
|
10
12
|
export type NavigationGuardMode = 'confirm' | 'save-draft' | 'off';
|
|
@@ -16,6 +18,10 @@ export interface EditorRootProps {
|
|
|
16
18
|
documentId?: string;
|
|
17
19
|
/** Where auto-corrections are loaded from and saved to, in priority order. Pass a stable array. */
|
|
18
20
|
autoCorrectStores?: AutoCorrectStore[];
|
|
21
|
+
/** Urdu normalisation applied as part of auto-correct. Diacritics are kept unless `removeDiacritics` is set. */
|
|
22
|
+
urduNormalization?: UrduNormalizationOptions;
|
|
23
|
+
/** Common punctuation fixes, and whether a straight " becomes ”. Both on by default. */
|
|
24
|
+
punctuation?: PunctuationOptions;
|
|
19
25
|
initialContent?: EditorInitialContent;
|
|
20
26
|
featureConfig?: EditorFeatureConfig;
|
|
21
27
|
featurePreset?: FeatureConfigPresetName;
|
package/dist/EditorRoot.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"EditorRoot.d.ts","sourceRoot":"","sources":["../src/EditorRoot.tsx"],"names":[],"mappings":"AACA,OAAO,KAAK,EAAE,oBAAoB,EAAE,MAAM,eAAe,CAAC;AAY1D,OAAO,KAAK,EAA8B,qBAAqB,EAAE,MAAM,SAAS,CAAC;AAEjF,OAAO,EAAyB,KAAK,QAAQ,EAAE,MAAM,iCAAiC,CAAC;AACvF,OAAO,EAAwB,KAAK,mBAAmB,EAAE,KAAK,uBAAuB,EAAE,MAAM,2BAA2B,CAAC;AAKzH,OAAO,EAAyB,KAAK,UAAU,EAAE,MAAM,SAAS,CAAC;AAOjE,OAAO,EAAgB,KAAK,gBAAgB,EAAE,MAAM,2BAA2B,CAAC;AAIhF,OAAO,EAAgC,KAAK,gBAAgB,EAAE,MAAM,iCAAiC,CAAC;
|
|
1
|
+
{"version":3,"file":"EditorRoot.d.ts","sourceRoot":"","sources":["../src/EditorRoot.tsx"],"names":[],"mappings":"AACA,OAAO,KAAK,EAAE,oBAAoB,EAAE,MAAM,eAAe,CAAC;AAY1D,OAAO,KAAK,EAA8B,qBAAqB,EAAE,MAAM,SAAS,CAAC;AAEjF,OAAO,EAAyB,KAAK,QAAQ,EAAE,MAAM,iCAAiC,CAAC;AACvF,OAAO,EAAwB,KAAK,mBAAmB,EAAE,KAAK,uBAAuB,EAAE,MAAM,2BAA2B,CAAC;AAKzH,OAAO,EAAyB,KAAK,UAAU,EAAE,MAAM,SAAS,CAAC;AAOjE,OAAO,EAAgB,KAAK,gBAAgB,EAAE,MAAM,2BAA2B,CAAC;AAIhF,OAAO,EAAgC,KAAK,gBAAgB,EAAE,MAAM,iCAAiC,CAAC;AACtG,OAAO,KAAK,EAAE,wBAAwB,EAAE,MAAM,+BAA+B,CAAC;AAC9E,OAAO,KAAK,EAAE,kBAAkB,EAAE,MAAM,gCAAgC,CAAC;AAczE,OAAO,EAAgC,KAAK,MAAM,EAAE,MAAM,QAAQ,CAAC;AAMnE,yEAAyE;AACzE,MAAM,MAAM,mBAAmB,GAAG,SAAS,GAAG,YAAY,GAAG,KAAK,CAAC;AAEnE,MAAM,WAAW,oBAAoB;IACnC,MAAM,EAAE,QAAQ,CAAC;IACjB,KAAK,EAAE,MAAM,CAAC;CACf;AAOD,MAAM,WAAW,eAAe;IAC9B,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,mGAAmG;IACnG,iBAAiB,CAAC,EAAE,gBAAgB,EAAE,CAAC;IACvC,gHAAgH;IAChH,iBAAiB,CAAC,EAAE,wBAAwB,CAAC;IAC7C,wFAAwF;IACxF,WAAW,CAAC,EAAE,kBAAkB,CAAC;IACjC,cAAc,CAAC,EAAE,oBAAoB,CAAC;IACtC,aAAa,CAAC,EAAE,mBAAmB,CAAC;IACpC,aAAa,CAAC,EAAE,uBAAuB,CAAC;IACxC,KAAK,CAAC,EAAE,oBAAoB,CAAC;IAC7B,WAAW,CAAC,EAAE,OAAO,GAAG,MAAM,CAAC;IAC/B;;;;;;OAMG;IACH,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB;;;;;;;OAOG;IACH,MAAM,CAAC,EAAE,MAAM,GAAG,MAAM,CAAC;IACzB,QAAQ,CAAC,EAAE,CAAC,KAAK,EAAE,qBAAqB,KAAK,IAAI,CAAC;IAClD;;;OAGG;IACH,MAAM,CAAC,EAAE,CAAC,OAAO,EAAE,MAAM,EAAE,MAAM,EAAE,QAAQ,KAAK,IAAI,CAAC;IACrD;;;;;OAKG;IACH,QAAQ,CAAC,EAAE,OAAO,CAAC;IACnB;;;;OAIG;IACH,aAAa,CAAC,EAAE,CAAC,IAAI,EAAE,IAAI,KAAK,OAAO,CAAC,MAAM,CAAC,CAAC;IAChD;;;;;;;OAOG;IACH,UAAU,CAAC,EAAE,CAAC,GAAG,EAAE,MAAM,KAAK,OAAO,CAAC,IAAI,CAAC,CAAC;IAC5C;;;;;;OAMG;IACH,WAAW,CAAC,EAAE,UAAU,EAAE,CAAC;IAC3B;;;;;;;;;OASG;IACH,QAAQ,CAAC,EAAE,OAAO,CAAC;IACnB;;;;;;;OAOG;IACH,YAAY,CAAC,EAAE,gBAAgB,CAAC;IAChC;;;;;;;;;;;;;;;OAeG;IACH,eAAe,CAAC,EAAE,mBAAmB,CAAC;IACtC,uGAAuG;IACvG,eAAe,CAAC,EAAE,CAAC,KAAK,EAAE;QAAE,UAAU,EAAE,MAAM,CAAC;QAAC,OAAO,EAAE,MAAM,CAAA;KAAE,KAAK,IAAI,CAAC;IAC3E,6EAA6E;IAC7E,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB;;;;;OAKG;IACH,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B;;;;OAIG;IACH,iBAAiB,CAAC,EAAE,MAAM,CAAC;CAC5B;AAED,MAAM,WAAW,SAAS;IACxB,UAAU,CAAC,MAAM,EAAE,QAAQ,GAAG,MAAM,CAAC;IACrC,UAAU,CAAC,KAAK,EAAE,MAAM,EAAE,MAAM,EAAE,QAAQ,GAAG,IAAI,CAAC;IAClD,iBAAiB,IAAI,OAAO,CAAC;IAC7B,cAAc,IAAI,OAAO,CAAC,OAAO,CAAC,CAAC;IACnC,KAAK,IAAI,IAAI,CAAC;CACf;AAQD,eAAO,MAAM,UAAU,uGA2ZrB,CAAC"}
|
package/dist/EditorRoot.js
CHANGED
|
@@ -48,7 +48,7 @@ function initialEditorStateJson(initialContent) {
|
|
|
48
48
|
const state = defaultFormatRegistry.parse(initialContent.format, initialContent.value);
|
|
49
49
|
return JSON.stringify(state);
|
|
50
50
|
}
|
|
51
|
-
export const EditorRoot = forwardRef(function EditorRoot({ documentId, autoCorrectStores, initialContent, featureConfig, featurePreset, theme, colorScheme, accentColor, locale = 'en', placeholder, height = '480px', onChange, onSave, showSave = Boolean(onSave), onImageUpload, fetchImage, fontOptions, autosave = true, restoreDraft = 'prompt', navigationGuard = 'confirm', onDraftRestored, autosaveDelayMs = DEFAULT_AUTOSAVE_DELAY_MS, autosaveMaxBytes = DEFAULT_AUTOSAVE_MAX_BYTES, autosaveMaxDrafts = DEFAULT_AUTOSAVE_MAX_DRAFTS, }, ref) {
|
|
51
|
+
export const EditorRoot = forwardRef(function EditorRoot({ documentId, autoCorrectStores, urduNormalization, punctuation, initialContent, featureConfig, featurePreset, theme, colorScheme, accentColor, locale = 'en', placeholder, height = '480px', onChange, onSave, showSave = Boolean(onSave), onImageUpload, fetchImage, fontOptions, autosave = true, restoreDraft = 'prompt', navigationGuard = 'confirm', onDraftRestored, autosaveDelayMs = DEFAULT_AUTOSAVE_DELAY_MS, autosaveMaxBytes = DEFAULT_AUTOSAVE_MAX_BYTES, autosaveMaxDrafts = DEFAULT_AUTOSAVE_MAX_DRAFTS, }, ref) {
|
|
52
52
|
const config = useMemo(() => resolveFeatureConfig(featureConfig, featurePreset), [featureConfig, featurePreset]);
|
|
53
53
|
const strings = useMemo(() => getStrings(locale), [locale]);
|
|
54
54
|
const resolvedPlaceholder = placeholder ?? strings.editor.placeholder;
|
|
@@ -293,7 +293,7 @@ export const EditorRoot = forwardRef(function EditorRoot({ documentId, autoCorre
|
|
|
293
293
|
const portalTargetSelector = `#${CSS.escape(portalTargetId)}`;
|
|
294
294
|
return (_jsx(EditorThemeProvider, { theme: theme, colorScheme: colorScheme, accentColor: accentColor, scopeElementId: portalTargetId, children: _jsxs("div", { className: "likhari-root", dir: dir, ref: rootElementRef, "data-document-id": documentId, style: { height: typeof height === 'number' ? `${height}px` : height }, children: [_jsx("div", { id: portalTargetId, className: "likhari-portal-target" }), _jsx(PortalTargetContext.Provider, { value: portalTargetSelector, children: _jsxs(UiStringsContext.Provider, { value: strings, children: [_jsx(ImageOptionsContext.Provider, { value: imageOptions, children: _jsxs(LexicalComposer, { initialConfig: initialConfig, children: [_jsx(Toolbar, { config: config, onSave: handleSave, findOpen: openPanel === 'find', onToggleFind: () => togglePanel('find'), spellOpen: openPanel === 'spell', onToggleSpell: () => togglePanel('spell'), autoCorrectOpen: openPanel === 'autocorrect', onToggleAutoCorrect: () => togglePanel('autocorrect'), isDirty: isDirty, showSave: showSave, fontOptions: fontOptions, direction: dir, locale: locale, drafts: autosave
|
|
295
295
|
? { currentId: draftId, maxBytes: autosaveMaxBytes, adoptOnRestore: !documentId, onRestored: handleDraftRestored }
|
|
296
|
-
: undefined }), documentId && (_jsx(DraftRestore, { draftId: documentId, mode: restoreDraft, initialJson: initialContentJsonRef.current, locale: locale, onRestored: handleDraftRestored })), _jsxs("div", { className: "likhari-canvas-frame", children: [config.findReplace && openPanel === 'find' && _jsx(FindReplaceBar, { strings: strings, dir: dir, onClose: () => setOpenPanel(null) }), config.language.spellCheck && openPanel === 'spell' && (_jsx(SpellcheckPanel, { strings: strings, dir: dir, onClose: () => setOpenPanel(null) })), config.language.autocorrect && openPanel === 'autocorrect' && (_jsx(AutoCorrectPanel, { strings: strings, dir: dir, stores: stores, onSaved: () => setAutoCorrectVersion((v) => v + 1), onClose: () => setOpenPanel(null) })), _jsx("div", { className: "likhari-canvas", children: _jsx(RichTextPlugin, { contentEditable: _jsx(ContentEditable, { className: "likhari-content-editable", dir: dir, spellCheck: false, "aria-label": strings.editor.contentLabel }), placeholder: _jsx("div", { className: "likhari-placeholder", children: resolvedPlaceholder }), ErrorBoundary: LexicalErrorBoundary }) })] }), config.history && _jsx(HistoryPlugin, {}), (config.lists.bullet || config.lists.numbered || config.lists.check) && _jsx(ListPlugin, {}), config.lists.check && _jsx(CheckListPlugin, {}), config.links && _jsx(LinkPlugin, {}), config.links && _jsx(LinkPastePlugin, {}), config.blocks.pageBreak && _jsx(PageBreakPlugin, {}), config.language.spellCheck && _jsx(SpellHighlightPlugin, {}), config.language.autocorrect && _jsx(AutoCorrectPlugin, { stores: stores, version: autoCorrectVersion, enabled: true }), config.columns && _jsx(LayoutPlugin, {}), config.footnotes && _jsx(FootnotePlugin, {}), config.poetry.enabled && _jsx(PoetryPlugin, {}), config.poetry.enabled && _jsx(PoetryResizer, {}), config.tables && _jsx(TablePlugin, { hasCellMerge: true, hasTabHandler: true }), config.blocks.horizontalRule && _jsx(HorizontalRulePlugin, {}), _jsx(EditorInstancePlugin, { instanceRef: editorInstanceRef }), _jsx(OnChangePlugin, { onChange: handleChange })] }) }), _jsx(LeaveDialog, { opened: leavePrompt !== null, canSave: Boolean(onSave), canSaveDraft: autosave || navigationGuard === 'save-draft', draftFailed: leavePrompt?.draftFailed ?? false, onSave: () => {
|
|
296
|
+
: undefined }), documentId && (_jsx(DraftRestore, { draftId: documentId, mode: restoreDraft, initialJson: initialContentJsonRef.current, locale: locale, onRestored: handleDraftRestored })), _jsxs("div", { className: "likhari-canvas-frame", children: [config.findReplace && openPanel === 'find' && _jsx(FindReplaceBar, { strings: strings, dir: dir, onClose: () => setOpenPanel(null) }), config.language.spellCheck && openPanel === 'spell' && (_jsx(SpellcheckPanel, { strings: strings, dir: dir, onClose: () => setOpenPanel(null) })), config.language.autocorrect && openPanel === 'autocorrect' && (_jsx(AutoCorrectPanel, { strings: strings, dir: dir, stores: stores, onSaved: () => setAutoCorrectVersion((v) => v + 1), onClose: () => setOpenPanel(null) })), _jsx("div", { className: "likhari-canvas", children: _jsx(RichTextPlugin, { contentEditable: _jsx(ContentEditable, { className: "likhari-content-editable", dir: dir, spellCheck: false, "aria-label": strings.editor.contentLabel }), placeholder: _jsx("div", { className: "likhari-placeholder", children: resolvedPlaceholder }), ErrorBoundary: LexicalErrorBoundary }) })] }), config.history && _jsx(HistoryPlugin, {}), (config.lists.bullet || config.lists.numbered || config.lists.check) && _jsx(ListPlugin, {}), config.lists.check && _jsx(CheckListPlugin, {}), config.links && _jsx(LinkPlugin, {}), config.links && _jsx(LinkPastePlugin, {}), config.blocks.pageBreak && _jsx(PageBreakPlugin, {}), config.language.spellCheck && _jsx(SpellHighlightPlugin, {}), config.language.autocorrect && _jsx(AutoCorrectPlugin, { stores: stores, version: autoCorrectVersion, enabled: true, urduNormalization: urduNormalization, punctuation: punctuation }), config.columns && _jsx(LayoutPlugin, {}), config.footnotes && _jsx(FootnotePlugin, {}), config.poetry.enabled && _jsx(PoetryPlugin, {}), config.poetry.enabled && _jsx(PoetryResizer, {}), config.tables && _jsx(TablePlugin, { hasCellMerge: true, hasTabHandler: true }), config.blocks.horizontalRule && _jsx(HorizontalRulePlugin, {}), _jsx(EditorInstancePlugin, { instanceRef: editorInstanceRef }), _jsx(OnChangePlugin, { onChange: handleChange })] }) }), _jsx(LeaveDialog, { opened: leavePrompt !== null, canSave: Boolean(onSave), canSaveDraft: autosave || navigationGuard === 'save-draft', draftFailed: leavePrompt?.draftFailed ?? false, onSave: () => {
|
|
297
297
|
handleSave();
|
|
298
298
|
settleLeave(true);
|
|
299
299
|
}, onSaveDraft: () => {
|
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
import { type LexicalCommand } from 'lexical';
|
|
2
|
+
import { type PunctuationOptions } from './punctuationRules.js';
|
|
2
3
|
import { type AutoCorrectStore } from './autoCorrectStores.js';
|
|
4
|
+
import { type UrduNormalizationOptions } from '../normalization/urduNormalize.js';
|
|
3
5
|
/** Keys that end a word: a space or punctuation, in Latin and Arabic-script forms. */
|
|
4
6
|
/** Corrects every word in the document, e.g. after pasting or loading content. */
|
|
5
7
|
export declare const CORRECT_DOCUMENT_COMMAND: LexicalCommand<void>;
|
|
@@ -10,9 +12,13 @@ export declare const CORRECT_DOCUMENT_COMMAND: LexicalCommand<void>;
|
|
|
10
12
|
* the table reloads). The table used depends on the block's direction: RTL
|
|
11
13
|
* blocks use the Urdu and Shahmukhi tables, LTR blocks the English one.
|
|
12
14
|
*/
|
|
13
|
-
export declare function AutoCorrectPlugin({ stores, version, enabled }: {
|
|
15
|
+
export declare function AutoCorrectPlugin({ stores, version, enabled, urduNormalization, punctuation, }: {
|
|
14
16
|
stores: AutoCorrectStore[];
|
|
15
17
|
version: number;
|
|
16
18
|
enabled: boolean;
|
|
19
|
+
/** Urdu normalisation for right-to-left text; character mapping always applies. */
|
|
20
|
+
urduNormalization?: UrduNormalizationOptions;
|
|
21
|
+
/** Common punctuation fixes (from the bundled Urdu rule list). */
|
|
22
|
+
punctuation?: PunctuationOptions;
|
|
17
23
|
}): null;
|
|
18
24
|
//# sourceMappingURL=AutoCorrectPlugin.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"AutoCorrectPlugin.d.ts","sourceRoot":"","sources":["../../src/autocorrect/AutoCorrectPlugin.tsx"],"names":[],"mappings":"AAEA,OAAO,EAA4E,KAAK,cAAc,EAAE,MAAM,SAAS,CAAC;AAExH,OAAO,EAAuB,KAAK,gBAAgB,EAAE,MAAM,qBAAqB,CAAC;
|
|
1
|
+
{"version":3,"file":"AutoCorrectPlugin.d.ts","sourceRoot":"","sources":["../../src/autocorrect/AutoCorrectPlugin.tsx"],"names":[],"mappings":"AAEA,OAAO,EAA4E,KAAK,cAAc,EAAE,MAAM,SAAS,CAAC;AAExH,OAAO,EAA0B,KAAK,kBAAkB,EAAE,MAAM,oBAAoB,CAAC;AACrF,OAAO,EAAuB,KAAK,gBAAgB,EAAE,MAAM,qBAAqB,CAAC;AACjF,OAAO,EAAiB,KAAK,wBAAwB,EAAE,MAAM,gCAAgC,CAAC;AAG9F,sFAAsF;AACtF,kFAAkF;AAClF,eAAO,MAAM,wBAAwB,EAAE,cAAc,CAAC,IAAI,CAA6C,CAAC;AAMxG;;;;;;GAMG;AACH,wBAAgB,iBAAiB,CAAC,EAChC,MAAM,EACN,OAAO,EACP,OAAO,EACP,iBAAiB,EACjB,WAAW,GACZ,EAAE;IACD,MAAM,EAAE,gBAAgB,EAAE,CAAC;IAC3B,OAAO,EAAE,MAAM,CAAC;IAChB,OAAO,EAAE,OAAO,CAAC;IACjB,mFAAmF;IACnF,iBAAiB,CAAC,EAAE,wBAAwB,CAAC;IAC7C,kEAAkE;IAClE,WAAW,CAAC,EAAE,kBAAkB,CAAC;CAClC,QAwFA"}
|
|
@@ -1,8 +1,10 @@
|
|
|
1
1
|
import { useEffect, useRef } from 'react';
|
|
2
2
|
import { useLexicalComposerContext } from '@lexical/react/LexicalComposerContext';
|
|
3
3
|
import { $getSelection, $isRangeSelection, COMMAND_PRIORITY_EDITOR, createCommand } from 'lexical';
|
|
4
|
-
import { $correctDocument, $correctWordBeforeCaret } from './autoCorrectActions.js';
|
|
4
|
+
import { $correctDocument, $correctPunctuationBeforeCaret, $correctPunctuationDocument, $correctWordBeforeCaret } from './autoCorrectActions.js';
|
|
5
|
+
import { activePunctuationRules } from './punctuationRules.js';
|
|
5
6
|
import { loadAutoCorrections } from './autoCorrectStores.js';
|
|
7
|
+
import { normalizeUrdu } from '../normalization/urduNormalize.js';
|
|
6
8
|
/** Keys that end a word: a space or punctuation, in Latin and Arabic-script forms. */
|
|
7
9
|
/** Corrects every word in the document, e.g. after pasting or loading content. */
|
|
8
10
|
export const CORRECT_DOCUMENT_COMMAND = createCommand('CORRECT_DOCUMENT_COMMAND');
|
|
@@ -16,8 +18,14 @@ const RTL_LANGUAGES = ['ur', 'pa-shahmukhi'];
|
|
|
16
18
|
* the table reloads). The table used depends on the block's direction: RTL
|
|
17
19
|
* blocks use the Urdu and Shahmukhi tables, LTR blocks the English one.
|
|
18
20
|
*/
|
|
19
|
-
export function AutoCorrectPlugin({ stores, version, enabled }) {
|
|
21
|
+
export function AutoCorrectPlugin({ stores, version, enabled, urduNormalization, punctuation, }) {
|
|
20
22
|
const [editor] = useLexicalComposerContext();
|
|
23
|
+
// Effects below run once per editor, so they reach the latest options through this ref.
|
|
24
|
+
const normalizeOptions = useRef(urduNormalization);
|
|
25
|
+
normalizeOptions.current = urduNormalization;
|
|
26
|
+
const punctuationRules = useRef(activePunctuationRules(punctuation));
|
|
27
|
+
punctuationRules.current = activePunctuationRules(punctuation);
|
|
28
|
+
const normalizeRtl = (text) => normalizeUrdu(text, normalizeOptions.current);
|
|
21
29
|
const tables = useRef(new Map());
|
|
22
30
|
const rtlTable = useRef(new Map());
|
|
23
31
|
useEffect(() => {
|
|
@@ -46,7 +54,8 @@ export function AutoCorrectPlugin({ stores, version, enabled }) {
|
|
|
46
54
|
return;
|
|
47
55
|
return editor.registerCommand(CORRECT_DOCUMENT_COMMAND, () => {
|
|
48
56
|
editor.update(() => {
|
|
49
|
-
$
|
|
57
|
+
$correctPunctuationDocument(punctuationRules.current);
|
|
58
|
+
$correctDocument(tables.current.get('en') ?? new Map(), rtlTable.current, normalizeRtl);
|
|
50
59
|
}, { discrete: true });
|
|
51
60
|
return true;
|
|
52
61
|
}, COMMAND_PRIORITY_EDITOR);
|
|
@@ -56,18 +65,24 @@ export function AutoCorrectPlugin({ stores, version, enabled }) {
|
|
|
56
65
|
return;
|
|
57
66
|
const tableFor = (rtl) => (rtl ? rtlTable.current : tables.current.get('en'));
|
|
58
67
|
const onKeyDown = (event) => {
|
|
59
|
-
if (event.isComposing ||
|
|
68
|
+
if (event.isComposing || event.key.length !== 1)
|
|
60
69
|
return;
|
|
61
|
-
|
|
70
|
+
const boundary = BOUNDARY_KEYS.has(event.key);
|
|
71
|
+
// The character is inserted after this event; correct once it is there.
|
|
62
72
|
setTimeout(() => {
|
|
63
73
|
editor.update(() => {
|
|
64
74
|
const selection = $getSelection();
|
|
65
75
|
if (!$isRangeSelection(selection))
|
|
66
76
|
return;
|
|
77
|
+
// Punctuation can end in any character (e.g. the letter ه), so every key is checked.
|
|
78
|
+
$correctPunctuationBeforeCaret(punctuationRules.current);
|
|
79
|
+
if (!boundary)
|
|
80
|
+
return;
|
|
67
81
|
const block = selection.anchor.getNode().getTopLevelElement();
|
|
68
|
-
const
|
|
82
|
+
const rtl = block?.getDirection() === 'rtl';
|
|
83
|
+
const table = tableFor(rtl);
|
|
69
84
|
if (table)
|
|
70
|
-
$correctWordBeforeCaret(table);
|
|
85
|
+
$correctWordBeforeCaret(table, rtl ? normalizeRtl : undefined);
|
|
71
86
|
}, { discrete: true });
|
|
72
87
|
}, 0);
|
|
73
88
|
};
|
|
@@ -1,15 +1,27 @@
|
|
|
1
|
+
import type { PunctuationRule } from './punctuationRules.js';
|
|
1
2
|
/**
|
|
2
3
|
* Corrects the word that ends just before the caret, when the character just
|
|
3
4
|
* typed (the word boundary: space or punctuation) is not a word character. The
|
|
4
5
|
* correction replaces only the word, and the caret stays right after the
|
|
5
6
|
* boundary. Returns true if a correction was made.
|
|
6
7
|
*/
|
|
7
|
-
export declare function $correctWordBeforeCaret(table: Map<string, string
|
|
8
|
+
export declare function $correctWordBeforeCaret(table: Map<string, string>, normalize?: (word: string) => string): boolean;
|
|
8
9
|
/**
|
|
9
10
|
* Corrects every whole word in the document, for text that never went through
|
|
10
11
|
* typing (pasted or loaded content). Words are found per paragraph or cell, so a
|
|
11
12
|
* word split by formatting is still one word. Each block uses the table for its
|
|
12
13
|
* direction. Returns how many words were corrected.
|
|
13
14
|
*/
|
|
14
|
-
export declare function $correctDocument(ltr: Map<string, string>, rtl: Map<string, string
|
|
15
|
+
export declare function $correctDocument(ltr: Map<string, string>, rtl: Map<string, string>, normalizeRtl?: (text: string) => string): number;
|
|
16
|
+
/**
|
|
17
|
+
* Applies a punctuation rule to the text just before the caret, when that text
|
|
18
|
+
* ends with a rule's incorrect form. Longest match first. The caret ends after
|
|
19
|
+
* the replacement.
|
|
20
|
+
*/
|
|
21
|
+
export declare function $correctPunctuationBeforeCaret(rules: PunctuationRule[]): boolean;
|
|
22
|
+
/**
|
|
23
|
+
* Applies the punctuation rules to every paragraph and cell in the document.
|
|
24
|
+
* Where two matches overlap, the longer one wins. Returns how many were applied.
|
|
25
|
+
*/
|
|
26
|
+
export declare function $correctPunctuationDocument(rules: PunctuationRule[]): number;
|
|
15
27
|
//# sourceMappingURL=autoCorrectActions.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"autoCorrectActions.d.ts","sourceRoot":"","sources":["../../src/autocorrect/autoCorrectActions.ts"],"names":[],"mappings":"
|
|
1
|
+
{"version":3,"file":"autoCorrectActions.d.ts","sourceRoot":"","sources":["../../src/autocorrect/autoCorrectActions.ts"],"names":[],"mappings":"AAGA,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,oBAAoB,CAAC;AAM1D;;;;;GAKG;AACH,wBAAgB,uBAAuB,CAAC,KAAK,EAAE,GAAG,CAAC,MAAM,EAAE,MAAM,CAAC,EAAE,SAAS,CAAC,EAAE,CAAC,IAAI,EAAE,MAAM,KAAK,MAAM,GAAG,OAAO,CAiCjH;AAED;;;;;GAKG;AACH,wBAAgB,gBAAgB,CAC9B,GAAG,EAAE,GAAG,CAAC,MAAM,EAAE,MAAM,CAAC,EACxB,GAAG,EAAE,GAAG,CAAC,MAAM,EAAE,MAAM,CAAC,EACxB,YAAY,CAAC,EAAE,CAAC,IAAI,EAAE,MAAM,KAAK,MAAM,GACtC,MAAM,CA8BR;AAQD;;;;GAIG;AACH,wBAAgB,8BAA8B,CAAC,KAAK,EAAE,eAAe,EAAE,GAAG,OAAO,CAqBhF;AAED;;;GAGG;AACH,wBAAgB,2BAA2B,CAAC,KAAK,EAAE,eAAe,EAAE,GAAG,MAAM,CA2B5E"}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { $getSelection, $isRangeSelection, $isTextNode } from 'lexical';
|
|
1
|
+
import { $getRoot, $getSelection, $isRangeSelection, $isTextNode } from 'lexical';
|
|
2
2
|
import { $locate, $replaceMatch, $textGroups } from '../find/findReplaceActions.js';
|
|
3
3
|
import { wordsIn } from '../spellcheck/spellDictionaries.js';
|
|
4
4
|
/** The part of a word that counts as word characters at its end. */
|
|
@@ -10,7 +10,7 @@ const WORD_CHAR = /[\p{L}\p{M}'’]/u;
|
|
|
10
10
|
* correction replaces only the word, and the caret stays right after the
|
|
11
11
|
* boundary. Returns true if a correction was made.
|
|
12
12
|
*/
|
|
13
|
-
export function $correctWordBeforeCaret(table) {
|
|
13
|
+
export function $correctWordBeforeCaret(table, normalize) {
|
|
14
14
|
const selection = $getSelection();
|
|
15
15
|
if (!$isRangeSelection(selection) || !selection.isCollapsed())
|
|
16
16
|
return false;
|
|
@@ -26,7 +26,7 @@ export function $correctWordBeforeCaret(table) {
|
|
|
26
26
|
const word = WORD_END.exec(body)?.[0];
|
|
27
27
|
if (!word)
|
|
28
28
|
return false;
|
|
29
|
-
const replacement =
|
|
29
|
+
const replacement = correctionFor(word, table, normalize);
|
|
30
30
|
if (replacement === undefined || replacement === word)
|
|
31
31
|
return false;
|
|
32
32
|
const start = body.length - word.length;
|
|
@@ -47,19 +47,97 @@ export function $correctWordBeforeCaret(table) {
|
|
|
47
47
|
* word split by formatting is still one word. Each block uses the table for its
|
|
48
48
|
* direction. Returns how many words were corrected.
|
|
49
49
|
*/
|
|
50
|
-
export function $correctDocument(ltr, rtl) {
|
|
50
|
+
export function $correctDocument(ltr, rtl, normalizeRtl) {
|
|
51
|
+
if (normalizeRtl) {
|
|
52
|
+
for (const node of $getRoot().getAllTextNodes()) {
|
|
53
|
+
if (node.getParentOrThrow().getDirection() !== 'rtl')
|
|
54
|
+
continue;
|
|
55
|
+
const text = node.getTextContent();
|
|
56
|
+
const normalized = normalizeRtl(text);
|
|
57
|
+
if (normalized !== text)
|
|
58
|
+
node.setTextContent(normalized);
|
|
59
|
+
}
|
|
60
|
+
}
|
|
51
61
|
let corrected = 0;
|
|
52
62
|
for (const group of $textGroups()) {
|
|
53
|
-
const
|
|
63
|
+
const rightToLeft = group[0].getParentOrThrow().getDirection() === 'rtl';
|
|
64
|
+
const table = rightToLeft ? rtl : ltr;
|
|
54
65
|
const text = group.map((node) => node.getTextContent()).join('');
|
|
55
|
-
const spans = wordsIn(text)
|
|
66
|
+
const spans = wordsIn(text)
|
|
67
|
+
.map((span) => ({ span, replacement: correctionFor(span.word, table, rightToLeft ? normalizeRtl : undefined) }))
|
|
68
|
+
.filter((item) => item.replacement !== undefined && item.replacement !== item.span.word);
|
|
56
69
|
// Right to left, so earlier positions stay valid as each word is replaced.
|
|
57
|
-
for (const span of spans.reverse()) {
|
|
70
|
+
for (const { span, replacement } of spans.reverse()) {
|
|
58
71
|
const start = $locate(group, span.start, false);
|
|
59
72
|
const end = $locate(group, span.end, true);
|
|
60
|
-
$replaceMatch({ anchorKey: start.key, anchorOffset: start.offset, focusKey: end.key, focusOffset: end.offset },
|
|
73
|
+
$replaceMatch({ anchorKey: start.key, anchorOffset: start.offset, focusKey: end.key, focusOffset: end.offset }, replacement);
|
|
61
74
|
corrected += 1;
|
|
62
75
|
}
|
|
63
76
|
}
|
|
64
77
|
return corrected;
|
|
65
78
|
}
|
|
79
|
+
/** What a word should become: its normalised form, then any table correction for that. */
|
|
80
|
+
function correctionFor(word, table, normalize) {
|
|
81
|
+
const normalized = normalize ? normalize(word) : word;
|
|
82
|
+
return table.get(normalized) ?? (normalized !== word ? normalized : undefined);
|
|
83
|
+
}
|
|
84
|
+
/**
|
|
85
|
+
* Applies a punctuation rule to the text just before the caret, when that text
|
|
86
|
+
* ends with a rule's incorrect form. Longest match first. The caret ends after
|
|
87
|
+
* the replacement.
|
|
88
|
+
*/
|
|
89
|
+
export function $correctPunctuationBeforeCaret(rules) {
|
|
90
|
+
const selection = $getSelection();
|
|
91
|
+
if (!$isRangeSelection(selection) || !selection.isCollapsed())
|
|
92
|
+
return false;
|
|
93
|
+
const anchor = selection.anchor;
|
|
94
|
+
const node = anchor.getNode();
|
|
95
|
+
if (!$isTextNode(node))
|
|
96
|
+
return false;
|
|
97
|
+
const before = node.getTextContent().slice(0, anchor.offset);
|
|
98
|
+
const rule = rules.find((r) => before.endsWith(r.incorrect) && (!r.completeWord || !WORD_CHAR.test(before.charAt(before.length - r.incorrect.length - 1) || ' ')));
|
|
99
|
+
if (!rule)
|
|
100
|
+
return false;
|
|
101
|
+
const start = before.length - rule.incorrect.length;
|
|
102
|
+
$replaceMatch({ anchorKey: node.getKey(), anchorOffset: start, focusKey: node.getKey(), focusOffset: before.length }, rule.correct);
|
|
103
|
+
const after = $getSelection();
|
|
104
|
+
if ($isRangeSelection(after)) {
|
|
105
|
+
const caret = start + rule.correct.length;
|
|
106
|
+
after.anchor.set(node.getKey(), caret, 'text');
|
|
107
|
+
after.focus.set(node.getKey(), caret, 'text');
|
|
108
|
+
}
|
|
109
|
+
return true;
|
|
110
|
+
}
|
|
111
|
+
/**
|
|
112
|
+
* Applies the punctuation rules to every paragraph and cell in the document.
|
|
113
|
+
* Where two matches overlap, the longer one wins. Returns how many were applied.
|
|
114
|
+
*/
|
|
115
|
+
export function $correctPunctuationDocument(rules) {
|
|
116
|
+
let applied = 0;
|
|
117
|
+
for (const group of $textGroups()) {
|
|
118
|
+
const text = group.map((node) => node.getTextContent()).join('');
|
|
119
|
+
const candidates = [];
|
|
120
|
+
for (const rule of rules) {
|
|
121
|
+
let index = text.indexOf(rule.incorrect);
|
|
122
|
+
while (index !== -1) {
|
|
123
|
+
candidates.push({ start: index, end: index + rule.incorrect.length, correct: rule.correct });
|
|
124
|
+
index = text.indexOf(rule.incorrect, index + 1);
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
// Longest first, then earliest; keep only matches that do not overlap a kept one.
|
|
128
|
+
candidates.sort((a, b) => b.end - b.start - (a.end - a.start) || a.start - b.start);
|
|
129
|
+
const kept = [];
|
|
130
|
+
for (const c of candidates) {
|
|
131
|
+
if (!kept.some((k) => c.start < k.end && k.start < c.end))
|
|
132
|
+
kept.push(c);
|
|
133
|
+
}
|
|
134
|
+
// Right to left, so earlier positions stay valid as each match is replaced.
|
|
135
|
+
for (const c of kept.sort((a, b) => b.start - a.start)) {
|
|
136
|
+
const start = $locate(group, c.start, false);
|
|
137
|
+
const end = $locate(group, c.end, true);
|
|
138
|
+
$replaceMatch({ anchorKey: start.key, anchorOffset: start.offset, focusKey: end.key, focusOffset: end.offset }, c.correct);
|
|
139
|
+
applied += 1;
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
return applied;
|
|
143
|
+
}
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Common Urdu punctuation and character fixes, as a snapshot of
|
|
3
|
+
* https://api.nawishta.co.uk/tools/ur/spellchecker/punctuation (fetched 2026-10-04).
|
|
4
|
+
* Bundled so the editor works offline and does not depend on that service;
|
|
5
|
+
* refresh the list when the service changes.
|
|
6
|
+
*/
|
|
7
|
+
export interface PunctuationRule {
|
|
8
|
+
incorrect: string;
|
|
9
|
+
correct: string;
|
|
10
|
+
/** When true, the match must be a whole word (not part of a longer one). */
|
|
11
|
+
completeWord: boolean;
|
|
12
|
+
}
|
|
13
|
+
export declare const URDU_PUNCTUATION_RULES: PunctuationRule[];
|
|
14
|
+
export interface PunctuationOptions {
|
|
15
|
+
/** Apply the punctuation fixes. On by default. */
|
|
16
|
+
enabled?: boolean;
|
|
17
|
+
/** Replace a straight double quote (") with a closing curly quote (”). On by default. */
|
|
18
|
+
straightDoubleQuote?: boolean;
|
|
19
|
+
}
|
|
20
|
+
/** The rules in use for the options, longest match first, so `۔"` wins over `"`. */
|
|
21
|
+
export declare function activePunctuationRules(options?: PunctuationOptions): PunctuationRule[];
|
|
22
|
+
//# sourceMappingURL=punctuationRules.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"punctuationRules.d.ts","sourceRoot":"","sources":["../../src/autocorrect/punctuationRules.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AACH,MAAM,WAAW,eAAe;IAC9B,SAAS,EAAE,MAAM,CAAC;IAClB,OAAO,EAAE,MAAM,CAAC;IAChB,4EAA4E;IAC5E,YAAY,EAAE,OAAO,CAAC;CACvB;AAED,eAAO,MAAM,sBAAsB,EAAE,eAAe,EAmCnD,CAAC;AAEF,MAAM,WAAW,kBAAkB;IACjC,kDAAkD;IAClD,OAAO,CAAC,EAAE,OAAO,CAAC;IAClB,yFAAyF;IACzF,mBAAmB,CAAC,EAAE,OAAO,CAAC;CAC/B;AAED,oFAAoF;AACpF,wBAAgB,sBAAsB,CAAC,OAAO,GAAE,kBAAuB,GAAG,eAAe,EAAE,CAK1F"}
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
export const URDU_PUNCTUATION_RULES = [
|
|
2
|
+
{ incorrect: ".", correct: "۔", completeWord: false },
|
|
3
|
+
{ incorrect: ",", correct: "،", completeWord: false },
|
|
4
|
+
{ incorrect: " ۔", correct: "۔", completeWord: false },
|
|
5
|
+
{ incorrect: " - ", correct: "۔ ", completeWord: false },
|
|
6
|
+
{ incorrect: " ،", correct: "،", completeWord: false },
|
|
7
|
+
{ incorrect: "?", correct: "؟", completeWord: false },
|
|
8
|
+
{ incorrect: " ؟", correct: "؟", completeWord: false },
|
|
9
|
+
{ incorrect: " !", correct: "!", completeWord: false },
|
|
10
|
+
{ incorrect: "( ", correct: "(", completeWord: false },
|
|
11
|
+
{ incorrect: " (", correct: "(", completeWord: false },
|
|
12
|
+
{ incorrect: "ه", correct: "ہ", completeWord: false },
|
|
13
|
+
{ incorrect: "ک", correct: "ک", completeWord: false },
|
|
14
|
+
{ incorrect: "ئو", correct: "ؤ", completeWord: false },
|
|
15
|
+
{ incorrect: "’’", correct: "”", completeWord: false },
|
|
16
|
+
{ incorrect: "‘‘", correct: "“", completeWord: false },
|
|
17
|
+
{ incorrect: "…", correct: "۔۔۔", completeWord: false },
|
|
18
|
+
{ incorrect: "——", correct: "۔۔۔", completeWord: false },
|
|
19
|
+
{ incorrect: "—", correct: "۔۔۔", completeWord: false },
|
|
20
|
+
{ incorrect: " ۔۔۔", correct: "۔۔۔ ", completeWord: false },
|
|
21
|
+
{ incorrect: "،،", correct: "“", completeWord: false },
|
|
22
|
+
{ incorrect: "۔\"", correct: "۔“", completeWord: false },
|
|
23
|
+
{ incorrect: "، \"", correct: "۔ ”", completeWord: false },
|
|
24
|
+
{ incorrect: "؟\"", correct: "؟“", completeWord: false },
|
|
25
|
+
{ incorrect: "!\"", correct: "!“", completeWord: false },
|
|
26
|
+
{ incorrect: "-", correct: "۔", completeWord: false },
|
|
27
|
+
{ incorrect: " ، ", correct: "، ", completeWord: false },
|
|
28
|
+
{ incorrect: " ۔", correct: "۔ ", completeWord: false },
|
|
29
|
+
{ incorrect: " : ", correct: ": ", completeWord: false },
|
|
30
|
+
{ incorrect: "آ", correct: "آ", completeWord: false },
|
|
31
|
+
{ incorrect: "ؤ", correct: "ؤ", completeWord: false },
|
|
32
|
+
{ incorrect: " ِ", correct: "ِ", completeWord: false },
|
|
33
|
+
{ incorrect: " ً", correct: "ً", completeWord: false },
|
|
34
|
+
{ incorrect: " :", correct: ":", completeWord: false },
|
|
35
|
+
{ incorrect: "” ", correct: "”", completeWord: false },
|
|
36
|
+
];
|
|
37
|
+
/** The rules in use for the options, longest match first, so `۔"` wins over `"`. */
|
|
38
|
+
export function activePunctuationRules(options = {}) {
|
|
39
|
+
if (options.enabled === false)
|
|
40
|
+
return [];
|
|
41
|
+
const rules = URDU_PUNCTUATION_RULES.filter((rule) => rule.incorrect !== rule.correct);
|
|
42
|
+
if (options.straightDoubleQuote !== false)
|
|
43
|
+
rules.push({ incorrect: '"', correct: '”', completeWord: false });
|
|
44
|
+
return rules.sort((a, b) => b.incorrect.length - a.incorrect.length);
|
|
45
|
+
}
|
package/dist/index.d.ts
CHANGED
|
@@ -11,4 +11,7 @@ export type { Locale, Strings } from './i18n/index.js';
|
|
|
11
11
|
export { registerSpellDictionary, hasSpellDictionary, type HunspellFiles, type SpellLanguage } from './spellcheck/spellDictionaries.js';
|
|
12
12
|
export { localStorageAutoCorrectStore, apiAutoCorrectStore, fileAutoCorrectStore } from './autocorrect/autoCorrectStores.js';
|
|
13
13
|
export type { AutoCorrectStore, AutoCorrectEntry } from './autocorrect/autoCorrectStores.js';
|
|
14
|
+
export { normalizeUrdu, normalizeUrduCharacters, removeUrduDiacritics, replaceUrduDigits } from './normalization/urduNormalize.js';
|
|
15
|
+
export type { UrduNormalizationOptions } from './normalization/urduNormalize.js';
|
|
16
|
+
export type { PunctuationOptions, PunctuationRule } from './autocorrect/punctuationRules.js';
|
|
14
17
|
//# sourceMappingURL=index.d.ts.map
|
package/dist/index.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,UAAU,EAAE,MAAM,cAAc,CAAC;AAC1C,YAAY,EAAE,eAAe,EAAE,SAAS,EAAE,oBAAoB,EAAE,mBAAmB,EAAE,MAAM,cAAc,CAAC;AAC1G,YAAY,EAAE,gBAAgB,EAAE,MAAM,2BAA2B,CAAC;AAClE,OAAO,EAAE,mBAAmB,EAAE,MAAM,sBAAsB,CAAC;AAC3D,OAAO,EAAE,SAAS,EAAE,gBAAgB,EAAE,YAAY,EAAE,MAAM,mBAAmB,CAAC;AAC9E,YAAY,EAAE,YAAY,EAAE,aAAa,EAAE,mBAAmB,EAAE,MAAM,mBAAmB,CAAC;AAC1F,OAAO,EAAE,oBAAoB,EAAE,qBAAqB,EAAE,aAAa,EAAE,MAAM,SAAS,CAAC;AACrF,YAAY,EAAE,UAAU,EAAE,MAAM,SAAS,CAAC;AAC1C,OAAO,EAAE,OAAO,EAAE,UAAU,EAAE,UAAU,EAAE,YAAY,EAAE,gBAAgB,EAAE,MAAM,QAAQ,CAAC;AACzF,YAAY,EAAE,MAAM,EAAE,OAAO,EAAE,MAAM,QAAQ,CAAC;AAE9C,OAAO,EAAE,uBAAuB,EAAE,kBAAkB,EAAE,KAAK,aAAa,EAAE,KAAK,aAAa,EAAE,MAAM,gCAAgC,CAAC;AACrI,OAAO,EAAE,4BAA4B,EAAE,mBAAmB,EAAE,oBAAoB,EAAE,MAAM,iCAAiC,CAAC;AAC1H,YAAY,EAAE,gBAAgB,EAAE,gBAAgB,EAAE,MAAM,iCAAiC,CAAC"}
|
|
1
|
+
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,UAAU,EAAE,MAAM,cAAc,CAAC;AAC1C,YAAY,EAAE,eAAe,EAAE,SAAS,EAAE,oBAAoB,EAAE,mBAAmB,EAAE,MAAM,cAAc,CAAC;AAC1G,YAAY,EAAE,gBAAgB,EAAE,MAAM,2BAA2B,CAAC;AAClE,OAAO,EAAE,mBAAmB,EAAE,MAAM,sBAAsB,CAAC;AAC3D,OAAO,EAAE,SAAS,EAAE,gBAAgB,EAAE,YAAY,EAAE,MAAM,mBAAmB,CAAC;AAC9E,YAAY,EAAE,YAAY,EAAE,aAAa,EAAE,mBAAmB,EAAE,MAAM,mBAAmB,CAAC;AAC1F,OAAO,EAAE,oBAAoB,EAAE,qBAAqB,EAAE,aAAa,EAAE,MAAM,SAAS,CAAC;AACrF,YAAY,EAAE,UAAU,EAAE,MAAM,SAAS,CAAC;AAC1C,OAAO,EAAE,OAAO,EAAE,UAAU,EAAE,UAAU,EAAE,YAAY,EAAE,gBAAgB,EAAE,MAAM,QAAQ,CAAC;AACzF,YAAY,EAAE,MAAM,EAAE,OAAO,EAAE,MAAM,QAAQ,CAAC;AAE9C,OAAO,EAAE,uBAAuB,EAAE,kBAAkB,EAAE,KAAK,aAAa,EAAE,KAAK,aAAa,EAAE,MAAM,gCAAgC,CAAC;AACrI,OAAO,EAAE,4BAA4B,EAAE,mBAAmB,EAAE,oBAAoB,EAAE,MAAM,iCAAiC,CAAC;AAC1H,YAAY,EAAE,gBAAgB,EAAE,gBAAgB,EAAE,MAAM,iCAAiC,CAAC;AAC1F,OAAO,EAAE,aAAa,EAAE,uBAAuB,EAAE,oBAAoB,EAAE,iBAAiB,EAAE,MAAM,+BAA+B,CAAC;AAChI,YAAY,EAAE,wBAAwB,EAAE,MAAM,+BAA+B,CAAC;AAC9E,YAAY,EAAE,kBAAkB,EAAE,eAAe,EAAE,MAAM,gCAAgC,CAAC"}
|
package/dist/index.js
CHANGED
|
@@ -5,3 +5,4 @@ export { DEFAULT_FONT_OPTIONS, URDU_WEB_FONT_OPTIONS, FONT_SIZES_PX } from './fo
|
|
|
5
5
|
export { STRINGS, getStrings, useStrings, useUiStrings, UiStringsContext } from './i18n/index.js';
|
|
6
6
|
export { registerSpellDictionary, hasSpellDictionary } from './spellcheck/spellDictionaries.js';
|
|
7
7
|
export { localStorageAutoCorrectStore, apiAutoCorrectStore, fileAutoCorrectStore } from './autocorrect/autoCorrectStores.js';
|
|
8
|
+
export { normalizeUrdu, normalizeUrduCharacters, removeUrduDiacritics, replaceUrduDigits } from './normalization/urduNormalize.js';
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Urdu character normalisation, ported from urduhack's `normalization/character.py`
|
|
3
|
+
* (MIT licence, https://github.com/urduhack/urduhack). The mapping replaces
|
|
4
|
+
* Arabic and presentation-form letters with the standard Urdu letter, and
|
|
5
|
+
* the tables below keep the same entries and order as the original, so the
|
|
6
|
+
* output matches it character for character.
|
|
7
|
+
*/
|
|
8
|
+
/** Replaces Arabic and presentation-form letters with standard Urdu letters, character by character. */
|
|
9
|
+
export declare function normalizeUrduCharacters(text: string): string;
|
|
10
|
+
export declare function removeUrduDiacritics(text: string): string;
|
|
11
|
+
/** Converts Urdu digits to English digits, or the reverse. */
|
|
12
|
+
export declare function replaceUrduDigits(text: string, to: 'english' | 'urdu'): string;
|
|
13
|
+
export interface UrduNormalizationOptions {
|
|
14
|
+
/** Remove harakat. Off by default, since they are meaningful in poetry and religious text. */
|
|
15
|
+
removeDiacritics?: boolean;
|
|
16
|
+
/** Digit style to convert to; 'keep' (the default) leaves digits as they are. */
|
|
17
|
+
digits?: 'keep' | 'english' | 'urdu';
|
|
18
|
+
}
|
|
19
|
+
/**
|
|
20
|
+
* The normalisation urduhack's `normalize` applies, with the diacritic and
|
|
21
|
+
* digit steps made optional. Character mapping always runs.
|
|
22
|
+
*/
|
|
23
|
+
export declare function normalizeUrdu(text: string, options?: UrduNormalizationOptions): string;
|
|
24
|
+
//# sourceMappingURL=urduNormalize.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"urduNormalize.d.ts","sourceRoot":"","sources":["../../src/normalization/urduNormalize.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AA8EH,wGAAwG;AACxG,wBAAgB,uBAAuB,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,CAI5D;AAED,wBAAgB,oBAAoB,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,CAIzD;AAED,8DAA8D;AAC9D,wBAAgB,iBAAiB,CAAC,IAAI,EAAE,MAAM,EAAE,EAAE,EAAE,SAAS,GAAG,MAAM,GAAG,MAAM,CAS9E;AAED,MAAM,WAAW,wBAAwB;IACvC,8FAA8F;IAC9F,gBAAgB,CAAC,EAAE,OAAO,CAAC;IAC3B,iFAAiF;IACjF,MAAM,CAAC,EAAE,MAAM,GAAG,SAAS,GAAG,MAAM,CAAC;CACtC;AAED;;;GAGG;AACH,wBAAgB,aAAa,CAAC,IAAI,EAAE,MAAM,EAAE,OAAO,GAAE,wBAA6B,GAAG,MAAM,CAK1F"}
|
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Urdu character normalisation, ported from urduhack's `normalization/character.py`
|
|
3
|
+
* (MIT licence, https://github.com/urduhack/urduhack). The mapping replaces
|
|
4
|
+
* Arabic and presentation-form letters with the standard Urdu letter, and
|
|
5
|
+
* the tables below keep the same entries and order as the original, so the
|
|
6
|
+
* output matches it character for character.
|
|
7
|
+
*/
|
|
8
|
+
const CORRECT_URDU_CHARACTERS = {
|
|
9
|
+
"آ": ["ﺁ", "ﺂ"],
|
|
10
|
+
"أ": ["ﺃ"],
|
|
11
|
+
"ا": ["ﺍ", "ﺎ"],
|
|
12
|
+
"ب": ["ﺏ", "ﺐ", "ﺑ", "ﺒ"],
|
|
13
|
+
"پ": ["ﭖ", "ﭘ", "ﭙ"],
|
|
14
|
+
"ت": ["ﺕ", "ﺖ", "ﺗ", "ﺘ"],
|
|
15
|
+
"ٹ": ["ﭦ", "ﭧ", "ﭨ", "ﭩ"],
|
|
16
|
+
"ث": ["ﺛ", "ﺜ", "ﺚ"],
|
|
17
|
+
"ج": ["ﺝ", "ﺞ", "ﺟ", "ﺠ"],
|
|
18
|
+
"ح": ["ﺡ", "ﺣ", "ﺤ", "ﺢ"],
|
|
19
|
+
"خ": ["ﺧ", "ﺨ", "ﺦ"],
|
|
20
|
+
"د": ["ﺩ", "ﺪ"],
|
|
21
|
+
"ذ": ["ﺬ", "ﺫ"],
|
|
22
|
+
"ر": ["ﺭ", "ﺮ"],
|
|
23
|
+
"ز": ["ﺯ", "ﺰ"],
|
|
24
|
+
"س": ["ﺱ", "ﺲ", "ﺳ", "ﺴ"],
|
|
25
|
+
"ش": ["ﺵ", "ﺶ", "ﺷ", "ﺸ"],
|
|
26
|
+
"ص": ["ﺹ", "ﺺ", "ﺻ", "ﺼ"],
|
|
27
|
+
"ض": ["ﺽ", "ﺾ", "ﺿ", "ﻀ"],
|
|
28
|
+
"ط": ["ﻃ", "ﻄ"],
|
|
29
|
+
"ظ": ["ﻅ", "ﻇ", "ﻈ"],
|
|
30
|
+
"ع": ["ﻉ", "ﻊ", "ﻋ", "ﻌ"],
|
|
31
|
+
"غ": ["ﻍ", "ﻏ", "ﻐ"],
|
|
32
|
+
"ف": ["ﻑ", "ﻒ", "ﻓ", "ﻔ"],
|
|
33
|
+
"ق": ["ﻕ", "ﻖ", "ﻗ", "ﻘ"],
|
|
34
|
+
"ل": ["ﻝ", "ﻞ", "ﻟ", "ﻠ"],
|
|
35
|
+
"م": ["ﻡ", "ﻢ", "ﻣ", "ﻤ"],
|
|
36
|
+
"ن": ["ﻥ", "ﻦ", "ﻧ", "ﻨ"],
|
|
37
|
+
"چ": ["ﭺ", "ﭻ", "ﭼ", "ﭽ"],
|
|
38
|
+
"ڈ": ["ﮈ", "ﮉ"],
|
|
39
|
+
"ڑ": ["ﮍ", "ﮌ"],
|
|
40
|
+
"ژ": ["ﮋ"],
|
|
41
|
+
"ک": ["ﮎ", "ﮏ", "ﮐ", "ﮑ", "ﻛ", "ك"],
|
|
42
|
+
"گ": ["ﮒ", "ﮓ", "ﮔ", "ﮕ"],
|
|
43
|
+
"ں": ["ﮞ", "ﮟ"],
|
|
44
|
+
"و": ["ﻮ", "ﻭ", "ﻮ"],
|
|
45
|
+
"ؤ": ["ﺅ"],
|
|
46
|
+
"ھ": ["ﮪ", "ﮬ", "ﮭ", "ﻬ", "ﻫ", "ﮫ"],
|
|
47
|
+
"ہ": ["ﻩ", "ﮦ", "ﻪ", "ﮧ", "ﮩ", "ﮨ", "ه"],
|
|
48
|
+
"ۂ": [],
|
|
49
|
+
"ۃ": ["ة"],
|
|
50
|
+
"ء": ["ﺀ"],
|
|
51
|
+
"ی": ["ﯼ", "ى", "ﯽ", "ﻰ", "ﻱ", "ﻲ", "ﯾ", "ﯿ", "ي"],
|
|
52
|
+
"ئ": ["ﺋ", "ﺌ"],
|
|
53
|
+
"ے": ["ﮮ", "ﮯ", "ﻳ", "ﻴ"],
|
|
54
|
+
"ۓ": [],
|
|
55
|
+
"۰": ["٠"],
|
|
56
|
+
"۱": ["١"],
|
|
57
|
+
"۲": ["٢"],
|
|
58
|
+
"۳": ["٣"],
|
|
59
|
+
"۴": ["٤"],
|
|
60
|
+
"۵": ["٥"],
|
|
61
|
+
"۶": ["٦"],
|
|
62
|
+
"۷": ["٧"],
|
|
63
|
+
"۸": ["٨"],
|
|
64
|
+
"۹": ["٩"],
|
|
65
|
+
"۔": [],
|
|
66
|
+
"؟": [],
|
|
67
|
+
"٫": [],
|
|
68
|
+
"،": [],
|
|
69
|
+
"لا": ["ﻻ", "ﻼ"],
|
|
70
|
+
"": ["ـ"],
|
|
71
|
+
};
|
|
72
|
+
const TRANSLATOR = new Map();
|
|
73
|
+
for (const [key, values] of Object.entries(CORRECT_URDU_CHARACTERS)) {
|
|
74
|
+
for (const value of values)
|
|
75
|
+
TRANSLATOR.set(value, key);
|
|
76
|
+
}
|
|
77
|
+
/** Combining diacritics (harakat and similar), as urduhack's URDU_DIACRITICS. */
|
|
78
|
+
const DIACRITICS = new Set(['َ', 'ً', 'ٰ', 'ِ', 'ُ', 'ٍ']);
|
|
79
|
+
const ENGLISH_DIGITS = ['0', '1', '2', '3', '4', '5', '6', '7', '8', '9'];
|
|
80
|
+
const URDU_DIGITS = ['۰', '۱', '۲', '۳', '۴', '۵', '۶', '۷', '۸', '۹'];
|
|
81
|
+
/** Replaces Arabic and presentation-form letters with standard Urdu letters, character by character. */
|
|
82
|
+
export function normalizeUrduCharacters(text) {
|
|
83
|
+
let out = '';
|
|
84
|
+
for (const char of text)
|
|
85
|
+
out += TRANSLATOR.get(char) ?? char;
|
|
86
|
+
return out;
|
|
87
|
+
}
|
|
88
|
+
export function removeUrduDiacritics(text) {
|
|
89
|
+
let out = '';
|
|
90
|
+
for (const char of text)
|
|
91
|
+
if (!DIACRITICS.has(char))
|
|
92
|
+
out += char;
|
|
93
|
+
return out;
|
|
94
|
+
}
|
|
95
|
+
/** Converts Urdu digits to English digits, or the reverse. */
|
|
96
|
+
export function replaceUrduDigits(text, to) {
|
|
97
|
+
const from = to === 'english' ? URDU_DIGITS : ENGLISH_DIGITS;
|
|
98
|
+
const target = to === 'english' ? ENGLISH_DIGITS : URDU_DIGITS;
|
|
99
|
+
let out = '';
|
|
100
|
+
for (const char of text) {
|
|
101
|
+
const index = from.indexOf(char);
|
|
102
|
+
out += index === -1 ? char : target[index];
|
|
103
|
+
}
|
|
104
|
+
return out;
|
|
105
|
+
}
|
|
106
|
+
/**
|
|
107
|
+
* The normalisation urduhack's `normalize` applies, with the diacritic and
|
|
108
|
+
* digit steps made optional. Character mapping always runs.
|
|
109
|
+
*/
|
|
110
|
+
export function normalizeUrdu(text, options = {}) {
|
|
111
|
+
let out = options.removeDiacritics ? removeUrduDiacritics(text) : text;
|
|
112
|
+
out = normalizeUrduCharacters(out);
|
|
113
|
+
if (options.digits === 'english' || options.digits === 'urdu')
|
|
114
|
+
out = replaceUrduDigits(out, options.digits);
|
|
115
|
+
return out;
|
|
116
|
+
}
|