@laisuk/opencc-fmmseg-wasm 0.3.4 → 0.3.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +59 -13
- package/bin/opencc.js +27 -11
- package/opencc_fmmseg_wasm.d.ts +2 -0
- package/opencc_fmmseg_wasm.js +18 -0
- package/opencc_fmmseg_wasm_bg.wasm +0 -0
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -21,6 +21,7 @@ Features:
|
|
|
21
21
|
* Traditional Chinese regional variants
|
|
22
22
|
* Japanese Shinjitai conversion support
|
|
23
23
|
* Chinese script detection (`zho_check`)
|
|
24
|
+
* Optional CJK Compatibility Ideograph normalization
|
|
24
25
|
* In-memory Office / EPUB document conversion
|
|
25
26
|
* Zero-dependency Node.js CLI
|
|
26
27
|
|
|
@@ -162,6 +163,41 @@ cc.convert("汉字", false);
|
|
|
162
163
|
|
|
163
164
|
---
|
|
164
165
|
|
|
166
|
+
### normalizeCompat
|
|
167
|
+
|
|
168
|
+
Normalize Unicode CJK Compatibility Ideographs before conversion.
|
|
169
|
+
|
|
170
|
+
```javascript
|
|
171
|
+
cc.normalizeCompat(text)
|
|
172
|
+
```
|
|
173
|
+
|
|
174
|
+
Parameters:
|
|
175
|
+
|
|
176
|
+
* `text`: input string
|
|
177
|
+
|
|
178
|
+
Returns:
|
|
179
|
+
|
|
180
|
+
* normalized string
|
|
181
|
+
|
|
182
|
+
Example:
|
|
183
|
+
|
|
184
|
+
```javascript
|
|
185
|
+
const cc = new OpenccWasm("t2s");
|
|
186
|
+
|
|
187
|
+
const input = "天龍八部書裡的喬峰是契丹人";
|
|
188
|
+
const normalized = cc.normalizeCompat(input);
|
|
189
|
+
|
|
190
|
+
console.log(normalized);
|
|
191
|
+
// 天龍八部書裡的喬峰是契丹人
|
|
192
|
+
|
|
193
|
+
console.log(cc.convert(normalized, false));
|
|
194
|
+
// 天龙八部书里的乔峰是契丹人
|
|
195
|
+
```
|
|
196
|
+
|
|
197
|
+
This is an optional pre-conversion pass for text that contains compatibility ideographs from Unicode compatibility ranges. Unmapped characters are preserved unchanged. Normal OpenCC conversion does not automatically run this pass, so call it explicitly when compatibility normalization is desired.
|
|
198
|
+
|
|
199
|
+
---
|
|
200
|
+
|
|
165
201
|
### detofu
|
|
166
202
|
|
|
167
203
|
Replace tofu-risk rare CJK extension characters with display-compatible fallbacks.
|
|
@@ -387,8 +423,8 @@ const cc = OpenccWasm.newWithCustomDicts("s2t", [
|
|
|
387
423
|
`Override` replaces the selected slot before inserting the provided pairs. It is powerful and should be used only when
|
|
388
424
|
the caller intentionally wants to discard built-in entries for that slot.
|
|
389
425
|
|
|
390
|
-
Custom dictionary specs identify the target dictionary slot by `DictSlot` name. Slot names are trimmed and normalized
|
|
391
|
-
case-insensitively for the known slots, so `"stphrases"`, `" STPhrases "`, and `"STPhrases"` all select
|
|
426
|
+
Custom dictionary specs identify the target dictionary slot by `DictSlot` name. Slot names are trimmed and normalized
|
|
427
|
+
case-insensitively for the known slots, so `"stphrases"`, `" STPhrases "`, and `"STPhrases"` all select
|
|
392
428
|
`STPhrases`. Canonical names are recommended in TypeScript code and docs:
|
|
393
429
|
|
|
394
430
|
```text
|
|
@@ -415,7 +451,7 @@ STPunctuations
|
|
|
415
451
|
TSPunctuations
|
|
416
452
|
```
|
|
417
453
|
|
|
418
|
-
Suffixes such as `.txt` are not accepted, even though case and surrounding whitespace are normalized. Use
|
|
454
|
+
Suffixes such as `.txt` are not accepted, even though case and surrounding whitespace are normalized. Use
|
|
419
455
|
`"STPhrases"` or `"stphrases"`, not `"STPhrases.txt"`.
|
|
420
456
|
|
|
421
457
|
Merge contract:
|
|
@@ -584,6 +620,8 @@ The package includes a zero-dependency Node.js CLI:
|
|
|
584
620
|
opencc-fmmseg convert -i input.txt -o output.txt -c s2t -p
|
|
585
621
|
opencc-fmmseg convert -i input.txt -o output.txt -c t2s -p --detofu all
|
|
586
622
|
echo "别随便录影侵犯个人隐私权" | opencc-fmmseg convert -c s2hkp
|
|
623
|
+
echo "天龍八部書裡的喬峰是契丹人" | opencc-fmmseg convert -c t2s --norm-compat
|
|
624
|
+
// 天龙八部书里的乔峰是契丹人
|
|
587
625
|
echo "這個細路哥很靈活" | opencc-fmmseg convert -c hk2sp --custom-dict hkphrasesrev:append:my_hk_dict.txt
|
|
588
626
|
// 这个小男孩很灵活
|
|
589
627
|
```
|
|
@@ -603,21 +641,23 @@ opencc-fmmseg office -i input.docx -o output.docx -c s2t -p --keep-font
|
|
|
603
641
|
### Text Conversion Options
|
|
604
642
|
|
|
605
643
|
```text
|
|
606
|
-
-i, --input <file> Input text file
|
|
607
|
-
-o, --output <file> Output text file
|
|
608
|
-
-c, --config <conversion> Conversion config
|
|
644
|
+
-i, --input <file> Input text file; stdin if omitted
|
|
645
|
+
-o, --output <file> Output text file; stdout if omitted
|
|
646
|
+
-c, --config <conversion> Conversion config (default: s2t)
|
|
609
647
|
-p, --punct Enable punctuation conversion
|
|
610
648
|
--detofu [level] Replace tofu-risk rare CJK extension chars after conversion
|
|
611
649
|
level: all | ext-b | ext-c | ext-d | ext-e | ext-f | ext-g | ext-h | ext-i
|
|
612
650
|
default when omitted value: all
|
|
613
|
-
--
|
|
651
|
+
--keep-ids Preserve complete IDS expressions during conversion (default: false)
|
|
652
|
+
-n, --norm-compat Normalize CJK Compatibility Ideographs before conversion (default: false)
|
|
653
|
+
-D, --custom-dict <slot:mode:file>
|
|
614
654
|
Load a custom dictionary.
|
|
615
655
|
May be specified multiple times.
|
|
616
656
|
Examples:
|
|
617
657
|
--custom-dict hkphrasesrev:append:my_hk_dict.txt
|
|
618
658
|
--custom-dict stphrases:override:terms.txt
|
|
619
|
-
--in-enc <encoding> Input encoding
|
|
620
|
-
--out-enc <encoding> Output encoding
|
|
659
|
+
--in-enc <encoding> Input encoding (default: utf8)
|
|
660
|
+
--out-enc <encoding> Output encoding (default: utf8)
|
|
621
661
|
```
|
|
622
662
|
|
|
623
663
|
Supported conversion configs:
|
|
@@ -632,12 +672,18 @@ tw2s, tw2sp, tw2t, tw2tp, hk2s, hk2sp, hk2t, jp2t, t2jp
|
|
|
632
672
|
```text
|
|
633
673
|
-i, --input <file> Input Office / EPUB file
|
|
634
674
|
-o, --output <file> Output file
|
|
635
|
-
-c, --config <conversion> Conversion config
|
|
675
|
+
-c, --config <conversion> Conversion config (default: s2t)
|
|
636
676
|
-p, --punct Enable punctuation conversion
|
|
637
|
-
--format <format>
|
|
638
|
-
--
|
|
639
|
-
--keep-font Preserve font-family information
|
|
677
|
+
-f, --format <format> docx | xlsx | pptx | odt | ods | odp | epub
|
|
678
|
+
-F, --convert-filename Convert generated output filename stem (default: false)
|
|
679
|
+
--keep-font Preserve font-family information (default)
|
|
640
680
|
--no-keep-font Do not preserve font-family information
|
|
681
|
+
--custom-dict <slot:mode:file>
|
|
682
|
+
Load a custom dictionary.
|
|
683
|
+
May be specified multiple times.
|
|
684
|
+
Examples:
|
|
685
|
+
--custom-dict hkphrasesrev:append:my_hk_dict.txt
|
|
686
|
+
--custom-dict stphrases:override:terms.txt
|
|
641
687
|
```
|
|
642
688
|
|
|
643
689
|
For `office`, the format is inferred from the input file extension when `--format` is omitted.
|
package/bin/opencc.js
CHANGED
|
@@ -57,7 +57,8 @@ Convert options:
|
|
|
57
57
|
level: all | ext-b | ext-c | ext-d | ext-e | ext-f | ext-g | ext-h | ext-i
|
|
58
58
|
default when omitted value: all
|
|
59
59
|
--keep-ids Preserve complete IDS expressions during conversion (default: false)
|
|
60
|
-
--
|
|
60
|
+
-n, --norm-compat Normalize CJK Compatibility Ideographs before conversion (default: false)
|
|
61
|
+
-D, --custom-dict <slot:mode:file>
|
|
61
62
|
Load a custom dictionary.
|
|
62
63
|
May be specified multiple times.
|
|
63
64
|
Examples:
|
|
@@ -75,8 +76,8 @@ Office options:
|
|
|
75
76
|
-o, --output <file> Output file
|
|
76
77
|
-c, --config <conversion> Conversion config (default: s2t)
|
|
77
78
|
-p, --punct Enable punctuation conversion
|
|
78
|
-
--format <format>
|
|
79
|
-
--convert-filename
|
|
79
|
+
-f, --format <format> docx | xlsx | pptx | odt | ods | odp | epub
|
|
80
|
+
-F, --convert-filename Convert generated output filename stem (default: false)
|
|
80
81
|
--keep-font Preserve font-family information (default)
|
|
81
82
|
--no-keep-font Do not preserve font-family information
|
|
82
83
|
--custom-dict <slot:mode:file>
|
|
@@ -126,12 +127,18 @@ function getArg(args, shortName, longName, defaultValue = null) {
|
|
|
126
127
|
return defaultValue;
|
|
127
128
|
}
|
|
128
129
|
|
|
129
|
-
function getArgs(args, longName) {
|
|
130
|
+
function getArgs(args, shortName, longName) {
|
|
130
131
|
const values = [];
|
|
132
|
+
const candidates = [];
|
|
133
|
+
|
|
134
|
+
if (shortName) candidates.push(shortName);
|
|
135
|
+
if (longName) candidates.push(longName);
|
|
136
|
+
|
|
137
|
+
if (candidates.length === 0) return values;
|
|
131
138
|
|
|
132
139
|
for (let i = 0; i < args.length; i++) {
|
|
133
|
-
if (args[i]
|
|
134
|
-
values.push(
|
|
140
|
+
if (candidates.includes(args[i]) && i + 1 < args.length) {
|
|
141
|
+
values.push(args[i + 1]);
|
|
135
142
|
i++;
|
|
136
143
|
}
|
|
137
144
|
}
|
|
@@ -344,7 +351,9 @@ async function runConvert(args) {
|
|
|
344
351
|
const outEnc = getArg(args, null, "--out-enc", "utf8");
|
|
345
352
|
const punct = hasFlag(args, "-p", "--punct");
|
|
346
353
|
const keepIds = hasFlag(args, null, "--keep-ids");
|
|
347
|
-
const
|
|
354
|
+
const normCompat = hasFlag(args, "-n", "--norm-compat");
|
|
355
|
+
const customDicts = getArgs(args, "-D", "--custom-dict")
|
|
356
|
+
.map(parseCustomDictSpec);
|
|
348
357
|
|
|
349
358
|
const detofuIndex = args.indexOf("--detofu");
|
|
350
359
|
const detofuEnabled = detofuIndex !== -1;
|
|
@@ -376,7 +385,12 @@ async function runConvert(args) {
|
|
|
376
385
|
console.error("Input text to convert, <Ctrl+Z>/<Ctrl+D> to submit:");
|
|
377
386
|
}
|
|
378
387
|
|
|
379
|
-
|
|
388
|
+
let inputText = readInputText(input, inEnc);
|
|
389
|
+
|
|
390
|
+
if (normCompat) {
|
|
391
|
+
inputText = cc.normalizeCompat(inputText);
|
|
392
|
+
}
|
|
393
|
+
|
|
380
394
|
let outputText = cc.convert(inputText, punct);
|
|
381
395
|
|
|
382
396
|
if (detofuEnabled) {
|
|
@@ -394,6 +408,7 @@ async function runConvert(args) {
|
|
|
394
408
|
}
|
|
395
409
|
|
|
396
410
|
const suffixParts = [];
|
|
411
|
+
if (normCompat) suffixParts.push("normalized");
|
|
397
412
|
if (detofuEnabled) suffixParts.push("detofu");
|
|
398
413
|
if (keepIds) suffixParts.push("keep-ids");
|
|
399
414
|
|
|
@@ -406,11 +421,12 @@ async function runOffice(args) {
|
|
|
406
421
|
const input = getArg(args, "-i", "--input");
|
|
407
422
|
let output = getArg(args, "-o", "--output");
|
|
408
423
|
const config = getArg(args, "-c", "--config", "s2t");
|
|
409
|
-
const explicitFormat = getArg(args,
|
|
424
|
+
const explicitFormat = getArg(args, "-f", "--format");
|
|
410
425
|
const punct = hasFlag(args, "-p", "--punct");
|
|
411
|
-
const convertFilename = hasFlag(args,
|
|
426
|
+
const convertFilename = hasFlag(args, "-F", "--convert-filename");
|
|
412
427
|
const keepFont = !hasFlag(args, null, "--no-keep-font");
|
|
413
|
-
const customDicts = getArgs(args, "--custom-dict")
|
|
428
|
+
const customDicts = getArgs(args, "-D", "--custom-dict")
|
|
429
|
+
.map(parseCustomDictSpec);
|
|
414
430
|
|
|
415
431
|
if (!input) {
|
|
416
432
|
throw new Error("Input file is missing.");
|
package/opencc_fmmseg_wasm.d.ts
CHANGED
|
@@ -49,6 +49,7 @@ export class OpenccWasm {
|
|
|
49
49
|
constructor(config?: string | null);
|
|
50
50
|
static newWithCustomDicts(config: string | null | undefined, specs: any): OpenccWasm;
|
|
51
51
|
static newWithEnum(config?: OpenccConfigWasm | null): OpenccWasm;
|
|
52
|
+
normalizeCompat(text: string): string;
|
|
52
53
|
setConfig(config: string): boolean;
|
|
53
54
|
setConfigEnum(config: OpenccConfigWasm): void;
|
|
54
55
|
setPreserveIds(value: boolean): void;
|
|
@@ -77,6 +78,7 @@ export interface InitOutput {
|
|
|
77
78
|
readonly openccwasm_new: (a: number, b: number) => [number, number, number];
|
|
78
79
|
readonly openccwasm_newWithCustomDicts: (a: number, b: number, c: any) => [number, number, number];
|
|
79
80
|
readonly openccwasm_newWithEnum: (a: number) => [number, number, number];
|
|
81
|
+
readonly openccwasm_normalizeCompat: (a: number, b: number, c: number) => [number, number];
|
|
80
82
|
readonly openccwasm_setConfig: (a: number, b: number, c: number) => number;
|
|
81
83
|
readonly openccwasm_setConfigEnum: (a: number, b: number) => void;
|
|
82
84
|
readonly openccwasm_setPreserveIds: (a: number, b: number) => void;
|
package/opencc_fmmseg_wasm.js
CHANGED
|
@@ -235,6 +235,24 @@ export class OpenccWasm {
|
|
|
235
235
|
}
|
|
236
236
|
return OpenccWasm.__wrap(ret[0]);
|
|
237
237
|
}
|
|
238
|
+
/**
|
|
239
|
+
* @param {string} text
|
|
240
|
+
* @returns {string}
|
|
241
|
+
*/
|
|
242
|
+
normalizeCompat(text) {
|
|
243
|
+
let deferred2_0;
|
|
244
|
+
let deferred2_1;
|
|
245
|
+
try {
|
|
246
|
+
const ptr0 = passStringToWasm0(text, wasm.__wbindgen_malloc, wasm.__wbindgen_realloc);
|
|
247
|
+
const len0 = WASM_VECTOR_LEN;
|
|
248
|
+
const ret = wasm.openccwasm_normalizeCompat(this.__wbg_ptr, ptr0, len0);
|
|
249
|
+
deferred2_0 = ret[0];
|
|
250
|
+
deferred2_1 = ret[1];
|
|
251
|
+
return getStringFromWasm0(ret[0], ret[1]);
|
|
252
|
+
} finally {
|
|
253
|
+
wasm.__wbindgen_free(deferred2_0, deferred2_1, 1);
|
|
254
|
+
}
|
|
255
|
+
}
|
|
238
256
|
/**
|
|
239
257
|
* @param {string} config
|
|
240
258
|
* @returns {boolean}
|
|
Binary file
|
package/package.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"name": "@laisuk/opencc-fmmseg-wasm",
|
|
3
3
|
"type": "module",
|
|
4
4
|
"description": "WebAssembly bindings for opencc-fmmseg, a high-performance OpenCC-compatible Simplified/Traditional Chinese converter.",
|
|
5
|
-
"version": "0.3.
|
|
5
|
+
"version": "0.3.6",
|
|
6
6
|
"license": "MIT",
|
|
7
7
|
"repository": {
|
|
8
8
|
"type": "git",
|