@laisuk/opencc-fmmseg-wasm 0.3.4 → 0.3.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -21,6 +21,7 @@ Features:
21
21
  * Traditional Chinese regional variants
22
22
  * Japanese Shinjitai conversion support
23
23
  * Chinese script detection (`zho_check`)
24
+ * Optional CJK Compatibility Ideograph normalization
24
25
  * In-memory Office / EPUB document conversion
25
26
  * Zero-dependency Node.js CLI
26
27
 
@@ -162,6 +163,41 @@ cc.convert("汉字", false);
162
163
 
163
164
  ---
164
165
 
166
+ ### normalizeCompat
167
+
168
+ Normalize Unicode CJK Compatibility Ideographs before conversion.
169
+
170
+ ```javascript
171
+ cc.normalizeCompat(text)
172
+ ```
173
+
174
+ Parameters:
175
+
176
+ * `text`: input string
177
+
178
+ Returns:
179
+
180
+ * normalized string
181
+
182
+ Example:
183
+
184
+ ```javascript
185
+ const cc = new OpenccWasm("t2s");
186
+
187
+ const input = "天龍八部書裡的喬峰是契丹人";
188
+ const normalized = cc.normalizeCompat(input);
189
+
190
+ console.log(normalized);
191
+ // 天龍八部書裡的喬峰是契丹人
192
+
193
+ console.log(cc.convert(normalized, false));
194
+ // 天龙八部书里的乔峰是契丹人
195
+ ```
196
+
197
+ This is an optional pre-conversion pass for text that contains compatibility ideographs from Unicode compatibility ranges. Unmapped characters are preserved unchanged. Normal OpenCC conversion does not automatically run this pass, so call it explicitly when compatibility normalization is desired.
198
+
199
+ ---
200
+
165
201
  ### detofu
166
202
 
167
203
  Replace tofu-risk rare CJK extension characters with display-compatible fallbacks.
@@ -387,8 +423,8 @@ const cc = OpenccWasm.newWithCustomDicts("s2t", [
387
423
  `Override` replaces the selected slot before inserting the provided pairs. It is powerful and should be used only when
388
424
  the caller intentionally wants to discard built-in entries for that slot.
389
425
 
390
- Custom dictionary specs identify the target dictionary slot by `DictSlot` name. Slot names are trimmed and normalized
391
- case-insensitively for the known slots, so `"stphrases"`, `" STPhrases "`, and `"STPhrases"` all select
426
+ Custom dictionary specs identify the target dictionary slot by `DictSlot` name. Slot names are trimmed and normalized
427
+ case-insensitively for the known slots, so `"stphrases"`, `" STPhrases "`, and `"STPhrases"` all select
392
428
  `STPhrases`. Canonical names are recommended in TypeScript code and docs:
393
429
 
394
430
  ```text
@@ -415,7 +451,7 @@ STPunctuations
415
451
  TSPunctuations
416
452
  ```
417
453
 
418
- Suffixes such as `.txt` are not accepted, even though case and surrounding whitespace are normalized. Use
454
+ Suffixes such as `.txt` are not accepted, even though case and surrounding whitespace are normalized. Use
419
455
  `"STPhrases"` or `"stphrases"`, not `"STPhrases.txt"`.
420
456
 
421
457
  Merge contract:
@@ -584,6 +620,8 @@ The package includes a zero-dependency Node.js CLI:
584
620
  opencc-fmmseg convert -i input.txt -o output.txt -c s2t -p
585
621
  opencc-fmmseg convert -i input.txt -o output.txt -c t2s -p --detofu all
586
622
  echo "别随便录影侵犯个人隐私权" | opencc-fmmseg convert -c s2hkp
623
+ echo "天龍八部書裡的喬峰是契丹人" | opencc-fmmseg convert -c t2s --norm-compat
624
+ // 天龙八部书里的乔峰是契丹人
587
625
  echo "這個細路哥很靈活" | opencc-fmmseg convert -c hk2sp --custom-dict hkphrasesrev:append:my_hk_dict.txt
588
626
  // 这个小男孩很灵活
589
627
  ```
@@ -603,21 +641,23 @@ opencc-fmmseg office -i input.docx -o output.docx -c s2t -p --keep-font
603
641
  ### Text Conversion Options
604
642
 
605
643
  ```text
606
- -i, --input <file> Input text file
607
- -o, --output <file> Output text file
608
- -c, --config <conversion> Conversion config
644
+ -i, --input <file> Input text file; stdin if omitted
645
+ -o, --output <file> Output text file; stdout if omitted
646
+ -c, --config <conversion> Conversion config (default: s2t)
609
647
  -p, --punct Enable punctuation conversion
610
648
  --detofu [level] Replace tofu-risk rare CJK extension chars after conversion
611
649
  level: all | ext-b | ext-c | ext-d | ext-e | ext-f | ext-g | ext-h | ext-i
612
650
  default when omitted value: all
613
- --custom-dict <slot:mode:file>
651
+ --keep-ids Preserve complete IDS expressions during conversion (default: false)
652
+ -n, --norm-compat Normalize CJK Compatibility Ideographs before conversion (default: false)
653
+ -D, --custom-dict <slot:mode:file>
614
654
  Load a custom dictionary.
615
655
  May be specified multiple times.
616
656
  Examples:
617
657
  --custom-dict hkphrasesrev:append:my_hk_dict.txt
618
658
  --custom-dict stphrases:override:terms.txt
619
- --in-enc <encoding> Input encoding
620
- --out-enc <encoding> Output encoding
659
+ --in-enc <encoding> Input encoding (default: utf8)
660
+ --out-enc <encoding> Output encoding (default: utf8)
621
661
  ```
622
662
 
623
663
  Supported conversion configs:
@@ -632,12 +672,18 @@ tw2s, tw2sp, tw2t, tw2tp, hk2s, hk2sp, hk2t, jp2t, t2jp
632
672
  ```text
633
673
  -i, --input <file> Input Office / EPUB file
634
674
  -o, --output <file> Output file
635
- -c, --config <conversion> Conversion config
675
+ -c, --config <conversion> Conversion config (default: s2t)
636
676
  -p, --punct Enable punctuation conversion
637
- --format <format> docx | xlsx | pptx | odt | ods | odp | epub
638
- --auto-ext Append extension to output if missing
639
- --keep-font Preserve font-family information
677
+ -f, --format <format> docx | xlsx | pptx | odt | ods | odp | epub
678
+ -F, --convert-filename Convert generated output filename stem (default: false)
679
+ --keep-font Preserve font-family information (default)
640
680
  --no-keep-font Do not preserve font-family information
681
+ --custom-dict <slot:mode:file>
682
+ Load a custom dictionary.
683
+ May be specified multiple times.
684
+ Examples:
685
+ --custom-dict hkphrasesrev:append:my_hk_dict.txt
686
+ --custom-dict stphrases:override:terms.txt
641
687
  ```
642
688
 
643
689
  For `office`, the format is inferred from the input file extension when `--format` is omitted.
package/bin/opencc.js CHANGED
@@ -57,7 +57,8 @@ Convert options:
57
57
  level: all | ext-b | ext-c | ext-d | ext-e | ext-f | ext-g | ext-h | ext-i
58
58
  default when omitted value: all
59
59
  --keep-ids Preserve complete IDS expressions during conversion (default: false)
60
- --custom-dict <slot:mode:file>
60
+ -n, --norm-compat Normalize CJK Compatibility Ideographs before conversion (default: false)
61
+ -D, --custom-dict <slot:mode:file>
61
62
  Load a custom dictionary.
62
63
  May be specified multiple times.
63
64
  Examples:
@@ -75,8 +76,8 @@ Office options:
75
76
  -o, --output <file> Output file
76
77
  -c, --config <conversion> Conversion config (default: s2t)
77
78
  -p, --punct Enable punctuation conversion
78
- --format <format> docx | xlsx | pptx | odt | ods | odp | epub
79
- --convert-filename Convert generated output filename stem (default: false)
79
+ -f, --format <format> docx | xlsx | pptx | odt | ods | odp | epub
80
+ -F, --convert-filename Convert generated output filename stem (default: false)
80
81
  --keep-font Preserve font-family information (default)
81
82
  --no-keep-font Do not preserve font-family information
82
83
  --custom-dict <slot:mode:file>
@@ -126,12 +127,18 @@ function getArg(args, shortName, longName, defaultValue = null) {
126
127
  return defaultValue;
127
128
  }
128
129
 
129
- function getArgs(args, longName) {
130
+ function getArgs(args, shortName, longName) {
130
131
  const values = [];
132
+ const candidates = [];
133
+
134
+ if (shortName) candidates.push(shortName);
135
+ if (longName) candidates.push(longName);
136
+
137
+ if (candidates.length === 0) return values;
131
138
 
132
139
  for (let i = 0; i < args.length; i++) {
133
- if (args[i] === longName && i + 1 < args.length) {
134
- values.push(parseCustomDictSpec(args[i + 1]));
140
+ if (candidates.includes(args[i]) && i + 1 < args.length) {
141
+ values.push(args[i + 1]);
135
142
  i++;
136
143
  }
137
144
  }
@@ -344,7 +351,9 @@ async function runConvert(args) {
344
351
  const outEnc = getArg(args, null, "--out-enc", "utf8");
345
352
  const punct = hasFlag(args, "-p", "--punct");
346
353
  const keepIds = hasFlag(args, null, "--keep-ids");
347
- const customDicts = getArgs(args, "--custom-dict");
354
+ const normCompat = hasFlag(args, "-n", "--norm-compat");
355
+ const customDicts = getArgs(args, "-D", "--custom-dict")
356
+ .map(parseCustomDictSpec);
348
357
 
349
358
  const detofuIndex = args.indexOf("--detofu");
350
359
  const detofuEnabled = detofuIndex !== -1;
@@ -376,7 +385,12 @@ async function runConvert(args) {
376
385
  console.error("Input text to convert, <Ctrl+Z>/<Ctrl+D> to submit:");
377
386
  }
378
387
 
379
- const inputText = readInputText(input, inEnc);
388
+ let inputText = readInputText(input, inEnc);
389
+
390
+ if (normCompat) {
391
+ inputText = cc.normalizeCompat(inputText);
392
+ }
393
+
380
394
  let outputText = cc.convert(inputText, punct);
381
395
 
382
396
  if (detofuEnabled) {
@@ -394,6 +408,7 @@ async function runConvert(args) {
394
408
  }
395
409
 
396
410
  const suffixParts = [];
411
+ if (normCompat) suffixParts.push("normalized");
397
412
  if (detofuEnabled) suffixParts.push("detofu");
398
413
  if (keepIds) suffixParts.push("keep-ids");
399
414
 
@@ -406,11 +421,12 @@ async function runOffice(args) {
406
421
  const input = getArg(args, "-i", "--input");
407
422
  let output = getArg(args, "-o", "--output");
408
423
  const config = getArg(args, "-c", "--config", "s2t");
409
- const explicitFormat = getArg(args, null, "--format");
424
+ const explicitFormat = getArg(args, "-f", "--format");
410
425
  const punct = hasFlag(args, "-p", "--punct");
411
- const convertFilename = hasFlag(args, null, "--convert-filename");
426
+ const convertFilename = hasFlag(args, "-F", "--convert-filename");
412
427
  const keepFont = !hasFlag(args, null, "--no-keep-font");
413
- const customDicts = getArgs(args, "--custom-dict");
428
+ const customDicts = getArgs(args, "-D", "--custom-dict")
429
+ .map(parseCustomDictSpec);
414
430
 
415
431
  if (!input) {
416
432
  throw new Error("Input file is missing.");
@@ -49,6 +49,7 @@ export class OpenccWasm {
49
49
  constructor(config?: string | null);
50
50
  static newWithCustomDicts(config: string | null | undefined, specs: any): OpenccWasm;
51
51
  static newWithEnum(config?: OpenccConfigWasm | null): OpenccWasm;
52
+ normalizeCompat(text: string): string;
52
53
  setConfig(config: string): boolean;
53
54
  setConfigEnum(config: OpenccConfigWasm): void;
54
55
  setPreserveIds(value: boolean): void;
@@ -77,6 +78,7 @@ export interface InitOutput {
77
78
  readonly openccwasm_new: (a: number, b: number) => [number, number, number];
78
79
  readonly openccwasm_newWithCustomDicts: (a: number, b: number, c: any) => [number, number, number];
79
80
  readonly openccwasm_newWithEnum: (a: number) => [number, number, number];
81
+ readonly openccwasm_normalizeCompat: (a: number, b: number, c: number) => [number, number];
80
82
  readonly openccwasm_setConfig: (a: number, b: number, c: number) => number;
81
83
  readonly openccwasm_setConfigEnum: (a: number, b: number) => void;
82
84
  readonly openccwasm_setPreserveIds: (a: number, b: number) => void;
@@ -235,6 +235,24 @@ export class OpenccWasm {
235
235
  }
236
236
  return OpenccWasm.__wrap(ret[0]);
237
237
  }
238
+ /**
239
+ * @param {string} text
240
+ * @returns {string}
241
+ */
242
+ normalizeCompat(text) {
243
+ let deferred2_0;
244
+ let deferred2_1;
245
+ try {
246
+ const ptr0 = passStringToWasm0(text, wasm.__wbindgen_malloc, wasm.__wbindgen_realloc);
247
+ const len0 = WASM_VECTOR_LEN;
248
+ const ret = wasm.openccwasm_normalizeCompat(this.__wbg_ptr, ptr0, len0);
249
+ deferred2_0 = ret[0];
250
+ deferred2_1 = ret[1];
251
+ return getStringFromWasm0(ret[0], ret[1]);
252
+ } finally {
253
+ wasm.__wbindgen_free(deferred2_0, deferred2_1, 1);
254
+ }
255
+ }
238
256
  /**
239
257
  * @param {string} config
240
258
  * @returns {boolean}
Binary file
package/package.json CHANGED
@@ -2,7 +2,7 @@
2
2
  "name": "@laisuk/opencc-fmmseg-wasm",
3
3
  "type": "module",
4
4
  "description": "WebAssembly bindings for opencc-fmmseg, a high-performance OpenCC-compatible Simplified/Traditional Chinese converter.",
5
- "version": "0.3.4",
5
+ "version": "0.3.6",
6
6
  "license": "MIT",
7
7
  "repository": {
8
8
  "type": "git",