@laisuk/opencc-fmmseg-wasm 0.3.9 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -12,25 +12,25 @@ This package provides high-quality Simplified Chinese ↔ Traditional Chinese co
12
12
 
13
13
  Features:
14
14
 
15
- * OpenCC-compatible conversion configs
16
- * Pure WebAssembly (no native binaries)
17
- * Browser-friendly
18
- * TypeScript-friendly APIs
19
- * Fast Rust backend
20
- * FMM-based phrase segmentation
21
- * Traditional Chinese regional variants
22
- * Japanese Shinjitai conversion support
23
- * Chinese script detection (`zho_check`)
24
- * Optional CJK Compatibility Ideograph normalization
25
- * In-memory Office / EPUB document conversion
26
- * Zero-dependency Node.js CLI
15
+ - OpenCC-compatible conversion configs
16
+ - Pure WebAssembly (no native binaries)
17
+ - Browser-friendly
18
+ - TypeScript-friendly APIs
19
+ - Fast Rust backend
20
+ - FMM-based phrase segmentation
21
+ - Traditional Chinese regional variants
22
+ - Japanese Shinjitai conversion support
23
+ - Chinese script detection (`zho_check`)
24
+ - Optional CJK Compatibility Ideograph normalization
25
+ - In-memory Office / EPUB document conversion
26
+ - Zero-dependency Node.js CLI
27
27
 
28
28
  Package profile:
29
29
 
30
- * 0 runtime dependencies
31
- * 1 WASM file
32
- * 18 conversion configs
33
- * 100% offline
30
+ - 0 runtime dependencies
31
+ - 1 WASM file
32
+ - 20 conversion configs
33
+ - 100% offline
34
34
 
35
35
  ---
36
36
 
@@ -52,10 +52,10 @@ import init, {
52
52
 
53
53
  await init();
54
54
 
55
- const cc = new OpenccWasm("s2t");
55
+ const cc = new OpenccWasm("t2s");
56
56
 
57
- console.log(cc.convert("汉字", false));
58
- // 漢字
57
+ console.log(cc.convert("漢字", false));
58
+ // 汉字
59
59
 
60
60
  console.log(cc.convertDetofu("儼驂騑於上路", false, DetofuLevelWasm.ExtB));
61
61
  // 俨骖騑于上路
@@ -63,56 +63,6 @@ console.log(cc.convertDetofu("儼驂騑於上路", false, DetofuLevelWasm.ExtB))
63
63
 
64
64
  ---
65
65
 
66
- ## Using Config Enums
67
-
68
- ```javascript
69
- import init, {
70
- OpenccWasm,
71
- OpenccConfigWasm
72
- } from "@laisuk/opencc-fmmseg-wasm";
73
-
74
- await init();
75
-
76
- const cc = OpenccWasm.newWithEnum(
77
- OpenccConfigWasm.S2hkp
78
- );
79
-
80
- console.log(cc.convert("别随便录影侵犯个人隐私权", false));
81
- // 別隨便錄影侵犯個人私隱權
82
- ```
83
-
84
- ---
85
-
86
- ## Supported Configs
87
-
88
- | Config | Enum | Description |
89
- |---------|--------------------------|-------------------------------------------------------|
90
- | `s2t` | `OpenccConfigWasm.S2t` | Simplified Chinese → Traditional Chinese |
91
- | `s2tw` | `OpenccConfigWasm.S2tw` | Simplified Chinese → Taiwan Traditional |
92
- | `s2twp` | `OpenccConfigWasm.S2twp` | Simplified Chinese → Taiwan Traditional (phrases) |
93
- | `s2hk` | `OpenccConfigWasm.S2hk` | Simplified Chinese → Hong Kong Traditional |
94
- | `s2hkp` | `OpenccConfigWasm.S2hkp` | Simplified Chinese → Hong Kong Traditional (phrases) |
95
- | `t2s` | `OpenccConfigWasm.T2s` | Traditional Chinese → Simplified Chinese |
96
- | `t2tw` | `OpenccConfigWasm.T2tw` | Traditional Chinese → Taiwan Traditional |
97
- | `t2twp` | `OpenccConfigWasm.T2twp` | Traditional Chinese → Taiwan Traditional (phrases) |
98
- | `t2hk` | `OpenccConfigWasm.T2hk` | Traditional Chinese → Hong Kong Traditional |
99
- | `t2hkp` | `OpenccConfigWasm.T2hkp` | Traditional Chinese → Hong Kong Traditional (phrases) |
100
- | `tw2s` | `OpenccConfigWasm.Tw2s` | Taiwan Traditional → Simplified Chinese |
101
- | `tw2sp` | `OpenccConfigWasm.Tw2sp` | Taiwan Traditional → Simplified Chinese (phrases) |
102
- | `tw2t` | `OpenccConfigWasm.Tw2t` | Taiwan Traditional → Traditional Chinese |
103
- | `tw2tp` | `OpenccConfigWasm.Tw2tp` | Taiwan Traditional → Traditional Chinese (phrases) |
104
- | `hk2s` | `OpenccConfigWasm.Hk2s` | Hong Kong Traditional → Simplified Chinese |
105
- | `hk2sp` | `OpenccConfigWasm.Hk2sp` | Hong Kong Traditional → Simplified Chinese (phrases) |
106
- | `hk2t` | `OpenccConfigWasm.Hk2t` | Hong Kong Traditional → Traditional Chinese |
107
- | `hk2tp` | `OpenccConfigWasm.Hk2tp` | Hong Kong Traditional → Traditional Chinese (phrases) |
108
- | `jp2t` | `OpenccConfigWasm.Jp2t` | Japanese Shinjitai → Traditional Chinese |
109
- | `t2jp` | `OpenccConfigWasm.T2jp` | Traditional Chinese → Japanese Shinjitai |
110
-
111
- The numeric enum values match the vendored Rust backend. Existing values are unchanged; `S2hkp = 17`, `Hk2sp = 18`,
112
- `T2hkp = 19`, and `Hk2tp = 20`.
113
-
114
- ---
115
-
116
66
  ## API
117
67
 
118
68
  ### Constructor
@@ -123,8 +73,8 @@ const cc = new OpenccWasm("s2t");
123
73
 
124
74
  Parameters:
125
75
 
126
- * `config` (optional): OpenCC config string
127
- * default: `"s2t"`
76
+ - `config` (optional): OpenCC config string
77
+ - default: `"s2t"`
128
78
 
129
79
  Example:
130
80
 
@@ -151,12 +101,12 @@ cc.convert(text, punctuation)
151
101
 
152
102
  Parameters:
153
103
 
154
- * `text`: input string
155
- * `punctuation`: whether to convert punctuation variants
104
+ - `text`: input string
105
+ - `punctuation`: whether to convert punctuation variants
156
106
 
157
107
  Returns:
158
108
 
159
- * converted string
109
+ - converted string
160
110
 
161
111
  Example:
162
112
 
@@ -166,6 +116,85 @@ cc.convert("汉字", false);
166
116
 
167
117
  ---
168
118
 
119
+ ### setConfig
120
+
121
+ ```javascript
122
+ cc.setConfig("t2s");
123
+ ```
124
+
125
+ Returns:
126
+
127
+ - `true` if valid
128
+ - `false` if invalid
129
+
130
+ ---
131
+
132
+ ### getConfig
133
+
134
+ ```javascript
135
+ cc.getConfig();
136
+ ```
137
+
138
+ Returns current config string.
139
+
140
+ ---
141
+
142
+ ### isValidConfig
143
+
144
+ ```javascript
145
+ OpenccWasm.isValidConfig("s2t");
146
+ ```
147
+
148
+ ---
149
+
150
+ ### getSupportedConfigs
151
+
152
+ ```javascript
153
+ OpenccWasm.getSupportedConfigs();
154
+ ```
155
+
156
+ Returns all supported config strings.
157
+
158
+ Includes `s2hkp`, `hk2sp`, `t2hkp`, and `hk2tp`.
159
+
160
+ ---
161
+
162
+ ### getAvailableSlots
163
+
164
+ ```javascript
165
+ const slots = OpenccWasm.getAvailableSlots();
166
+ ```
167
+
168
+ Returns all canonical dictionary slot names accepted by `newWithCustomDicts` as a string array. The list is sourced from
169
+ the core `DictSlot` definitions, so callers can use it to populate selectors or validate custom dictionary input without
170
+ maintaining their own slot list.
171
+
172
+ ```javascript
173
+ if (!OpenccWasm.getAvailableSlots().includes(slot)) {
174
+ throw new Error(`Unsupported dictionary slot: ${slot}`);
175
+ }
176
+ ```
177
+
178
+ ---
179
+
180
+ ### zhoCheck
181
+
182
+ Detect Chinese script type.
183
+
184
+ ```javascript
185
+ cc.zhoCheck(text);
186
+ ```
187
+
188
+ Returns:
189
+
190
+ | Value | Meaning |
191
+ |-------|---------------------|
192
+ | `0` | Unknown / mixed |
193
+ | `1` | Traditional Chinese |
194
+ | `2` | Simplified Chinese |
195
+
196
+ ---
197
+
169
198
  ### normalizeCompat
170
199
 
171
200
  Normalize Unicode CJK Compatibility Ideographs before conversion.
@@ -176,11 +205,11 @@ cc.normalizeCompat(text)
176
205
 
177
206
  Parameters:
178
207
 
179
- * `text`: input string
208
+ - `text`: input string
180
209
 
181
210
  Returns:
182
211
 
183
- * normalized string
212
+ - normalized string
184
213
 
185
214
  Example:
186
215
 
@@ -197,9 +226,89 @@ console.log(cc.convert(normalized, false));
197
226
  // 天龙八部书里的乔峰是契丹人
198
227
  ```
199
228
 
200
- This is an optional pre-conversion pass for text that contains compatibility ideographs from Unicode compatibility
201
- ranges. Unmapped characters are preserved unchanged. Normal OpenCC conversion does not automatically run this pass, so
202
- call it explicitly when compatibility normalization is desired.
229
+ This is an optional pre-conversion pass for text that contains CJK Compatibility Ideographs. Unmapped characters are
230
+ preserved unchanged. Normal OpenCC conversion does not automatically run this pass, so call it explicitly when
231
+ compatibility normalization is desired.
232
+
233
+ ### normalizeUnicodeCompat
234
+
235
+ Normalize additional Unicode compatibility forms, CJK radicals, allographs, legacy glyphs, and selected
236
+ compatibility-like punctuation before conversion.
237
+
238
+ ```javascript
239
+ cc.normalizeUnicodeCompat(text)
240
+ ```
241
+
242
+ Parameters:
243
+
244
+ - `text`: input string
245
+
246
+ Returns:
247
+
248
+ - normalized string
249
+
250
+ Example:
251
+
252
+ ```javascript
253
+ const cc = new OpenccWasm("t2s");
254
+
255
+ const input = "聼聼竒羙⽟䂖甁噐⾳";
256
+ const normalized = cc.normalizeUnicodeCompat(input);
257
+
258
+ console.log(normalized);
259
+ // 聽聽奇美玉石瓶器音
260
+
261
+ console.log(cc.convert(normalized, false));
262
+ // 听听奇美玉石瓶器音
263
+ ```
264
+
265
+ This pass uses the extended Unicode compatibility table and is separate from `normalizeCompat()`. It is useful for text
266
+ containing radical forms, historical or allographic Han forms, and other compatibility-like characters that are not
267
+ covered by the CJK Compatibility Ideograph ranges.
268
+
269
+ Unmapped characters are preserved unchanged.
270
+
271
+ ### normalizeCompatExtended
272
+
273
+ Apply complete compatibility normalization before conversion.
274
+
275
+ ```javascript
276
+ cc.normalizeCompatExtended(text)
277
+ ```
278
+
279
+ Parameters:
280
+
281
+ - `text`: input string
282
+
283
+ Returns:
284
+
285
+ - normalized string
286
+
287
+ Example:
288
+
289
+ ```javascript
290
+ const cc = new OpenccWasm("t2s");
291
+
292
+ const input = "天龍八部書裡的聼眾";
293
+ const normalized = cc.normalizeCompatExtended(input);
294
+
295
+ console.log(normalized);
296
+ // 天龍八部書裡的聽眾
297
+
298
+ console.log(cc.convert(normalized, false));
299
+ // 天龙八部书里的听众
300
+ ```
301
+
302
+ `normalizeCompatExtended()` combines the extended Unicode compatibility table with CJK Compatibility Ideograph
303
+ normalization. Use this when input may contain characters handled by either normalization set.
304
+
305
+ The normalization order is:
306
+
307
+ 1. extended Unicode compatibility normalization;
308
+ 2. CJK Compatibility Ideograph normalization.
309
+
310
+ Normal OpenCC conversion does not automatically perform compatibility normalization. Call this method explicitly before
311
+ `convert()` when complete compatibility normalization is desired.
203
312
 
204
313
  ---
205
314
 
@@ -213,12 +322,12 @@ cc.detofu(text, level)
213
322
 
214
323
  Parameters:
215
324
 
216
- * `text`: input string
217
- * `level`: `DetofuLevelWasm` threshold for the CJK extension ranges to replace
325
+ - `text`: input string
326
+ - `level`: `DetofuLevelWasm` threshold for the CJK extension ranges to replace
218
327
 
219
328
  Returns:
220
329
 
221
- * detofu-safe string
330
+ - detofu-safe string
222
331
 
223
332
  Supported levels:
224
333
 
@@ -265,13 +374,13 @@ cc.convertDetofu(text, punctuation, level)
265
374
 
266
375
  Parameters:
267
376
 
268
- * `text`: input string
269
- * `punctuation`: whether to convert punctuation variants
270
- * `level`: `DetofuLevelWasm` threshold for the CJK extension ranges to replace
377
+ - `text`: input string
378
+ - `punctuation`: whether to convert punctuation variants
379
+ - `level`: `DetofuLevelWasm` threshold for the CJK extension ranges to replace
271
380
 
272
381
  Returns:
273
382
 
274
- * converted detofu-safe string
383
+ - converted detofu-safe string
275
384
 
276
385
  Example:
277
386
 
@@ -282,85 +391,6 @@ cc.convertDetofu("儼驂騑於上路", false, DetofuLevelWasm.ExtB);
282
391
 
283
392
  ---
284
393
 
285
- ### setConfig
286
-
287
- ```javascript
288
- cc.setConfig("t2s");
289
- ```
290
-
291
- Returns:
292
-
293
- * `true` if valid
294
- * `false` if invalid
295
-
296
- ---
297
-
298
- ### getConfig
299
-
300
- ```javascript
301
- cc.getConfig();
302
- ```
303
-
304
- Returns current config string.
305
-
306
- ---
307
-
308
- ### isValidConfig
309
-
310
- ```javascript
311
- OpenccWasm.isValidConfig("s2t");
312
- ```
313
-
314
- ---
315
-
316
- ### getSupportedConfigs
317
-
318
- ```javascript
319
- OpenccWasm.getSupportedConfigs();
320
- ```
321
-
322
- Returns all supported config strings.
323
-
324
- Includes `s2hkp`, `hk2sp`, `t2hkp`, and `hk2tp`.
325
-
326
- ---
327
-
328
- ### getAvailableSlots
329
-
330
- ```javascript
331
- const slots = OpenccWasm.getAvailableSlots();
332
- ```
333
-
334
- Returns all canonical dictionary slot names accepted by `newWithCustomDicts` as a string array. The list is sourced from
335
- the core `DictSlot` definitions, so callers can use it to populate selectors or validate custom dictionary input without
336
- maintaining their own slot list.
337
-
338
- ```javascript
339
- if (!OpenccWasm.getAvailableSlots().includes(slot)) {
340
- throw new Error(`Unsupported dictionary slot: ${slot}`);
341
- }
342
- ```
343
-
344
- ---
345
-
346
- ### zhoCheck
347
-
348
- Detect Chinese script type.
349
-
350
- ```javascript
351
- cc.zhoCheck(text);
352
- ```
353
-
354
- Returns:
355
-
356
- | Value | Meaning |
357
- |-------|---------------------|
358
- | `0` | Unknown / mixed |
359
- | `1` | Traditional Chinese |
360
- | `2` | Simplified Chinese |
361
-
362
- ---
363
-
364
394
  ### newWithCustomDicts
365
395
 
366
396
  Construct a converter with in-memory custom dictionary pairs.
@@ -371,8 +401,8 @@ const cc = OpenccWasm.newWithCustomDicts(config, specs);
371
401
 
372
402
  Parameters:
373
403
 
374
- * `config`: OpenCC config string, such as `"s2t"`
375
- * `specs`: array of custom dictionary specs
404
+ - `config`: OpenCC config string, such as `"s2t"`
405
+ - `specs`: array of custom dictionary specs
376
406
 
377
407
  TypeScript-style spec shape:
378
408
 
@@ -455,14 +485,14 @@ Suffixes such as `.txt` are not accepted, even though case and surrounding white
455
485
 
456
486
  Merge contract:
457
487
 
458
- * Custom dictionaries are loaded from in-memory pairs only; no file I/O is involved.
459
- * The embedded compressed CBOR dictionary is loaded first.
460
- * Custom specs are applied to `DictionaryMaxlength` before `OpenCC::from_dictionary(...)`.
461
- * Conversion hot paths remain immutable after construction.
462
- * `Append` mode merges into the selected slot.
463
- * Duplicate or conflicting keys use last-wins semantics.
464
- * `Override` mode clears the selected slot first, then inserts the provided custom pairs.
465
- * Multiple specs are applied in array order.
488
+ - Custom dictionaries are loaded from in-memory pairs only; no file I/O is involved.
489
+ - The embedded compressed CBOR dictionary is loaded first.
490
+ - Custom specs are applied to `DictionaryMaxlength` before `OpenCC::from_dictionary(...)`.
491
+ - Conversion hot paths remain immutable after construction.
492
+ - `Append` mode merges into the selected slot.
493
+ - Duplicate or conflicting keys use last-wins semantics.
494
+ - `Override` mode clears the selected slot first, then inserts the provided custom pairs.
495
+ - Multiple specs are applied in array order.
466
496
 
467
497
  This API is useful for browser apps, user-defined terminology, database-loaded terms, generated dictionaries,
468
498
  `localStorage` or `IndexedDB` terms, testing, and embedded WASM environments. Customization happens at construction
@@ -470,6 +500,56 @@ time, not during conversion.
470
500
 
471
501
  ---
472
502
 
503
+ ## Supported Configs
504
+
505
+ | Config | Enum | Description |
506
+ |---------|--------------------------|-------------------------------------------------------|
507
+ | `s2t` | `OpenccConfigWasm.S2t` | Simplified Chinese → Traditional Chinese |
508
+ | `s2tw` | `OpenccConfigWasm.S2tw` | Simplified Chinese → Taiwan Traditional |
509
+ | `s2twp` | `OpenccConfigWasm.S2twp` | Simplified Chinese → Taiwan Traditional (phrases) |
510
+ | `s2hk` | `OpenccConfigWasm.S2hk` | Simplified Chinese → Hong Kong Traditional |
511
+ | `s2hkp` | `OpenccConfigWasm.S2hkp` | Simplified Chinese → Hong Kong Traditional (phrases) |
512
+ | `t2s` | `OpenccConfigWasm.T2s` | Traditional Chinese → Simplified Chinese |
513
+ | `t2tw` | `OpenccConfigWasm.T2tw` | Traditional Chinese → Taiwan Traditional |
514
+ | `t2twp` | `OpenccConfigWasm.T2twp` | Traditional Chinese → Taiwan Traditional (phrases) |
515
+ | `t2hk` | `OpenccConfigWasm.T2hk` | Traditional Chinese → Hong Kong Traditional |
516
+ | `t2hkp` | `OpenccConfigWasm.T2hkp` | Traditional Chinese → Hong Kong Traditional (phrases) |
517
+ | `tw2s` | `OpenccConfigWasm.Tw2s` | Taiwan Traditional → Simplified Chinese |
518
+ | `tw2sp` | `OpenccConfigWasm.Tw2sp` | Taiwan Traditional → Simplified Chinese (phrases) |
519
+ | `tw2t` | `OpenccConfigWasm.Tw2t` | Taiwan Traditional → Traditional Chinese |
520
+ | `tw2tp` | `OpenccConfigWasm.Tw2tp` | Taiwan Traditional → Traditional Chinese (phrases) |
521
+ | `hk2s` | `OpenccConfigWasm.Hk2s` | Hong Kong Traditional → Simplified Chinese |
522
+ | `hk2sp` | `OpenccConfigWasm.Hk2sp` | Hong Kong Traditional → Simplified Chinese (phrases) |
523
+ | `hk2t` | `OpenccConfigWasm.Hk2t` | Hong Kong Traditional → Traditional Chinese |
524
+ | `hk2tp` | `OpenccConfigWasm.Hk2tp` | Hong Kong Traditional → Traditional Chinese (phrases) |
525
+ | `jp2t` | `OpenccConfigWasm.Jp2t` | Japanese Shinjitai → Traditional Chinese |
526
+ | `t2jp` | `OpenccConfigWasm.T2jp` | Traditional Chinese → Japanese Shinjitai |
527
+
528
+ The numeric enum values match the vendored Rust backend. Existing values are unchanged; `S2hkp = 17`, `Hk2sp = 18`,
529
+ `T2hkp = 19`, and `Hk2tp = 20`.
530
+
531
+ ---
532
+
533
+ ## Using Config Enums
534
+
535
+ ```javascript
536
+ import init, {
537
+ OpenccWasm,
538
+ OpenccConfigWasm
539
+ } from "@laisuk/opencc-fmmseg-wasm";
540
+
541
+ await init();
542
+
543
+ const cc = OpenccWasm.newWithEnum(
544
+ OpenccConfigWasm.S2hkp
545
+ );
546
+
547
+ console.log(cc.convert("别随便录影侵犯个人隐私权", false));
548
+ // 別隨便錄影侵犯個人私隱權
549
+ ```
550
+
551
+ ---
552
+
473
553
  ## Office / EPUB Conversion
474
554
 
475
555
  Office and EPUB conversion runs fully locally in the browser or Node.js. Files are passed in and returned as bytes;
@@ -493,14 +573,14 @@ cc.convertOfficeBytes(inputBytes, format, punctuation, keepFont)
493
573
 
494
574
  Parameters:
495
575
 
496
- * `inputBytes`: `Uint8Array` document bytes
497
- * `format`: `docx`, `xlsx`, `pptx`, `odt`, `ods`, `odp`, or `epub`
498
- * `punctuation`: whether to convert punctuation variants
499
- * `keepFont`: whether to preserve font declarations where supported
576
+ - `inputBytes`: `Uint8Array` document bytes
577
+ - `format`: `docx`, `xlsx`, `pptx`, `odt`, `ods`, `odp`, or `epub`
578
+ - `punctuation`: whether to convert punctuation variants
579
+ - `keepFont`: whether to preserve font declarations where supported
500
580
 
501
581
  Returns:
502
582
 
503
- * converted output bytes
583
+ - converted output bytes
504
584
 
505
585
  The older free function remains available for compatibility:
506
586
 
@@ -649,6 +729,7 @@ opencc-fmmseg office -i input.docx -o output.docx -c s2t -p --keep-font
649
729
  default when omitted value: all
650
730
  --keep-ids Preserve complete IDS expressions during conversion (default: false)
651
731
  -n, --norm-compat Normalize CJK Compatibility Ideographs before conversion (default: false)
732
+ -E, --norm-compat-extended Normalize extended Unicode compatibility forms before conversion (default: false)
652
733
  -D, --custom-dict <slot:mode:file>
653
734
  Load a custom dictionary.
654
735
  May be specified multiple times.
@@ -708,18 +789,18 @@ The WASM-facing enum is exported as `OpenccConfigWasm`, alongside `OpenccWasm`.
708
789
 
709
790
  ## Performance Notes
710
791
 
711
- * WebAssembly build disables Rayon parallelism by default.
712
- * Dictionaries are embedded into the WASM binary.
713
- * Browser caching significantly improves subsequent loads.
792
+ - WebAssembly build disables Rayon parallelism by default.
793
+ - Dictionaries are embedded into the WASM binary.
794
+ - Browser caching significantly improves subsequent loads.
714
795
 
715
796
  ---
716
797
 
717
798
  ## Related Projects
718
799
 
719
- * Rust backend: https://github.com/laisuk/opencc-fmmseg
720
- * C API: https://github.com/laisuk/opencc-fmmseg/tree/master/capi/opencc-fmmseg-capi
721
- * .NET: https://github.com/laisuk/OpenccNet
722
- * Python: https://github.com/laisuk/opencc_purepy
800
+ - Rust backend: https://github.com/laisuk/opencc-fmmseg
801
+ - C API: https://github.com/laisuk/opencc-fmmseg/tree/master/capi/opencc-fmmseg-capi
802
+ - .NET: https://github.com/laisuk/OpenccNet
803
+ - Python: https://github.com/laisuk/opencc_purepy
723
804
 
724
805
  ---
725
806
 
package/bin/opencc.js CHANGED
@@ -58,6 +58,7 @@ Convert options:
58
58
  default when omitted value: all
59
59
  --keep-ids Preserve complete IDS expressions during conversion (default: false)
60
60
  -n, --norm-compat Normalize CJK Compatibility Ideographs before conversion (default: false)
61
+ -E, --norm-compat-extended Normalize extended Unicode compatibility forms before conversion (default: false)
61
62
  -D, --custom-dict <slot:mode:file>
62
63
  Load a custom dictionary.
63
64
  May be specified multiple times.
@@ -88,7 +89,7 @@ Office options:
88
89
  -F, --convert-filename Convert generated output filename stem (default: false)
89
90
  --keep-font Preserve font-family information (default)
90
91
  --no-keep-font Do not preserve font-family information
91
- --custom-dict <slot:mode:file>
92
+ -D, --custom-dict <slot:mode:file>
92
93
  Load a custom dictionary.
93
94
  May be specified multiple times.
94
95
  Examples:
@@ -107,6 +108,7 @@ Examples:
107
108
  npx opencc-fmmseg convert -i a.txt -o b.txt -c t2s --detofu ext-c
108
109
  echo "⿰氵漢" | npx opencc-fmmseg convert -c t2s
109
110
  echo "⿰氵漢" | npx opencc-fmmseg convert -c t2s --keep-ids
111
+ echo "聼聼竒羙⽟䂖甁噐⾳" | npx opencc-fmmseg convert -c t2s -E
110
112
 
111
113
  npx opencc-fmmseg office -i a.docx -o b.docx -c s2t -p
112
114
  npx opencc-fmmseg office -i a.epub -c s2tw
@@ -189,6 +191,24 @@ function validateInputFile(filePath) {
189
191
  }
190
192
  }
191
193
 
194
+ function ensureDistinctPaths(input, output) {
195
+ if (!input || !output) {
196
+ return;
197
+ }
198
+
199
+ let inputPath = path.resolve(input);
200
+ let outputPath = path.resolve(output);
201
+
202
+ if (process.platform === "win32") {
203
+ inputPath = inputPath.toLowerCase();
204
+ outputPath = outputPath.toLowerCase();
205
+ }
206
+
207
+ if (inputPath === outputPath) {
208
+ throw new Error("Input and output files must be different.");
209
+ }
210
+ }
211
+
192
212
  function parseCustomDictSpec(value) {
193
213
  const first = value.indexOf(":");
194
214
  const second = value.indexOf(":", first + 1);
@@ -450,6 +470,8 @@ function parseDetofuLevel(value) {
450
470
  async function runConvert(args) {
451
471
  const input = getArg(args, "-i", "--input");
452
472
  const output = getArg(args, "-o", "--output");
473
+ ensureDistinctPaths(input, output);
474
+
453
475
  const config = getArg(args, "-c", "--config", "s2t");
454
476
  const inEnc = validateEncoding(
455
477
  getArg(args, null, "--in-enc", "utf8"),
@@ -463,6 +485,7 @@ async function runConvert(args) {
463
485
  const punct = hasFlag(args, "-p", "--punct");
464
486
  const keepIds = hasFlag(args, null, "--keep-ids");
465
487
  const normCompat = hasFlag(args, "-n", "--norm-compat");
488
+ const normCompatExtended = hasFlag(args, "-E", "--norm-compat-extended");
466
489
  const customDicts = getArgs(args, "-D", "--custom-dict")
467
490
  .map(parseCustomDictSpec);
468
491
 
@@ -498,7 +521,9 @@ async function runConvert(args) {
498
521
 
499
522
  let inputText = readInputText(input, inEnc);
500
523
 
501
- if (normCompat) {
524
+ if (normCompatExtended) {
525
+ inputText = cc.normalizeCompatExtended(inputText);
526
+ } else if (normCompat) {
502
527
  inputText = cc.normalizeCompat(inputText);
503
528
  }
504
529
 
@@ -519,10 +544,20 @@ async function runConvert(args) {
519
544
  }
520
545
 
521
546
  const suffixParts = [];
522
- if (normCompat) suffixParts.push("normalized");
547
+
548
+ if (normCompatExtended) {
549
+ suffixParts.push("norm-compat-extended");
550
+ } else if (normCompat) {
551
+ suffixParts.push("norm-compat");
552
+ }
553
+
523
554
  if (detofuEnabled) suffixParts.push("detofu");
524
555
  if (keepIds) suffixParts.push("keep-ids");
525
556
 
557
+ for (const spec of customDicts) {
558
+ suffixParts.push(`custom:${spec.slot}:${spec.mode}`);
559
+ }
560
+
526
561
  const suffix = suffixParts.length ? `, ${suffixParts.join(", ")}` : "";
527
562
  console.error(`Conversion completed (${cc.getConfig()}${suffix}): ${inFrom} -> ${outTo}`);
528
563
  }
@@ -561,6 +596,8 @@ async function runOffice(args) {
561
596
  output = applyOutputExtension(output, officeFormat);
562
597
  }
563
598
 
599
+ ensureDistinctPaths(input, output);
600
+
564
601
  const inputBytes = fs.readFileSync(input);
565
602
 
566
603
  const outputBytes = cc.convertOfficeBytes(