@laisuk/opencc-fmmseg-wasm 0.3.9 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +268 -187
- package/bin/opencc.js +40 -3
- package/opencc_fmmseg_wasm.d.ts +217 -2
- package/opencc_fmmseg_wasm.js +237 -47
- package/opencc_fmmseg_wasm_bg.wasm +0 -0
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -12,25 +12,25 @@ This package provides high-quality Simplified Chinese ↔ Traditional Chinese co
|
|
|
12
12
|
|
|
13
13
|
Features:
|
|
14
14
|
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
15
|
+
- OpenCC-compatible conversion configs
|
|
16
|
+
- Pure WebAssembly (no native binaries)
|
|
17
|
+
- Browser-friendly
|
|
18
|
+
- TypeScript-friendly APIs
|
|
19
|
+
- Fast Rust backend
|
|
20
|
+
- FMM-based phrase segmentation
|
|
21
|
+
- Traditional Chinese regional variants
|
|
22
|
+
- Japanese Shinjitai conversion support
|
|
23
|
+
- Chinese script detection (`zho_check`)
|
|
24
|
+
- Optional CJK Compatibility Ideograph normalization
|
|
25
|
+
- In-memory Office / EPUB document conversion
|
|
26
|
+
- Zero-dependency Node.js CLI
|
|
27
27
|
|
|
28
28
|
Package profile:
|
|
29
29
|
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
30
|
+
- 0 runtime dependencies
|
|
31
|
+
- 1 WASM file
|
|
32
|
+
- 20 conversion configs
|
|
33
|
+
- 100% offline
|
|
34
34
|
|
|
35
35
|
---
|
|
36
36
|
|
|
@@ -52,10 +52,10 @@ import init, {
|
|
|
52
52
|
|
|
53
53
|
await init();
|
|
54
54
|
|
|
55
|
-
const cc = new OpenccWasm("
|
|
55
|
+
const cc = new OpenccWasm("t2s");
|
|
56
56
|
|
|
57
|
-
console.log(cc.convert("
|
|
58
|
-
//
|
|
57
|
+
console.log(cc.convert("漢字", false));
|
|
58
|
+
// 汉字
|
|
59
59
|
|
|
60
60
|
console.log(cc.convertDetofu("儼驂騑於上路", false, DetofuLevelWasm.ExtB));
|
|
61
61
|
// 俨骖騑于上路
|
|
@@ -63,56 +63,6 @@ console.log(cc.convertDetofu("儼驂騑於上路", false, DetofuLevelWasm.ExtB))
|
|
|
63
63
|
|
|
64
64
|
---
|
|
65
65
|
|
|
66
|
-
## Using Config Enums
|
|
67
|
-
|
|
68
|
-
```javascript
|
|
69
|
-
import init, {
|
|
70
|
-
OpenccWasm,
|
|
71
|
-
OpenccConfigWasm
|
|
72
|
-
} from "@laisuk/opencc-fmmseg-wasm";
|
|
73
|
-
|
|
74
|
-
await init();
|
|
75
|
-
|
|
76
|
-
const cc = OpenccWasm.newWithEnum(
|
|
77
|
-
OpenccConfigWasm.S2hkp
|
|
78
|
-
);
|
|
79
|
-
|
|
80
|
-
console.log(cc.convert("别随便录影侵犯个人隐私权", false));
|
|
81
|
-
// 別隨便錄影侵犯個人私隱權
|
|
82
|
-
```
|
|
83
|
-
|
|
84
|
-
---
|
|
85
|
-
|
|
86
|
-
## Supported Configs
|
|
87
|
-
|
|
88
|
-
| Config | Enum | Description |
|
|
89
|
-
|---------|--------------------------|-------------------------------------------------------|
|
|
90
|
-
| `s2t` | `OpenccConfigWasm.S2t` | Simplified Chinese → Traditional Chinese |
|
|
91
|
-
| `s2tw` | `OpenccConfigWasm.S2tw` | Simplified Chinese → Taiwan Traditional |
|
|
92
|
-
| `s2twp` | `OpenccConfigWasm.S2twp` | Simplified Chinese → Taiwan Traditional (phrases) |
|
|
93
|
-
| `s2hk` | `OpenccConfigWasm.S2hk` | Simplified Chinese → Hong Kong Traditional |
|
|
94
|
-
| `s2hkp` | `OpenccConfigWasm.S2hkp` | Simplified Chinese → Hong Kong Traditional (phrases) |
|
|
95
|
-
| `t2s` | `OpenccConfigWasm.T2s` | Traditional Chinese → Simplified Chinese |
|
|
96
|
-
| `t2tw` | `OpenccConfigWasm.T2tw` | Traditional Chinese → Taiwan Traditional |
|
|
97
|
-
| `t2twp` | `OpenccConfigWasm.T2twp` | Traditional Chinese → Taiwan Traditional (phrases) |
|
|
98
|
-
| `t2hk` | `OpenccConfigWasm.T2hk` | Traditional Chinese → Hong Kong Traditional |
|
|
99
|
-
| `t2hkp` | `OpenccConfigWasm.T2hkp` | Traditional Chinese → Hong Kong Traditional (phrases) |
|
|
100
|
-
| `tw2s` | `OpenccConfigWasm.Tw2s` | Taiwan Traditional → Simplified Chinese |
|
|
101
|
-
| `tw2sp` | `OpenccConfigWasm.Tw2sp` | Taiwan Traditional → Simplified Chinese (phrases) |
|
|
102
|
-
| `tw2t` | `OpenccConfigWasm.Tw2t` | Taiwan Traditional → Traditional Chinese |
|
|
103
|
-
| `tw2tp` | `OpenccConfigWasm.Tw2tp` | Taiwan Traditional → Traditional Chinese (phrases) |
|
|
104
|
-
| `hk2s` | `OpenccConfigWasm.Hk2s` | Hong Kong Traditional → Simplified Chinese |
|
|
105
|
-
| `hk2sp` | `OpenccConfigWasm.Hk2sp` | Hong Kong Traditional → Simplified Chinese (phrases) |
|
|
106
|
-
| `hk2t` | `OpenccConfigWasm.Hk2t` | Hong Kong Traditional → Traditional Chinese |
|
|
107
|
-
| `hk2tp` | `OpenccConfigWasm.Hk2tp` | Hong Kong Traditional → Traditional Chinese (phrases) |
|
|
108
|
-
| `jp2t` | `OpenccConfigWasm.Jp2t` | Japanese Shinjitai → Traditional Chinese |
|
|
109
|
-
| `t2jp` | `OpenccConfigWasm.T2jp` | Traditional Chinese → Japanese Shinjitai |
|
|
110
|
-
|
|
111
|
-
The numeric enum values match the vendored Rust backend. Existing values are unchanged; `S2hkp = 17`, `Hk2sp = 18`,
|
|
112
|
-
`T2hkp = 19`, and `Hk2tp = 20`.
|
|
113
|
-
|
|
114
|
-
---
|
|
115
|
-
|
|
116
66
|
## API
|
|
117
67
|
|
|
118
68
|
### Constructor
|
|
@@ -123,8 +73,8 @@ const cc = new OpenccWasm("s2t");
|
|
|
123
73
|
|
|
124
74
|
Parameters:
|
|
125
75
|
|
|
126
|
-
|
|
127
|
-
|
|
76
|
+
- `config` (optional): OpenCC config string
|
|
77
|
+
- default: `"s2t"`
|
|
128
78
|
|
|
129
79
|
Example:
|
|
130
80
|
|
|
@@ -151,12 +101,12 @@ cc.convert(text, punctuation)
|
|
|
151
101
|
|
|
152
102
|
Parameters:
|
|
153
103
|
|
|
154
|
-
|
|
155
|
-
|
|
104
|
+
- `text`: input string
|
|
105
|
+
- `punctuation`: whether to convert punctuation variants
|
|
156
106
|
|
|
157
107
|
Returns:
|
|
158
108
|
|
|
159
|
-
|
|
109
|
+
- converted string
|
|
160
110
|
|
|
161
111
|
Example:
|
|
162
112
|
|
|
@@ -166,6 +116,85 @@ cc.convert("汉字", false);
|
|
|
166
116
|
|
|
167
117
|
---
|
|
168
118
|
|
|
119
|
+
### setConfig
|
|
120
|
+
|
|
121
|
+
```javascript
|
|
122
|
+
cc.setConfig("t2s");
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
Returns:
|
|
126
|
+
|
|
127
|
+
- `true` if valid
|
|
128
|
+
- `false` if invalid
|
|
129
|
+
|
|
130
|
+
---
|
|
131
|
+
|
|
132
|
+
### getConfig
|
|
133
|
+
|
|
134
|
+
```javascript
|
|
135
|
+
cc.getConfig();
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
Returns current config string.
|
|
139
|
+
|
|
140
|
+
---
|
|
141
|
+
|
|
142
|
+
### isValidConfig
|
|
143
|
+
|
|
144
|
+
```javascript
|
|
145
|
+
OpenccWasm.isValidConfig("s2t");
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
---
|
|
149
|
+
|
|
150
|
+
### getSupportedConfigs
|
|
151
|
+
|
|
152
|
+
```javascript
|
|
153
|
+
OpenccWasm.getSupportedConfigs();
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
Returns all supported config strings.
|
|
157
|
+
|
|
158
|
+
Includes `s2hkp`, `hk2sp`, `t2hkp`, and `hk2tp`.
|
|
159
|
+
|
|
160
|
+
---
|
|
161
|
+
|
|
162
|
+
### getAvailableSlots
|
|
163
|
+
|
|
164
|
+
```javascript
|
|
165
|
+
const slots = OpenccWasm.getAvailableSlots();
|
|
166
|
+
```
|
|
167
|
+
|
|
168
|
+
Returns all canonical dictionary slot names accepted by `newWithCustomDicts` as a string array. The list is sourced from
|
|
169
|
+
the core `DictSlot` definitions, so callers can use it to populate selectors or validate custom dictionary input without
|
|
170
|
+
maintaining their own slot list.
|
|
171
|
+
|
|
172
|
+
```javascript
|
|
173
|
+
if (!OpenccWasm.getAvailableSlots().includes(slot)) {
|
|
174
|
+
throw new Error(`Unsupported dictionary slot: ${slot}`);
|
|
175
|
+
}
|
|
176
|
+
```
|
|
177
|
+
|
|
178
|
+
---
|
|
179
|
+
|
|
180
|
+
### zhoCheck
|
|
181
|
+
|
|
182
|
+
Detect Chinese script type.
|
|
183
|
+
|
|
184
|
+
```javascript
|
|
185
|
+
cc.zhoCheck(text);
|
|
186
|
+
```
|
|
187
|
+
|
|
188
|
+
Returns:
|
|
189
|
+
|
|
190
|
+
| Value | Meaning |
|
|
191
|
+
|-------|---------------------|
|
|
192
|
+
| `0` | Unknown / mixed |
|
|
193
|
+
| `1` | Traditional Chinese |
|
|
194
|
+
| `2` | Simplified Chinese |
|
|
195
|
+
|
|
196
|
+
---
|
|
197
|
+
|
|
169
198
|
### normalizeCompat
|
|
170
199
|
|
|
171
200
|
Normalize Unicode CJK Compatibility Ideographs before conversion.
|
|
@@ -176,11 +205,11 @@ cc.normalizeCompat(text)
|
|
|
176
205
|
|
|
177
206
|
Parameters:
|
|
178
207
|
|
|
179
|
-
|
|
208
|
+
- `text`: input string
|
|
180
209
|
|
|
181
210
|
Returns:
|
|
182
211
|
|
|
183
|
-
|
|
212
|
+
- normalized string
|
|
184
213
|
|
|
185
214
|
Example:
|
|
186
215
|
|
|
@@ -197,9 +226,89 @@ console.log(cc.convert(normalized, false));
|
|
|
197
226
|
// 天龙八部书里的乔峰是契丹人
|
|
198
227
|
```
|
|
199
228
|
|
|
200
|
-
This is an optional pre-conversion pass for text that contains
|
|
201
|
-
|
|
202
|
-
|
|
229
|
+
This is an optional pre-conversion pass for text that contains CJK Compatibility Ideographs. Unmapped characters are
|
|
230
|
+
preserved unchanged. Normal OpenCC conversion does not automatically run this pass, so call it explicitly when
|
|
231
|
+
compatibility normalization is desired.
|
|
232
|
+
|
|
233
|
+
### normalizeUnicodeCompat
|
|
234
|
+
|
|
235
|
+
Normalize additional Unicode compatibility forms, CJK radicals, allographs, legacy glyphs, and selected
|
|
236
|
+
compatibility-like punctuation before conversion.
|
|
237
|
+
|
|
238
|
+
```javascript
|
|
239
|
+
cc.normalizeUnicodeCompat(text)
|
|
240
|
+
```
|
|
241
|
+
|
|
242
|
+
Parameters:
|
|
243
|
+
|
|
244
|
+
- `text`: input string
|
|
245
|
+
|
|
246
|
+
Returns:
|
|
247
|
+
|
|
248
|
+
- normalized string
|
|
249
|
+
|
|
250
|
+
Example:
|
|
251
|
+
|
|
252
|
+
```javascript
|
|
253
|
+
const cc = new OpenccWasm("t2s");
|
|
254
|
+
|
|
255
|
+
const input = "聼聼竒羙⽟䂖甁噐⾳";
|
|
256
|
+
const normalized = cc.normalizeUnicodeCompat(input);
|
|
257
|
+
|
|
258
|
+
console.log(normalized);
|
|
259
|
+
// 聽聽奇美玉石瓶器音
|
|
260
|
+
|
|
261
|
+
console.log(cc.convert(normalized, false));
|
|
262
|
+
// 听听奇美玉石瓶器音
|
|
263
|
+
```
|
|
264
|
+
|
|
265
|
+
This pass uses the extended Unicode compatibility table and is separate from `normalizeCompat()`. It is useful for text
|
|
266
|
+
containing radical forms, historical or allographic Han forms, and other compatibility-like characters that are not
|
|
267
|
+
covered by the CJK Compatibility Ideograph ranges.
|
|
268
|
+
|
|
269
|
+
Unmapped characters are preserved unchanged.
|
|
270
|
+
|
|
271
|
+
### normalizeCompatExtended
|
|
272
|
+
|
|
273
|
+
Apply complete compatibility normalization before conversion.
|
|
274
|
+
|
|
275
|
+
```javascript
|
|
276
|
+
cc.normalizeCompatExtended(text)
|
|
277
|
+
```
|
|
278
|
+
|
|
279
|
+
Parameters:
|
|
280
|
+
|
|
281
|
+
- `text`: input string
|
|
282
|
+
|
|
283
|
+
Returns:
|
|
284
|
+
|
|
285
|
+
- normalized string
|
|
286
|
+
|
|
287
|
+
Example:
|
|
288
|
+
|
|
289
|
+
```javascript
|
|
290
|
+
const cc = new OpenccWasm("t2s");
|
|
291
|
+
|
|
292
|
+
const input = "天龍八部書裡的聼眾";
|
|
293
|
+
const normalized = cc.normalizeCompatExtended(input);
|
|
294
|
+
|
|
295
|
+
console.log(normalized);
|
|
296
|
+
// 天龍八部書裡的聽眾
|
|
297
|
+
|
|
298
|
+
console.log(cc.convert(normalized, false));
|
|
299
|
+
// 天龙八部书里的听众
|
|
300
|
+
```
|
|
301
|
+
|
|
302
|
+
`normalizeCompatExtended()` combines the extended Unicode compatibility table with CJK Compatibility Ideograph
|
|
303
|
+
normalization. Use this when input may contain characters handled by either normalization set.
|
|
304
|
+
|
|
305
|
+
The normalization order is:
|
|
306
|
+
|
|
307
|
+
1. extended Unicode compatibility normalization;
|
|
308
|
+
2. CJK Compatibility Ideograph normalization.
|
|
309
|
+
|
|
310
|
+
Normal OpenCC conversion does not automatically perform compatibility normalization. Call this method explicitly before
|
|
311
|
+
`convert()` when complete compatibility normalization is desired.
|
|
203
312
|
|
|
204
313
|
---
|
|
205
314
|
|
|
@@ -213,12 +322,12 @@ cc.detofu(text, level)
|
|
|
213
322
|
|
|
214
323
|
Parameters:
|
|
215
324
|
|
|
216
|
-
|
|
217
|
-
|
|
325
|
+
- `text`: input string
|
|
326
|
+
- `level`: `DetofuLevelWasm` threshold for the CJK extension ranges to replace
|
|
218
327
|
|
|
219
328
|
Returns:
|
|
220
329
|
|
|
221
|
-
|
|
330
|
+
- detofu-safe string
|
|
222
331
|
|
|
223
332
|
Supported levels:
|
|
224
333
|
|
|
@@ -265,13 +374,13 @@ cc.convertDetofu(text, punctuation, level)
|
|
|
265
374
|
|
|
266
375
|
Parameters:
|
|
267
376
|
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
377
|
+
- `text`: input string
|
|
378
|
+
- `punctuation`: whether to convert punctuation variants
|
|
379
|
+
- `level`: `DetofuLevelWasm` threshold for the CJK extension ranges to replace
|
|
271
380
|
|
|
272
381
|
Returns:
|
|
273
382
|
|
|
274
|
-
|
|
383
|
+
- converted detofu-safe string
|
|
275
384
|
|
|
276
385
|
Example:
|
|
277
386
|
|
|
@@ -282,85 +391,6 @@ cc.convertDetofu("儼驂騑於上路", false, DetofuLevelWasm.ExtB);
|
|
|
282
391
|
|
|
283
392
|
---
|
|
284
393
|
|
|
285
|
-
### setConfig
|
|
286
|
-
|
|
287
|
-
```javascript
|
|
288
|
-
cc.setConfig("t2s");
|
|
289
|
-
```
|
|
290
|
-
|
|
291
|
-
Returns:
|
|
292
|
-
|
|
293
|
-
* `true` if valid
|
|
294
|
-
* `false` if invalid
|
|
295
|
-
|
|
296
|
-
---
|
|
297
|
-
|
|
298
|
-
### getConfig
|
|
299
|
-
|
|
300
|
-
```javascript
|
|
301
|
-
cc.getConfig();
|
|
302
|
-
```
|
|
303
|
-
|
|
304
|
-
Returns current config string.
|
|
305
|
-
|
|
306
|
-
---
|
|
307
|
-
|
|
308
|
-
### isValidConfig
|
|
309
|
-
|
|
310
|
-
```javascript
|
|
311
|
-
OpenccWasm.isValidConfig("s2t");
|
|
312
|
-
```
|
|
313
|
-
|
|
314
|
-
---
|
|
315
|
-
|
|
316
|
-
### getSupportedConfigs
|
|
317
|
-
|
|
318
|
-
```javascript
|
|
319
|
-
OpenccWasm.getSupportedConfigs();
|
|
320
|
-
```
|
|
321
|
-
|
|
322
|
-
Returns all supported config strings.
|
|
323
|
-
|
|
324
|
-
Includes `s2hkp`, `hk2sp`, `t2hkp`, and `hk2tp`.
|
|
325
|
-
|
|
326
|
-
---
|
|
327
|
-
|
|
328
|
-
### getAvailableSlots
|
|
329
|
-
|
|
330
|
-
```javascript
|
|
331
|
-
const slots = OpenccWasm.getAvailableSlots();
|
|
332
|
-
```
|
|
333
|
-
|
|
334
|
-
Returns all canonical dictionary slot names accepted by `newWithCustomDicts` as a string array. The list is sourced from
|
|
335
|
-
the core `DictSlot` definitions, so callers can use it to populate selectors or validate custom dictionary input without
|
|
336
|
-
maintaining their own slot list.
|
|
337
|
-
|
|
338
|
-
```javascript
|
|
339
|
-
if (!OpenccWasm.getAvailableSlots().includes(slot)) {
|
|
340
|
-
throw new Error(`Unsupported dictionary slot: ${slot}`);
|
|
341
|
-
}
|
|
342
|
-
```
|
|
343
|
-
|
|
344
|
-
---
|
|
345
|
-
|
|
346
|
-
### zhoCheck
|
|
347
|
-
|
|
348
|
-
Detect Chinese script type.
|
|
349
|
-
|
|
350
|
-
```javascript
|
|
351
|
-
cc.zhoCheck(text);
|
|
352
|
-
```
|
|
353
|
-
|
|
354
|
-
Returns:
|
|
355
|
-
|
|
356
|
-
| Value | Meaning |
|
|
357
|
-
|-------|---------------------|
|
|
358
|
-
| `0` | Unknown / mixed |
|
|
359
|
-
| `1` | Traditional Chinese |
|
|
360
|
-
| `2` | Simplified Chinese |
|
|
361
|
-
|
|
362
|
-
---
|
|
363
|
-
|
|
364
394
|
### newWithCustomDicts
|
|
365
395
|
|
|
366
396
|
Construct a converter with in-memory custom dictionary pairs.
|
|
@@ -371,8 +401,8 @@ const cc = OpenccWasm.newWithCustomDicts(config, specs);
|
|
|
371
401
|
|
|
372
402
|
Parameters:
|
|
373
403
|
|
|
374
|
-
|
|
375
|
-
|
|
404
|
+
- `config`: OpenCC config string, such as `"s2t"`
|
|
405
|
+
- `specs`: array of custom dictionary specs
|
|
376
406
|
|
|
377
407
|
TypeScript-style spec shape:
|
|
378
408
|
|
|
@@ -455,14 +485,14 @@ Suffixes such as `.txt` are not accepted, even though case and surrounding white
|
|
|
455
485
|
|
|
456
486
|
Merge contract:
|
|
457
487
|
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
488
|
+
- Custom dictionaries are loaded from in-memory pairs only; no file I/O is involved.
|
|
489
|
+
- The embedded compressed CBOR dictionary is loaded first.
|
|
490
|
+
- Custom specs are applied to `DictionaryMaxlength` before `OpenCC::from_dictionary(...)`.
|
|
491
|
+
- Conversion hot paths remain immutable after construction.
|
|
492
|
+
- `Append` mode merges into the selected slot.
|
|
493
|
+
- Duplicate or conflicting keys use last-wins semantics.
|
|
494
|
+
- `Override` mode clears the selected slot first, then inserts the provided custom pairs.
|
|
495
|
+
- Multiple specs are applied in array order.
|
|
466
496
|
|
|
467
497
|
This API is useful for browser apps, user-defined terminology, database-loaded terms, generated dictionaries,
|
|
468
498
|
`localStorage` or `IndexedDB` terms, testing, and embedded WASM environments. Customization happens at construction
|
|
@@ -470,6 +500,56 @@ time, not during conversion.
|
|
|
470
500
|
|
|
471
501
|
---
|
|
472
502
|
|
|
503
|
+
## Supported Configs
|
|
504
|
+
|
|
505
|
+
| Config | Enum | Description |
|
|
506
|
+
|---------|--------------------------|-------------------------------------------------------|
|
|
507
|
+
| `s2t` | `OpenccConfigWasm.S2t` | Simplified Chinese → Traditional Chinese |
|
|
508
|
+
| `s2tw` | `OpenccConfigWasm.S2tw` | Simplified Chinese → Taiwan Traditional |
|
|
509
|
+
| `s2twp` | `OpenccConfigWasm.S2twp` | Simplified Chinese → Taiwan Traditional (phrases) |
|
|
510
|
+
| `s2hk` | `OpenccConfigWasm.S2hk` | Simplified Chinese → Hong Kong Traditional |
|
|
511
|
+
| `s2hkp` | `OpenccConfigWasm.S2hkp` | Simplified Chinese → Hong Kong Traditional (phrases) |
|
|
512
|
+
| `t2s` | `OpenccConfigWasm.T2s` | Traditional Chinese → Simplified Chinese |
|
|
513
|
+
| `t2tw` | `OpenccConfigWasm.T2tw` | Traditional Chinese → Taiwan Traditional |
|
|
514
|
+
| `t2twp` | `OpenccConfigWasm.T2twp` | Traditional Chinese → Taiwan Traditional (phrases) |
|
|
515
|
+
| `t2hk` | `OpenccConfigWasm.T2hk` | Traditional Chinese → Hong Kong Traditional |
|
|
516
|
+
| `t2hkp` | `OpenccConfigWasm.T2hkp` | Traditional Chinese → Hong Kong Traditional (phrases) |
|
|
517
|
+
| `tw2s` | `OpenccConfigWasm.Tw2s` | Taiwan Traditional → Simplified Chinese |
|
|
518
|
+
| `tw2sp` | `OpenccConfigWasm.Tw2sp` | Taiwan Traditional → Simplified Chinese (phrases) |
|
|
519
|
+
| `tw2t` | `OpenccConfigWasm.Tw2t` | Taiwan Traditional → Traditional Chinese |
|
|
520
|
+
| `tw2tp` | `OpenccConfigWasm.Tw2tp` | Taiwan Traditional → Traditional Chinese (phrases) |
|
|
521
|
+
| `hk2s` | `OpenccConfigWasm.Hk2s` | Hong Kong Traditional → Simplified Chinese |
|
|
522
|
+
| `hk2sp` | `OpenccConfigWasm.Hk2sp` | Hong Kong Traditional → Simplified Chinese (phrases) |
|
|
523
|
+
| `hk2t` | `OpenccConfigWasm.Hk2t` | Hong Kong Traditional → Traditional Chinese |
|
|
524
|
+
| `hk2tp` | `OpenccConfigWasm.Hk2tp` | Hong Kong Traditional → Traditional Chinese (phrases) |
|
|
525
|
+
| `jp2t` | `OpenccConfigWasm.Jp2t` | Japanese Shinjitai → Traditional Chinese |
|
|
526
|
+
| `t2jp` | `OpenccConfigWasm.T2jp` | Traditional Chinese → Japanese Shinjitai |
|
|
527
|
+
|
|
528
|
+
The numeric enum values match the vendored Rust backend. Existing values are unchanged; `S2hkp = 17`, `Hk2sp = 18`,
|
|
529
|
+
`T2hkp = 19`, and `Hk2tp = 20`.
|
|
530
|
+
|
|
531
|
+
---
|
|
532
|
+
|
|
533
|
+
## Using Config Enums
|
|
534
|
+
|
|
535
|
+
```javascript
|
|
536
|
+
import init, {
|
|
537
|
+
OpenccWasm,
|
|
538
|
+
OpenccConfigWasm
|
|
539
|
+
} from "@laisuk/opencc-fmmseg-wasm";
|
|
540
|
+
|
|
541
|
+
await init();
|
|
542
|
+
|
|
543
|
+
const cc = OpenccWasm.newWithEnum(
|
|
544
|
+
OpenccConfigWasm.S2hkp
|
|
545
|
+
);
|
|
546
|
+
|
|
547
|
+
console.log(cc.convert("别随便录影侵犯个人隐私权", false));
|
|
548
|
+
// 別隨便錄影侵犯個人私隱權
|
|
549
|
+
```
|
|
550
|
+
|
|
551
|
+
---
|
|
552
|
+
|
|
473
553
|
## Office / EPUB Conversion
|
|
474
554
|
|
|
475
555
|
Office and EPUB conversion runs fully locally in the browser or Node.js. Files are passed in and returned as bytes;
|
|
@@ -493,14 +573,14 @@ cc.convertOfficeBytes(inputBytes, format, punctuation, keepFont)
|
|
|
493
573
|
|
|
494
574
|
Parameters:
|
|
495
575
|
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
576
|
+
- `inputBytes`: `Uint8Array` document bytes
|
|
577
|
+
- `format`: `docx`, `xlsx`, `pptx`, `odt`, `ods`, `odp`, or `epub`
|
|
578
|
+
- `punctuation`: whether to convert punctuation variants
|
|
579
|
+
- `keepFont`: whether to preserve font declarations where supported
|
|
500
580
|
|
|
501
581
|
Returns:
|
|
502
582
|
|
|
503
|
-
|
|
583
|
+
- converted output bytes
|
|
504
584
|
|
|
505
585
|
The older free function remains available for compatibility:
|
|
506
586
|
|
|
@@ -649,6 +729,7 @@ opencc-fmmseg office -i input.docx -o output.docx -c s2t -p --keep-font
|
|
|
649
729
|
default when omitted value: all
|
|
650
730
|
--keep-ids Preserve complete IDS expressions during conversion (default: false)
|
|
651
731
|
-n, --norm-compat Normalize CJK Compatibility Ideographs before conversion (default: false)
|
|
732
|
+
-E, --norm-compat-extended Normalize extended Unicode compatibility forms before conversion (default: false)
|
|
652
733
|
-D, --custom-dict <slot:mode:file>
|
|
653
734
|
Load a custom dictionary.
|
|
654
735
|
May be specified multiple times.
|
|
@@ -708,18 +789,18 @@ The WASM-facing enum is exported as `OpenccConfigWasm`, alongside `OpenccWasm`.
|
|
|
708
789
|
|
|
709
790
|
## Performance Notes
|
|
710
791
|
|
|
711
|
-
|
|
712
|
-
|
|
713
|
-
|
|
792
|
+
- WebAssembly build disables Rayon parallelism by default.
|
|
793
|
+
- Dictionaries are embedded into the WASM binary.
|
|
794
|
+
- Browser caching significantly improves subsequent loads.
|
|
714
795
|
|
|
715
796
|
---
|
|
716
797
|
|
|
717
798
|
## Related Projects
|
|
718
799
|
|
|
719
|
-
|
|
720
|
-
|
|
721
|
-
|
|
722
|
-
|
|
800
|
+
- Rust backend: https://github.com/laisuk/opencc-fmmseg
|
|
801
|
+
- C API: https://github.com/laisuk/opencc-fmmseg/tree/master/capi/opencc-fmmseg-capi
|
|
802
|
+
- .NET: https://github.com/laisuk/OpenccNet
|
|
803
|
+
- Python: https://github.com/laisuk/opencc_purepy
|
|
723
804
|
|
|
724
805
|
---
|
|
725
806
|
|
package/bin/opencc.js
CHANGED
|
@@ -58,6 +58,7 @@ Convert options:
|
|
|
58
58
|
default when omitted value: all
|
|
59
59
|
--keep-ids Preserve complete IDS expressions during conversion (default: false)
|
|
60
60
|
-n, --norm-compat Normalize CJK Compatibility Ideographs before conversion (default: false)
|
|
61
|
+
-E, --norm-compat-extended Normalize extended Unicode compatibility forms before conversion (default: false)
|
|
61
62
|
-D, --custom-dict <slot:mode:file>
|
|
62
63
|
Load a custom dictionary.
|
|
63
64
|
May be specified multiple times.
|
|
@@ -88,7 +89,7 @@ Office options:
|
|
|
88
89
|
-F, --convert-filename Convert generated output filename stem (default: false)
|
|
89
90
|
--keep-font Preserve font-family information (default)
|
|
90
91
|
--no-keep-font Do not preserve font-family information
|
|
91
|
-
--custom-dict <slot:mode:file>
|
|
92
|
+
-D, --custom-dict <slot:mode:file>
|
|
92
93
|
Load a custom dictionary.
|
|
93
94
|
May be specified multiple times.
|
|
94
95
|
Examples:
|
|
@@ -107,6 +108,7 @@ Examples:
|
|
|
107
108
|
npx opencc-fmmseg convert -i a.txt -o b.txt -c t2s --detofu ext-c
|
|
108
109
|
echo "⿰氵漢" | npx opencc-fmmseg convert -c t2s
|
|
109
110
|
echo "⿰氵漢" | npx opencc-fmmseg convert -c t2s --keep-ids
|
|
111
|
+
echo "聼聼竒羙⽟䂖甁噐⾳" | npx opencc-fmmseg convert -c t2s -E
|
|
110
112
|
|
|
111
113
|
npx opencc-fmmseg office -i a.docx -o b.docx -c s2t -p
|
|
112
114
|
npx opencc-fmmseg office -i a.epub -c s2tw
|
|
@@ -189,6 +191,24 @@ function validateInputFile(filePath) {
|
|
|
189
191
|
}
|
|
190
192
|
}
|
|
191
193
|
|
|
194
|
+
function ensureDistinctPaths(input, output) {
|
|
195
|
+
if (!input || !output) {
|
|
196
|
+
return;
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
let inputPath = path.resolve(input);
|
|
200
|
+
let outputPath = path.resolve(output);
|
|
201
|
+
|
|
202
|
+
if (process.platform === "win32") {
|
|
203
|
+
inputPath = inputPath.toLowerCase();
|
|
204
|
+
outputPath = outputPath.toLowerCase();
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
if (inputPath === outputPath) {
|
|
208
|
+
throw new Error("Input and output files must be different.");
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
|
|
192
212
|
function parseCustomDictSpec(value) {
|
|
193
213
|
const first = value.indexOf(":");
|
|
194
214
|
const second = value.indexOf(":", first + 1);
|
|
@@ -450,6 +470,8 @@ function parseDetofuLevel(value) {
|
|
|
450
470
|
async function runConvert(args) {
|
|
451
471
|
const input = getArg(args, "-i", "--input");
|
|
452
472
|
const output = getArg(args, "-o", "--output");
|
|
473
|
+
ensureDistinctPaths(input, output);
|
|
474
|
+
|
|
453
475
|
const config = getArg(args, "-c", "--config", "s2t");
|
|
454
476
|
const inEnc = validateEncoding(
|
|
455
477
|
getArg(args, null, "--in-enc", "utf8"),
|
|
@@ -463,6 +485,7 @@ async function runConvert(args) {
|
|
|
463
485
|
const punct = hasFlag(args, "-p", "--punct");
|
|
464
486
|
const keepIds = hasFlag(args, null, "--keep-ids");
|
|
465
487
|
const normCompat = hasFlag(args, "-n", "--norm-compat");
|
|
488
|
+
const normCompatExtended = hasFlag(args, "-E", "--norm-compat-extended");
|
|
466
489
|
const customDicts = getArgs(args, "-D", "--custom-dict")
|
|
467
490
|
.map(parseCustomDictSpec);
|
|
468
491
|
|
|
@@ -498,7 +521,9 @@ async function runConvert(args) {
|
|
|
498
521
|
|
|
499
522
|
let inputText = readInputText(input, inEnc);
|
|
500
523
|
|
|
501
|
-
if (
|
|
524
|
+
if (normCompatExtended) {
|
|
525
|
+
inputText = cc.normalizeCompatExtended(inputText);
|
|
526
|
+
} else if (normCompat) {
|
|
502
527
|
inputText = cc.normalizeCompat(inputText);
|
|
503
528
|
}
|
|
504
529
|
|
|
@@ -519,10 +544,20 @@ async function runConvert(args) {
|
|
|
519
544
|
}
|
|
520
545
|
|
|
521
546
|
const suffixParts = [];
|
|
522
|
-
|
|
547
|
+
|
|
548
|
+
if (normCompatExtended) {
|
|
549
|
+
suffixParts.push("norm-compat-extended");
|
|
550
|
+
} else if (normCompat) {
|
|
551
|
+
suffixParts.push("norm-compat");
|
|
552
|
+
}
|
|
553
|
+
|
|
523
554
|
if (detofuEnabled) suffixParts.push("detofu");
|
|
524
555
|
if (keepIds) suffixParts.push("keep-ids");
|
|
525
556
|
|
|
557
|
+
for (const spec of customDicts) {
|
|
558
|
+
suffixParts.push(`custom:${spec.slot}:${spec.mode}`);
|
|
559
|
+
}
|
|
560
|
+
|
|
526
561
|
const suffix = suffixParts.length ? `, ${suffixParts.join(", ")}` : "";
|
|
527
562
|
console.error(`Conversion completed (${cc.getConfig()}${suffix}): ${inFrom} -> ${outTo}`);
|
|
528
563
|
}
|
|
@@ -561,6 +596,8 @@ async function runOffice(args) {
|
|
|
561
596
|
output = applyOutputExtension(output, officeFormat);
|
|
562
597
|
}
|
|
563
598
|
|
|
599
|
+
ensureDistinctPaths(input, output);
|
|
600
|
+
|
|
564
601
|
const inputBytes = fs.readFileSync(input);
|
|
565
602
|
|
|
566
603
|
const outputBytes = cc.convertOfficeBytes(
|