@laisuk/opencc-fmmseg-wasm 0.3.3 → 0.3.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -21,6 +21,7 @@ Features:
21
21
  * Traditional Chinese regional variants
22
22
  * Japanese Shinjitai conversion support
23
23
  * Chinese script detection (`zho_check`)
24
+ * Optional CJK Compatibility Ideograph normalization
24
25
  * In-memory Office / EPUB document conversion
25
26
  * Zero-dependency Node.js CLI
26
27
 
@@ -162,6 +163,41 @@ cc.convert("汉字", false);
162
163
 
163
164
  ---
164
165
 
166
+ ### normalizeCompat
167
+
168
+ Normalize Unicode CJK Compatibility Ideographs before conversion.
169
+
170
+ ```javascript
171
+ cc.normalizeCompat(text)
172
+ ```
173
+
174
+ Parameters:
175
+
176
+ * `text`: input string
177
+
178
+ Returns:
179
+
180
+ * normalized string
181
+
182
+ Example:
183
+
184
+ ```javascript
185
+ const cc = new OpenccWasm("t2s");
186
+
187
+ const input = "天龍八部書裡的喬峰是契丹人";
188
+ const normalized = cc.normalizeCompat(input);
189
+
190
+ console.log(normalized);
191
+ // 天龍八部書裡的喬峰是契丹人
192
+
193
+ console.log(cc.convert(normalized, false));
194
+ // 天龙八部书里的乔峰是契丹人
195
+ ```
196
+
197
+ This is an optional pre-conversion pass for text that contains compatibility ideographs from Unicode compatibility ranges. Unmapped characters are preserved unchanged. Normal OpenCC conversion does not automatically run this pass, so call it explicitly when compatibility normalization is desired.
198
+
199
+ ---
200
+
165
201
  ### detofu
166
202
 
167
203
  Replace tofu-risk rare CJK extension characters with display-compatible fallbacks.
@@ -387,7 +423,9 @@ const cc = OpenccWasm.newWithCustomDicts("s2t", [
387
423
  `Override` replaces the selected slot before inserting the provided pairs. It is powerful and should be used only when
388
424
  the caller intentionally wants to discard built-in entries for that slot.
389
425
 
390
- Custom dictionary specs use strict canonical `DictSlot` names:
426
+ Custom dictionary specs identify the target dictionary slot by `DictSlot` name. Slot names are trimmed and normalized
427
+ case-insensitively for the known slots, so `"stphrases"`, `" STPhrases "`, and `"STPhrases"` all select
428
+ `STPhrases`. Canonical names are recommended in TypeScript code and docs:
391
429
 
392
430
  ```text
393
431
  STPhrases
@@ -406,15 +444,15 @@ HKVariants
406
444
  HKVariantsPhrases
407
445
  HKVariantsRev
408
446
  HKVariantsRevPhrases
409
- JPShinjitaiCharacters
410
- JPShinjitaiPhrases
411
- JPVariants
412
- JPVariantsRev
447
+ JPSCharacters
448
+ JPSCharactersRev
449
+ JPSPhrases
413
450
  STPunctuations
414
451
  TSPunctuations
415
452
  ```
416
453
 
417
- Suffixes such as `.txt` are not accepted. Use `"STPhrases"`, not `"STPhrases.txt"`.
454
+ Suffixes such as `.txt` are not accepted, even though case and surrounding whitespace are normalized. Use
455
+ `"STPhrases"` or `"stphrases"`, not `"STPhrases.txt"`.
418
456
 
419
457
  Merge contract:
420
458
 
@@ -447,15 +485,17 @@ docx, xlsx, pptx, odt, ods, odp, epub
447
485
  File size is limited by available browser or Node.js memory, but there is no upload or server-side limit. Font
448
486
  preservation is supported with the `keepFont` option.
449
487
 
488
+ Use the instance method when possible. It reuses the converter configuration and any custom dictionaries already held by
489
+ the `OpenccWasm` instance.
490
+
450
491
  ```javascript
451
- convert_office_bytes(inputBytes, format, config, punctuation, keepFont)
492
+ cc.convertOfficeBytes(inputBytes, format, punctuation, keepFont)
452
493
  ```
453
494
 
454
495
  Parameters:
455
496
 
456
497
  * `inputBytes`: `Uint8Array` document bytes
457
498
  * `format`: `docx`, `xlsx`, `pptx`, `odt`, `ods`, `odp`, or `epub`
458
- * `config`: OpenCC config string, such as `"s2t"`
459
499
  * `punctuation`: whether to convert punctuation variants
460
500
  * `keepFont`: whether to preserve font declarations where supported
461
501
 
@@ -463,20 +503,26 @@ Returns:
463
503
 
464
504
  * converted output bytes
465
505
 
506
+ The older free function remains available for compatibility:
507
+
508
+ ```javascript
509
+ convert_office_bytes(inputBytes, format, config, punctuation, keepFont)
510
+ ```
511
+
466
512
  ### Browser Office Example
467
513
 
468
514
  ```javascript
469
- import init, {convert_office_bytes} from "@laisuk/opencc-fmmseg-wasm";
515
+ import init, {OpenccWasm} from "@laisuk/opencc-fmmseg-wasm";
470
516
 
471
517
  await init();
472
518
 
519
+ const cc = new OpenccWasm("s2t");
473
520
  const file = document.querySelector("input[type=file]").files[0];
474
521
  const inputBytes = new Uint8Array(await file.arrayBuffer());
475
522
 
476
- const outputBytes = convert_office_bytes(
523
+ const outputBytes = cc.convertOfficeBytes(
477
524
  inputBytes,
478
525
  "docx",
479
- "s2t",
480
526
  true,
481
527
  true
482
528
  );
@@ -496,16 +542,16 @@ URL.revokeObjectURL(a.href);
496
542
 
497
543
  ```javascript
498
544
  import fs from "fs";
499
- import init, {convert_office_bytes} from "@laisuk/opencc-fmmseg-wasm";
545
+ import init, {OpenccWasm} from "@laisuk/opencc-fmmseg-wasm";
500
546
 
501
547
  await init();
502
548
 
549
+ const cc = new OpenccWasm("s2t");
503
550
  const inputBytes = fs.readFileSync("input.docx");
504
551
 
505
- const outputBytes = convert_office_bytes(
552
+ const outputBytes = cc.convertOfficeBytes(
506
553
  inputBytes,
507
554
  "docx",
508
- "s2t",
509
555
  true,
510
556
  true
511
557
  );
@@ -574,6 +620,18 @@ The package includes a zero-dependency Node.js CLI:
574
620
  opencc-fmmseg convert -i input.txt -o output.txt -c s2t -p
575
621
  opencc-fmmseg convert -i input.txt -o output.txt -c t2s -p --detofu all
576
622
  echo "别随便录影侵犯个人隐私权" | opencc-fmmseg convert -c s2hkp
623
+ echo "天龍八部書裡的喬峰是契丹人" | opencc-fmmseg convert -c t2s --norm-compat
624
+ // 天龙八部书里的乔峰是契丹人
625
+ echo "這個細路哥很靈活" | opencc-fmmseg convert -c hk2sp --custom-dict hkphrasesrev:append:my_hk_dict.txt
626
+ // 这个小男孩很灵活
627
+ ```
628
+
629
+ my_hk_dict.txt:
630
+
631
+ ```
632
+ # Custom Dictionary
633
+
634
+ 細路哥 小男孩
577
635
  ```
578
636
 
579
637
  ```bash
@@ -583,15 +641,23 @@ opencc-fmmseg office -i input.docx -o output.docx -c s2t -p --keep-font
583
641
  ### Text Conversion Options
584
642
 
585
643
  ```text
586
- -i, --input <file> Input text file
587
- -o, --output <file> Output text file
588
- -c, --config <conversion> Conversion config
644
+ -i, --input <file> Input text file; stdin if omitted
645
+ -o, --output <file> Output text file; stdout if omitted
646
+ -c, --config <conversion> Conversion config (default: s2t)
589
647
  -p, --punct Enable punctuation conversion
590
648
  --detofu [level] Replace tofu-risk rare CJK extension chars after conversion
591
649
  level: all | ext-b | ext-c | ext-d | ext-e | ext-f | ext-g | ext-h | ext-i
592
650
  default when omitted value: all
593
- --in-enc <encoding> Input encoding
594
- --out-enc <encoding> Output encoding
651
+ --keep-ids Preserve complete IDS expressions during conversion (default: false)
652
+ -n, --norm-compat Normalize CJK Compatibility Ideographs before conversion (default: false)
653
+ -D, --custom-dict <slot:mode:file>
654
+ Load a custom dictionary.
655
+ May be specified multiple times.
656
+ Examples:
657
+ --custom-dict hkphrasesrev:append:my_hk_dict.txt
658
+ --custom-dict stphrases:override:terms.txt
659
+ --in-enc <encoding> Input encoding (default: utf8)
660
+ --out-enc <encoding> Output encoding (default: utf8)
595
661
  ```
596
662
 
597
663
  Supported conversion configs:
@@ -606,12 +672,18 @@ tw2s, tw2sp, tw2t, tw2tp, hk2s, hk2sp, hk2t, jp2t, t2jp
606
672
  ```text
607
673
  -i, --input <file> Input Office / EPUB file
608
674
  -o, --output <file> Output file
609
- -c, --config <conversion> Conversion config
675
+ -c, --config <conversion> Conversion config (default: s2t)
610
676
  -p, --punct Enable punctuation conversion
611
- --format <format> docx | xlsx | pptx | odt | ods | odp | epub
612
- --auto-ext Append extension to output if missing
613
- --keep-font Preserve font-family information
677
+ -f, --format <format> docx | xlsx | pptx | odt | ods | odp | epub
678
+ -F, --convert-filename Convert generated output filename stem (default: false)
679
+ --keep-font Preserve font-family information (default)
614
680
  --no-keep-font Do not preserve font-family information
681
+ --custom-dict <slot:mode:file>
682
+ Load a custom dictionary.
683
+ May be specified multiple times.
684
+ Examples:
685
+ --custom-dict hkphrasesrev:append:my_hk_dict.txt
686
+ --custom-dict stphrases:override:terms.txt
615
687
  ```
616
688
 
617
689
  For `office`, the format is inferred from the input file extension when `--format` is omitted.
package/bin/opencc.js CHANGED
@@ -6,8 +6,7 @@ import process from "process";
6
6
 
7
7
  import init, {
8
8
  OpenccWasm,
9
- DetofuLevelWasm,
10
- convert_office_bytes
9
+ DetofuLevelWasm
11
10
  } from "../opencc_fmmseg_wasm.js";
12
11
 
13
12
  const OFFICE_FORMATS = new Set([
@@ -58,6 +57,13 @@ Convert options:
58
57
  level: all | ext-b | ext-c | ext-d | ext-e | ext-f | ext-g | ext-h | ext-i
59
58
  default when omitted value: all
60
59
  --keep-ids Preserve complete IDS expressions during conversion (default: false)
60
+ -n, --norm-compat Normalize CJK Compatibility Ideographs before conversion (default: false)
61
+ -D, --custom-dict <slot:mode:file>
62
+ Load a custom dictionary.
63
+ May be specified multiple times.
64
+ Examples:
65
+ --custom-dict hkphrasesrev:append:my_hk_dict.txt
66
+ --custom-dict stphrases:override:terms.txt
61
67
  --in-enc <encoding> Input encoding (default: utf8)
62
68
  --out-enc <encoding> Output encoding (default: utf8)
63
69
 
@@ -70,10 +76,16 @@ Office options:
70
76
  -o, --output <file> Output file
71
77
  -c, --config <conversion> Conversion config (default: s2t)
72
78
  -p, --punct Enable punctuation conversion
73
- --format <format> docx | xlsx | pptx | odt | ods | odp | epub
74
- --auto-ext Append extension to output if missing
79
+ -f, --format <format> docx | xlsx | pptx | odt | ods | odp | epub
80
+ -F, --convert-filename Convert generated output filename stem (default: false)
75
81
  --keep-font Preserve font-family information (default)
76
82
  --no-keep-font Do not preserve font-family information
83
+ --custom-dict <slot:mode:file>
84
+ Load a custom dictionary.
85
+ May be specified multiple times.
86
+ Examples:
87
+ --custom-dict hkphrasesrev:append:my_hk_dict.txt
88
+ --custom-dict stphrases:override:terms.txt
77
89
 
78
90
  General options:
79
91
  -h, --help Show help
@@ -85,9 +97,12 @@ Examples:
85
97
  echo "别随便录影侵犯个人隐私权" | npx opencc-fmmseg convert -c s2hkp
86
98
  npx opencc-fmmseg convert -i a.txt -o b.txt -c t2s --detofu
87
99
  npx opencc-fmmseg convert -i a.txt -o b.txt -c t2s --detofu ext-c
100
+ echo "⿰氵漢" | npx opencc-fmmseg convert -c t2s
101
+ echo "⿰氵漢" | npx opencc-fmmseg convert -c t2s --keep-ids
88
102
 
89
103
  npx opencc-fmmseg office -i a.docx -o b.docx -c s2t -p
90
- npx opencc-fmmseg office -i a.epub -c s2tw --auto-ext
104
+ npx opencc-fmmseg office -i a.epub -c s2tw
105
+ npx opencc-fmmseg office -i 软件手册.docx -c s2t --convert-filename
91
106
  `);
92
107
  }
93
108
 
@@ -112,6 +127,97 @@ function getArg(args, shortName, longName, defaultValue = null) {
112
127
  return defaultValue;
113
128
  }
114
129
 
130
+ function getArgs(args, shortName, longName) {
131
+ const values = [];
132
+ const candidates = [];
133
+
134
+ if (shortName) candidates.push(shortName);
135
+ if (longName) candidates.push(longName);
136
+
137
+ if (candidates.length === 0) return values;
138
+
139
+ for (let i = 0; i < args.length; i++) {
140
+ if (candidates.includes(args[i]) && i + 1 < args.length) {
141
+ values.push(args[i + 1]);
142
+ i++;
143
+ }
144
+ }
145
+
146
+ return values;
147
+ }
148
+
149
+ function parseCustomDictSpec(value) {
150
+ const first = value.indexOf(":");
151
+ const second = value.indexOf(":", first + 1);
152
+
153
+ if (first < 0 || second < 0) {
154
+ throw new Error(
155
+ `Invalid custom dictionary specification: ${value}\n` +
156
+ "Expected: <slot>:<append|override>:<file>"
157
+ );
158
+ }
159
+
160
+ const slot = value.substring(0, first).trim();
161
+ const mode = value.substring(first + 1, second).trim();
162
+ const file = value.substring(second + 1).trim();
163
+
164
+ if (!slot) throw new Error("Custom dictionary slot is empty.");
165
+ if (!mode) throw new Error("Custom dictionary mode is empty.");
166
+ if (!file) throw new Error("Custom dictionary file is empty.");
167
+
168
+ return {
169
+ slot,
170
+ mode,
171
+ pairs: loadCustomDictPairs(file)
172
+ };
173
+ }
174
+
175
+ function loadCustomDictPairs(file) {
176
+ if (!fs.existsSync(file)) {
177
+ throw new Error(`Custom dictionary file not found: ${file}`);
178
+ }
179
+
180
+ const text = fs.readFileSync(file, "utf8");
181
+ const lines = text.split(/\r?\n/);
182
+ const pairs = [];
183
+
184
+ for (let i = 0; i < lines.length; i++) {
185
+ let line = lines[i].replace(/\r$/, "");
186
+
187
+ if (i === 0 && line.charCodeAt(0) === 0xfeff) {
188
+ line = line.slice(1);
189
+ }
190
+
191
+ line = line.trimEnd();
192
+
193
+ if (!line || line.startsWith("#")) {
194
+ continue;
195
+ }
196
+
197
+ const tab = line.indexOf("\t");
198
+
199
+ if (tab < 0) {
200
+ throw new Error(
201
+ `Invalid custom dictionary file ${file}:${i + 1}: missing TAB separator`
202
+ );
203
+ }
204
+
205
+ const source = line.substring(0, tab);
206
+ const values = line.substring(tab + 1).trim().split(/\s+/);
207
+ const target = values[0] || "";
208
+
209
+ if (!source || !target) {
210
+ throw new Error(
211
+ `Invalid custom dictionary file ${file}:${i + 1}: empty source or target`
212
+ );
213
+ }
214
+
215
+ pairs.push([source, target]);
216
+ }
217
+
218
+ return pairs;
219
+ }
220
+
115
221
  function hasFlag(args, shortName, longName) {
116
222
  return (
117
223
  (shortName && args.includes(shortName)) ||
@@ -167,24 +273,20 @@ function inferOfficeFormat(inputFile, explicitFormat) {
167
273
  return ext;
168
274
  }
169
275
 
170
- function makeDefaultOfficeOutput(inputFile, officeFormat, autoExt) {
276
+ function makeDefaultOfficeOutput(inputFile, officeFormat, convertFilename, cc, config, punct) {
171
277
  const parsed = path.parse(inputFile);
172
- const ext = autoExt && OFFICE_FORMATS.has(officeFormat)
173
- ? `.${officeFormat}`
174
- : parsed.ext;
278
+ const stem = convertFilename
279
+ ? cc.convert(parsed.name, punct)
280
+ : parsed.name;
175
281
 
176
282
  return path.join(
177
283
  parsed.dir || process.cwd(),
178
- `${parsed.name}_converted${ext}`
284
+ `${stem}_converted.${officeFormat}`
179
285
  );
180
286
  }
181
287
 
182
- function applyAutoExt(outputFile, officeFormat, autoExt) {
183
- if (!autoExt || path.extname(outputFile)) {
184
- return outputFile;
185
- }
186
-
187
- if (!OFFICE_FORMATS.has(officeFormat)) {
288
+ function applyOutputExtension(outputFile, officeFormat) {
289
+ if (path.extname(outputFile)) {
188
290
  return outputFile;
189
291
  }
190
292
 
@@ -249,6 +351,9 @@ async function runConvert(args) {
249
351
  const outEnc = getArg(args, null, "--out-enc", "utf8");
250
352
  const punct = hasFlag(args, "-p", "--punct");
251
353
  const keepIds = hasFlag(args, null, "--keep-ids");
354
+ const normCompat = hasFlag(args, "-n", "--norm-compat");
355
+ const customDicts = getArgs(args, "-D", "--custom-dict")
356
+ .map(parseCustomDictSpec);
252
357
 
253
358
  const detofuIndex = args.indexOf("--detofu");
254
359
  const detofuEnabled = detofuIndex !== -1;
@@ -266,7 +371,10 @@ async function runConvert(args) {
266
371
 
267
372
  await ensureWasmInitialized();
268
373
 
269
- const cc = new OpenccWasm(config);
374
+ // const cc = new OpenccWasm(config);
375
+ const cc = customDicts.length === 0
376
+ ? new OpenccWasm(config)
377
+ : OpenccWasm.newWithCustomDicts(config, customDicts);
270
378
 
271
379
  if (keepIds) {
272
380
  cc.setPreserveIds(true);
@@ -277,7 +385,12 @@ async function runConvert(args) {
277
385
  console.error("Input text to convert, <Ctrl+Z>/<Ctrl+D> to submit:");
278
386
  }
279
387
 
280
- const inputText = readInputText(input, inEnc);
388
+ let inputText = readInputText(input, inEnc);
389
+
390
+ if (normCompat) {
391
+ inputText = cc.normalizeCompat(inputText);
392
+ }
393
+
281
394
  let outputText = cc.convert(inputText, punct);
282
395
 
283
396
  if (detofuEnabled) {
@@ -295,11 +408,12 @@ async function runConvert(args) {
295
408
  }
296
409
 
297
410
  const suffixParts = [];
411
+ if (normCompat) suffixParts.push("normalized");
298
412
  if (detofuEnabled) suffixParts.push("detofu");
299
413
  if (keepIds) suffixParts.push("keep-ids");
300
414
 
301
415
  const suffix = suffixParts.length ? `, ${suffixParts.join(", ")}` : "";
302
- console.error(`Conversion completed (${config}${suffix}): ${inFrom} -> ${outTo}`);
416
+ console.error(`Conversion completed (${cc.getConfig()}${suffix}): ${inFrom} -> ${outTo}`);
303
417
  }
304
418
  }
305
419
 
@@ -307,10 +421,12 @@ async function runOffice(args) {
307
421
  const input = getArg(args, "-i", "--input");
308
422
  let output = getArg(args, "-o", "--output");
309
423
  const config = getArg(args, "-c", "--config", "s2t");
310
- const explicitFormat = getArg(args, null, "--format");
424
+ const explicitFormat = getArg(args, "-f", "--format");
311
425
  const punct = hasFlag(args, "-p", "--punct");
312
- const autoExt = hasFlag(args, null, "--auto-ext");
426
+ const convertFilename = hasFlag(args, "-F", "--convert-filename");
313
427
  const keepFont = !hasFlag(args, null, "--no-keep-font");
428
+ const customDicts = getArgs(args, "-D", "--custom-dict")
429
+ .map(parseCustomDictSpec);
314
430
 
315
431
  if (!input) {
316
432
  throw new Error("Input file is missing.");
@@ -322,18 +438,23 @@ async function runOffice(args) {
322
438
 
323
439
  const officeFormat = inferOfficeFormat(input, explicitFormat);
324
440
 
441
+ await ensureWasmInitialized();
442
+
443
+ // const cc = new OpenccWasm(config);
444
+ const cc = customDicts.length === 0
445
+ ? new OpenccWasm(config)
446
+ : OpenccWasm.newWithCustomDicts(config, customDicts);
447
+
325
448
  if (!output) {
326
- output = makeDefaultOfficeOutput(input, officeFormat, autoExt);
449
+ output = makeDefaultOfficeOutput(input, officeFormat, convertFilename, cc, config, punct);
327
450
  console.error(`Output file not specified. Using: ${output}`);
328
451
  } else {
329
- output = applyAutoExt(output, officeFormat, autoExt);
452
+ output = applyOutputExtension(output, officeFormat);
330
453
  }
331
454
 
332
- await ensureWasmInitialized();
333
-
334
455
  const inputBytes = fs.readFileSync(input);
335
456
 
336
- const outputBytes = convert_office_bytes(
457
+ const outputBytes = cc.convertOfficeBytes(
337
458
  inputBytes,
338
459
  officeFormat,
339
460
  config,
@@ -343,7 +464,7 @@ async function runOffice(args) {
343
464
 
344
465
  fs.writeFileSync(output, outputBytes);
345
466
 
346
- console.error(`Conversion completed (${config}, ${officeFormat}): ${input} -> ${output}`);
467
+ console.error(`Conversion completed (${cc.getConfig()}, ${officeFormat}): ${input} -> ${output}`);
347
468
  }
348
469
 
349
470
  async function main() {
@@ -38,6 +38,7 @@ export class OpenccWasm {
38
38
  [Symbol.dispose](): void;
39
39
  convert(text: string, punctuation: boolean): string;
40
40
  convertDetofu(text: string, punctuation: boolean, level: DetofuLevelWasm): string;
41
+ convertOfficeBytes(input: Uint8Array, format: string, punctuation: boolean, keep_font: boolean): Uint8Array;
41
42
  debugPing(): string;
42
43
  detofu(text: string, level: DetofuLevelWasm): string;
43
44
  getConfig(): string;
@@ -48,6 +49,7 @@ export class OpenccWasm {
48
49
  constructor(config?: string | null);
49
50
  static newWithCustomDicts(config: string | null | undefined, specs: any): OpenccWasm;
50
51
  static newWithEnum(config?: OpenccConfigWasm | null): OpenccWasm;
52
+ normalizeCompat(text: string): string;
51
53
  setConfig(config: string): boolean;
52
54
  setConfigEnum(config: OpenccConfigWasm): void;
53
55
  setPreserveIds(value: boolean): void;
@@ -65,6 +67,7 @@ export interface InitOutput {
65
67
  readonly convert_office_bytes: (a: number, b: number, c: number, d: number, e: number, f: number, g: number, h: number) => [number, number, number, number];
66
68
  readonly openccwasm_convert: (a: number, b: number, c: number, d: number) => [number, number];
67
69
  readonly openccwasm_convertDetofu: (a: number, b: number, c: number, d: number, e: number) => [number, number];
70
+ readonly openccwasm_convertOfficeBytes: (a: number, b: number, c: number, d: number, e: number, f: number, g: number) => [number, number, number, number];
68
71
  readonly openccwasm_debugPing: (a: number) => [number, number];
69
72
  readonly openccwasm_detofu: (a: number, b: number, c: number, d: number) => [number, number];
70
73
  readonly openccwasm_getConfig: (a: number) => [number, number];
@@ -75,6 +78,7 @@ export interface InitOutput {
75
78
  readonly openccwasm_new: (a: number, b: number) => [number, number, number];
76
79
  readonly openccwasm_newWithCustomDicts: (a: number, b: number, c: any) => [number, number, number];
77
80
  readonly openccwasm_newWithEnum: (a: number) => [number, number, number];
81
+ readonly openccwasm_normalizeCompat: (a: number, b: number, c: number) => [number, number];
78
82
  readonly openccwasm_setConfig: (a: number, b: number, c: number) => number;
79
83
  readonly openccwasm_setConfigEnum: (a: number, b: number) => void;
80
84
  readonly openccwasm_setPreserveIds: (a: number, b: number) => void;
@@ -94,6 +94,26 @@ export class OpenccWasm {
94
94
  wasm.__wbindgen_free(deferred2_0, deferred2_1, 1);
95
95
  }
96
96
  }
97
+ /**
98
+ * @param {Uint8Array} input
99
+ * @param {string} format
100
+ * @param {boolean} punctuation
101
+ * @param {boolean} keep_font
102
+ * @returns {Uint8Array}
103
+ */
104
+ convertOfficeBytes(input, format, punctuation, keep_font) {
105
+ const ptr0 = passArray8ToWasm0(input, wasm.__wbindgen_malloc);
106
+ const len0 = WASM_VECTOR_LEN;
107
+ const ptr1 = passStringToWasm0(format, wasm.__wbindgen_malloc, wasm.__wbindgen_realloc);
108
+ const len1 = WASM_VECTOR_LEN;
109
+ const ret = wasm.openccwasm_convertOfficeBytes(this.__wbg_ptr, ptr0, len0, ptr1, len1, punctuation, keep_font);
110
+ if (ret[3]) {
111
+ throw takeFromExternrefTable0(ret[2]);
112
+ }
113
+ var v3 = getArrayU8FromWasm0(ret[0], ret[1]).slice();
114
+ wasm.__wbindgen_free(ret[0], ret[1] * 1, 1);
115
+ return v3;
116
+ }
97
117
  /**
98
118
  * @returns {string}
99
119
  */
@@ -215,6 +235,24 @@ export class OpenccWasm {
215
235
  }
216
236
  return OpenccWasm.__wrap(ret[0]);
217
237
  }
238
+ /**
239
+ * @param {string} text
240
+ * @returns {string}
241
+ */
242
+ normalizeCompat(text) {
243
+ let deferred2_0;
244
+ let deferred2_1;
245
+ try {
246
+ const ptr0 = passStringToWasm0(text, wasm.__wbindgen_malloc, wasm.__wbindgen_realloc);
247
+ const len0 = WASM_VECTOR_LEN;
248
+ const ret = wasm.openccwasm_normalizeCompat(this.__wbg_ptr, ptr0, len0);
249
+ deferred2_0 = ret[0];
250
+ deferred2_1 = ret[1];
251
+ return getStringFromWasm0(ret[0], ret[1]);
252
+ } finally {
253
+ wasm.__wbindgen_free(deferred2_0, deferred2_1, 1);
254
+ }
255
+ }
218
256
  /**
219
257
  * @param {string} config
220
258
  * @returns {boolean}
@@ -291,7 +329,7 @@ export function convert_office_bytes(input, format, config, punctuation, keep_fo
291
329
  function __wbg_get_imports() {
292
330
  const import0 = {
293
331
  __proto__: null,
294
- __wbg_Error_ef53bc310eb298a0: function(arg0, arg1) {
332
+ __wbg_Error_92b29b0548f8b746: function(arg0, arg1) {
295
333
  const ret = Error(getStringFromWasm0(arg0, arg1));
296
334
  return ret;
297
335
  },
@@ -302,46 +340,46 @@ function __wbg_get_imports() {
302
340
  getDataViewMemory0().setInt32(arg0 + 4 * 1, len1, true);
303
341
  getDataViewMemory0().setInt32(arg0 + 4 * 0, ptr1, true);
304
342
  },
305
- __wbg___wbindgen_boolean_get_1a45e2c38d4d41b9: function(arg0) {
343
+ __wbg___wbindgen_boolean_get_fa956cfa2d1bd751: function(arg0) {
306
344
  const v = arg0;
307
345
  const ret = typeof(v) === 'boolean' ? v : undefined;
308
346
  return isLikeNone(ret) ? 0xFFFFFF : ret ? 1 : 0;
309
347
  },
310
- __wbg___wbindgen_debug_string_0accd80f45e5faa2: function(arg0, arg1) {
348
+ __wbg___wbindgen_debug_string_c25d447a39f5578f: function(arg0, arg1) {
311
349
  const ret = debugString(arg1);
312
350
  const ptr1 = passStringToWasm0(ret, wasm.__wbindgen_malloc, wasm.__wbindgen_realloc);
313
351
  const len1 = WASM_VECTOR_LEN;
314
352
  getDataViewMemory0().setInt32(arg0 + 4 * 1, len1, true);
315
353
  getDataViewMemory0().setInt32(arg0 + 4 * 0, ptr1, true);
316
354
  },
317
- __wbg___wbindgen_in_70a403a56e771704: function(arg0, arg1) {
355
+ __wbg___wbindgen_in_aca499c5de7ff5e5: function(arg0, arg1) {
318
356
  const ret = arg0 in arg1;
319
357
  return ret;
320
358
  },
321
- __wbg___wbindgen_is_function_754e9f305ff6029e: function(arg0) {
359
+ __wbg___wbindgen_is_function_1ff95bcc5517c252: function(arg0) {
322
360
  const ret = typeof(arg0) === 'function';
323
361
  return ret;
324
362
  },
325
- __wbg___wbindgen_is_object_56732c2bc353f41d: function(arg0) {
363
+ __wbg___wbindgen_is_object_a27215656b807791: function(arg0) {
326
364
  const val = arg0;
327
365
  const ret = typeof(val) === 'object' && val !== null;
328
366
  return ret;
329
367
  },
330
- __wbg___wbindgen_is_undefined_67b456be8673d3d7: function(arg0) {
368
+ __wbg___wbindgen_is_undefined_c05833b95a3cf397: function(arg0) {
331
369
  const ret = arg0 === undefined;
332
370
  return ret;
333
371
  },
334
- __wbg___wbindgen_jsval_loose_eq_2c56564c75129511: function(arg0, arg1) {
372
+ __wbg___wbindgen_jsval_loose_eq_db4c3b15f63fc170: function(arg0, arg1) {
335
373
  const ret = arg0 == arg1;
336
374
  return ret;
337
375
  },
338
- __wbg___wbindgen_number_get_9bb1761122181af2: function(arg0, arg1) {
376
+ __wbg___wbindgen_number_get_394265ed1e1b84ee: function(arg0, arg1) {
339
377
  const obj = arg1;
340
378
  const ret = typeof(obj) === 'number' ? obj : undefined;
341
379
  getDataViewMemory0().setFloat64(arg0 + 8 * 1, isLikeNone(ret) ? 0 : ret, true);
342
380
  getDataViewMemory0().setInt32(arg0 + 4 * 0, !isLikeNone(ret), true);
343
381
  },
344
- __wbg___wbindgen_string_get_72bdf95d3ae505b1: function(arg0, arg1) {
382
+ __wbg___wbindgen_string_get_b0ca35b86a603356: function(arg0, arg1) {
345
383
  const obj = arg1;
346
384
  const ret = typeof(obj) === 'string' ? obj : undefined;
347
385
  var ptr1 = isLikeNone(ret) ? 0 : passStringToWasm0(ret, wasm.__wbindgen_malloc, wasm.__wbindgen_realloc);
@@ -349,22 +387,22 @@ function __wbg_get_imports() {
349
387
  getDataViewMemory0().setInt32(arg0 + 4 * 1, len1, true);
350
388
  getDataViewMemory0().setInt32(arg0 + 4 * 0, ptr1, true);
351
389
  },
352
- __wbg___wbindgen_throw_1506f2235d1bdba0: function(arg0, arg1) {
390
+ __wbg___wbindgen_throw_344f42d3211c4765: function(arg0, arg1) {
353
391
  throw new Error(getStringFromWasm0(arg0, arg1));
354
392
  },
355
- __wbg_call_8a89609d89f6608a: function() { return handleError(function (arg0, arg1) {
393
+ __wbg_call_8a2dd23819f8a60a: function() { return handleError(function (arg0, arg1) {
356
394
  const ret = arg0.call(arg1);
357
395
  return ret;
358
396
  }, arguments); },
359
- __wbg_done_60cf307fcc680536: function(arg0) {
397
+ __wbg_done_89b2b13e91a60321: function(arg0) {
360
398
  const ret = arg0.done;
361
399
  return ret;
362
400
  },
363
- __wbg_get_1f8f054ddbaa7db2: function() { return handleError(function (arg0, arg1) {
401
+ __wbg_get_c7eb1f358a7654df: function() { return handleError(function (arg0, arg1) {
364
402
  const ret = Reflect.get(arg0, arg1);
365
403
  return ret;
366
404
  }, arguments); },
367
- __wbg_get_unchecked_33f6e5c9e2f2d6b2: function(arg0, arg1) {
405
+ __wbg_get_unchecked_6e0ad6d2a41b06f6: function(arg0, arg1) {
368
406
  const ret = arg0[arg1 >>> 0];
369
407
  return ret;
370
408
  },
@@ -372,7 +410,7 @@ function __wbg_get_imports() {
372
410
  const ret = arg0[arg1];
373
411
  return ret;
374
412
  },
375
- __wbg_instanceof_ArrayBuffer_8f49811467741499: function(arg0) {
413
+ __wbg_instanceof_ArrayBuffer_4480b9e0068a8adb: function(arg0) {
376
414
  let result;
377
415
  try {
378
416
  result = arg0 instanceof ArrayBuffer;
@@ -382,7 +420,7 @@ function __wbg_get_imports() {
382
420
  const ret = result;
383
421
  return ret;
384
422
  },
385
- __wbg_instanceof_Uint8Array_86f30649f63ef9c2: function(arg0) {
423
+ __wbg_instanceof_Uint8Array_309b927aaf7a3fc7: function(arg0) {
386
424
  let result;
387
425
  try {
388
426
  result = arg0 instanceof Uint8Array;
@@ -392,38 +430,38 @@ function __wbg_get_imports() {
392
430
  const ret = result;
393
431
  return ret;
394
432
  },
395
- __wbg_isArray_67c2c9c4313f4448: function(arg0) {
433
+ __wbg_isArray_0677c962b281d01a: function(arg0) {
396
434
  const ret = Array.isArray(arg0);
397
435
  return ret;
398
436
  },
399
- __wbg_iterator_8732428d309e270e: function() {
437
+ __wbg_iterator_6f722e4a93058b71: function() {
400
438
  const ret = Symbol.iterator;
401
439
  return ret;
402
440
  },
403
- __wbg_length_4a591ecaa01354d9: function(arg0) {
441
+ __wbg_length_1f0964f4a5e2c6d8: function(arg0) {
404
442
  const ret = arg0.length;
405
443
  return ret;
406
444
  },
407
- __wbg_length_66f1a4b2e9026940: function(arg0) {
445
+ __wbg_length_370319915dc99107: function(arg0) {
408
446
  const ret = arg0.length;
409
447
  return ret;
410
448
  },
411
- __wbg_new_578aeef4b6b94378: function(arg0) {
449
+ __wbg_new_cd45aabdf6073e84: function(arg0) {
412
450
  const ret = new Uint8Array(arg0);
413
451
  return ret;
414
452
  },
415
- __wbg_next_9e03acdf51c4960d: function(arg0) {
453
+ __wbg_next_6dbf2c0ac8cde20f: function(arg0) {
416
454
  const ret = arg0.next;
417
455
  return ret;
418
456
  },
419
- __wbg_next_eb8ca7351fa27906: function() { return handleError(function (arg0) {
457
+ __wbg_next_71f2aa1cb3d1e37e: function() { return handleError(function (arg0) {
420
458
  const ret = arg0.next();
421
459
  return ret;
422
460
  }, arguments); },
423
- __wbg_prototypesetcall_3249fc62a0fafa30: function(arg0, arg1, arg2) {
461
+ __wbg_prototypesetcall_4770620bbe4688a0: function(arg0, arg1, arg2) {
424
462
  Uint8Array.prototype.set.call(getArrayU8FromWasm0(arg0, arg1), arg2);
425
463
  },
426
- __wbg_value_f3625092ee4b37f4: function(arg0) {
464
+ __wbg_value_a5d5488a9589444a: function(arg0) {
427
465
  const ret = arg0.value;
428
466
  return ret;
429
467
  },
Binary file
package/package.json CHANGED
@@ -2,7 +2,7 @@
2
2
  "name": "@laisuk/opencc-fmmseg-wasm",
3
3
  "type": "module",
4
4
  "description": "WebAssembly bindings for opencc-fmmseg, a high-performance OpenCC-compatible Simplified/Traditional Chinese converter.",
5
- "version": "0.3.3",
5
+ "version": "0.3.5",
6
6
  "license": "MIT",
7
7
  "repository": {
8
8
  "type": "git",