@qvac/translation-nmtcpp 0.15.0 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -298,11 +298,19 @@ The three GPU control keys each accept a camelCase alias alongside the snake_cas
298
298
  | Key | Alias | Type | Description |
299
299
  |-----|-------|------|-------------|
300
300
  | `use_gpu` | `useGPU` | boolean | Enable GPU inference. When `false` (default), only the CPU backend is used. Bergamot is CPU-only by design — this flag is effectively a no-op for Bergamot. |
301
- | `gpu_backend` | `gpuBackend` | string | Case-insensitive **substring** match against the ggml device name (e.g. `"vulkan"`, `"vulkan0"`, `"opencl"`, `"metal"`). When set, the selector runs a single explicit pass and picks the first non-CPU device whose name contains the substring. When unset, the default gated selection runs (see [Backends](#backends)). Explicit `"opencl"` bypasses the build-time `USE_OPENCL` guard — an informed opt-in. |
301
+ | `gpu_backend` | `gpuBackend` | string | Case-insensitive **substring** match against eligible Vulkan, Metal, OpenCL, or CUDA device names (e.g. `"vulkan"`, `"vulkan0"`, `"opencl"`, `"metal"`). When unset, the default gated selection runs (see [Backends](#backends)). Any explicit selector that resolves to OpenCL bypasses the build-time `USE_OPENCL` guard — an informed opt-in. |
302
302
  | `gpu_device` | `gpuDevice` | int | Ordinal within the matching devices. Defaults to `0` (first match). Example: `{gpu_backend: "vulkan", gpu_device: 1}` picks the second Vulkan adapter. |
303
303
  | `backendsDir` | — | string | Path to the directory containing the runtime backend shared libraries (`libqvac-ggml-vulkan.so`, etc.). Defaults to `<package>/prebuilds` when unset, which is where `npm install` places the shipped prebuilds. Must be an absolute path; paths with `..` segments or unresolvable symlinks are rejected with a warning and fall back to the default prebuilds directory. |
304
304
  | `openclCacheDir` | — | string | **Android only.** Writable directory the OpenCL backend uses for its JIT kernel cache (forwarded via `GGML_OPENCL_CACHE_DIR`). Must be an absolute path; paths with `..` segments are rejected. The OpenCL backend falls back to a non-writable relative path if this is unset, which `ggml_abort()`s during init inside the app sandbox — always provide an app-writable path when exercising OpenCL on Android. |
305
305
 
306
+ GPU execution is limited to Vulkan, Metal/MTL, OpenCL, and CUDA devices reported as GPU or integrated GPU. RPC, ROCm/HIP, SYCL, MUSA, and unknown backend families fall back to CPU — RPC because translation has no split-mode support and cannot execute over it. CUDA is admitted ahead of its planned Fabric build so that build does not require another selector change. An explicit selector only narrows this eligible set; it cannot enable an unsupported family.
307
+
308
+ Within the eligible set, dedicated GPUs are ordered ahead of integrated ones, registry order preserved within each class, and `gpu_device` is an ordinal into that ordered list.
309
+
310
+ OpenCL is admitted as a family, not narrowed to Adreno. The selector inspects only the device and registry name, never the vendor or description, so any GPU or integrated-GPU OpenCL device is eligible. What gates it is the build-time `USE_OPENCL` guard, which exists because Adreno is the only OpenCL target this package has been validated on; an explicit `gpu_backend` selector bypasses that guard as an informed opt-in.
311
+
312
+ ACCEL and META devices are never selected. Translation runs on the single selected compute device plus CPU — it exposes no layer or tensor split mode, so no secondary accelerator is added to the backend list.
313
+
306
314
  > **Tip:** Use `model.getActiveBackendName()` after `load()` to confirm which backend actually took the request — see [Additional Features](#additional-features). The GGML scheduler silently falls back to CPU when no usable GPU ICD is registered, and this is the only way to detect that.
307
315
 
308
316
  ### 4. Create Model Instance
@@ -942,7 +950,7 @@ bare-make generate -D USE_BERGAMOT=OFF
942
950
  At runtime, the addon picks a ggml compute device using the `use_gpu`, `gpu_backend`, and `gpu_device` config keys described in [Backend & GPU Settings](#backend--gpu-settings). When `use_gpu` is true and `gpu_backend` is **not** set, the selector falls back to a default gated pass:
943
951
 
944
952
  1. If built with `USE_OPENCL=ON`, prefer an OpenCL device first.
945
- 2. Otherwise (and as a fallback in the `ON` case) pick any non-CPU device. When `USE_OPENCL=OFF` (the default), OpenCL-named devices are also excluded from the fallback — the OpenCL backend still loads as a shared library but is never selected automatically.
953
+ 2. Otherwise (and as a fallback in the `ON` case) pick an eligible Vulkan, Metal, or CUDA device. When `USE_OPENCL=OFF` (the default), OpenCL devices are excluded from the fallback — the OpenCL backend still loads as a shared library but is never selected automatically.
946
954
 
947
955
  An explicit `config.gpu_backend: 'opencl'` always bypasses the `USE_OPENCL` guard and selects OpenCL directly. The flag gates *automatic* selection only; caller-explicit requests are honored.
948
956
 
package/index.d.ts CHANGED
@@ -105,7 +105,7 @@ declare namespace TranslationNmtcpp {
105
105
  modelType: TranslationNmtcppModelTypes[keyof TranslationNmtcppModelTypes];
106
106
  pivotConfig?: Record<string, unknown>;
107
107
  /**
108
- * Enable GPU (non-CPU) compute backend. Read once at load() time.
108
+ * Enable an eligible Vulkan, Metal, OpenCL, or CUDA compute backend.
109
109
  * Bergamot is CPU-only by design — this flag is a no-op for that backend.
110
110
  *
111
111
  * `use_gpu` mirrors the C-struct field (`nmt_context_params::use_gpu`)
@@ -117,10 +117,10 @@ declare namespace TranslationNmtcpp {
117
117
  use_gpu?: boolean;
118
118
  useGPU?: boolean;
119
119
  /**
120
- * Case-insensitive substring filter over the ggml device name when selecting
120
+ * Case-insensitive substring filter over eligible ggml device names when selecting
121
121
  * a compute backend (e.g. "vulkan", "vulkan0", "opencl", "metal"). When set,
122
122
  * replaces the default gated selector with a single explicit pass.
123
- * An explicit "opencl" bypasses the build-time USE_OPENCL guard.
123
+ * Any explicit selector resolving to OpenCL bypasses the build-time USE_OPENCL guard.
124
124
  *
125
125
  * `gpu_backend` mirrors the C-struct field and is the primary key.
126
126
  * `gpuBackend` is the camelCase alias matching the sibling-addon convention.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@qvac/translation-nmtcpp",
3
- "version": "0.15.0",
3
+ "version": "0.16.0",
4
4
  "description": "translation addon for qvac",
5
5
  "addon": true,
6
6
  "engines": {
@@ -95,7 +95,7 @@
95
95
  },
96
96
  "dependencies": {
97
97
  "@qvac/error": "^0.1.0",
98
- "@qvac/fabric": "^0.13.0",
98
+ "@qvac/fabric": "^0.14.0",
99
99
  "@qvac/infer-base": "^0.6.2",
100
100
  "@qvac/logging": "^0.1.0",
101
101
  "bare-fs": "^4.5.1",
@@ -7,7 +7,7 @@
7
7
  ],
8
8
  "current_versions": [
9
9
  {
10
- "version": "0.15"
10
+ "version": "0.16"
11
11
  }
12
12
  ],
13
13
  "exported_symbols": [
@@ -3717,6 +3717,7 @@
3717
3717
  "__ZN4absl12lts_2025081418debugging_internal22StackTraceWorksForTestEv",
3718
3718
  "__ZNK4YAML6Stream13StreamInUtf16Ev",
3719
3719
  "__ZN6marian4castE12IntrusivePtrINS_9ChainableIS0_INS_10TensorBaseEEEEENS_4TypeE",
3720
+ "__Z18nmtSelectGpuDeviceRK19NmtBackendInterfacebRKNSt3__112basic_stringIcNS2_11char_traitsIcEENS2_9allocatorIcEEEEiPKcb",
3720
3721
  "__ZN6google8protobuf8internal24RepeatedStringTypeTraits23GetDefaultRepeatedFieldEv",
3721
3722
  "__ZNK6marian4data24BinaryShortlistGenerator14saveBlobToFileERKNSt3__112basic_stringIcNS2_11char_traitsIcEENS2_9allocatorIcEEEE",
3722
3723
  "__ZN13sentencepiece10ModelProto9MergeFromERKS0_",
@@ -7,7 +7,7 @@
7
7
  ],
8
8
  "current_versions": [
9
9
  {
10
- "version": "0.15"
10
+ "version": "0.16"
11
11
  }
12
12
  ],
13
13
  "exported_symbols": [
@@ -5841,6 +5841,7 @@
5841
5841
  "__ZN6marian4Node14init_dependentEv",
5842
5842
  "__ZN6Pathie4Path8sanitizeEv",
5843
5843
  "__ZNK6marian4data24BinaryShortlistGenerator14saveBlobToFileERKNSt3__112basic_stringIcNS2_11char_traitsIcEENS2_9allocatorIcEEEE",
5844
+ "__Z18nmtSelectGpuDeviceRK19NmtBackendInterfacebRKNSt3__112basic_stringIcNS2_11char_traitsIcEENS2_9allocatorIcEEEEiPKcb",
5844
5845
  "__tr_align",
5845
5846
  "__ZN6marian8bergamot25getVocabsMemoryFromConfigENSt3__110shared_ptrINS_7OptionsEEERNS1_6vectorINS2_INS0_13AlignedVectorIcEEEENS1_9allocatorIS8_EEEE",
5846
5847
  "__ZNK6google8protobuf8internal12ExtensionSet13ExtensionTypeEi",
@@ -7,7 +7,7 @@
7
7
  ],
8
8
  "current_versions": [
9
9
  {
10
- "version": "0.15"
10
+ "version": "0.16"
11
11
  }
12
12
  ],
13
13
  "exported_symbols": [
@@ -1118,13 +1118,13 @@
1118
1118
  "_pcre2_get_mark_8",
1119
1119
  "__ZN3ruy14Kernel8bitNeonERKNS_16KernelParams8bitILi4ELi4EEE",
1120
1120
  "__ZN4absl12lts_2025081418container_internal28SetHashtablezSampleParameterEi",
1121
- "__ZN6marian3cpu11ConcatenateE12IntrusivePtrINS_10TensorBaseEERKNSt3__16vectorIS3_NS4_9allocatorIS3_EEEEi",
1122
1121
  "__ZN13sentencepiece10normalizer10Normalizer4InitEv",
1123
1122
  "__ZN6google8protobuf8internal26DuplicateIfNonNullInternalEPNS0_11MessageLiteE",
1124
- "_pcre2_set_match_limit_8",
1123
+ "__ZN6marian3cpu11ConcatenateE12IntrusivePtrINS_10TensorBaseEERKNSt3__16vectorIS3_NS4_9allocatorIS3_EEEEi",
1125
1124
  "__ZN6google8protobuf8internal12ExtensionSet13MutableStringEihPKNS0_15FieldDescriptorE",
1126
- "__ZNK13sentencepiece23SentencePieceNormalizer23mutable_normalizer_specEv",
1127
1125
  "__ZNK6google8protobuf8internal12ExtensionSet26GetPrototypeForLazyMessageEPKNS0_11MessageLiteEi",
1126
+ "_pcre2_set_match_limit_8",
1127
+ "__ZNK13sentencepiece23SentencePieceNormalizer23mutable_normalizer_specEv",
1128
1128
  "__ZNK3ruy14PrepackedCache7KeyHashclERKNS0_3KeyE",
1129
1129
  "__ZNK6Pathie4Path5dglobERKNSt3__112basic_stringIcNS1_11char_traitsIcEENS1_9allocatorIcEEEEi",
1130
1130
  "__ZN6google8protobuf8internal12FieldSkipper11SkipMessageEPNS0_2io16CodedInputStreamE",
@@ -3697,6 +3697,7 @@
3697
3697
  "__ZN4absl12lts_2025081418debugging_internal22StackTraceWorksForTestEv",
3698
3698
  "__ZNK4YAML6Stream13StreamInUtf16Ev",
3699
3699
  "__ZN6marian4castE12IntrusivePtrINS_9ChainableIS0_INS_10TensorBaseEEEEENS_4TypeE",
3700
+ "__Z18nmtSelectGpuDeviceRK19NmtBackendInterfacebRKNSt3__112basic_stringIcNS2_11char_traitsIcEENS2_9allocatorIcEEEEiPKcb",
3700
3701
  "__ZN6google8protobuf8internal24RepeatedStringTypeTraits23GetDefaultRepeatedFieldEv",
3701
3702
  "__ZNK6marian4data24BinaryShortlistGenerator14saveBlobToFileERKNSt3__112basic_stringIcNS2_11char_traitsIcEENS2_9allocatorIcEEEE",
3702
3703
  "__ZN13sentencepiece10ModelProto9MergeFromERKS0_",
@@ -4142,8 +4143,8 @@
4142
4143
  "__ZN6google8protobuf16RepeatedPtrFieldINSt3__112basic_stringIcNS2_11char_traitsIcEENS2_9allocatorIcEEEEE2atEi",
4143
4144
  "__ZNK6google8protobuf13RepeatedFieldIfE4dataEv",
4144
4145
  "__ZN6google8protobuf13RepeatedFieldIfE12InternalSwapEPS2_",
4145
- "__ZN4absl12lts_2025081419str_format_internal13FormatArgImpl8DispatchIaEEbNS2_4DataENS1_24FormatConversionSpecImplEPv",
4146
4146
  "__ZN6marian15fastopt_helpers2AsIdE5applyERKNS_7FastOptE",
4147
+ "__ZN4absl12lts_2025081419str_format_internal13FormatArgImpl8DispatchIaEEbNS2_4DataENS1_24FormatConversionSpecImplEPv",
4147
4148
  "__ZN4absl12lts_2025081419str_format_internal13ConvertIntArgIjEEbT_NS1_24FormatConversionSpecImplEPNS1_14FormatSinkImplE",
4148
4149
  "__ZN6google8protobuf13RepeatedFieldIbE3AddERKb",
4149
4150
  "__ZN6google8protobuf13RepeatedFieldIbEC1Ev",
@@ -7,7 +7,7 @@
7
7
  ],
8
8
  "current_versions": [
9
9
  {
10
- "version": "0.15"
10
+ "version": "0.16"
11
11
  }
12
12
  ],
13
13
  "exported_symbols": [
@@ -1118,13 +1118,13 @@
1118
1118
  "_pcre2_get_mark_8",
1119
1119
  "__ZN3ruy14Kernel8bitNeonERKNS_16KernelParams8bitILi4ELi4EEE",
1120
1120
  "__ZN4absl12lts_2025081418container_internal28SetHashtablezSampleParameterEi",
1121
- "__ZN6marian3cpu11ConcatenateE12IntrusivePtrINS_10TensorBaseEERKNSt3__16vectorIS3_NS4_9allocatorIS3_EEEEi",
1122
1121
  "__ZN13sentencepiece10normalizer10Normalizer4InitEv",
1123
1122
  "__ZN6google8protobuf8internal26DuplicateIfNonNullInternalEPNS0_11MessageLiteE",
1124
- "_pcre2_set_match_limit_8",
1123
+ "__ZN6marian3cpu11ConcatenateE12IntrusivePtrINS_10TensorBaseEERKNSt3__16vectorIS3_NS4_9allocatorIS3_EEEEi",
1125
1124
  "__ZN6google8protobuf8internal12ExtensionSet13MutableStringEihPKNS0_15FieldDescriptorE",
1126
- "__ZNK13sentencepiece23SentencePieceNormalizer23mutable_normalizer_specEv",
1127
1125
  "__ZNK6google8protobuf8internal12ExtensionSet26GetPrototypeForLazyMessageEPKNS0_11MessageLiteEi",
1126
+ "_pcre2_set_match_limit_8",
1127
+ "__ZNK13sentencepiece23SentencePieceNormalizer23mutable_normalizer_specEv",
1128
1128
  "__ZNK3ruy14PrepackedCache7KeyHashclERKNS0_3KeyE",
1129
1129
  "__ZNK6Pathie4Path5dglobERKNSt3__112basic_stringIcNS1_11char_traitsIcEENS1_9allocatorIcEEEEi",
1130
1130
  "__ZN6google8protobuf8internal12FieldSkipper11SkipMessageEPNS0_2io16CodedInputStreamE",
@@ -3697,6 +3697,7 @@
3697
3697
  "__ZN4absl12lts_2025081418debugging_internal22StackTraceWorksForTestEv",
3698
3698
  "__ZNK4YAML6Stream13StreamInUtf16Ev",
3699
3699
  "__ZN6marian4castE12IntrusivePtrINS_9ChainableIS0_INS_10TensorBaseEEEEENS_4TypeE",
3700
+ "__Z18nmtSelectGpuDeviceRK19NmtBackendInterfacebRKNSt3__112basic_stringIcNS2_11char_traitsIcEENS2_9allocatorIcEEEEiPKcb",
3700
3701
  "__ZN6google8protobuf8internal24RepeatedStringTypeTraits23GetDefaultRepeatedFieldEv",
3701
3702
  "__ZNK6marian4data24BinaryShortlistGenerator14saveBlobToFileERKNSt3__112basic_stringIcNS2_11char_traitsIcEENS2_9allocatorIcEEEE",
3702
3703
  "__ZN13sentencepiece10ModelProto9MergeFromERKS0_",
@@ -4142,8 +4143,8 @@
4142
4143
  "__ZN6google8protobuf16RepeatedPtrFieldINSt3__112basic_stringIcNS2_11char_traitsIcEENS2_9allocatorIcEEEEE2atEi",
4143
4144
  "__ZNK6google8protobuf13RepeatedFieldIfE4dataEv",
4144
4145
  "__ZN6google8protobuf13RepeatedFieldIfE12InternalSwapEPS2_",
4145
- "__ZN4absl12lts_2025081419str_format_internal13FormatArgImpl8DispatchIaEEbNS2_4DataENS1_24FormatConversionSpecImplEPv",
4146
4146
  "__ZN6marian15fastopt_helpers2AsIdE5applyERKNS_7FastOptE",
4147
+ "__ZN4absl12lts_2025081419str_format_internal13FormatArgImpl8DispatchIaEEbNS2_4DataENS1_24FormatConversionSpecImplEPv",
4147
4148
  "__ZN4absl12lts_2025081419str_format_internal13ConvertIntArgIjEEbT_NS1_24FormatConversionSpecImplEPNS1_14FormatSinkImplE",
4148
4149
  "__ZN6google8protobuf13RepeatedFieldIbE3AddERKb",
4149
4150
  "__ZN6google8protobuf13RepeatedFieldIbEC1Ev",
@@ -7,7 +7,7 @@
7
7
  ],
8
8
  "current_versions": [
9
9
  {
10
- "version": "0.15"
10
+ "version": "0.16"
11
11
  }
12
12
  ],
13
13
  "exported_symbols": [
@@ -5814,6 +5814,7 @@
5814
5814
  "__ZN6marian4Node14init_dependentEv",
5815
5815
  "__ZN6Pathie4Path8sanitizeEv",
5816
5816
  "__ZNK6marian4data24BinaryShortlistGenerator14saveBlobToFileERKNSt3__112basic_stringIcNS2_11char_traitsIcEENS2_9allocatorIcEEEE",
5817
+ "__Z18nmtSelectGpuDeviceRK19NmtBackendInterfacebRKNSt3__112basic_stringIcNS2_11char_traitsIcEENS2_9allocatorIcEEEEiPKcb",
5817
5818
  "__tr_align",
5818
5819
  "__ZN6marian8bergamot25getVocabsMemoryFromConfigENSt3__110shared_ptrINS_7OptionsEEERNS1_6vectorINS2_INS0_13AlignedVectorIcEEEENS1_9allocatorIS8_EEEE",
5819
5820
  "__ZNK6google8protobuf8internal12ExtensionSet13ExtensionTypeEi",
@@ -541,25 +541,56 @@ function prestagedModelPath(modelName) {
541
541
  return staged ? staged.src : null
542
542
  }
543
543
 
544
- // Require the exact host-recorded byte count on both sides of the app copy so
545
- // truncated staged files fall through to the normal presigned-S3 download.
544
+ // iOS kills an app that dirties more than 4 GiB in 24h, and there the staged
545
+ // file already sits in the app's own writable Documents dir — so copying it
546
+ // into the model dir spends that whole budget for nothing. Hardlink instead:
547
+ // same inode, zero bytes. On Android the staging dir is a different filesystem,
548
+ // so link() fails EXDEV and we fall back to the copy that has always run there.
549
+ // `link`/`copy` are injectable so that fallback is unit-testable.
550
+ function linkOrCopySync({ src, dest, link = fs.linkSync, copy = fs.copyFileSync }) {
551
+ // Same path in and out — the staged file IS the destination (a caller whose
552
+ // model dir is testDir itself). Deleting first would destroy the staged model
553
+ // and leave both link() and copy() failing ENOENT; the copy this replaced was
554
+ // a harmless no-op here.
555
+ if (path.resolve(src) === path.resolve(dest)) return 'link'
556
+
557
+ try {
558
+ fs.unlinkSync(dest)
559
+ } catch (_) {}
560
+
561
+ try {
562
+ link(src, dest)
563
+ return 'link'
564
+ } catch (err) {
565
+ console.log(
566
+ `[prestage] hardlink failed on ${platform} (${err.message}); falling back to a byte copy`
567
+ )
568
+ }
569
+
570
+ copy(src, dest)
571
+ return 'copy'
572
+ }
573
+
574
+ // Require the exact host-recorded byte count on both sides of the staging step
575
+ // so truncated staged files fall through to the normal presigned-S3 download.
546
576
  function copyPrestagedModel(modelName, destPath, minBytes) {
547
577
  const staged = readPrestagedModel(modelName)
548
578
  if (!staged || staged.expectedSize < minBytes) return false
549
579
  try {
550
580
  const dir = path.dirname(destPath)
551
581
  if (!fs.existsSync(dir)) fs.mkdirSync(dir, { recursive: true })
552
- fs.copyFileSync(staged.src, destPath)
582
+ const how = linkOrCopySync({ src: staged.src, dest: destPath })
553
583
  const size = fs.statSync(destPath).size
554
584
  if (size === staged.expectedSize) {
555
585
  console.log(
556
- `[prestage] Using pre-staged model ${modelName} (${(size / 1024 / 1024).toFixed(1)}MB)`
586
+ `[prestage] Using pre-staged model ${modelName} (${(size / 1024 / 1024).toFixed(1)}MB, ` +
587
+ `${how === 'link' ? 'hardlinked' : 'copied'})`
557
588
  )
558
589
  return true
559
590
  }
560
591
  fs.unlinkSync(destPath)
561
592
  } catch (err) {
562
- console.log(`[prestage] copy of ${modelName} failed: ${err.message}`)
593
+ console.log(`[prestage] staging of ${modelName} failed: ${err.message}`)
563
594
  try {
564
595
  fs.unlinkSync(destPath)
565
596
  } catch (_) {}
@@ -1248,6 +1279,7 @@ module.exports = {
1248
1279
  ensureIndicTransModel,
1249
1280
  ensureBergamotModel,
1250
1281
  copyPrestagedModel,
1282
+ linkOrCopySync,
1251
1283
  prestagedModelPath,
1252
1284
 
1253
1285
  // Utilities