@elizaos/plugin-local-inference 2.0.0-beta.1 → 2.0.3-beta.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (893) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +157 -0
  3. package/dist/actions/generate-media.d.ts +59 -0
  4. package/dist/actions/generate-media.d.ts.map +1 -0
  5. package/dist/actions/identify-speaker.d.ts +23 -0
  6. package/dist/actions/identify-speaker.d.ts.map +1 -0
  7. package/dist/actions/transcription-control.d.ts +29 -0
  8. package/dist/actions/transcription-control.d.ts.map +1 -0
  9. package/dist/adapters/capacitor-llama/environment.d.ts +12 -0
  10. package/dist/adapters/capacitor-llama/environment.d.ts.map +1 -0
  11. package/dist/adapters/capacitor-llama/index.browser.d.ts +9 -0
  12. package/dist/adapters/capacitor-llama/index.browser.d.ts.map +1 -0
  13. package/dist/adapters/capacitor-llama/index.d.ts +18 -0
  14. package/dist/adapters/capacitor-llama/index.d.ts.map +1 -0
  15. package/dist/adapters/capacitor-llama/loader.d.ts +35 -0
  16. package/dist/adapters/capacitor-llama/loader.d.ts.map +1 -0
  17. package/dist/adapters/capacitor-llama/native-voice-capture.d.ts +70 -0
  18. package/dist/adapters/capacitor-llama/native-voice-capture.d.ts.map +1 -0
  19. package/dist/adapters/capacitor-llama/structured-output.d.ts +62 -0
  20. package/dist/adapters/capacitor-llama/structured-output.d.ts.map +1 -0
  21. package/dist/adapters/capacitor-llama/text-streaming.d.ts +24 -0
  22. package/dist/adapters/capacitor-llama/text-streaming.d.ts.map +1 -0
  23. package/dist/adapters/capacitor-llama/types.d.ts +338 -0
  24. package/dist/adapters/capacitor-llama/types.d.ts.map +1 -0
  25. package/dist/adapters/capacitor-llama/voice-turn.d.ts +86 -0
  26. package/dist/adapters/capacitor-llama/voice-turn.d.ts.map +1 -0
  27. package/dist/backends/apple-foundation.d.ts +56 -0
  28. package/dist/backends/apple-foundation.d.ts.map +1 -0
  29. package/dist/index.d.ts +8 -37
  30. package/dist/index.d.ts.map +1 -0
  31. package/dist/index.js +38979 -430
  32. package/dist/index.js.map +217 -0
  33. package/dist/local-inference-routes.d.ts +47 -0
  34. package/dist/local-inference-routes.d.ts.map +1 -0
  35. package/dist/provider.d.ts +21 -0
  36. package/dist/provider.d.ts.map +1 -0
  37. package/dist/routes/compat-helpers.d.ts +18 -0
  38. package/dist/routes/compat-helpers.d.ts.map +1 -0
  39. package/dist/routes/family-member-route.d.ts +62 -0
  40. package/dist/routes/family-member-route.d.ts.map +1 -0
  41. package/dist/routes/index.d.ts +20 -0
  42. package/dist/routes/index.d.ts.map +1 -0
  43. package/dist/routes/index.js +42040 -0
  44. package/dist/routes/index.js.map +236 -0
  45. package/dist/routes/live-diarization-route.d.ts +33 -0
  46. package/dist/routes/live-diarization-route.d.ts.map +1 -0
  47. package/dist/routes/local-inference-asr-route.d.ts +4 -0
  48. package/dist/routes/local-inference-asr-route.d.ts.map +1 -0
  49. package/dist/routes/local-inference-asr-transcribe.d.ts +20 -0
  50. package/dist/routes/local-inference-asr-transcribe.d.ts.map +1 -0
  51. package/dist/routes/local-inference-compat-routes.d.ts +16 -0
  52. package/dist/routes/local-inference-compat-routes.d.ts.map +1 -0
  53. package/dist/routes/local-inference-tts-route.d.ts +7 -0
  54. package/dist/routes/local-inference-tts-route.d.ts.map +1 -0
  55. package/dist/routes/native-pcm-turn-route.d.ts +3 -0
  56. package/dist/routes/native-pcm-turn-route.d.ts.map +1 -0
  57. package/dist/routes/transcript-audio-store.d.ts +15 -0
  58. package/dist/routes/transcript-audio-store.d.ts.map +1 -0
  59. package/dist/routes/transcripts-routes.d.ts +44 -0
  60. package/dist/routes/transcripts-routes.d.ts.map +1 -0
  61. package/dist/routes/voice-first-run-routes.d.ts +62 -0
  62. package/dist/routes/voice-first-run-routes.d.ts.map +1 -0
  63. package/dist/routes/voice-models-routes.d.ts +62 -0
  64. package/dist/routes/voice-models-routes.d.ts.map +1 -0
  65. package/dist/routes/voice-profile-plugin-routes.d.ts +19 -0
  66. package/dist/routes/voice-profile-plugin-routes.d.ts.map +1 -0
  67. package/dist/routes/voice-profiles-management-routes.d.ts +52 -0
  68. package/dist/routes/voice-profiles-management-routes.d.ts.map +1 -0
  69. package/dist/routes/voice-speaker-profile-routes.d.ts +57 -0
  70. package/dist/routes/voice-speaker-profile-routes.d.ts.map +1 -0
  71. package/dist/runtime/embedding-manager-support.d.ts +77 -0
  72. package/dist/runtime/embedding-manager-support.d.ts.map +1 -0
  73. package/dist/runtime/embedding-presets.d.ts +16 -0
  74. package/dist/runtime/embedding-presets.d.ts.map +1 -0
  75. package/dist/runtime/embedding-warmup-policy.d.ts +14 -0
  76. package/dist/runtime/embedding-warmup-policy.d.ts.map +1 -0
  77. package/dist/runtime/ensure-local-inference-handler.d.ts +70 -0
  78. package/dist/runtime/ensure-local-inference-handler.d.ts.map +1 -0
  79. package/dist/runtime/index.d.ts +15 -0
  80. package/dist/runtime/index.d.ts.map +1 -0
  81. package/dist/runtime/index.js +38768 -0
  82. package/dist/runtime/index.js.map +217 -0
  83. package/dist/runtime/mobile-local-inference-gate.d.ts +63 -0
  84. package/dist/runtime/mobile-local-inference-gate.d.ts.map +1 -0
  85. package/dist/runtime/voice-entity-binding.d.ts +113 -0
  86. package/dist/runtime/voice-entity-binding.d.ts.map +1 -0
  87. package/dist/services/active-model.d.ts +310 -0
  88. package/dist/services/active-model.d.ts.map +1 -0
  89. package/dist/services/asr-provenance.d.ts +5 -0
  90. package/dist/services/asr-provenance.d.ts.map +1 -0
  91. package/dist/services/assignments.d.ts +84 -0
  92. package/dist/services/assignments.d.ts.map +1 -0
  93. package/dist/services/backend-selector.d.ts +55 -0
  94. package/dist/services/backend-selector.d.ts.map +1 -0
  95. package/dist/services/backend.d.ts +440 -0
  96. package/dist/services/backend.d.ts.map +1 -0
  97. package/dist/services/bionic-host-loader.d.ts +67 -0
  98. package/dist/services/bionic-host-loader.d.ts.map +1 -0
  99. package/dist/services/bundled-models.d.ts +34 -0
  100. package/dist/services/bundled-models.d.ts.map +1 -0
  101. package/dist/services/cache-bridge.d.ts +206 -0
  102. package/dist/services/cache-bridge.d.ts.map +1 -0
  103. package/dist/services/catalog.d.ts +10 -0
  104. package/dist/services/catalog.d.ts.map +1 -0
  105. package/dist/services/checkpoint-client.d.ts +109 -0
  106. package/dist/services/checkpoint-client.d.ts.map +1 -0
  107. package/dist/services/checkpoint-manager.d.ts +217 -0
  108. package/dist/services/checkpoint-manager.d.ts.map +1 -0
  109. package/dist/services/cloud-fallback.d.ts +102 -0
  110. package/dist/services/cloud-fallback.d.ts.map +1 -0
  111. package/dist/services/context-fit.d.ts +36 -0
  112. package/dist/services/context-fit.d.ts.map +1 -0
  113. package/dist/services/conversation-registry.d.ts +142 -0
  114. package/dist/services/conversation-registry.d.ts.map +1 -0
  115. package/dist/services/desktop-fused-ffi-backend-runtime.d.ts +111 -0
  116. package/dist/services/desktop-fused-ffi-backend-runtime.d.ts.map +1 -0
  117. package/dist/services/device-bridge.d.ts +188 -0
  118. package/dist/services/device-bridge.d.ts.map +1 -0
  119. package/dist/services/device-resource-metrics.d.ts +149 -0
  120. package/dist/services/device-resource-metrics.d.ts.map +1 -0
  121. package/dist/services/device-tier.d.ts +133 -0
  122. package/dist/services/device-tier.d.ts.map +1 -0
  123. package/dist/services/downloader.d.ts +94 -0
  124. package/dist/services/downloader.d.ts.map +1 -0
  125. package/dist/services/engine.d.ts +579 -0
  126. package/dist/services/engine.d.ts.map +1 -0
  127. package/dist/services/ensure-local-artifacts.d.ts +82 -0
  128. package/dist/services/ensure-local-artifacts.d.ts.map +1 -0
  129. package/dist/services/external-scanner.d.ts +17 -0
  130. package/dist/services/external-scanner.d.ts.map +1 -0
  131. package/dist/services/ffi-llm-mock.d.ts +90 -0
  132. package/dist/services/ffi-llm-mock.d.ts.map +1 -0
  133. package/dist/services/ffi-llm-streaming-abi.d.ts +318 -0
  134. package/dist/services/ffi-llm-streaming-abi.d.ts.map +1 -0
  135. package/dist/services/ffi-streaming-backend.d.ts +201 -0
  136. package/dist/services/ffi-streaming-backend.d.ts.map +1 -0
  137. package/dist/services/ffi-streaming-runner.d.ts +146 -0
  138. package/dist/services/ffi-streaming-runner.d.ts.map +1 -0
  139. package/dist/services/gpu-autotune.d.ts +150 -0
  140. package/dist/services/gpu-autotune.d.ts.map +1 -0
  141. package/dist/services/gpu-detect.d.ts +56 -0
  142. package/dist/services/gpu-detect.d.ts.map +1 -0
  143. package/dist/services/handler-registry.d.ts +72 -0
  144. package/dist/services/handler-registry.d.ts.map +1 -0
  145. package/dist/services/hardware.d.ts +63 -0
  146. package/dist/services/hardware.d.ts.map +1 -0
  147. package/dist/services/image-description-runtime.d.ts +14 -0
  148. package/dist/services/image-description-runtime.d.ts.map +1 -0
  149. package/dist/services/imagegen/aosp-unavailable.d.ts +134 -0
  150. package/dist/services/imagegen/aosp-unavailable.d.ts.map +1 -0
  151. package/dist/services/imagegen/backend-selector.d.ts +118 -0
  152. package/dist/services/imagegen/backend-selector.d.ts.map +1 -0
  153. package/dist/services/imagegen/coreml-unavailable.d.ts +105 -0
  154. package/dist/services/imagegen/coreml-unavailable.d.ts.map +1 -0
  155. package/dist/services/imagegen/errors.d.ts +16 -0
  156. package/dist/services/imagegen/errors.d.ts.map +1 -0
  157. package/dist/services/imagegen/index.d.ts +58 -0
  158. package/dist/services/imagegen/index.d.ts.map +1 -0
  159. package/dist/services/imagegen/mflux.d.ts +74 -0
  160. package/dist/services/imagegen/mflux.d.ts.map +1 -0
  161. package/dist/services/imagegen/sd-cpp.d.ts +181 -0
  162. package/dist/services/imagegen/sd-cpp.d.ts.map +1 -0
  163. package/dist/services/imagegen/tensorrt-unavailable.d.ts +83 -0
  164. package/dist/services/imagegen/tensorrt-unavailable.d.ts.map +1 -0
  165. package/dist/services/imagegen/types.d.ts +181 -0
  166. package/dist/services/imagegen/types.d.ts.map +1 -0
  167. package/dist/services/index.d.ts +31 -0
  168. package/dist/services/index.d.ts.map +1 -0
  169. package/dist/services/index.js +39453 -0
  170. package/dist/services/index.js.map +227 -0
  171. package/dist/services/inference-capabilities.d.ts +132 -0
  172. package/dist/services/inference-capabilities.d.ts.map +1 -0
  173. package/dist/services/inference-telemetry.d.ts +59 -0
  174. package/dist/services/inference-telemetry.d.ts.map +1 -0
  175. package/dist/services/ios-llama-streaming.d.ts +119 -0
  176. package/dist/services/ios-llama-streaming.d.ts.map +1 -0
  177. package/dist/services/kv-spill.d.ts +189 -0
  178. package/dist/services/kv-spill.d.ts.map +1 -0
  179. package/dist/services/latency-trace.d.ts +346 -0
  180. package/dist/services/latency-trace.d.ts.map +1 -0
  181. package/dist/services/lib-target.d.ts +55 -0
  182. package/dist/services/lib-target.d.ts.map +1 -0
  183. package/dist/services/live-signals.d.ts +86 -0
  184. package/dist/services/live-signals.d.ts.map +1 -0
  185. package/dist/services/llama-server-metrics.d.ts +114 -0
  186. package/dist/services/llama-server-metrics.d.ts.map +1 -0
  187. package/dist/services/llm-streaming-binding.d.ts +96 -0
  188. package/dist/services/llm-streaming-binding.d.ts.map +1 -0
  189. package/dist/services/load-args.d.ts +82 -0
  190. package/dist/services/load-args.d.ts.map +1 -0
  191. package/dist/services/manifest/index.d.ts +4 -0
  192. package/dist/services/manifest/index.d.ts.map +1 -0
  193. package/dist/services/manifest/schema.d.ts +903 -0
  194. package/dist/services/manifest/schema.d.ts.map +1 -0
  195. package/dist/services/manifest/types.d.ts +32 -0
  196. package/dist/services/manifest/types.d.ts.map +1 -0
  197. package/dist/services/manifest/validator.d.ts +66 -0
  198. package/dist/services/manifest/validator.d.ts.map +1 -0
  199. package/dist/services/memory-arbiter.d.ts +348 -0
  200. package/dist/services/memory-arbiter.d.ts.map +1 -0
  201. package/dist/services/memory-benchmark.d.ts +76 -0
  202. package/dist/services/memory-benchmark.d.ts.map +1 -0
  203. package/dist/services/memory-monitor.d.ts +128 -0
  204. package/dist/services/memory-monitor.d.ts.map +1 -0
  205. package/dist/services/memory-pressure.d.ts +130 -0
  206. package/dist/services/memory-pressure.d.ts.map +1 -0
  207. package/dist/services/mtp-doctor.d.ts +13 -0
  208. package/dist/services/mtp-doctor.d.ts.map +1 -0
  209. package/dist/services/network-policy.d.ts +127 -0
  210. package/dist/services/network-policy.d.ts.map +1 -0
  211. package/dist/services/paths.d.ts +6 -0
  212. package/dist/services/paths.d.ts.map +1 -0
  213. package/dist/services/planner-skeleton.d.ts +124 -0
  214. package/dist/services/planner-skeleton.d.ts.map +1 -0
  215. package/dist/services/providers.d.ts +38 -0
  216. package/dist/services/providers.d.ts.map +1 -0
  217. package/dist/services/ram-budget.d.ts +110 -0
  218. package/dist/services/ram-budget.d.ts.map +1 -0
  219. package/dist/services/readiness.d.ts +9 -0
  220. package/dist/services/readiness.d.ts.map +1 -0
  221. package/dist/services/recommendation.d.ts +111 -0
  222. package/dist/services/recommendation.d.ts.map +1 -0
  223. package/dist/services/registry.d.ts +33 -0
  224. package/dist/services/registry.d.ts.map +1 -0
  225. package/dist/services/router-handler.d.ts +92 -0
  226. package/dist/services/router-handler.d.ts.map +1 -0
  227. package/dist/services/routing-policy.d.ts +92 -0
  228. package/dist/services/routing-policy.d.ts.map +1 -0
  229. package/dist/services/routing-preferences.d.ts +8 -0
  230. package/dist/services/routing-preferences.d.ts.map +1 -0
  231. package/dist/services/runtime-target.d.ts +98 -0
  232. package/dist/services/runtime-target.d.ts.map +1 -0
  233. package/dist/services/service.d.ts +128 -0
  234. package/dist/services/service.d.ts.map +1 -0
  235. package/dist/services/session-pool.d.ts +72 -0
  236. package/dist/services/session-pool.d.ts.map +1 -0
  237. package/dist/services/structured-output/deterministic-repair.d.ts +23 -0
  238. package/dist/services/structured-output/deterministic-repair.d.ts.map +1 -0
  239. package/dist/services/structured-output/index.d.ts +2 -0
  240. package/dist/services/structured-output/index.d.ts.map +1 -0
  241. package/dist/services/structured-output.d.ts +311 -0
  242. package/dist/services/structured-output.d.ts.map +1 -0
  243. package/dist/services/system-memory.d.ts +33 -0
  244. package/dist/services/system-memory.d.ts.map +1 -0
  245. package/dist/services/types.d.ts +19 -0
  246. package/dist/services/types.d.ts.map +1 -0
  247. package/dist/services/verify-on-device.d.ts +34 -0
  248. package/dist/services/verify-on-device.d.ts.map +1 -0
  249. package/dist/services/verify.d.ts +8 -0
  250. package/dist/services/verify.d.ts.map +1 -0
  251. package/dist/services/vision/aosp-unavailable.d.ts +115 -0
  252. package/dist/services/vision/aosp-unavailable.d.ts.map +1 -0
  253. package/dist/services/vision/capacitor-llama.d.ts +99 -0
  254. package/dist/services/vision/capacitor-llama.d.ts.map +1 -0
  255. package/dist/services/vision/cloud-fallback.d.ts +47 -0
  256. package/dist/services/vision/cloud-fallback.d.ts.map +1 -0
  257. package/dist/services/vision/hash.d.ts +71 -0
  258. package/dist/services/vision/hash.d.ts.map +1 -0
  259. package/dist/services/vision/index.d.ts +95 -0
  260. package/dist/services/vision/index.d.ts.map +1 -0
  261. package/dist/services/vision/llama-server.d.ts +73 -0
  262. package/dist/services/vision/llama-server.d.ts.map +1 -0
  263. package/dist/services/vision/types.d.ts +162 -0
  264. package/dist/services/vision/types.d.ts.map +1 -0
  265. package/dist/services/vision/vast-fallback.d.ts +18 -0
  266. package/dist/services/vision/vast-fallback.d.ts.map +1 -0
  267. package/dist/services/vision-embedding-cache.d.ts +98 -0
  268. package/dist/services/vision-embedding-cache.d.ts.map +1 -0
  269. package/dist/services/voice/__test-helpers__/fake-ffi.d.ts +27 -0
  270. package/dist/services/voice/__test-helpers__/fake-ffi.d.ts.map +1 -0
  271. package/dist/services/voice/__test-helpers__/synthetic-speech.d.ts +66 -0
  272. package/dist/services/voice/__test-helpers__/synthetic-speech.d.ts.map +1 -0
  273. package/dist/services/voice/acoustic-speaker-attribution.d.ts +61 -0
  274. package/dist/services/voice/acoustic-speaker-attribution.d.ts.map +1 -0
  275. package/dist/services/voice/audio-frame-consumer.d.ts +294 -0
  276. package/dist/services/voice/audio-frame-consumer.d.ts.map +1 -0
  277. package/dist/services/voice/barge-in.d.ts +112 -0
  278. package/dist/services/voice/barge-in.d.ts.map +1 -0
  279. package/dist/services/voice/cancellation-coordinator.d.ts +127 -0
  280. package/dist/services/voice/cancellation-coordinator.d.ts.map +1 -0
  281. package/dist/services/voice/checkpoint-manager.d.ts +199 -0
  282. package/dist/services/voice/checkpoint-manager.d.ts.map +1 -0
  283. package/dist/services/voice/checkpoint-policy.d.ts +178 -0
  284. package/dist/services/voice/checkpoint-policy.d.ts.map +1 -0
  285. package/dist/services/voice/corpus-augment.d.ts +111 -0
  286. package/dist/services/voice/corpus-augment.d.ts.map +1 -0
  287. package/dist/services/voice/corpus-generator.d.ts +134 -0
  288. package/dist/services/voice/corpus-generator.d.ts.map +1 -0
  289. package/dist/services/voice/diarization-error-rate.d.ts +40 -0
  290. package/dist/services/voice/diarization-error-rate.d.ts.map +1 -0
  291. package/dist/services/voice/e2e-harness.d.ts +297 -0
  292. package/dist/services/voice/e2e-harness.d.ts.map +1 -0
  293. package/dist/services/voice/eager-context-builder.d.ts +170 -0
  294. package/dist/services/voice/eager-context-builder.d.ts.map +1 -0
  295. package/dist/services/voice/echo-delay.d.ts +67 -0
  296. package/dist/services/voice/echo-delay.d.ts.map +1 -0
  297. package/dist/services/voice/echo-metrics.d.ts +7 -0
  298. package/dist/services/voice/echo-metrics.d.ts.map +1 -0
  299. package/dist/services/voice/echo-reference-buffer.d.ts +65 -0
  300. package/dist/services/voice/echo-reference-buffer.d.ts.map +1 -0
  301. package/dist/services/voice/eliza1-eot-scorer.d.ts +124 -0
  302. package/dist/services/voice/eliza1-eot-scorer.d.ts.map +1 -0
  303. package/dist/services/voice/embedding-server.d.ts +37 -0
  304. package/dist/services/voice/embedding-server.d.ts.map +1 -0
  305. package/dist/services/voice/embedding.d.ts +132 -0
  306. package/dist/services/voice/embedding.d.ts.map +1 -0
  307. package/dist/services/voice/emotion-attribution.d.ts +68 -0
  308. package/dist/services/voice/emotion-attribution.d.ts.map +1 -0
  309. package/dist/services/voice/engine-bridge.d.ts +762 -0
  310. package/dist/services/voice/engine-bridge.d.ts.map +1 -0
  311. package/dist/services/voice/eot-classifier-ggml.d.ts +179 -0
  312. package/dist/services/voice/eot-classifier-ggml.d.ts.map +1 -0
  313. package/dist/services/voice/eot-classifier.d.ts +211 -0
  314. package/dist/services/voice/eot-classifier.d.ts.map +1 -0
  315. package/dist/services/voice/errors.d.ts +20 -0
  316. package/dist/services/voice/errors.d.ts.map +1 -0
  317. package/dist/services/voice/expressive-tags.d.ts +158 -0
  318. package/dist/services/voice/expressive-tags.d.ts.map +1 -0
  319. package/dist/services/voice/ffi-bindings.d.ts +696 -0
  320. package/dist/services/voice/ffi-bindings.d.ts.map +1 -0
  321. package/dist/services/voice/first-line-cache.d.ts +181 -0
  322. package/dist/services/voice/first-line-cache.d.ts.map +1 -0
  323. package/dist/services/voice/fused-eot-scorer.d.ts +51 -0
  324. package/dist/services/voice/fused-eot-scorer.d.ts.map +1 -0
  325. package/dist/services/voice/index.d.ts +96 -0
  326. package/dist/services/voice/index.d.ts.map +1 -0
  327. package/dist/services/voice/kokoro/index.d.ts +24 -0
  328. package/dist/services/voice/kokoro/index.d.ts.map +1 -0
  329. package/dist/services/voice/kokoro/kokoro-backend.d.ts +87 -0
  330. package/dist/services/voice/kokoro/kokoro-backend.d.ts.map +1 -0
  331. package/dist/services/voice/kokoro/kokoro-engine-discovery.d.ts +58 -0
  332. package/dist/services/voice/kokoro/kokoro-engine-discovery.d.ts.map +1 -0
  333. package/dist/services/voice/kokoro/kokoro-ffi-runtime.d.ts +75 -0
  334. package/dist/services/voice/kokoro/kokoro-ffi-runtime.d.ts.map +1 -0
  335. package/dist/services/voice/kokoro/kokoro-runtime.d.ts +100 -0
  336. package/dist/services/voice/kokoro/kokoro-runtime.d.ts.map +1 -0
  337. package/dist/services/voice/kokoro/phoneme-stream.d.ts +51 -0
  338. package/dist/services/voice/kokoro/phoneme-stream.d.ts.map +1 -0
  339. package/dist/services/voice/kokoro/phonemizer.d.ts +50 -0
  340. package/dist/services/voice/kokoro/phonemizer.d.ts.map +1 -0
  341. package/dist/services/voice/kokoro/pick-runtime.d.ts +61 -0
  342. package/dist/services/voice/kokoro/pick-runtime.d.ts.map +1 -0
  343. package/dist/services/voice/kokoro/runtime-selection.d.ts +31 -0
  344. package/dist/services/voice/kokoro/runtime-selection.d.ts.map +1 -0
  345. package/dist/services/voice/kokoro/types.d.ts +82 -0
  346. package/dist/services/voice/kokoro/types.d.ts.map +1 -0
  347. package/dist/services/voice/kokoro/voice-presets.d.ts +23 -0
  348. package/dist/services/voice/kokoro/voice-presets.d.ts.map +1 -0
  349. package/dist/services/voice/kokoro/voices.d.ts +30 -0
  350. package/dist/services/voice/kokoro/voices.d.ts.map +1 -0
  351. package/dist/services/voice/lifecycle.d.ts +135 -0
  352. package/dist/services/voice/lifecycle.d.ts.map +1 -0
  353. package/dist/services/voice/live-diarization-session.d.ts +196 -0
  354. package/dist/services/voice/live-diarization-session.d.ts.map +1 -0
  355. package/dist/services/voice/metric-math.d.ts +10 -0
  356. package/dist/services/voice/metric-math.d.ts.map +1 -0
  357. package/dist/services/voice/mic-source.d.ts +136 -0
  358. package/dist/services/voice/mic-source.d.ts.map +1 -0
  359. package/dist/services/voice/nlms-echo-canceller.d.ts +137 -0
  360. package/dist/services/voice/nlms-echo-canceller.d.ts.map +1 -0
  361. package/dist/services/voice/optimistic-policy.d.ts +109 -0
  362. package/dist/services/voice/optimistic-policy.d.ts.map +1 -0
  363. package/dist/services/voice/optimistic-rollback.d.ts +151 -0
  364. package/dist/services/voice/optimistic-rollback.d.ts.map +1 -0
  365. package/dist/services/voice/partial-stabilizer.d.ts +73 -0
  366. package/dist/services/voice/partial-stabilizer.d.ts.map +1 -0
  367. package/dist/services/voice/phoneme-tokenizer.d.ts +49 -0
  368. package/dist/services/voice/phoneme-tokenizer.d.ts.map +1 -0
  369. package/dist/services/voice/phrase-cache.d.ts +76 -0
  370. package/dist/services/voice/phrase-cache.d.ts.map +1 -0
  371. package/dist/services/voice/phrase-chunker.d.ts +62 -0
  372. package/dist/services/voice/phrase-chunker.d.ts.map +1 -0
  373. package/dist/services/voice/pipeline-impls.d.ts +151 -0
  374. package/dist/services/voice/pipeline-impls.d.ts.map +1 -0
  375. package/dist/services/voice/pipeline.d.ts +216 -0
  376. package/dist/services/voice/pipeline.d.ts.map +1 -0
  377. package/dist/services/voice/prefill-client.d.ts +123 -0
  378. package/dist/services/voice/prefill-client.d.ts.map +1 -0
  379. package/dist/services/voice/prefix-preserving-queue.d.ts +113 -0
  380. package/dist/services/voice/prefix-preserving-queue.d.ts.map +1 -0
  381. package/dist/services/voice/profile-store.d.ts +248 -0
  382. package/dist/services/voice/profile-store.d.ts.map +1 -0
  383. package/dist/services/voice/ring-buffer.d.ts +40 -0
  384. package/dist/services/voice/ring-buffer.d.ts.map +1 -0
  385. package/dist/services/voice/rollback-queue.d.ts +24 -0
  386. package/dist/services/voice/rollback-queue.d.ts.map +1 -0
  387. package/dist/services/voice/samantha-preset-placeholder.d.ts +67 -0
  388. package/dist/services/voice/samantha-preset-placeholder.d.ts.map +1 -0
  389. package/dist/services/voice/samantha-preset-regenerator.d.ts +87 -0
  390. package/dist/services/voice/samantha-preset-regenerator.d.ts.map +1 -0
  391. package/dist/services/voice/scheduler.d.ts +146 -0
  392. package/dist/services/voice/scheduler.d.ts.map +1 -0
  393. package/dist/services/voice/self-voice-imprint.d.ts +33 -0
  394. package/dist/services/voice/self-voice-imprint.d.ts.map +1 -0
  395. package/dist/services/voice/shared-resources.d.ts +204 -0
  396. package/dist/services/voice/shared-resources.d.ts.map +1 -0
  397. package/dist/services/voice/speaker/attribution-pipeline.d.ts +74 -0
  398. package/dist/services/voice/speaker/attribution-pipeline.d.ts.map +1 -0
  399. package/dist/services/voice/speaker/diarizer-fused.d.ts +59 -0
  400. package/dist/services/voice/speaker/diarizer-fused.d.ts.map +1 -0
  401. package/dist/services/voice/speaker/diarizer.d.ts +75 -0
  402. package/dist/services/voice/speaker/diarizer.d.ts.map +1 -0
  403. package/dist/services/voice/speaker/encoder-fused.d.ts +60 -0
  404. package/dist/services/voice/speaker/encoder-fused.d.ts.map +1 -0
  405. package/dist/services/voice/speaker/encoder-ggml.d.ts +33 -0
  406. package/dist/services/voice/speaker/encoder-ggml.d.ts.map +1 -0
  407. package/dist/services/voice/speaker/encoder.d.ts +37 -0
  408. package/dist/services/voice/speaker/encoder.d.ts.map +1 -0
  409. package/dist/services/voice/speaker-imprint.d.ts +83 -0
  410. package/dist/services/voice/speaker-imprint.d.ts.map +1 -0
  411. package/dist/services/voice/speaker-preset-cache.d.ts +77 -0
  412. package/dist/services/voice/speaker-preset-cache.d.ts.map +1 -0
  413. package/dist/services/voice/streaming-asr/streaming-pipeline-adapter.d.ts +160 -0
  414. package/dist/services/voice/streaming-asr/streaming-pipeline-adapter.d.ts.map +1 -0
  415. package/dist/services/voice/system-audio-sink.d.ts +73 -0
  416. package/dist/services/voice/system-audio-sink.d.ts.map +1 -0
  417. package/dist/services/voice/transcriber.d.ts +244 -0
  418. package/dist/services/voice/transcriber.d.ts.map +1 -0
  419. package/dist/services/voice/transcript-knowledge.d.ts +37 -0
  420. package/dist/services/voice/transcript-knowledge.d.ts.map +1 -0
  421. package/dist/services/voice/transcript-service.d.ts +60 -0
  422. package/dist/services/voice/transcript-service.d.ts.map +1 -0
  423. package/dist/services/voice/transcript-store.d.ts +64 -0
  424. package/dist/services/voice/transcript-store.d.ts.map +1 -0
  425. package/dist/services/voice/turn-controller.d.ts +183 -0
  426. package/dist/services/voice/turn-controller.d.ts.map +1 -0
  427. package/dist/services/voice/types.d.ts +643 -0
  428. package/dist/services/voice/types.d.ts.map +1 -0
  429. package/dist/services/voice/vad.d.ts +283 -0
  430. package/dist/services/voice/vad.d.ts.map +1 -0
  431. package/dist/services/voice/voice-budget.d.ts +241 -0
  432. package/dist/services/voice/voice-budget.d.ts.map +1 -0
  433. package/dist/services/voice/voice-emotion-classifier.d.ts +95 -0
  434. package/dist/services/voice/voice-emotion-classifier.d.ts.map +1 -0
  435. package/dist/services/voice/voice-preload-predictor.d.ts +76 -0
  436. package/dist/services/voice/voice-preload-predictor.d.ts.map +1 -0
  437. package/dist/services/voice/voice-preset-format.d.ts +158 -0
  438. package/dist/services/voice/voice-preset-format.d.ts.map +1 -0
  439. package/dist/services/voice/voice-profile-artifact.d.ts +116 -0
  440. package/dist/services/voice/voice-profile-artifact.d.ts.map +1 -0
  441. package/dist/services/voice/voice-profile-routes.d.ts +83 -0
  442. package/dist/services/voice/voice-profile-routes.d.ts.map +1 -0
  443. package/dist/services/voice/voice-scenario.d.ts +131 -0
  444. package/dist/services/voice/voice-scenario.d.ts.map +1 -0
  445. package/dist/services/voice/voice-state-machine.d.ts +364 -0
  446. package/dist/services/voice/voice-state-machine.d.ts.map +1 -0
  447. package/dist/services/voice/voice-workbench-report.d.ts +117 -0
  448. package/dist/services/voice/voice-workbench-report.d.ts.map +1 -0
  449. package/dist/services/voice/wake-word-ggml.d.ts +100 -0
  450. package/dist/services/voice/wake-word-ggml.d.ts.map +1 -0
  451. package/dist/services/voice/wake-word.d.ts +255 -0
  452. package/dist/services/voice/wake-word.d.ts.map +1 -0
  453. package/dist/services/voice/wav-codec.d.ts +11 -0
  454. package/dist/services/voice/wav-codec.d.ts.map +1 -0
  455. package/dist/services/voice/workbench-entrypoint.d.ts +42 -0
  456. package/dist/services/voice/workbench-entrypoint.d.ts.map +1 -0
  457. package/dist/services/voice/workbench-headless-runner.d.ts +102 -0
  458. package/dist/services/voice/workbench-headless-runner.d.ts.map +1 -0
  459. package/dist/services/voice/workbench-logic-services.d.ts +36 -0
  460. package/dist/services/voice/workbench-logic-services.d.ts.map +1 -0
  461. package/dist/services/voice/workbench-real-services.d.ts +17 -0
  462. package/dist/services/voice/workbench-real-services.d.ts.map +1 -0
  463. package/dist/services/voice/workbench-scenarios.d.ts +24 -0
  464. package/dist/services/voice/workbench-scenarios.d.ts.map +1 -0
  465. package/dist/services/voice/wrap-with-first-line-cache.d.ts +70 -0
  466. package/dist/services/voice/wrap-with-first-line-cache.d.ts.map +1 -0
  467. package/dist/services/voice-model-updater.d.ts +240 -0
  468. package/dist/services/voice-model-updater.d.ts.map +1 -0
  469. package/dist/services/voice-prewarm.d.ts +3 -0
  470. package/dist/services/voice-prewarm.d.ts.map +1 -0
  471. package/dist/voice-workbench.d.ts +18 -0
  472. package/dist/voice-workbench.d.ts.map +1 -0
  473. package/dist/voice-workbench.js +5259 -0
  474. package/dist/voice-workbench.js.map +34 -0
  475. package/package.json +101 -15
  476. package/registry-entry.json +137 -0
  477. package/src/actions/generate-media.ts +647 -0
  478. package/src/actions/identify-speaker.ts +171 -0
  479. package/src/actions/transcription-control.test.ts +100 -0
  480. package/src/actions/transcription-control.ts +127 -0
  481. package/src/adapters/capacitor-llama/__tests__/compat-behavior.test.ts +218 -0
  482. package/src/adapters/capacitor-llama/__tests__/index.test.ts +68 -0
  483. package/src/adapters/capacitor-llama/__tests__/structured-output.test.ts +215 -0
  484. package/src/adapters/capacitor-llama/__tests__/text-streaming.test.ts +174 -0
  485. package/src/adapters/capacitor-llama/__tests__/voice-turn.test.ts +293 -0
  486. package/src/adapters/capacitor-llama/environment.ts +71 -0
  487. package/src/adapters/capacitor-llama/index.browser.ts +83 -0
  488. package/src/adapters/capacitor-llama/index.ts +831 -0
  489. package/src/adapters/capacitor-llama/loader.ts +109 -0
  490. package/src/adapters/capacitor-llama/native-voice-capture.ts +140 -0
  491. package/src/adapters/capacitor-llama/structured-output.ts +165 -0
  492. package/src/adapters/capacitor-llama/text-streaming.ts +227 -0
  493. package/src/adapters/capacitor-llama/types.ts +374 -0
  494. package/src/adapters/capacitor-llama/voice-turn.ts +178 -0
  495. package/src/backends/apple-foundation.ts +127 -0
  496. package/src/index.ts +62 -0
  497. package/src/local-inference-routes.test.ts +390 -0
  498. package/src/local-inference-routes.ts +1625 -0
  499. package/src/provider.ts +1111 -0
  500. package/src/routes/compat-helpers.ts +275 -0
  501. package/src/routes/family-member-route.ts +353 -0
  502. package/src/routes/index.ts +61 -0
  503. package/src/routes/live-diarization-route.test.ts +347 -0
  504. package/src/routes/live-diarization-route.ts +198 -0
  505. package/src/routes/local-inference-asr-route.test.ts +246 -0
  506. package/src/routes/local-inference-asr-route.ts +166 -0
  507. package/src/routes/local-inference-asr-transcribe.test.ts +118 -0
  508. package/src/routes/local-inference-asr-transcribe.ts +97 -0
  509. package/src/routes/local-inference-compat-routes.test.ts +485 -0
  510. package/src/routes/local-inference-compat-routes.ts +775 -0
  511. package/src/routes/local-inference-tts-route.test.ts +179 -0
  512. package/src/routes/local-inference-tts-route.ts +230 -0
  513. package/src/routes/native-pcm-turn-route.test.ts +136 -0
  514. package/src/routes/native-pcm-turn-route.ts +121 -0
  515. package/src/routes/transcript-audio-store.ts +27 -0
  516. package/src/routes/transcripts-routes.test.ts +195 -0
  517. package/src/routes/transcripts-routes.ts +191 -0
  518. package/src/routes/voice-first-run-routes.ts +524 -0
  519. package/src/routes/voice-models-routes.ts +554 -0
  520. package/src/routes/voice-profile-plugin-routes.ts +138 -0
  521. package/src/routes/voice-profiles-management-routes.ts +476 -0
  522. package/src/routes/voice-speaker-profile-routes.ts +199 -0
  523. package/src/runtime/aosp-llama-loader-selection.test.ts +80 -0
  524. package/src/runtime/bionic-wire-encoding.test.ts +147 -0
  525. package/src/runtime/capacitor-llama.d.ts +25 -0
  526. package/src/runtime/embedding-manager-support.ts +497 -0
  527. package/src/runtime/embedding-presets.ts +81 -0
  528. package/src/runtime/embedding-warmup-policy.test.ts +53 -0
  529. package/src/runtime/embedding-warmup-policy.ts +48 -0
  530. package/src/runtime/ensure-local-inference-handler.test.ts +726 -0
  531. package/src/runtime/ensure-local-inference-handler.ts +1640 -0
  532. package/src/runtime/index.ts +36 -0
  533. package/src/runtime/mobile-local-inference-gate.test.ts +152 -0
  534. package/src/runtime/mobile-local-inference-gate.ts +99 -0
  535. package/src/runtime/voice-entity-binding.transcript.test.ts +98 -0
  536. package/src/runtime/voice-entity-binding.ts +368 -0
  537. package/src/runtime/voice-speaker-entity-contract.test.ts +149 -0
  538. package/src/services/README.md +71 -0
  539. package/src/services/__tests__/backend-selector.precedence.test.ts +333 -0
  540. package/src/services/__tests__/backend-selector.test.ts +101 -0
  541. package/src/services/__tests__/checkpoint-manager.test.ts +376 -0
  542. package/src/services/__tests__/gpu-autotune.test.ts +400 -0
  543. package/src/services/__tests__/llm-streaming-binding.test.ts +85 -0
  544. package/src/services/__tests__/planner-grammar.test.ts +372 -0
  545. package/src/services/__tests__/runtime-target.test.ts +176 -0
  546. package/src/services/active-model-context-fit.test.ts +125 -0
  547. package/src/services/active-model-switch-rollback.test.ts +183 -0
  548. package/src/services/active-model.ts +1416 -0
  549. package/src/services/asr-provenance.ts +68 -0
  550. package/src/services/assignment-validation.test.ts +118 -0
  551. package/src/services/assignments.test.ts +106 -0
  552. package/src/services/assignments.ts +278 -0
  553. package/src/services/backend-selector.ts +95 -0
  554. package/src/services/backend.test.ts +84 -0
  555. package/src/services/backend.ts +791 -0
  556. package/src/services/bionic-host-loader.test.ts +226 -0
  557. package/src/services/bionic-host-loader.ts +252 -0
  558. package/src/services/bundled-models.ts +129 -0
  559. package/src/services/cache-bridge.test.ts +516 -0
  560. package/src/services/cache-bridge.ts +423 -0
  561. package/src/services/catalog.test.ts +259 -0
  562. package/src/services/catalog.ts +33 -0
  563. package/src/services/checkpoint-client.ts +258 -0
  564. package/src/services/checkpoint-manager.ts +474 -0
  565. package/src/services/cloud-fallback.ts +230 -0
  566. package/src/services/context-fit.test.ts +121 -0
  567. package/src/services/context-fit.ts +113 -0
  568. package/src/services/conversation-registry.test.ts +235 -0
  569. package/src/services/conversation-registry.ts +264 -0
  570. package/src/services/desktop-fused-ffi-backend-runtime.ts +431 -0
  571. package/src/services/device-bridge.ts +1237 -0
  572. package/src/services/device-resource-metrics.test.ts +98 -0
  573. package/src/services/device-resource-metrics.ts +346 -0
  574. package/src/services/device-tier.test.ts +458 -0
  575. package/src/services/device-tier.ts +502 -0
  576. package/src/services/downloader.test.ts +888 -0
  577. package/src/services/downloader.ts +1039 -0
  578. package/src/services/engine-direct-bundle.test.ts +90 -0
  579. package/src/services/engine-streaming.test.ts +80 -0
  580. package/src/services/engine.ts +2096 -0
  581. package/src/services/ensure-local-artifacts.integration.test.ts +273 -0
  582. package/src/services/ensure-local-artifacts.test.ts +368 -0
  583. package/src/services/ensure-local-artifacts.ts +351 -0
  584. package/src/services/external-scanner.ts +312 -0
  585. package/src/services/ffi-llm-mock.ts +354 -0
  586. package/src/services/ffi-llm-streaming-abi.ts +445 -0
  587. package/src/services/ffi-streaming-backend.ts +418 -0
  588. package/src/services/ffi-streaming-runner.test.ts +220 -0
  589. package/src/services/ffi-streaming-runner.ts +407 -0
  590. package/src/services/ffi-unload-ordering.test.ts +166 -0
  591. package/src/services/fused-eliza1-no-regression.test.ts +144 -0
  592. package/src/services/gpu-autotune.ts +534 -0
  593. package/src/services/gpu-detect.ts +139 -0
  594. package/src/services/handler-registry.ts +240 -0
  595. package/src/services/hardware.test.ts +236 -0
  596. package/src/services/hardware.ts +438 -0
  597. package/src/services/image-description-runtime.test.ts +61 -0
  598. package/src/services/image-description-runtime.ts +118 -0
  599. package/src/services/imagegen/aosp-unavailable.ts +229 -0
  600. package/src/services/imagegen/backend-selector.test.ts +190 -0
  601. package/src/services/imagegen/backend-selector.ts +277 -0
  602. package/src/services/imagegen/coreml-unavailable.ts +237 -0
  603. package/src/services/imagegen/errors.ts +40 -0
  604. package/src/services/imagegen/index.ts +144 -0
  605. package/src/services/imagegen/mflux.ts +313 -0
  606. package/src/services/imagegen/sd-cpp.ts +715 -0
  607. package/src/services/imagegen/tensorrt-unavailable.ts +295 -0
  608. package/src/services/imagegen/types.ts +193 -0
  609. package/src/services/index.ts +229 -0
  610. package/src/services/inference-capabilities.test.ts +75 -0
  611. package/src/services/inference-capabilities.ts +204 -0
  612. package/src/services/inference-telemetry.ts +143 -0
  613. package/src/services/ios-llama-streaming.ts +248 -0
  614. package/src/services/kv-spill.test.ts +222 -0
  615. package/src/services/kv-spill.ts +357 -0
  616. package/src/services/latency-trace.test.ts +266 -0
  617. package/src/services/latency-trace.ts +844 -0
  618. package/src/services/lib-target.test.ts +145 -0
  619. package/src/services/lib-target.ts +102 -0
  620. package/src/services/live-signals.test.ts +132 -0
  621. package/src/services/live-signals.ts +177 -0
  622. package/src/services/llama-server-metrics.test.ts +168 -0
  623. package/src/services/llama-server-metrics.ts +304 -0
  624. package/src/services/llm-streaming-binding.ts +136 -0
  625. package/src/services/load-args.ts +81 -0
  626. package/src/services/manifest/eliza-1.manifest.v1.json +790 -0
  627. package/src/services/manifest/index.ts +72 -0
  628. package/src/services/manifest/manifest.test.ts +791 -0
  629. package/src/services/manifest/schema.ts +761 -0
  630. package/src/services/manifest/types.ts +61 -0
  631. package/src/services/manifest/validator.ts +633 -0
  632. package/src/services/memory-arbiter.test.ts +558 -0
  633. package/src/services/memory-arbiter.ts +991 -0
  634. package/src/services/memory-benchmark.test.ts +91 -0
  635. package/src/services/memory-benchmark.ts +354 -0
  636. package/src/services/memory-monitor.test.ts +232 -0
  637. package/src/services/memory-monitor.ts +309 -0
  638. package/src/services/memory-pressure.ts +414 -0
  639. package/src/services/mtp-doctor.ts +86 -0
  640. package/src/services/network-policy.ts +346 -0
  641. package/src/services/paths.ts +25 -0
  642. package/src/services/planner-skeleton.ts +175 -0
  643. package/src/services/providers.ts +507 -0
  644. package/src/services/ram-budget-cache.test.ts +164 -0
  645. package/src/services/ram-budget.ts +309 -0
  646. package/src/services/readiness.test.ts +87 -0
  647. package/src/services/readiness.ts +238 -0
  648. package/src/services/recommendation.test.ts +216 -0
  649. package/src/services/recommendation.ts +671 -0
  650. package/src/services/registry.ts +157 -0
  651. package/src/services/required-kernels-gate.test.ts +64 -0
  652. package/src/services/router-handler.test.ts +45 -0
  653. package/src/services/router-handler.ts +426 -0
  654. package/src/services/routing-policy.test.ts +352 -0
  655. package/src/services/routing-policy.ts +367 -0
  656. package/src/services/routing-preferences.ts +17 -0
  657. package/src/services/runtime-target.ts +154 -0
  658. package/src/services/service.test.ts +223 -0
  659. package/src/services/service.ts +750 -0
  660. package/src/services/session-pool.ts +153 -0
  661. package/src/services/structured-output/deterministic-repair.test.ts +169 -0
  662. package/src/services/structured-output/deterministic-repair.ts +443 -0
  663. package/src/services/structured-output/index.ts +4 -0
  664. package/src/services/structured-output.test.ts +483 -0
  665. package/src/services/structured-output.ts +712 -0
  666. package/src/services/system-memory.test.ts +47 -0
  667. package/src/services/system-memory.ts +67 -0
  668. package/src/services/transcription-priority.test.ts +211 -0
  669. package/src/services/types.ts +59 -0
  670. package/src/services/verify-on-device.test.ts +87 -0
  671. package/src/services/verify-on-device.ts +127 -0
  672. package/src/services/verify.ts +13 -0
  673. package/src/services/vision/aosp-unavailable.ts +163 -0
  674. package/src/services/vision/capacitor-llama.ts +255 -0
  675. package/src/services/vision/cloud-fallback.test.ts +243 -0
  676. package/src/services/vision/cloud-fallback.ts +268 -0
  677. package/src/services/vision/fallback-chain.test.ts +86 -0
  678. package/src/services/vision/hash.ts +157 -0
  679. package/src/services/vision/index.ts +251 -0
  680. package/src/services/vision/llama-server.ts +177 -0
  681. package/src/services/vision/types.ts +163 -0
  682. package/src/services/vision/vast-fallback.ts +127 -0
  683. package/src/services/vision-embedding-cache.ts +189 -0
  684. package/src/services/voice/VOICE_WORKBENCH.md +133 -0
  685. package/src/services/voice/__fixtures__/voice-workbench-logic-baseline.json +180 -0
  686. package/src/services/voice/__test-helpers__/fake-ffi.ts +94 -0
  687. package/src/services/voice/__test-helpers__/synthetic-speech.ts +194 -0
  688. package/src/services/voice/__tests__/checkpoint-manager.test.ts +241 -0
  689. package/src/services/voice/__tests__/checkpoint-policy.test.ts +270 -0
  690. package/src/services/voice/__tests__/eager-context-builder.test.ts +257 -0
  691. package/src/services/voice/__tests__/eliza1-eot-scorer.test.ts +288 -0
  692. package/src/services/voice/__tests__/eot-classifier.test.ts +431 -0
  693. package/src/services/voice/__tests__/optimistic-rollback.test.ts +312 -0
  694. package/src/services/voice/__tests__/prefill-client.test.ts +266 -0
  695. package/src/services/voice/__tests__/prefix-preserving-queue.test.ts +208 -0
  696. package/src/services/voice/__tests__/streaming-asr.test.ts +450 -0
  697. package/src/services/voice/__tests__/streaming-transcriber.test.ts +339 -0
  698. package/src/services/voice/__tests__/turn-detector-resolver.test.ts +195 -0
  699. package/src/services/voice/__tests__/voice-state-machine-prefill.test.ts +275 -0
  700. package/src/services/voice/__tests__/voice-state-machine.test.ts +354 -0
  701. package/src/services/voice/acoustic-speaker-attribution.test.ts +165 -0
  702. package/src/services/voice/acoustic-speaker-attribution.ts +336 -0
  703. package/src/services/voice/asr-timed.real.test.ts +139 -0
  704. package/src/services/voice/audio-frame-consumer.test.ts +669 -0
  705. package/src/services/voice/audio-frame-consumer.ts +651 -0
  706. package/src/services/voice/barge-in.test.ts +244 -0
  707. package/src/services/voice/barge-in.ts +335 -0
  708. package/src/services/voice/cancellation-coordinator.test.ts +196 -0
  709. package/src/services/voice/cancellation-coordinator.ts +269 -0
  710. package/src/services/voice/checkpoint-manager.ts +401 -0
  711. package/src/services/voice/checkpoint-policy.ts +336 -0
  712. package/src/services/voice/composite-eot-classifier.test.ts +59 -0
  713. package/src/services/voice/corpus-augment.test.ts +276 -0
  714. package/src/services/voice/corpus-augment.ts +451 -0
  715. package/src/services/voice/corpus-generator.test.ts +201 -0
  716. package/src/services/voice/corpus-generator.ts +413 -0
  717. package/src/services/voice/diarization-error-rate.greedy.test.ts +140 -0
  718. package/src/services/voice/diarization-error-rate.test.ts +100 -0
  719. package/src/services/voice/diarization-error-rate.ts +249 -0
  720. package/src/services/voice/e2e-harness.der.test.ts +94 -0
  721. package/src/services/voice/e2e-harness.respond-eot-entity.test.ts +277 -0
  722. package/src/services/voice/e2e-harness.security-echo.test.ts +103 -0
  723. package/src/services/voice/e2e-harness.test.ts +182 -0
  724. package/src/services/voice/e2e-harness.ts +902 -0
  725. package/src/services/voice/eager-context-builder.ts +262 -0
  726. package/src/services/voice/echo-delay.test.ts +118 -0
  727. package/src/services/voice/echo-delay.ts +135 -0
  728. package/src/services/voice/echo-metrics.test.ts +17 -0
  729. package/src/services/voice/echo-metrics.ts +20 -0
  730. package/src/services/voice/echo-reference-buffer.test.ts +86 -0
  731. package/src/services/voice/echo-reference-buffer.ts +165 -0
  732. package/src/services/voice/eliza1-eot-scorer.ts +242 -0
  733. package/src/services/voice/embedding-server.ts +200 -0
  734. package/src/services/voice/embedding.test.ts +131 -0
  735. package/src/services/voice/embedding.ts +242 -0
  736. package/src/services/voice/emotion-attribution.test.ts +129 -0
  737. package/src/services/voice/emotion-attribution.ts +361 -0
  738. package/src/services/voice/engine-bridge-cancellation.test.ts +422 -0
  739. package/src/services/voice/engine-bridge-transcript-join.test.ts +278 -0
  740. package/src/services/voice/engine-bridge.test.ts +384 -0
  741. package/src/services/voice/engine-bridge.ts +2343 -0
  742. package/src/services/voice/eot-classifier-ggml.ts +569 -0
  743. package/src/services/voice/eot-classifier.test.ts +98 -0
  744. package/src/services/voice/eot-classifier.ts +422 -0
  745. package/src/services/voice/errors.ts +34 -0
  746. package/src/services/voice/expressive-tags.asr.test.ts +77 -0
  747. package/src/services/voice/expressive-tags.test.ts +102 -0
  748. package/src/services/voice/expressive-tags.ts +405 -0
  749. package/src/services/voice/ffi-bindings.test.ts +735 -0
  750. package/src/services/voice/ffi-bindings.ts +3387 -0
  751. package/src/services/voice/first-line-cache.ts +725 -0
  752. package/src/services/voice/fused-eot-scorer.ts +139 -0
  753. package/src/services/voice/index.ts +502 -0
  754. package/src/services/voice/kokoro/__tests__/kokoro-backend.test.ts +262 -0
  755. package/src/services/voice/kokoro/__tests__/kokoro-engine-bridge.real.test.ts +236 -0
  756. package/src/services/voice/kokoro/__tests__/kokoro-engine-bridge.test.ts +60 -0
  757. package/src/services/voice/kokoro/__tests__/kokoro-engine-discovery.test.ts +277 -0
  758. package/src/services/voice/kokoro/__tests__/kokoro-ffi-runtime.test.ts +235 -0
  759. package/src/services/voice/kokoro/__tests__/kokoro-runtime.test.ts +95 -0
  760. package/src/services/voice/kokoro/__tests__/phonemizer.test.ts +53 -0
  761. package/src/services/voice/kokoro/__tests__/runtime-selection.test.ts +67 -0
  762. package/src/services/voice/kokoro/__tests__/voices.test.ts +57 -0
  763. package/src/services/voice/kokoro/index.ts +79 -0
  764. package/src/services/voice/kokoro/kokoro-backend.ts +223 -0
  765. package/src/services/voice/kokoro/kokoro-engine-discovery.ts +177 -0
  766. package/src/services/voice/kokoro/kokoro-ffi-runtime.ts +233 -0
  767. package/src/services/voice/kokoro/kokoro-runtime.ts +170 -0
  768. package/src/services/voice/kokoro/phoneme-stream.ts +123 -0
  769. package/src/services/voice/kokoro/phonemizer.ts +344 -0
  770. package/src/services/voice/kokoro/pick-runtime.test.ts +91 -0
  771. package/src/services/voice/kokoro/pick-runtime.ts +130 -0
  772. package/src/services/voice/kokoro/runtime-selection.ts +64 -0
  773. package/src/services/voice/kokoro/types.ts +95 -0
  774. package/src/services/voice/kokoro/voice-presets.ts +129 -0
  775. package/src/services/voice/kokoro/voices.ts +64 -0
  776. package/src/services/voice/lifecycle.test.ts +315 -0
  777. package/src/services/voice/lifecycle.ts +301 -0
  778. package/src/services/voice/live-diarization-session.echo.test.ts +232 -0
  779. package/src/services/voice/live-diarization-session.ts +622 -0
  780. package/src/services/voice/metric-math.test.ts +61 -0
  781. package/src/services/voice/metric-math.ts +25 -0
  782. package/src/services/voice/mic-source.test.ts +210 -0
  783. package/src/services/voice/mic-source.ts +503 -0
  784. package/src/services/voice/nlms-echo-canceller.test.ts +244 -0
  785. package/src/services/voice/nlms-echo-canceller.ts +317 -0
  786. package/src/services/voice/optimistic-policy.power-source.test.ts +36 -0
  787. package/src/services/voice/optimistic-policy.test.ts +101 -0
  788. package/src/services/voice/optimistic-policy.ts +192 -0
  789. package/src/services/voice/optimistic-rollback.ts +343 -0
  790. package/src/services/voice/partial-stabilizer.test.ts +68 -0
  791. package/src/services/voice/partial-stabilizer.ts +140 -0
  792. package/src/services/voice/phoneme-tokenizer.ts +158 -0
  793. package/src/services/voice/phrase-cache.test.ts +242 -0
  794. package/src/services/voice/phrase-cache.ts +186 -0
  795. package/src/services/voice/phrase-chunker.test.ts +239 -0
  796. package/src/services/voice/phrase-chunker.ts +281 -0
  797. package/src/services/voice/pipeline-impls.l6.test.ts +110 -0
  798. package/src/services/voice/pipeline-impls.test.ts +292 -0
  799. package/src/services/voice/pipeline-impls.ts +315 -0
  800. package/src/services/voice/pipeline.ts +504 -0
  801. package/src/services/voice/prefill-client.ts +316 -0
  802. package/src/services/voice/prefix-preserving-queue.ts +162 -0
  803. package/src/services/voice/profile-store.ts +887 -0
  804. package/src/services/voice/real-audio-decode.test.ts +148 -0
  805. package/src/services/voice/research/VOICE_8785_ASSESSMENT.md +141 -0
  806. package/src/services/voice/research/VOICE_PIPELINE_RESEARCH_2026.md +117 -0
  807. package/src/services/voice/research/VOICE_VALIDATION_RUNBOOK.md +135 -0
  808. package/src/services/voice/ring-buffer.test.ts +129 -0
  809. package/src/services/voice/ring-buffer.ts +123 -0
  810. package/src/services/voice/rollback-queue.ts +74 -0
  811. package/src/services/voice/samantha-preset-placeholder.test.ts +97 -0
  812. package/src/services/voice/samantha-preset-placeholder.ts +148 -0
  813. package/src/services/voice/samantha-preset-regenerator.ts +393 -0
  814. package/src/services/voice/samantha-preset-regenerator.wav.test.ts +90 -0
  815. package/src/services/voice/scheduler.t2.test.ts +141 -0
  816. package/src/services/voice/scheduler.ts +927 -0
  817. package/src/services/voice/self-voice-imprint.test.ts +59 -0
  818. package/src/services/voice/self-voice-imprint.ts +102 -0
  819. package/src/services/voice/shared-resources.ts +343 -0
  820. package/src/services/voice/speaker/attribution-pipeline.test.ts +221 -0
  821. package/src/services/voice/speaker/attribution-pipeline.ts +449 -0
  822. package/src/services/voice/speaker/diarizer-fused.real.test.ts +100 -0
  823. package/src/services/voice/speaker/diarizer-fused.ts +154 -0
  824. package/src/services/voice/speaker/diarizer.ts +218 -0
  825. package/src/services/voice/speaker/encoder-fused.real.test.ts +113 -0
  826. package/src/services/voice/speaker/encoder-fused.ts +138 -0
  827. package/src/services/voice/speaker/encoder-ggml.test.ts +59 -0
  828. package/src/services/voice/speaker/encoder-ggml.ts +79 -0
  829. package/src/services/voice/speaker/encoder.ts +105 -0
  830. package/src/services/voice/speaker-imprint.test.ts +185 -0
  831. package/src/services/voice/speaker-imprint.ts +312 -0
  832. package/src/services/voice/speaker-preset-cache.test.ts +154 -0
  833. package/src/services/voice/speaker-preset-cache.ts +195 -0
  834. package/src/services/voice/streaming-asr/streaming-pipeline-adapter.ts +292 -0
  835. package/src/services/voice/system-audio-sink.test.ts +29 -0
  836. package/src/services/voice/system-audio-sink.ts +366 -0
  837. package/src/services/voice/transcriber.asr-backend.test.ts +76 -0
  838. package/src/services/voice/transcriber.test.ts +392 -0
  839. package/src/services/voice/transcriber.ts +704 -0
  840. package/src/services/voice/transcript-knowledge.test.ts +68 -0
  841. package/src/services/voice/transcript-knowledge.ts +75 -0
  842. package/src/services/voice/transcript-service.test.ts +195 -0
  843. package/src/services/voice/transcript-service.ts +205 -0
  844. package/src/services/voice/transcript-store.test.ts +189 -0
  845. package/src/services/voice/transcript-store.ts +164 -0
  846. package/src/services/voice/turn-controller.test.ts +575 -0
  847. package/src/services/voice/turn-controller.ts +596 -0
  848. package/src/services/voice/types.ts +699 -0
  849. package/src/services/voice/vad.test.ts +498 -0
  850. package/src/services/voice/vad.ts +832 -0
  851. package/src/services/voice/vad.v1-v4.test.ts +222 -0
  852. package/src/services/voice/voice-budget.test.ts +415 -0
  853. package/src/services/voice/voice-budget.ts +635 -0
  854. package/src/services/voice/voice-duet.test.ts +375 -0
  855. package/src/services/voice/voice-emotion-classifier.test.ts +210 -0
  856. package/src/services/voice/voice-emotion-classifier.ts +273 -0
  857. package/src/services/voice/voice-hardening.fuzz.test.ts +116 -0
  858. package/src/services/voice/voice-preload-predictor.test.ts +130 -0
  859. package/src/services/voice/voice-preload-predictor.ts +113 -0
  860. package/src/services/voice/voice-preset-format.fuzz.test.ts +89 -0
  861. package/src/services/voice/voice-preset-format.test.ts +75 -0
  862. package/src/services/voice/voice-preset-format.ts +713 -0
  863. package/src/services/voice/voice-preset-generator.test.ts +89 -0
  864. package/src/services/voice/voice-profile-artifact.test.ts +138 -0
  865. package/src/services/voice/voice-profile-artifact.ts +518 -0
  866. package/src/services/voice/voice-profile-routes.test.ts +429 -0
  867. package/src/services/voice/voice-profile-routes.ts +425 -0
  868. package/src/services/voice/voice-scenario.test.ts +159 -0
  869. package/src/services/voice/voice-scenario.ts +280 -0
  870. package/src/services/voice/voice-scenario.turn-helpers.test.ts +77 -0
  871. package/src/services/voice/voice-state-machine.ts +727 -0
  872. package/src/services/voice/voice-workbench-report.test.ts +168 -0
  873. package/src/services/voice/voice-workbench-report.ts +367 -0
  874. package/src/services/voice/voice-workbench.test.ts +158 -0
  875. package/src/services/voice/voice.test.ts +1070 -0
  876. package/src/services/voice/wake-word-ggml.ts +319 -0
  877. package/src/services/voice/wake-word.test.ts +298 -0
  878. package/src/services/voice/wake-word.ts +554 -0
  879. package/src/services/voice/wav-codec.fuzz.test.ts +59 -0
  880. package/src/services/voice/wav-codec.test.ts +32 -0
  881. package/src/services/voice/wav-codec.ts +101 -0
  882. package/src/services/voice/workbench-entrypoint.test.ts +55 -0
  883. package/src/services/voice/workbench-entrypoint.ts +88 -0
  884. package/src/services/voice/workbench-headless-runner.test.ts +162 -0
  885. package/src/services/voice/workbench-headless-runner.ts +396 -0
  886. package/src/services/voice/workbench-logic-services.test.ts +225 -0
  887. package/src/services/voice/workbench-logic-services.ts +184 -0
  888. package/src/services/voice/workbench-real-services.ts +629 -0
  889. package/src/services/voice/workbench-scenarios.ts +407 -0
  890. package/src/services/voice/wrap-with-first-line-cache.ts +267 -0
  891. package/src/services/voice-model-updater.ts +724 -0
  892. package/src/services/voice-prewarm.ts +51 -0
  893. package/src/voice-workbench.ts +71 -0
@@ -0,0 +1,713 @@
1
+ /**
2
+ * Binary format for `cache/voice-preset-*.bin`.
3
+ *
4
+ * Two versions are supported:
5
+ *
6
+ * v1 (`magic='ELZ1', version=1`) — legacy two-section layout used by the
7
+ * initial Kokoro-style placeholder. Carries a Float32 speaker embedding +
8
+ * a phrase-cache seed list. Still read for back-compat (older bundles only
9
+ * contain v1).
10
+ *
11
+ * v2 (`magic='ELZ1', version=2`) — superset adopted for the OmniVoice
12
+ * freeze. Adds three OmniVoice-specific sections that the v1 layout had
13
+ * no room for: pre-encoded `ref_audio_tokens` (int32, shape
14
+ * `[K, ref_T]`), a UTF-8 `ref_text` transcript of the reference clip, and
15
+ * a closed-vocabulary `instruct` string (the resolved VoiceDesign
16
+ * attributes). v2 readers handle v1 files transparently (the new sections
17
+ * default to empty). A v1 reader applied to a v2 file fails fast on
18
+ * `truncated-header` because the v2 header is larger.
19
+ *
20
+ * Layout (little-endian throughout):
21
+ *
22
+ * v1 header (24 bytes):
23
+ * +0 4 bytes magic 'ELZ1' (0x315A4C45)
24
+ * +4 4 bytes format version (uint32) — 1
25
+ * +8 4 bytes speaker embedding offset (uint32)
26
+ * +12 4 bytes speaker embedding byte length (uint32)
27
+ * +16 4 bytes phrase cache seed offset (uint32)
28
+ * +20 4 bytes phrase cache seed byte length (uint32)
29
+ *
30
+ * v2 header (64 bytes — additive, all section descriptors are
31
+ * `(offset:uint32, length:uint32)` pairs):
32
+ * +0 4 bytes magic 'ELZ1' (0x315A4C45)
33
+ * +4 4 bytes format version (uint32) — 2
34
+ * +8 4 bytes speaker embedding offset
35
+ * +12 4 bytes speaker embedding byte length
36
+ * +16 4 bytes phrase cache seed offset
37
+ * +20 4 bytes phrase cache seed byte length
38
+ * +24 4 bytes ref_audio_tokens offset
39
+ * +28 4 bytes ref_audio_tokens byte length
40
+ * +32 4 bytes ref_text offset
41
+ * +36 4 bytes ref_text byte length
42
+ * +40 4 bytes instruct offset
43
+ * +44 4 bytes instruct byte length
44
+ * +48 4 bytes metadata offset
45
+ * +52 4 bytes metadata byte length
46
+ * +56 4 bytes reserved (must be 0)
47
+ * +60 4 bytes reserved (must be 0)
48
+ *
49
+ * `ref_audio_tokens` payload (v2):
50
+ * +0 4 bytes K — codebook count (uint32, OmniVoice = 8)
51
+ * +4 4 bytes ref_T — frames per codebook (uint32)
52
+ * +8 ... int32 LE codebook samples, row-major shape `[K, ref_T]`
53
+ *
54
+ * `ref_text` payload (v2): raw UTF-8 bytes (no NUL terminator).
55
+ * `instruct` payload (v2): raw UTF-8 bytes (closed VoiceDesign vocabulary).
56
+ * `metadata` payload (v2): raw UTF-8 JSON bytes (codec sha256, corpus
57
+ * hash, etc.); the runtime never relies on
58
+ * metadata for correctness.
59
+ *
60
+ * Phrase cache seed payload (v1 + v2, identical):
61
+ * uint32 LE N (phrase count)
62
+ * for each phrase:
63
+ * uint16 LE text_byte_len
64
+ * uint8[] canonicalized text (UTF-8)
65
+ * uint32 LE sample_rate
66
+ * uint32 LE pcm_byte_len
67
+ * uint8[] PCM (Float32 LE samples)
68
+ *
69
+ * Per-section invariants:
70
+ * - Section bounds may not overlap the header.
71
+ * - Section bounds must fit within the file length.
72
+ * - A `length=0` section is allowed (means "absent"); the corresponding
73
+ * output field is an empty `Float32Array` / `Int32Array` / empty string.
74
+ * - `embedding.length % 4 == 0` (Float32).
75
+ * - `ref_audio_tokens.length` ≥ 8 (the two header words K, ref_T) and the
76
+ * payload is `8 + K*ref_T*4` bytes.
77
+ */
78
+
79
+ export const VOICE_PRESET_MAGIC = 0x315a4c45; // 'ELZ1'
80
+
81
+ /** Header byte counts. */
82
+ export const VOICE_PRESET_HEADER_BYTES_V1 = 24;
83
+ export const VOICE_PRESET_HEADER_BYTES_V2 = 64;
84
+
85
+ /** Supported format versions. v2 is the canonical write path. */
86
+ export const VOICE_PRESET_VERSION_V1 = 1;
87
+ export const VOICE_PRESET_VERSION_V2 = 2;
88
+ export const VOICE_PRESET_VERSION_CURRENT = VOICE_PRESET_VERSION_V2;
89
+
90
+ export interface VoicePresetSeedPhrase {
91
+ /** Canonicalized text (lowercase, single-spaced, trimmed). */
92
+ text: string;
93
+ sampleRate: number;
94
+ pcm: Float32Array;
95
+ }
96
+
97
+ /**
98
+ * OmniVoice reference-audio-tokens payload. `K` is the codebook count (=8 for
99
+ * OmniVoice / HiggsAudioV2) and `refT` is the number of frames per codebook.
100
+ * `tokens` is row-major: codebook `k`, frame `t` is at `tokens[k*refT + t]`.
101
+ * An empty payload (refT=0, K=0, tokens length 0) is valid and means "no
102
+ * reference audio bound to this preset" (instruct-only voice).
103
+ */
104
+ export interface RefAudioTokens {
105
+ K: number;
106
+ refT: number;
107
+ tokens: Int32Array;
108
+ }
109
+
110
+ export interface VoicePresetFile {
111
+ version: number;
112
+ embedding: Float32Array;
113
+ phrases: ReadonlyArray<VoicePresetSeedPhrase>;
114
+ /** v2 only — empty for v1 files. */
115
+ refAudioTokens: RefAudioTokens;
116
+ /** v2 only — empty for v1 files. */
117
+ refText: string;
118
+ /** v2 only — empty for v1 files. */
119
+ instruct: string;
120
+ /** v2 only — parsed JSON object, empty `{}` for v1 files. */
121
+ metadata: Record<string, unknown>;
122
+ }
123
+
124
+ export class VoicePresetFormatError extends Error {
125
+ constructor(
126
+ message: string,
127
+ readonly code:
128
+ | "bad-magic"
129
+ | "bad-version"
130
+ | "truncated-header"
131
+ | "truncated-section"
132
+ | "bad-section-bounds"
133
+ | "bad-phrase-record"
134
+ | "bad-embedding-length"
135
+ | "bad-ref-tokens"
136
+ | "bad-metadata"
137
+ | "bad-utf8",
138
+ ) {
139
+ super(message);
140
+ this.name = "VoicePresetFormatError";
141
+ }
142
+ }
143
+
144
+ interface SectionView {
145
+ offset: number;
146
+ length: number;
147
+ }
148
+
149
+ interface ParsedHeader {
150
+ version: number;
151
+ headerBytes: number;
152
+ embedding: SectionView;
153
+ phrases: SectionView;
154
+ refAudioTokens: SectionView;
155
+ refText: SectionView;
156
+ instruct: SectionView;
157
+ metadata: SectionView;
158
+ }
159
+
160
+ const EMPTY_SECTION: SectionView = Object.freeze({ offset: 0, length: 0 });
161
+
162
+ function checkSectionBounds(
163
+ sec: SectionView,
164
+ fileLen: number,
165
+ headerBytes: number,
166
+ ): void {
167
+ if (sec.length === 0) return;
168
+ if (sec.offset < headerBytes) {
169
+ throw new VoicePresetFormatError(
170
+ `voice preset section overlaps header (offset=${sec.offset} < header=${headerBytes})`,
171
+ "bad-section-bounds",
172
+ );
173
+ }
174
+ if (sec.offset + sec.length > fileLen) {
175
+ throw new VoicePresetFormatError(
176
+ `voice preset section bounds exceed file length`,
177
+ "bad-section-bounds",
178
+ );
179
+ }
180
+ }
181
+
182
+ function readHeader(view: DataView): ParsedHeader {
183
+ if (view.byteLength < VOICE_PRESET_HEADER_BYTES_V1) {
184
+ throw new VoicePresetFormatError(
185
+ `voice preset file truncated: header needs ${VOICE_PRESET_HEADER_BYTES_V1} bytes, got ${view.byteLength}`,
186
+ "truncated-header",
187
+ );
188
+ }
189
+ const magic = view.getUint32(0, true);
190
+ if (magic !== VOICE_PRESET_MAGIC) {
191
+ throw new VoicePresetFormatError(
192
+ `voice preset bad magic: expected 0x${VOICE_PRESET_MAGIC.toString(16)}, got 0x${magic.toString(16)}`,
193
+ "bad-magic",
194
+ );
195
+ }
196
+ const version = view.getUint32(4, true);
197
+ if (
198
+ version !== VOICE_PRESET_VERSION_V1 &&
199
+ version !== VOICE_PRESET_VERSION_V2
200
+ ) {
201
+ throw new VoicePresetFormatError(
202
+ `voice preset unsupported version: ${version} (this build supports 1 and 2)`,
203
+ "bad-version",
204
+ );
205
+ }
206
+ const headerBytes =
207
+ version === VOICE_PRESET_VERSION_V2
208
+ ? VOICE_PRESET_HEADER_BYTES_V2
209
+ : VOICE_PRESET_HEADER_BYTES_V1;
210
+ if (view.byteLength < headerBytes) {
211
+ throw new VoicePresetFormatError(
212
+ `voice preset file truncated: v${version} header needs ${headerBytes} bytes, got ${view.byteLength}`,
213
+ "truncated-header",
214
+ );
215
+ }
216
+
217
+ const embedding: SectionView = {
218
+ offset: view.getUint32(8, true),
219
+ length: view.getUint32(12, true),
220
+ };
221
+ const phrases: SectionView = {
222
+ offset: view.getUint32(16, true),
223
+ length: view.getUint32(20, true),
224
+ };
225
+
226
+ let refAudioTokens = EMPTY_SECTION;
227
+ let refText = EMPTY_SECTION;
228
+ let instruct = EMPTY_SECTION;
229
+ let metadata = EMPTY_SECTION;
230
+ if (version === VOICE_PRESET_VERSION_V2) {
231
+ refAudioTokens = {
232
+ offset: view.getUint32(24, true),
233
+ length: view.getUint32(28, true),
234
+ };
235
+ refText = {
236
+ offset: view.getUint32(32, true),
237
+ length: view.getUint32(36, true),
238
+ };
239
+ instruct = {
240
+ offset: view.getUint32(40, true),
241
+ length: view.getUint32(44, true),
242
+ };
243
+ metadata = {
244
+ offset: view.getUint32(48, true),
245
+ length: view.getUint32(52, true),
246
+ };
247
+ // Reserved words must be zero — fail closed on accidental reuse.
248
+ const r0 = view.getUint32(56, true);
249
+ const r1 = view.getUint32(60, true);
250
+ if (r0 !== 0 || r1 !== 0) {
251
+ throw new VoicePresetFormatError(
252
+ `voice preset v2 reserved header words must be 0 (got ${r0}, ${r1})`,
253
+ "bad-section-bounds",
254
+ );
255
+ }
256
+ }
257
+
258
+ const fileLen = view.byteLength;
259
+ checkSectionBounds(embedding, fileLen, headerBytes);
260
+ checkSectionBounds(phrases, fileLen, headerBytes);
261
+ checkSectionBounds(refAudioTokens, fileLen, headerBytes);
262
+ checkSectionBounds(refText, fileLen, headerBytes);
263
+ checkSectionBounds(instruct, fileLen, headerBytes);
264
+ checkSectionBounds(metadata, fileLen, headerBytes);
265
+
266
+ return {
267
+ version,
268
+ headerBytes,
269
+ embedding,
270
+ phrases,
271
+ refAudioTokens,
272
+ refText,
273
+ instruct,
274
+ metadata,
275
+ };
276
+ }
277
+
278
+ function copyFloat32(
279
+ bytes: Uint8Array,
280
+ /** Offset relative to `bytes` (i.e. relative to bytes.byteOffset). */
281
+ relativeOffset: number,
282
+ byteLength: number,
283
+ ): Float32Array {
284
+ // The source byte offset is not guaranteed to be 4-aligned in the file
285
+ // buffer, so we copy raw bytes into a fresh ArrayBuffer first.
286
+ const aligned = new Uint8Array(byteLength);
287
+ aligned.set(bytes.subarray(relativeOffset, relativeOffset + byteLength));
288
+ return new Float32Array(aligned.buffer, 0, byteLength / 4);
289
+ }
290
+
291
+ function copyInt32(
292
+ bytes: Uint8Array,
293
+ relativeOffset: number,
294
+ byteLength: number,
295
+ ): Int32Array {
296
+ const aligned = new Uint8Array(byteLength);
297
+ aligned.set(bytes.subarray(relativeOffset, relativeOffset + byteLength));
298
+ return new Int32Array(aligned.buffer, 0, byteLength / 4);
299
+ }
300
+
301
+ function readEmbedding(bytes: Uint8Array, sec: SectionView): Float32Array {
302
+ if (sec.length === 0) return new Float32Array(0);
303
+ if (sec.length % 4 !== 0) {
304
+ throw new VoicePresetFormatError(
305
+ `voice preset embedding length ${sec.length} is not a multiple of 4`,
306
+ "bad-embedding-length",
307
+ );
308
+ }
309
+ return copyFloat32(bytes, sec.offset, sec.length);
310
+ }
311
+
312
+ function readRefAudioTokens(
313
+ bytes: Uint8Array,
314
+ sec: SectionView,
315
+ ): RefAudioTokens {
316
+ if (sec.length === 0) {
317
+ return { K: 0, refT: 0, tokens: new Int32Array(0) };
318
+ }
319
+ if (sec.length < 8) {
320
+ throw new VoicePresetFormatError(
321
+ `voice preset ref_audio_tokens section truncated (need ≥ 8 bytes, got ${sec.length})`,
322
+ "bad-ref-tokens",
323
+ );
324
+ }
325
+ const view = new DataView(
326
+ bytes.buffer,
327
+ bytes.byteOffset + sec.offset,
328
+ sec.length,
329
+ );
330
+ const K = view.getUint32(0, true);
331
+ const refT = view.getUint32(4, true);
332
+ const tokenBytes = sec.length - 8;
333
+ if (tokenBytes % 4 !== 0) {
334
+ throw new VoicePresetFormatError(
335
+ `voice preset ref_audio_tokens payload bytes ${tokenBytes} is not a multiple of 4`,
336
+ "bad-ref-tokens",
337
+ );
338
+ }
339
+ const expected = K * refT * 4;
340
+ if (tokenBytes !== expected) {
341
+ throw new VoicePresetFormatError(
342
+ `voice preset ref_audio_tokens shape mismatch: K=${K}, ref_T=${refT}, expected ${expected} bytes, got ${tokenBytes}`,
343
+ "bad-ref-tokens",
344
+ );
345
+ }
346
+ const tokens =
347
+ tokenBytes === 0
348
+ ? new Int32Array(0)
349
+ : copyInt32(bytes, sec.offset + 8, tokenBytes);
350
+ return { K, refT, tokens };
351
+ }
352
+
353
+ /** Strict UTF-8 decode that surfaces invalid bytes as the format's own error
354
+ * rather than leaking a raw TextDecoder TypeError past the defensive boundary. */
355
+ function decodeUtf8Strict(bytes: Uint8Array): string {
356
+ try {
357
+ return new TextDecoder("utf-8", { fatal: true }).decode(bytes);
358
+ } catch {
359
+ throw new VoicePresetFormatError(
360
+ "voice preset contains invalid UTF-8 text",
361
+ "bad-utf8",
362
+ );
363
+ }
364
+ }
365
+
366
+ function readUtf8(bytes: Uint8Array, sec: SectionView): string {
367
+ if (sec.length === 0) return "";
368
+ const slice = bytes.subarray(sec.offset, sec.offset + sec.length);
369
+ return decodeUtf8Strict(slice);
370
+ }
371
+
372
+ function readMetadata(
373
+ bytes: Uint8Array,
374
+ sec: SectionView,
375
+ ): Record<string, unknown> {
376
+ if (sec.length === 0) return {};
377
+ const text = readUtf8(bytes, sec);
378
+ let parsed: unknown;
379
+ try {
380
+ parsed = JSON.parse(text);
381
+ } catch (err) {
382
+ throw new VoicePresetFormatError(
383
+ `voice preset metadata is not valid JSON: ${(err as Error).message}`,
384
+ "bad-metadata",
385
+ );
386
+ }
387
+ if (parsed === null || typeof parsed !== "object" || Array.isArray(parsed)) {
388
+ throw new VoicePresetFormatError(
389
+ `voice preset metadata must be a JSON object`,
390
+ "bad-metadata",
391
+ );
392
+ }
393
+ return parsed as Record<string, unknown>;
394
+ }
395
+
396
+ function readPhrases(
397
+ bytes: Uint8Array,
398
+ sec: SectionView,
399
+ ): VoicePresetSeedPhrase[] {
400
+ if (sec.length === 0) return [];
401
+ const view = new DataView(
402
+ bytes.buffer,
403
+ bytes.byteOffset + sec.offset,
404
+ sec.length,
405
+ );
406
+ let pos = 0;
407
+ if (sec.length < 4) {
408
+ throw new VoicePresetFormatError(
409
+ "voice preset phrase section truncated before count",
410
+ "truncated-section",
411
+ );
412
+ }
413
+ const count = view.getUint32(pos, true);
414
+ pos += 4;
415
+ const out: VoicePresetSeedPhrase[] = [];
416
+ for (let i = 0; i < count; i++) {
417
+ if (pos + 2 > sec.length) {
418
+ throw new VoicePresetFormatError(
419
+ `voice preset phrase #${i}: truncated before text length`,
420
+ "bad-phrase-record",
421
+ );
422
+ }
423
+ const textLen = view.getUint16(pos, true);
424
+ pos += 2;
425
+ if (pos + textLen > sec.length) {
426
+ throw new VoicePresetFormatError(
427
+ `voice preset phrase #${i}: text overruns section`,
428
+ "bad-phrase-record",
429
+ );
430
+ }
431
+ const textBytes = new Uint8Array(
432
+ bytes.buffer,
433
+ bytes.byteOffset + sec.offset + pos,
434
+ textLen,
435
+ );
436
+ const text = decodeUtf8Strict(textBytes);
437
+ pos += textLen;
438
+ if (pos + 8 > sec.length) {
439
+ throw new VoicePresetFormatError(
440
+ `voice preset phrase #${i}: truncated before sample_rate/pcm_len`,
441
+ "bad-phrase-record",
442
+ );
443
+ }
444
+ const sampleRate = view.getUint32(pos, true);
445
+ pos += 4;
446
+ const pcmByteLen = view.getUint32(pos, true);
447
+ pos += 4;
448
+ if (pcmByteLen % 4 !== 0) {
449
+ throw new VoicePresetFormatError(
450
+ `voice preset phrase #${i}: pcm byte length ${pcmByteLen} is not a multiple of 4`,
451
+ "bad-phrase-record",
452
+ );
453
+ }
454
+ if (pos + pcmByteLen > sec.length) {
455
+ throw new VoicePresetFormatError(
456
+ `voice preset phrase #${i}: pcm overruns section`,
457
+ "bad-phrase-record",
458
+ );
459
+ }
460
+ const pcm = copyFloat32(bytes, sec.offset + pos, pcmByteLen);
461
+ pos += pcmByteLen;
462
+ out.push({ text, sampleRate, pcm });
463
+ }
464
+ return out;
465
+ }
466
+
467
+ /**
468
+ * Parse a voice-preset binary blob. Throws `VoicePresetFormatError` on any
469
+ * malformed input — this is the single defensive boundary for the format.
470
+ * Supports both v1 and v2 files. For v1 files the v2-only fields are
471
+ * returned as their empty equivalents.
472
+ */
473
+ export function readVoicePresetFile(bytes: Uint8Array): VoicePresetFile {
474
+ const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength);
475
+ const header = readHeader(view);
476
+ return {
477
+ version: header.version,
478
+ embedding: readEmbedding(bytes, header.embedding),
479
+ phrases: readPhrases(bytes, header.phrases),
480
+ refAudioTokens: readRefAudioTokens(bytes, header.refAudioTokens),
481
+ refText: readUtf8(bytes, header.refText),
482
+ instruct: readUtf8(bytes, header.instruct),
483
+ metadata: readMetadata(bytes, header.metadata),
484
+ };
485
+ }
486
+
487
+ /**
488
+ * Serialize a voice preset to the v1 binary format. The output is a fresh
489
+ * `Uint8Array` ready to be written to disk.
490
+ *
491
+ * Use this only when the caller deliberately wants the legacy v1 shape (e.g.
492
+ * the existing Kokoro-style placeholder builder). New code should call
493
+ * `writeVoicePresetFileV2`.
494
+ */
495
+ export function writeVoicePresetFile(file: {
496
+ embedding: Float32Array;
497
+ phrases: ReadonlyArray<VoicePresetSeedPhrase>;
498
+ }): Uint8Array {
499
+ const encoder = new TextEncoder();
500
+ const encodedTexts = file.phrases.map((p) => encoder.encode(p.text));
501
+
502
+ const embBytes = file.embedding.byteLength;
503
+ let phrBytes = 4; // count
504
+ for (let i = 0; i < file.phrases.length; i++) {
505
+ const t = encodedTexts[i];
506
+ if (t.byteLength > 0xffff) {
507
+ throw new VoicePresetFormatError(
508
+ `phrase #${i} text too long (${t.byteLength} bytes, max 65535)`,
509
+ "bad-phrase-record",
510
+ );
511
+ }
512
+ phrBytes += 2 + t.byteLength + 4 + 4 + file.phrases[i].pcm.byteLength;
513
+ }
514
+
515
+ const embOff = VOICE_PRESET_HEADER_BYTES_V1;
516
+ const phrOff = embOff + embBytes;
517
+ const total = phrOff + phrBytes;
518
+
519
+ const out = new Uint8Array(total);
520
+ const view = new DataView(out.buffer);
521
+ view.setUint32(0, VOICE_PRESET_MAGIC, true);
522
+ view.setUint32(4, VOICE_PRESET_VERSION_V1, true);
523
+ view.setUint32(8, embOff, true);
524
+ view.setUint32(12, embBytes, true);
525
+ view.setUint32(16, phrOff, true);
526
+ view.setUint32(20, phrBytes, true);
527
+
528
+ // Embedding
529
+ out.set(
530
+ new Uint8Array(
531
+ file.embedding.buffer,
532
+ file.embedding.byteOffset,
533
+ file.embedding.byteLength,
534
+ ),
535
+ embOff,
536
+ );
537
+
538
+ // Phrases
539
+ writePhraseSection(out, view, phrOff, file.phrases, encodedTexts);
540
+
541
+ return out;
542
+ }
543
+
544
+ function writePhraseSection(
545
+ out: Uint8Array,
546
+ view: DataView,
547
+ startOff: number,
548
+ phrases: ReadonlyArray<VoicePresetSeedPhrase>,
549
+ encodedTexts: Uint8Array[],
550
+ ): void {
551
+ let pos = startOff;
552
+ view.setUint32(pos, phrases.length, true);
553
+ pos += 4;
554
+ for (let i = 0; i < phrases.length; i++) {
555
+ const t = encodedTexts[i];
556
+ const phrase = phrases[i];
557
+ view.setUint16(pos, t.byteLength, true);
558
+ pos += 2;
559
+ out.set(t, pos);
560
+ pos += t.byteLength;
561
+ view.setUint32(pos, phrase.sampleRate, true);
562
+ pos += 4;
563
+ view.setUint32(pos, phrase.pcm.byteLength, true);
564
+ pos += 4;
565
+ out.set(
566
+ new Uint8Array(
567
+ phrase.pcm.buffer,
568
+ phrase.pcm.byteOffset,
569
+ phrase.pcm.byteLength,
570
+ ),
571
+ pos,
572
+ );
573
+ pos += phrase.pcm.byteLength;
574
+ }
575
+ }
576
+
577
+ /**
578
+ * Write a voice preset in the v2 (additive) layout. Used by the OmniVoice
579
+ * freeze pipeline (`freeze-voice.mjs`) and other producers that need to
580
+ * persist `refAudioTokens` / `refText` / `instruct` alongside the v1
581
+ * embedding + phrase-seed sections.
582
+ *
583
+ * Any field that the caller doesn't need to persist can be omitted (or
584
+ * passed empty). The on-disk section is then written as length=0 and is
585
+ * read back as the empty equivalent.
586
+ */
587
+ export function writeVoicePresetFileV2(file: {
588
+ embedding?: Float32Array;
589
+ phrases?: ReadonlyArray<VoicePresetSeedPhrase>;
590
+ refAudioTokens?: RefAudioTokens;
591
+ refText?: string;
592
+ instruct?: string;
593
+ metadata?: Record<string, unknown>;
594
+ }): Uint8Array {
595
+ const embedding = file.embedding ?? new Float32Array(0);
596
+ const phrases = file.phrases ?? [];
597
+ const refAudioTokens = file.refAudioTokens ?? {
598
+ K: 0,
599
+ refT: 0,
600
+ tokens: new Int32Array(0),
601
+ };
602
+ const refText = file.refText ?? "";
603
+ const instruct = file.instruct ?? "";
604
+ const metadata = file.metadata ?? {};
605
+
606
+ if (refAudioTokens.K * refAudioTokens.refT !== refAudioTokens.tokens.length) {
607
+ throw new VoicePresetFormatError(
608
+ `ref_audio_tokens shape mismatch: K=${refAudioTokens.K}, ref_T=${refAudioTokens.refT}, but tokens.length=${refAudioTokens.tokens.length}`,
609
+ "bad-ref-tokens",
610
+ );
611
+ }
612
+
613
+ const encoder = new TextEncoder();
614
+ const encodedTexts = phrases.map((p) => encoder.encode(p.text));
615
+ const encodedRefText = encoder.encode(refText);
616
+ const encodedInstruct = encoder.encode(instruct);
617
+ const encodedMetadata =
618
+ Object.keys(metadata).length === 0
619
+ ? new Uint8Array(0)
620
+ : encoder.encode(JSON.stringify(metadata));
621
+
622
+ // Compute payload sizes up-front so we can lay out section offsets.
623
+ const embBytes = embedding.byteLength;
624
+ let phrBytes = phrases.length === 0 && encodedTexts.length === 0 ? 0 : 4;
625
+ if (phrBytes > 0) {
626
+ for (let i = 0; i < phrases.length; i++) {
627
+ const t = encodedTexts[i];
628
+ if (t.byteLength > 0xffff) {
629
+ throw new VoicePresetFormatError(
630
+ `phrase #${i} text too long (${t.byteLength} bytes, max 65535)`,
631
+ "bad-phrase-record",
632
+ );
633
+ }
634
+ phrBytes += 2 + t.byteLength + 4 + 4 + phrases[i].pcm.byteLength;
635
+ }
636
+ }
637
+ const refTokensBytes =
638
+ refAudioTokens.tokens.length === 0 && refAudioTokens.K === 0
639
+ ? 0
640
+ : 8 + refAudioTokens.tokens.byteLength;
641
+
642
+ // Lay out sections in declared order. Empty sections claim no space and
643
+ // are recorded as (offset=0, length=0).
644
+ let cursor = VOICE_PRESET_HEADER_BYTES_V2;
645
+ const embOff = embBytes > 0 ? cursor : 0;
646
+ cursor += embBytes;
647
+ const phrOff = phrBytes > 0 ? cursor : 0;
648
+ cursor += phrBytes;
649
+ const refTokensOff = refTokensBytes > 0 ? cursor : 0;
650
+ cursor += refTokensBytes;
651
+ const refTextOff = encodedRefText.byteLength > 0 ? cursor : 0;
652
+ cursor += encodedRefText.byteLength;
653
+ const instructOff = encodedInstruct.byteLength > 0 ? cursor : 0;
654
+ cursor += encodedInstruct.byteLength;
655
+ const metadataOff = encodedMetadata.byteLength > 0 ? cursor : 0;
656
+ cursor += encodedMetadata.byteLength;
657
+
658
+ const total = cursor;
659
+ const out = new Uint8Array(total);
660
+ const view = new DataView(out.buffer);
661
+
662
+ view.setUint32(0, VOICE_PRESET_MAGIC, true);
663
+ view.setUint32(4, VOICE_PRESET_VERSION_V2, true);
664
+ view.setUint32(8, embOff, true);
665
+ view.setUint32(12, embBytes, true);
666
+ view.setUint32(16, phrOff, true);
667
+ view.setUint32(20, phrBytes, true);
668
+ view.setUint32(24, refTokensOff, true);
669
+ view.setUint32(28, refTokensBytes, true);
670
+ view.setUint32(32, refTextOff, true);
671
+ view.setUint32(36, encodedRefText.byteLength, true);
672
+ view.setUint32(40, instructOff, true);
673
+ view.setUint32(44, encodedInstruct.byteLength, true);
674
+ view.setUint32(48, metadataOff, true);
675
+ view.setUint32(52, encodedMetadata.byteLength, true);
676
+ view.setUint32(56, 0, true);
677
+ view.setUint32(60, 0, true);
678
+
679
+ if (embBytes > 0) {
680
+ out.set(
681
+ new Uint8Array(embedding.buffer, embedding.byteOffset, embBytes),
682
+ embOff,
683
+ );
684
+ }
685
+ if (phrBytes > 0) {
686
+ writePhraseSection(out, view, phrOff, phrases, encodedTexts);
687
+ }
688
+ if (refTokensBytes > 0) {
689
+ view.setUint32(refTokensOff, refAudioTokens.K, true);
690
+ view.setUint32(refTokensOff + 4, refAudioTokens.refT, true);
691
+ if (refAudioTokens.tokens.byteLength > 0) {
692
+ out.set(
693
+ new Uint8Array(
694
+ refAudioTokens.tokens.buffer,
695
+ refAudioTokens.tokens.byteOffset,
696
+ refAudioTokens.tokens.byteLength,
697
+ ),
698
+ refTokensOff + 8,
699
+ );
700
+ }
701
+ }
702
+ if (encodedRefText.byteLength > 0) {
703
+ out.set(encodedRefText, refTextOff);
704
+ }
705
+ if (encodedInstruct.byteLength > 0) {
706
+ out.set(encodedInstruct, instructOff);
707
+ }
708
+ if (encodedMetadata.byteLength > 0) {
709
+ out.set(encodedMetadata, metadataOff);
710
+ }
711
+
712
+ return out;
713
+ }