@iternio/react-native-auto-play 0.5.10 → 0.5.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. package/android/src/main/java/com/margelo/nitro/swe/iternio/reactnativeautoplay/HybridVoice.kt +3 -1
  2. package/android/src/main/java/com/margelo/nitro/swe/iternio/reactnativeautoplay/VoiceInputManager.kt +60 -15
  3. package/android/src/main/java/com/margelo/nitro/swe/iternio/reactnativeautoplay/utils/G711.kt +77 -0
  4. package/ios/hybrid/HybridVoice.swift +4 -2
  5. package/ios/utils/G711.swift +83 -0
  6. package/ios/utils/VoiceInputManager.swift +17 -5
  7. package/lib/hybrid/HybridVoice.d.ts +1 -0
  8. package/lib/hybrid/HybridVoice.js +3 -2
  9. package/lib/specs/Voice.nitro.d.ts +2 -2
  10. package/lib/types/Voice.d.ts +6 -0
  11. package/nitrogen/generated/android/c++/JHybridVoiceSpec.cpp +7 -3
  12. package/nitrogen/generated/android/c++/JHybridVoiceSpec.hpp +1 -1
  13. package/nitrogen/generated/android/c++/JVoiceAudioEncoding.hpp +61 -0
  14. package/nitrogen/generated/android/kotlin/com/margelo/nitro/swe/iternio/reactnativeautoplay/HybridVoiceSpec.kt +3 -3
  15. package/nitrogen/generated/android/kotlin/com/margelo/nitro/swe/iternio/reactnativeautoplay/VoiceAudioEncoding.kt +24 -0
  16. package/nitrogen/generated/ios/ReactNativeAutoPlay-Swift-Cxx-Bridge.hpp +18 -0
  17. package/nitrogen/generated/ios/ReactNativeAutoPlay-Swift-Cxx-Umbrella.hpp +3 -0
  18. package/nitrogen/generated/ios/c++/HybridVoiceSpecSwift.hpp +5 -2
  19. package/nitrogen/generated/ios/swift/HybridVoiceSpec.swift +1 -1
  20. package/nitrogen/generated/ios/swift/HybridVoiceSpec_cxx.swift +2 -2
  21. package/nitrogen/generated/ios/swift/VoiceAudioEncoding.swift +44 -0
  22. package/nitrogen/generated/shared/c++/HybridVoiceSpec.hpp +4 -1
  23. package/nitrogen/generated/shared/c++/VoiceAudioEncoding.hpp +80 -0
  24. package/package.json +1 -1
  25. package/src/hybrid/HybridVoice.ts +4 -1
  26. package/src/specs/Voice.nitro.ts +3 -2
  27. package/src/types/Voice.ts +7 -0
@@ -70,7 +70,8 @@ class HybridVoice : HybridVoiceSpec() {
70
70
  onChunk: ((chunk: VoiceInputChunk) -> Unit)?,
71
71
  language: String?,
72
72
  startSoundUri: String?,
73
- endSoundUri: String?
73
+ endSoundUri: String?,
74
+ encoding: VoiceAudioEncoding?
74
75
  ): Promise<VoiceInputResult> {
75
76
  return Promise.async {
76
77
  if (Build.VERSION.SDK_INT < Build.VERSION_CODES.O) {
@@ -89,6 +90,7 @@ class HybridVoice : HybridVoiceSpec() {
89
90
  language = language,
90
91
  startSoundUri = startSoundUri,
91
92
  endSoundUri = endSoundUri,
93
+ encoding = encoding ?: VoiceAudioEncoding.LINEAR16,
92
94
  )
93
95
  } finally {
94
96
  voiceInputManager = null
@@ -27,6 +27,7 @@ import androidx.core.net.toUri
27
27
  import com.facebook.react.bridge.UiThreadUtil
28
28
  import com.margelo.nitro.NitroModules
29
29
  import com.margelo.nitro.core.ArrayBuffer
30
+ import com.margelo.nitro.swe.iternio.reactnativeautoplay.utils.G711
30
31
  import com.margelo.nitro.swe.iternio.reactnativeautoplay.utils.ThreadUtil
31
32
  import kotlinx.coroutines.CoroutineScope
32
33
  import kotlinx.coroutines.Dispatchers
@@ -81,6 +82,7 @@ class VoiceInputManager(
81
82
  language: String? = null,
82
83
  startSoundUri: String? = null,
83
84
  endSoundUri: String? = null,
85
+ encoding: VoiceAudioEncoding = VoiceAudioEncoding.LINEAR16,
84
86
  ): VoiceInputResult {
85
87
  cancelledByUser = false
86
88
  if (!requestAudioFocus()) {
@@ -93,19 +95,19 @@ class VoiceInputManager(
93
95
  if (SpeechRecognizer.isRecognitionAvailable(context)) {
94
96
  if (carContext != null) {
95
97
  if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.TIRAMISU) {
96
- startSTTFromCarAudio(silenceThresholdMs, maxDurationMs, onChunk, language)
98
+ startSTTFromCarAudio(silenceThresholdMs, maxDurationMs, encoding, onChunk, language)
97
99
  } else {
98
100
  // Car connected but API < 33: EXTRA_AUDIO_SOURCE unavailable, fall back to PCM
99
- startPCM(silenceThresholdMs, maxDurationMs, onChunk)
101
+ startPCM(silenceThresholdMs, maxDurationMs, encoding, onChunk)
100
102
  }
101
103
  } else {
102
104
  ThreadUtil.postOnUiAndAwait { startSTT(context, onChunk, language) }.getOrThrow()
103
105
  }
104
106
  } else {
105
- startPCM(silenceThresholdMs, maxDurationMs, onChunk)
107
+ startPCM(silenceThresholdMs, maxDurationMs, encoding, onChunk)
106
108
  }
107
109
  } else {
108
- startPCM(silenceThresholdMs, maxDurationMs, onChunk)
110
+ startPCM(silenceThresholdMs, maxDurationMs, encoding, onChunk)
109
111
  }
110
112
  startSoundJob?.join()
111
113
  if (cancelledByUser) throw VoiceInputCancelledException()
@@ -182,6 +184,7 @@ class VoiceInputManager(
182
184
  private suspend fun startSTTFromCarAudio(
183
185
  silenceThresholdMs: Long,
184
186
  maxDurationMs: Long,
187
+ encoding: VoiceAudioEncoding,
185
188
  onChunk: ((chunk: VoiceInputChunk) -> Unit)?,
186
189
  language: String?
187
190
  ): VoiceInputResult {
@@ -225,7 +228,9 @@ class VoiceInputManager(
225
228
  return try {
226
229
  sttDeferred.await()
227
230
  } catch (_: Exception) {
228
- val directBuffer = ByteBuffer.allocateDirect(pcmBytes.size).put(pcmBytes).rewind() as ByteBuffer
231
+ val encodedBytes = encodeBytes(pcmBytes, encoding)
232
+ val directBuffer =
233
+ ByteBuffer.allocateDirect(encodedBytes.size).put(encodedBytes).rewind() as ByteBuffer
229
234
  VoiceInputResult(transcription = null, audio = ArrayBuffer.wrap(directBuffer))
230
235
  }
231
236
  }
@@ -315,14 +320,36 @@ class VoiceInputManager(
315
320
  private suspend fun startPCM(
316
321
  silenceThresholdMs: Long,
317
322
  maxDurationMs: Long,
323
+ encoding: VoiceAudioEncoding,
318
324
  onChunk: ((chunk: VoiceInputChunk) -> Unit)?,
319
325
  ): VoiceInputResult {
320
- val pcmBytes = recordPCM(silenceThresholdMs, maxDurationMs, onChunk)
326
+ // Only the plain-PCM result/onChunk are re-encoded — STT paths (startSTT,
327
+ // startSTTFromCarAudio) always stay LINEAR16 since they feed the recognizer directly.
328
+ val wrappedOnChunk: ((VoiceInputChunk) -> Unit)? = onChunk?.let { cb ->
329
+ { chunk -> cb(encodeChunk(chunk, encoding)) }
330
+ }
331
+ val pcmBytes = recordPCM(silenceThresholdMs, maxDurationMs, wrappedOnChunk)
332
+ val encodedBytes = encodeBytes(pcmBytes, encoding)
321
333
  val directBuffer =
322
- ByteBuffer.allocateDirect(pcmBytes.size).put(pcmBytes).rewind() as ByteBuffer
334
+ ByteBuffer.allocateDirect(encodedBytes.size).put(encodedBytes).rewind() as ByteBuffer
323
335
  return VoiceInputResult(transcription = null, audio = ArrayBuffer.wrap(directBuffer))
324
336
  }
325
337
 
338
+ private fun encodeBytes(pcm16le: ByteArray, encoding: VoiceAudioEncoding): ByteArray =
339
+ when (encoding) {
340
+ VoiceAudioEncoding.LINEAR16 -> pcm16le
341
+ VoiceAudioEncoding.MULAW -> G711.encodeUlaw(pcm16le)
342
+ VoiceAudioEncoding.ALAW -> G711.encodeAlaw(pcm16le)
343
+ }
344
+
345
+ private fun encodeChunk(chunk: VoiceInputChunk, encoding: VoiceAudioEncoding): VoiceInputChunk {
346
+ val audio = chunk.audio
347
+ if (audio == null || encoding == VoiceAudioEncoding.LINEAR16) return chunk
348
+ val encoded = encodeBytes(audio.toByteArray(), encoding)
349
+ val direct = ByteBuffer.allocateDirect(encoded.size).put(encoded).rewind() as ByteBuffer
350
+ return VoiceInputChunk(partial = chunk.partial, audio = ArrayBuffer.wrap(direct))
351
+ }
352
+
326
353
  @SuppressLint("MissingPermission")
327
354
  @RequiresApi(Build.VERSION_CODES.O)
328
355
  private suspend fun recordPCM(
@@ -365,6 +392,16 @@ class VoiceInputManager(
365
392
  }
366
393
 
367
394
  val outputStream = ByteArrayOutputStream()
395
+ val pendingChunk = onChunk?.let { ByteArrayOutputStream() }
396
+
397
+ fun flushPendingChunk() {
398
+ val pending = pendingChunk ?: return
399
+ if (pending.size() == 0) return
400
+ val bytes = pending.toByteArray()
401
+ pending.reset()
402
+ val direct = ByteBuffer.allocateDirect(bytes.size).put(bytes).rewind() as ByteBuffer
403
+ onChunk?.invoke(VoiceInputChunk(partial = null, audio = ArrayBuffer.wrap(direct)))
404
+ }
368
405
 
369
406
  recordingJob = scope.launch {
370
407
  val buffer = ByteArray(bufferSize)
@@ -380,19 +417,22 @@ class VoiceInputManager(
380
417
  ) ?: -1
381
418
 
382
419
  if (read < 0) {
383
- // Whenever the user dismisses the microphone on the car screen, the next call to read will return -1
384
- cancelledByUser = carAudioRecord != null && read == -1
420
+ // Whenever the user dismisses the microphone on the car screen, the next call to read will return -1.
421
+ // But calling stop() ourselves also races CarAudioRecord's internal stream close against an
422
+ // in-flight read(), which can likewise surface as -1 — isRecording is already false by then
423
+ // (set before stopRecording() is called), so it distinguishes the two cases.
424
+ cancelledByUser = carAudioRecord != null && read == -1 && isRecording
385
425
  break
386
426
  }
387
427
 
388
428
  if (read > 0) {
389
429
  outputStream.write(buffer, 0, read)
390
430
 
391
- onChunk?.let { cb ->
392
- val chunk = ByteArray(read) { buffer[it] }
393
- val direct =
394
- ByteBuffer.allocateDirect(read).put(chunk).rewind() as ByteBuffer
395
- cb(VoiceInputChunk(partial = null, audio = ArrayBuffer.wrap(direct)))
431
+ pendingChunk?.let { pending ->
432
+ pending.write(buffer, 0, read)
433
+ if (pending.size() >= CHUNK_EMIT_BYTES) {
434
+ flushPendingChunk()
435
+ }
396
436
  }
397
437
 
398
438
  val now = System.currentTimeMillis()
@@ -430,6 +470,7 @@ class VoiceInputManager(
430
470
  }
431
471
  }
432
472
  } finally {
473
+ flushPendingChunk()
433
474
  releaseResources()
434
475
  val captured = pcmContinuation
435
476
  pcmContinuation = null
@@ -439,6 +480,10 @@ class VoiceInputManager(
439
480
  }
440
481
 
441
482
  fun stop() {
483
+ // Mark as no longer recording before triggering any teardown side effects, so anything
484
+ // racing against a concurrent read() sees this is an app-initiated stop.
485
+ isRecording = false
486
+
442
487
  // STT path: stopListening() triggers onResults/onError which resolves the continuation
443
488
  activeSpeechRecognizer?.let { recognizer ->
444
489
  UiThreadUtil.runOnUiThread {
@@ -446,7 +491,6 @@ class VoiceInputManager(
446
491
  }
447
492
  }
448
493
  // PCM path and car-audio STT pump
449
- isRecording = false
450
494
  carAudioRecord?.stopRecording()
451
495
  audioRecord?.stop()
452
496
  }
@@ -552,6 +596,7 @@ class VoiceInputManager(
552
596
  private const val WARMUP_MS = 500L
553
597
  private const val SAMPLE_RATE = 16_000
554
598
  private const val PHONE_BUFFER_SIZE = 3_200 // ~100ms at 16kHz/16-bit/mono
599
+ private const val CHUNK_EMIT_BYTES = 3_200 // ~100ms at 16kHz/16-bit/mono, batches onChunk callbacks
555
600
 
556
601
  fun hasVoiceInputPermission(): Boolean {
557
602
  val context = NitroModules.applicationContext ?: return false
@@ -0,0 +1,77 @@
1
+ package com.margelo.nitro.swe.iternio.reactnativeautoplay.utils
2
+
3
+ /** Encodes 16-bit signed little-endian PCM to G.711 µ-law/A-law (ITU-T reference algorithm). */
4
+ object G711 {
5
+ private const val ULAW_BIAS = 0x84
6
+ private const val ULAW_CLIP = 8159
7
+
8
+ private val ULAW_EXP_LUT = intArrayOf(
9
+ 0, 1, 2, 2, 3, 3, 3, 3, 4, 4, 4, 4, 4, 4, 4, 4,
10
+ 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5,
11
+ 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6,
12
+ 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6,
13
+ 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
14
+ 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
15
+ 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
16
+ 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
17
+ )
18
+
19
+ private val ALAW_SEG_END = intArrayOf(0x1F, 0x3F, 0x7F, 0xFF, 0x1FF, 0x3FF, 0x7FF, 0xFFF)
20
+
21
+ private fun search(value: Int, table: IntArray): Int {
22
+ for (i in table.indices) {
23
+ if (value <= table[i]) return i
24
+ }
25
+ return table.size
26
+ }
27
+
28
+ private fun linearToUlaw(sampleIn: Int): Byte {
29
+ var sample = sampleIn
30
+ val sign = (sample shr 8) and 0x80
31
+ if (sign != 0) sample = -sample
32
+ if (sample > ULAW_CLIP) sample = ULAW_CLIP
33
+ sample += ULAW_BIAS
34
+ val exponent = ULAW_EXP_LUT[(sample shr 7) and 0xFF]
35
+ val mantissa = (sample shr (exponent + 3)) and 0x0F
36
+ var ulawByte = (sign or (exponent shl 4) or mantissa).inv()
37
+ if (ulawByte == 0) ulawByte = 0x02
38
+ return ulawByte.toByte()
39
+ }
40
+
41
+ private fun linearToAlaw(sampleIn: Int): Byte {
42
+ var sample = sampleIn shr 3
43
+ val mask: Int
44
+ if (sample >= 0) {
45
+ mask = 0xD5
46
+ } else {
47
+ mask = 0x55
48
+ sample = -sample - 1
49
+ }
50
+
51
+ val seg = search(sample, ALAW_SEG_END)
52
+ val alawByte = if (seg >= 8) {
53
+ 0x7F xor mask
54
+ } else {
55
+ var aval = seg shl 4
56
+ aval = if (seg < 2) aval or ((sample shr 1) and 0x0F) else aval or ((sample shr seg) and 0x0F)
57
+ aval xor mask
58
+ }
59
+ return alawByte.toByte()
60
+ }
61
+
62
+ /** Encodes 16-bit signed little-endian PCM bytes to 8-bit G.711. Drops a trailing odd byte, if any. */
63
+ fun encode(pcm16le: ByteArray, encoding: (Int) -> Byte): ByteArray {
64
+ val sampleCount = pcm16le.size / 2
65
+ val out = ByteArray(sampleCount)
66
+ for (i in 0 until sampleCount) {
67
+ val lo = pcm16le[i * 2].toInt() and 0xFF
68
+ val hi = pcm16le[i * 2 + 1].toInt()
69
+ val sample = (hi shl 8) or lo
70
+ out[i] = encoding(sample)
71
+ }
72
+ return out
73
+ }
74
+
75
+ fun encodeUlaw(pcm16le: ByteArray): ByteArray = encode(pcm16le, ::linearToUlaw)
76
+ fun encodeAlaw(pcm16le: ByteArray): ByteArray = encode(pcm16le, ::linearToAlaw)
77
+ }
@@ -38,7 +38,8 @@ class HybridVoice: HybridVoiceSpec {
38
38
  onChunk: ((_ chunk: VoiceInputChunk) -> Void)?,
39
39
  language: String?,
40
40
  startSoundUri: String?,
41
- endSoundUri: String?
41
+ endSoundUri: String?,
42
+ encoding: VoiceAudioEncoding?
42
43
  ) throws -> Promise<VoiceInputResult> {
43
44
  return Promise.async {
44
45
  let interfaceController = try? await RootModule.withInterfaceController { $0 }
@@ -59,7 +60,8 @@ class HybridVoice: HybridVoiceSpec {
59
60
  onChunk: onChunk,
60
61
  language: language,
61
62
  startSoundUri: startSoundUri,
62
- endSoundUri: endSoundUri
63
+ endSoundUri: endSoundUri,
64
+ encoding: encoding ?? .linear16
63
65
  )
64
66
  }
65
67
  }
@@ -0,0 +1,83 @@
1
+ import Foundation
2
+
3
+ /// Encodes 16-bit signed little-endian PCM to G.711 µ-law/A-law (ITU-T reference algorithm).
4
+ enum G711 {
5
+ private static let ulawBias = 0x84
6
+ private static let ulawClip = 8159
7
+
8
+ private static let ulawExpLut: [Int] = [
9
+ 0, 1, 2, 2, 3, 3, 3, 3, 4, 4, 4, 4, 4, 4, 4, 4,
10
+ 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5,
11
+ 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6,
12
+ 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6,
13
+ 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
14
+ 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
15
+ 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
16
+ 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
17
+ ]
18
+
19
+ private static let alawSegEnd = [0x1F, 0x3F, 0x7F, 0xFF, 0x1FF, 0x3FF, 0x7FF, 0xFFF]
20
+
21
+ private static func search(_ value: Int, _ table: [Int]) -> Int {
22
+ for (i, boundary) in table.enumerated() where value <= boundary {
23
+ return i
24
+ }
25
+ return table.count
26
+ }
27
+
28
+ private static func linearToUlaw(_ sampleIn: Int) -> UInt8 {
29
+ var sample = sampleIn
30
+ let sign = (sample >> 8) & 0x80
31
+ if sign != 0 { sample = -sample }
32
+ if sample > ulawClip { sample = ulawClip }
33
+ sample += ulawBias
34
+ let exponent = ulawExpLut[(sample >> 7) & 0xFF]
35
+ let mantissa = (sample >> (exponent + 3)) & 0x0F
36
+ var ulawByte = ~(sign | (exponent << 4) | mantissa)
37
+ ulawByte &= 0xFF
38
+ if ulawByte == 0 { ulawByte = 0x02 }
39
+ return UInt8(ulawByte)
40
+ }
41
+
42
+ private static func linearToAlaw(_ sampleIn: Int) -> UInt8 {
43
+ var sample = sampleIn >> 3
44
+ let mask: Int
45
+ if sample >= 0 {
46
+ mask = 0xD5
47
+ }
48
+ else {
49
+ mask = 0x55
50
+ sample = -sample - 1
51
+ }
52
+
53
+ let seg = search(sample, alawSegEnd)
54
+ let alawByte: Int
55
+ if seg >= 8 {
56
+ alawByte = 0x7F ^ mask
57
+ }
58
+ else {
59
+ var aval = seg << 4
60
+ aval |= seg < 2 ? (sample >> 1) & 0x0F : (sample >> seg) & 0x0F
61
+ alawByte = aval ^ mask
62
+ }
63
+ return UInt8(alawByte & 0xFF)
64
+ }
65
+
66
+ /// Encodes 16-bit signed little-endian PCM data to 8-bit G.711. Drops a trailing odd byte, if any.
67
+ private static func encode(_ pcm16le: Data, _ encoding: (Int) -> UInt8) -> Data {
68
+ let sampleCount = pcm16le.count / 2
69
+ var out = [UInt8](repeating: 0, count: sampleCount)
70
+ pcm16le.withUnsafeBytes { (raw: UnsafeRawBufferPointer) in
71
+ for i in 0..<sampleCount {
72
+ let lo = Int(raw[i * 2])
73
+ let hi = Int(Int8(bitPattern: raw[i * 2 + 1]))
74
+ let sample = (hi << 8) | lo
75
+ out[i] = encoding(sample)
76
+ }
77
+ }
78
+ return Data(out)
79
+ }
80
+
81
+ static func encodeUlaw(_ pcm16le: Data) -> Data { encode(pcm16le, linearToUlaw) }
82
+ static func encodeAlaw(_ pcm16le: Data) -> Data { encode(pcm16le, linearToAlaw) }
83
+ }
@@ -69,6 +69,9 @@ class VoiceInputManager {
69
69
  private var silenceStart: Date?
70
70
  private var firstBufferContinuation: CheckedContinuation<Void, Never>?
71
71
 
72
+ // PCM result/onChunk audio encoding — STT transcription itself is unaffected
73
+ private var encoding: VoiceAudioEncoding = .linear16
74
+
72
75
  private static let sampleRate: Double = 16_000
73
76
  private static let tapBufferSize: AVAudioFrameCount = 4_096
74
77
  private static let silenceAmplitudeThreshold = 500
@@ -94,8 +97,10 @@ class VoiceInputManager {
94
97
  onChunk: ((_ chunk: VoiceInputChunk) -> Void)?,
95
98
  language: String?,
96
99
  startSoundUri: String?,
97
- endSoundUri: String?
100
+ endSoundUri: String?,
101
+ encoding: VoiceAudioEncoding
98
102
  ) async throws -> VoiceInputResult {
103
+ self.encoding = encoding
99
104
  stopLock.withLock {
100
105
  cancelledByUser = false
101
106
  }
@@ -355,9 +360,8 @@ class VoiceInputManager {
355
360
 
356
361
  // PCM chunk callback
357
362
  if activeRecognitionRequest == nil, let onChunk {
358
- if let chunkBuffer = try? ArrayBuffer.copy(
359
- data: newSamples.withUnsafeBufferPointer { Data(buffer: $0) }
360
- ) {
363
+ let pcmData = newSamples.withUnsafeBufferPointer { Data(buffer: $0) }
364
+ if let chunkBuffer = try? ArrayBuffer.copy(data: self.encodeAudio(pcmData)) {
361
365
  onChunk(VoiceInputChunk(partial: nil, audio: chunkBuffer))
362
366
  }
363
367
  }
@@ -424,10 +428,18 @@ class VoiceInputManager {
424
428
 
425
429
  private func makePCMResult(from samples: [Int16]) -> VoiceInputResult {
426
430
  let data = samples.withUnsafeBufferPointer { Data(buffer: $0) }
427
- let buffer = try? ArrayBuffer.copy(data: data)
431
+ let buffer = try? ArrayBuffer.copy(data: encodeAudio(data))
428
432
  return VoiceInputResult(transcription: nil, audio: buffer)
429
433
  }
430
434
 
435
+ private func encodeAudio(_ pcm16le: Data) -> Data {
436
+ switch encoding {
437
+ case .linear16: return pcm16le
438
+ case .mulaw: return G711.encodeUlaw(pcm16le)
439
+ case .alaw: return G711.encodeAlaw(pcm16le)
440
+ }
441
+ }
442
+
431
443
  // CPVoiceControlState enforces a maximum image size of 150x150 points.
432
444
  private static let voiceImageMaxSize = CGSize(width: 150, height: 150)
433
445
 
@@ -39,6 +39,7 @@ export declare const HybridVoice: {
39
39
  * @param preferSpeechToText request STT transcription instead of raw PCM
40
40
  * @param onChunk optional streaming callback
41
41
  * @param language specify the language for the SpeechRecognizer, falls back to system language if not set
42
+ * @param encoding PCM encoding for onChunk audio and the final result (default LINEAR16)
42
43
  */
43
44
  startVoiceInput: StartVoiceInput;
44
45
  /**
@@ -3,11 +3,11 @@ import { NitroModules } from 'react-native-nitro-modules';
3
3
  import { NitroImageUtil } from '../utils/NitroImage';
4
4
  const _native = NitroModules.createHybridObject('Voice');
5
5
  const startVoiceInput = async (options) => {
6
- const { onChunk, silenceThresholdMs, maxDurationMs, listeningText, listeningImage, preferSpeechToText, language, startSound, endSound, } = options ?? {};
6
+ const { onChunk, silenceThresholdMs, maxDurationMs, listeningText, listeningImage, preferSpeechToText, language, startSound, endSound, encoding, } = options ?? {};
7
7
  const listeningImageRepeats = listeningImage?.type === 'asset' ? listeningImage.repeats : undefined;
8
8
  const startSoundUri = startSound != null ? Image.resolveAssetSource(startSound).uri : undefined;
9
9
  const endSoundUri = endSound != null ? Image.resolveAssetSource(endSound).uri : undefined;
10
- return await _native.startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, NitroImageUtil.convert(listeningImage), listeningImageRepeats, preferSpeechToText, onChunk, language, startSoundUri, endSoundUri);
10
+ return await _native.startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, NitroImageUtil.convert(listeningImage), listeningImageRepeats, preferSpeechToText, onChunk, language, startSoundUri, endSoundUri, encoding);
11
11
  };
12
12
  export const HybridVoice = {
13
13
  /**
@@ -45,6 +45,7 @@ export const HybridVoice = {
45
45
  * @param preferSpeechToText request STT transcription instead of raw PCM
46
46
  * @param onChunk optional streaming callback
47
47
  * @param language specify the language for the SpeechRecognizer, falls back to system language if not set
48
+ * @param encoding PCM encoding for onChunk audio and the final result (default LINEAR16)
48
49
  */
49
50
  startVoiceInput,
50
51
  /**
@@ -1,5 +1,5 @@
1
1
  import type { HybridObject } from 'react-native-nitro-modules';
2
- import type { VoiceInputChunk, VoiceInputResult } from '../types/Voice';
2
+ import type { VoiceAudioEncoding, VoiceInputChunk, VoiceInputResult } from '../types/Voice';
3
3
  import type { NitroImage } from '../utils/NitroImage';
4
4
  export interface Voice extends HybridObject<{
5
5
  android: 'kotlin';
@@ -7,6 +7,6 @@ export interface Voice extends HybridObject<{
7
7
  }> {
8
8
  hasVoiceInputPermission(): boolean;
9
9
  requestVoiceInputPermission(): Promise<boolean>;
10
- startVoiceInput(silenceThresholdMs?: number, maxDurationMs?: number, listeningText?: string, listeningImage?: NitroImage, listeningImageRepeats?: boolean, preferSpeechToText?: boolean, onChunk?: (chunk: VoiceInputChunk) => void, language?: string, startSoundUri?: string, endSoundUri?: string): Promise<VoiceInputResult>;
10
+ startVoiceInput(silenceThresholdMs?: number, maxDurationMs?: number, listeningText?: string, listeningImage?: NitroImage, listeningImageRepeats?: boolean, preferSpeechToText?: boolean, onChunk?: (chunk: VoiceInputChunk) => void, language?: string, startSoundUri?: string, endSoundUri?: string, encoding?: VoiceAudioEncoding): Promise<VoiceInputResult>;
11
11
  stopVoiceInput(): void;
12
12
  }
@@ -1,4 +1,8 @@
1
1
  import type { VoiceInputImage } from './Image';
2
+ /** PCM sample encoding for the `audio` payload of chunks and the final result.
3
+ * `LINEAR16` is raw 16-bit signed PCM
4
+ * `MULAW`/`ALAW` are G.711 8-bit companded encodings, halving payload size */
5
+ export type VoiceAudioEncoding = 'LINEAR16' | 'MULAW' | 'ALAW';
2
6
  export interface VoiceInputChunk {
3
7
  partial?: string;
4
8
  audio?: ArrayBuffer;
@@ -18,6 +22,8 @@ export interface VoiceInputOptions {
18
22
  preferSpeechToText?: boolean;
19
23
  onChunk?: (chunk: VoiceInputChunk) => void;
20
24
  language?: string;
25
+ /** PCM encoding for onChunk audio and the final result. Defaults to LINEAR16. */
26
+ encoding?: VoiceAudioEncoding;
21
27
  /** Sound played just before recording starts. Pass a Metro asset: `require('./beep_start.wav')`. */
22
28
  startSound?: number;
23
29
  /** Sound played just after recording stops. Pass a Metro asset: `require('./beep_end.wav')`. */
@@ -19,6 +19,8 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct RemoteImage
19
19
  namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct NitroColor; }
20
20
  // Forward declaration of `VoiceInputChunk` to properly resolve imports.
21
21
  namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct VoiceInputChunk; }
22
+ // Forward declaration of `VoiceAudioEncoding` to properly resolve imports.
23
+ namespace margelo::nitro::swe::iternio::reactnativeautoplay { enum class VoiceAudioEncoding; }
22
24
 
23
25
  #include <NitroModules/Promise.hpp>
24
26
  #include <NitroModules/JPromise.hpp>
@@ -43,6 +45,8 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct VoiceInputC
43
45
  #include "JFunc_void_VoiceInputChunk.hpp"
44
46
  #include <NitroModules/JNICallable.hpp>
45
47
  #include "JVoiceInputChunk.hpp"
48
+ #include "VoiceAudioEncoding.hpp"
49
+ #include "JVoiceAudioEncoding.hpp"
46
50
 
47
51
  namespace margelo::nitro::swe::iternio::reactnativeautoplay {
48
52
 
@@ -98,9 +102,9 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
98
102
  return __promise;
99
103
  }();
100
104
  }
101
- std::shared_ptr<Promise<VoiceInputResult>> JHybridVoiceSpec::startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri) {
102
- static const auto method = _javaPart->javaClassStatic()->getMethod<jni::local_ref<JPromise::javaobject>(jni::alias_ref<jni::JDouble> /* silenceThresholdMs */, jni::alias_ref<jni::JDouble> /* maxDurationMs */, jni::alias_ref<jni::JString> /* listeningText */, jni::alias_ref<JVariant_GlyphImage_AssetImage_RemoteImage> /* listeningImage */, jni::alias_ref<jni::JBoolean> /* listeningImageRepeats */, jni::alias_ref<jni::JBoolean> /* preferSpeechToText */, jni::alias_ref<JFunc_void_VoiceInputChunk::javaobject> /* onChunk */, jni::alias_ref<jni::JString> /* language */, jni::alias_ref<jni::JString> /* startSoundUri */, jni::alias_ref<jni::JString> /* endSoundUri */)>("startVoiceInput_cxx");
103
- auto __result = method(_javaPart, silenceThresholdMs.has_value() ? jni::JDouble::valueOf(silenceThresholdMs.value()) : nullptr, maxDurationMs.has_value() ? jni::JDouble::valueOf(maxDurationMs.value()) : nullptr, listeningText.has_value() ? jni::make_jstring(listeningText.value()) : nullptr, listeningImage.has_value() ? JVariant_GlyphImage_AssetImage_RemoteImage::fromCpp(listeningImage.value()) : nullptr, listeningImageRepeats.has_value() ? jni::JBoolean::valueOf(listeningImageRepeats.value()) : nullptr, preferSpeechToText.has_value() ? jni::JBoolean::valueOf(preferSpeechToText.value()) : nullptr, onChunk.has_value() ? JFunc_void_VoiceInputChunk_cxx::fromCpp(onChunk.value()) : nullptr, language.has_value() ? jni::make_jstring(language.value()) : nullptr, startSoundUri.has_value() ? jni::make_jstring(startSoundUri.value()) : nullptr, endSoundUri.has_value() ? jni::make_jstring(endSoundUri.value()) : nullptr);
105
+ std::shared_ptr<Promise<VoiceInputResult>> JHybridVoiceSpec::startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri, std::optional<VoiceAudioEncoding> encoding) {
106
+ static const auto method = _javaPart->javaClassStatic()->getMethod<jni::local_ref<JPromise::javaobject>(jni::alias_ref<jni::JDouble> /* silenceThresholdMs */, jni::alias_ref<jni::JDouble> /* maxDurationMs */, jni::alias_ref<jni::JString> /* listeningText */, jni::alias_ref<JVariant_GlyphImage_AssetImage_RemoteImage> /* listeningImage */, jni::alias_ref<jni::JBoolean> /* listeningImageRepeats */, jni::alias_ref<jni::JBoolean> /* preferSpeechToText */, jni::alias_ref<JFunc_void_VoiceInputChunk::javaobject> /* onChunk */, jni::alias_ref<jni::JString> /* language */, jni::alias_ref<jni::JString> /* startSoundUri */, jni::alias_ref<jni::JString> /* endSoundUri */, jni::alias_ref<JVoiceAudioEncoding> /* encoding */)>("startVoiceInput_cxx");
107
+ auto __result = method(_javaPart, silenceThresholdMs.has_value() ? jni::JDouble::valueOf(silenceThresholdMs.value()) : nullptr, maxDurationMs.has_value() ? jni::JDouble::valueOf(maxDurationMs.value()) : nullptr, listeningText.has_value() ? jni::make_jstring(listeningText.value()) : nullptr, listeningImage.has_value() ? JVariant_GlyphImage_AssetImage_RemoteImage::fromCpp(listeningImage.value()) : nullptr, listeningImageRepeats.has_value() ? jni::JBoolean::valueOf(listeningImageRepeats.value()) : nullptr, preferSpeechToText.has_value() ? jni::JBoolean::valueOf(preferSpeechToText.value()) : nullptr, onChunk.has_value() ? JFunc_void_VoiceInputChunk_cxx::fromCpp(onChunk.value()) : nullptr, language.has_value() ? jni::make_jstring(language.value()) : nullptr, startSoundUri.has_value() ? jni::make_jstring(startSoundUri.value()) : nullptr, endSoundUri.has_value() ? jni::make_jstring(endSoundUri.value()) : nullptr, encoding.has_value() ? JVoiceAudioEncoding::fromCpp(encoding.value()) : nullptr);
104
108
  return [&]() {
105
109
  auto __promise = Promise<VoiceInputResult>::create();
106
110
  __result->cthis()->addOnResolvedListener([=](const jni::alias_ref<jni::JObject>& __boxedResult) {
@@ -56,7 +56,7 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
56
56
  // Methods
57
57
  bool hasVoiceInputPermission() override;
58
58
  std::shared_ptr<Promise<bool>> requestVoiceInputPermission() override;
59
- std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri) override;
59
+ std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri, std::optional<VoiceAudioEncoding> encoding) override;
60
60
  void stopVoiceInput() override;
61
61
 
62
62
  private:
@@ -0,0 +1,61 @@
1
+ ///
2
+ /// JVoiceAudioEncoding.hpp
3
+ /// This file was generated by nitrogen. DO NOT MODIFY THIS FILE.
4
+ /// https://github.com/mrousavy/nitro
5
+ /// Copyright © Marc Rousavy @ Margelo
6
+ ///
7
+
8
+ #pragma once
9
+
10
+ #include <fbjni/fbjni.h>
11
+ #include "VoiceAudioEncoding.hpp"
12
+
13
+ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
14
+
15
+ using namespace facebook;
16
+
17
+ /**
18
+ * The C++ JNI bridge between the C++ enum "VoiceAudioEncoding" and the the Kotlin enum "VoiceAudioEncoding".
19
+ */
20
+ struct JVoiceAudioEncoding final: public jni::JavaClass<JVoiceAudioEncoding> {
21
+ public:
22
+ static constexpr auto kJavaDescriptor = "Lcom/margelo/nitro/swe/iternio/reactnativeautoplay/VoiceAudioEncoding;";
23
+
24
+ public:
25
+ /**
26
+ * Convert this Java/Kotlin-based enum to the C++ enum VoiceAudioEncoding.
27
+ */
28
+ [[maybe_unused]]
29
+ [[nodiscard]]
30
+ VoiceAudioEncoding toCpp() const {
31
+ static const auto clazz = javaClassStatic();
32
+ static const auto fieldOrdinal = clazz->getField<int>("value");
33
+ int ordinal = this->getFieldValue(fieldOrdinal);
34
+ return static_cast<VoiceAudioEncoding>(ordinal);
35
+ }
36
+
37
+ public:
38
+ /**
39
+ * Create a Java/Kotlin-based enum with the given C++ enum's value.
40
+ */
41
+ [[maybe_unused]]
42
+ static jni::alias_ref<JVoiceAudioEncoding> fromCpp(VoiceAudioEncoding value) {
43
+ static const auto clazz = javaClassStatic();
44
+ switch (value) {
45
+ case VoiceAudioEncoding::LINEAR16:
46
+ static const auto fieldLINEAR16 = clazz->getStaticField<JVoiceAudioEncoding>("LINEAR16");
47
+ return clazz->getStaticFieldValue(fieldLINEAR16);
48
+ case VoiceAudioEncoding::MULAW:
49
+ static const auto fieldMULAW = clazz->getStaticField<JVoiceAudioEncoding>("MULAW");
50
+ return clazz->getStaticFieldValue(fieldMULAW);
51
+ case VoiceAudioEncoding::ALAW:
52
+ static const auto fieldALAW = clazz->getStaticField<JVoiceAudioEncoding>("ALAW");
53
+ return clazz->getStaticFieldValue(fieldALAW);
54
+ default:
55
+ std::string stringValue = std::to_string(static_cast<int>(value));
56
+ throw std::invalid_argument("Invalid enum value (" + stringValue + "!");
57
+ }
58
+ }
59
+ };
60
+
61
+ } // namespace margelo::nitro::swe::iternio::reactnativeautoplay
@@ -37,12 +37,12 @@ abstract class HybridVoiceSpec: HybridObject() {
37
37
  @Keep
38
38
  abstract fun requestVoiceInputPermission(): Promise<Boolean>
39
39
 
40
- abstract fun startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Boolean?, preferSpeechToText: Boolean?, onChunk: ((chunk: VoiceInputChunk) -> Unit)?, language: String?, startSoundUri: String?, endSoundUri: String?): Promise<VoiceInputResult>
40
+ abstract fun startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Boolean?, preferSpeechToText: Boolean?, onChunk: ((chunk: VoiceInputChunk) -> Unit)?, language: String?, startSoundUri: String?, endSoundUri: String?, encoding: VoiceAudioEncoding?): Promise<VoiceInputResult>
41
41
 
42
42
  @DoNotStrip
43
43
  @Keep
44
- private fun startVoiceInput_cxx(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Boolean?, preferSpeechToText: Boolean?, onChunk: Func_void_VoiceInputChunk?, language: String?, startSoundUri: String?, endSoundUri: String?): Promise<VoiceInputResult> {
45
- val __result = startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, listeningImage, listeningImageRepeats, preferSpeechToText, onChunk?.let { it }, language, startSoundUri, endSoundUri)
44
+ private fun startVoiceInput_cxx(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Boolean?, preferSpeechToText: Boolean?, onChunk: Func_void_VoiceInputChunk?, language: String?, startSoundUri: String?, endSoundUri: String?, encoding: VoiceAudioEncoding?): Promise<VoiceInputResult> {
45
+ val __result = startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, listeningImage, listeningImageRepeats, preferSpeechToText, onChunk?.let { it }, language, startSoundUri, endSoundUri, encoding)
46
46
  return __result
47
47
  }
48
48
 
@@ -0,0 +1,24 @@
1
+ ///
2
+ /// VoiceAudioEncoding.kt
3
+ /// This file was generated by nitrogen. DO NOT MODIFY THIS FILE.
4
+ /// https://github.com/mrousavy/nitro
5
+ /// Copyright © Marc Rousavy @ Margelo
6
+ ///
7
+
8
+ package com.margelo.nitro.swe.iternio.reactnativeautoplay
9
+
10
+ import androidx.annotation.Keep
11
+ import com.facebook.proguard.annotations.DoNotStrip
12
+
13
+ /**
14
+ * Represents the JavaScript enum/union "VoiceAudioEncoding".
15
+ */
16
+ @DoNotStrip
17
+ @Keep
18
+ enum class VoiceAudioEncoding(@DoNotStrip @Keep val value: Int) {
19
+ LINEAR16(0),
20
+ MULAW(1),
21
+ ALAW(2);
22
+
23
+ companion object
24
+ }
@@ -128,6 +128,8 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay { enum class TurnTyp
128
128
  namespace margelo::nitro::swe::iternio::reactnativeautoplay { enum class VisibilityState; }
129
129
  // Forward declaration of `VisibleTravelEstimate` to properly resolve imports.
130
130
  namespace margelo::nitro::swe::iternio::reactnativeautoplay { enum class VisibleTravelEstimate; }
131
+ // Forward declaration of `VoiceAudioEncoding` to properly resolve imports.
132
+ namespace margelo::nitro::swe::iternio::reactnativeautoplay { enum class VoiceAudioEncoding; }
131
133
  // Forward declaration of `VoiceInputChunk` to properly resolve imports.
132
134
  namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct VoiceInputChunk; }
133
135
  // Forward declaration of `VoiceInputResult` to properly resolve imports.
@@ -217,6 +219,7 @@ namespace ReactNativeAutoPlay { class HybridVoiceSpec_cxx; }
217
219
  #include "TurnType.hpp"
218
220
  #include "VisibilityState.hpp"
219
221
  #include "VisibleTravelEstimate.hpp"
222
+ #include "VoiceAudioEncoding.hpp"
220
223
  #include "VoiceInputChunk.hpp"
221
224
  #include "VoiceInputResult.hpp"
222
225
  #include "ZoomEvent.hpp"
@@ -1680,6 +1683,21 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay::bridge::swift {
1680
1683
  return optional.value();
1681
1684
  }
1682
1685
 
1686
+ // pragma MARK: std::optional<VoiceAudioEncoding>
1687
+ /**
1688
+ * Specialized version of `std::optional<VoiceAudioEncoding>`.
1689
+ */
1690
+ using std__optional_VoiceAudioEncoding_ = std::optional<VoiceAudioEncoding>;
1691
+ inline std::optional<VoiceAudioEncoding> create_std__optional_VoiceAudioEncoding_(const VoiceAudioEncoding& value) noexcept {
1692
+ return std::optional<VoiceAudioEncoding>(value);
1693
+ }
1694
+ inline bool has_value_std__optional_VoiceAudioEncoding_(const std::optional<VoiceAudioEncoding>& optional) noexcept {
1695
+ return optional.has_value();
1696
+ }
1697
+ inline VoiceAudioEncoding get_std__optional_VoiceAudioEncoding_(const std::optional<VoiceAudioEncoding>& optional) noexcept {
1698
+ return optional.value();
1699
+ }
1700
+
1683
1701
  // pragma MARK: std::shared_ptr<HybridVoiceSpec>
1684
1702
  /**
1685
1703
  * Specialized version of `std::shared_ptr<HybridVoiceSpec>`.
@@ -150,6 +150,8 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay { enum class TurnTyp
150
150
  namespace margelo::nitro::swe::iternio::reactnativeautoplay { enum class VisibilityState; }
151
151
  // Forward declaration of `VisibleTravelEstimate` to properly resolve imports.
152
152
  namespace margelo::nitro::swe::iternio::reactnativeautoplay { enum class VisibleTravelEstimate; }
153
+ // Forward declaration of `VoiceAudioEncoding` to properly resolve imports.
154
+ namespace margelo::nitro::swe::iternio::reactnativeautoplay { enum class VoiceAudioEncoding; }
153
155
  // Forward declaration of `VoiceInputChunk` to properly resolve imports.
154
156
  namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct VoiceInputChunk; }
155
157
  // Forward declaration of `VoiceInputResult` to properly resolve imports.
@@ -229,6 +231,7 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay { enum class ZoomEve
229
231
  #include "TurnType.hpp"
230
232
  #include "VisibilityState.hpp"
231
233
  #include "VisibleTravelEstimate.hpp"
234
+ #include "VoiceAudioEncoding.hpp"
232
235
  #include "VoiceInputChunk.hpp"
233
236
  #include "VoiceInputResult.hpp"
234
237
  #include "ZoomEvent.hpp"
@@ -26,6 +26,8 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct RemoteImage
26
26
  namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct NitroColor; }
27
27
  // Forward declaration of `VoiceInputChunk` to properly resolve imports.
28
28
  namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct VoiceInputChunk; }
29
+ // Forward declaration of `VoiceAudioEncoding` to properly resolve imports.
30
+ namespace margelo::nitro::swe::iternio::reactnativeautoplay { enum class VoiceAudioEncoding; }
29
31
 
30
32
  #include <NitroModules/Promise.hpp>
31
33
  #include "VoiceInputResult.hpp"
@@ -40,6 +42,7 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct VoiceInputC
40
42
  #include "NitroColor.hpp"
41
43
  #include "VoiceInputChunk.hpp"
42
44
  #include <functional>
45
+ #include "VoiceAudioEncoding.hpp"
43
46
 
44
47
  #include "ReactNativeAutoPlay-Swift-Cxx-Umbrella.hpp"
45
48
 
@@ -107,8 +110,8 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
107
110
  auto __value = std::move(__result.value());
108
111
  return __value;
109
112
  }
110
- inline std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri) override {
111
- auto __result = _swiftPart.startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, listeningImage, listeningImageRepeats, preferSpeechToText, onChunk, language, startSoundUri, endSoundUri);
113
+ inline std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri, std::optional<VoiceAudioEncoding> encoding) override {
114
+ auto __result = _swiftPart.startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, listeningImage, listeningImageRepeats, preferSpeechToText, onChunk, language, startSoundUri, endSoundUri, encoding);
112
115
  if (__result.hasError()) [[unlikely]] {
113
116
  std::rethrow_exception(__result.error());
114
117
  }
@@ -15,7 +15,7 @@ public protocol HybridVoiceSpec_protocol: HybridObject {
15
15
  // Methods
16
16
  func hasVoiceInputPermission() throws -> Bool
17
17
  func requestVoiceInputPermission() throws -> Promise<Bool>
18
- func startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Bool?, preferSpeechToText: Bool?, onChunk: ((_ chunk: VoiceInputChunk) -> Void)?, language: String?, startSoundUri: String?, endSoundUri: String?) throws -> Promise<VoiceInputResult>
18
+ func startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Bool?, preferSpeechToText: Bool?, onChunk: ((_ chunk: VoiceInputChunk) -> Void)?, language: String?, startSoundUri: String?, endSoundUri: String?, encoding: VoiceAudioEncoding?) throws -> Promise<VoiceInputResult>
19
19
  func stopVoiceInput() throws -> Void
20
20
  }
21
21
 
@@ -156,7 +156,7 @@ open class HybridVoiceSpec_cxx {
156
156
  }
157
157
 
158
158
  @inline(__always)
159
- public final func startVoiceInput(silenceThresholdMs: bridge.std__optional_double_, maxDurationMs: bridge.std__optional_double_, listeningText: bridge.std__optional_std__string_, listeningImage: bridge.std__optional_std__variant_GlyphImage__AssetImage__RemoteImage__, listeningImageRepeats: bridge.std__optional_bool_, preferSpeechToText: bridge.std__optional_bool_, onChunk: bridge.std__optional_std__function_void_const_VoiceInputChunk_____chunk______, language: bridge.std__optional_std__string_, startSoundUri: bridge.std__optional_std__string_, endSoundUri: bridge.std__optional_std__string_) -> bridge.Result_std__shared_ptr_Promise_VoiceInputResult___ {
159
+ public final func startVoiceInput(silenceThresholdMs: bridge.std__optional_double_, maxDurationMs: bridge.std__optional_double_, listeningText: bridge.std__optional_std__string_, listeningImage: bridge.std__optional_std__variant_GlyphImage__AssetImage__RemoteImage__, listeningImageRepeats: bridge.std__optional_bool_, preferSpeechToText: bridge.std__optional_bool_, onChunk: bridge.std__optional_std__function_void_const_VoiceInputChunk_____chunk______, language: bridge.std__optional_std__string_, startSoundUri: bridge.std__optional_std__string_, endSoundUri: bridge.std__optional_std__string_, encoding: bridge.std__optional_VoiceAudioEncoding_) -> bridge.Result_std__shared_ptr_Promise_VoiceInputResult___ {
160
160
  do {
161
161
  let __result = try self.__implementation.startVoiceInput(silenceThresholdMs: { () -> Double? in
162
162
  if bridge.has_value_std__optional_double_(silenceThresholdMs) {
@@ -248,7 +248,7 @@ open class HybridVoiceSpec_cxx {
248
248
  } else {
249
249
  return nil
250
250
  }
251
- }())
251
+ }(), encoding: encoding.value)
252
252
  let __resultCpp = { () -> bridge.std__shared_ptr_Promise_VoiceInputResult__ in
253
253
  let __promise = bridge.create_std__shared_ptr_Promise_VoiceInputResult__()
254
254
  let __promiseHolder = bridge.wrap_std__shared_ptr_Promise_VoiceInputResult__(__promise)
@@ -0,0 +1,44 @@
1
+ ///
2
+ /// VoiceAudioEncoding.swift
3
+ /// This file was generated by nitrogen. DO NOT MODIFY THIS FILE.
4
+ /// https://github.com/mrousavy/nitro
5
+ /// Copyright © Marc Rousavy @ Margelo
6
+ ///
7
+
8
+ /**
9
+ * Represents the JS union `VoiceAudioEncoding`, backed by a C++ enum.
10
+ */
11
+ public typealias VoiceAudioEncoding = margelo.nitro.swe.iternio.reactnativeautoplay.VoiceAudioEncoding
12
+
13
+ public extension VoiceAudioEncoding {
14
+ /**
15
+ * Get a VoiceAudioEncoding for the given String value, or
16
+ * return `nil` if the given value was invalid/unknown.
17
+ */
18
+ init?(fromString string: String) {
19
+ switch string {
20
+ case "LINEAR16":
21
+ self = .linear16
22
+ case "MULAW":
23
+ self = .mulaw
24
+ case "ALAW":
25
+ self = .alaw
26
+ default:
27
+ return nil
28
+ }
29
+ }
30
+
31
+ /**
32
+ * Get the String value this VoiceAudioEncoding represents.
33
+ */
34
+ var stringValue: String {
35
+ switch self {
36
+ case .linear16:
37
+ return "LINEAR16"
38
+ case .mulaw:
39
+ return "MULAW"
40
+ case .alaw:
41
+ return "ALAW"
42
+ }
43
+ }
44
+ }
@@ -23,6 +23,8 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct AssetImage;
23
23
  namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct RemoteImage; }
24
24
  // Forward declaration of `VoiceInputChunk` to properly resolve imports.
25
25
  namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct VoiceInputChunk; }
26
+ // Forward declaration of `VoiceAudioEncoding` to properly resolve imports.
27
+ namespace margelo::nitro::swe::iternio::reactnativeautoplay { enum class VoiceAudioEncoding; }
26
28
 
27
29
  #include <NitroModules/Promise.hpp>
28
30
  #include "VoiceInputResult.hpp"
@@ -34,6 +36,7 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct VoiceInputC
34
36
  #include <variant>
35
37
  #include "VoiceInputChunk.hpp"
36
38
  #include <functional>
39
+ #include "VoiceAudioEncoding.hpp"
37
40
 
38
41
  namespace margelo::nitro::swe::iternio::reactnativeautoplay {
39
42
 
@@ -68,7 +71,7 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
68
71
  // Methods
69
72
  virtual bool hasVoiceInputPermission() = 0;
70
73
  virtual std::shared_ptr<Promise<bool>> requestVoiceInputPermission() = 0;
71
- virtual std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri) = 0;
74
+ virtual std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri, std::optional<VoiceAudioEncoding> encoding) = 0;
72
75
  virtual void stopVoiceInput() = 0;
73
76
 
74
77
  protected:
@@ -0,0 +1,80 @@
1
+ ///
2
+ /// VoiceAudioEncoding.hpp
3
+ /// This file was generated by nitrogen. DO NOT MODIFY THIS FILE.
4
+ /// https://github.com/mrousavy/nitro
5
+ /// Copyright © Marc Rousavy @ Margelo
6
+ ///
7
+
8
+ #pragma once
9
+
10
+ #if __has_include(<NitroModules/NitroHash.hpp>)
11
+ #include <NitroModules/NitroHash.hpp>
12
+ #else
13
+ #error NitroModules cannot be found! Are you sure you installed NitroModules properly?
14
+ #endif
15
+ #if __has_include(<NitroModules/JSIConverter.hpp>)
16
+ #include <NitroModules/JSIConverter.hpp>
17
+ #else
18
+ #error NitroModules cannot be found! Are you sure you installed NitroModules properly?
19
+ #endif
20
+ #if __has_include(<NitroModules/NitroDefines.hpp>)
21
+ #include <NitroModules/NitroDefines.hpp>
22
+ #else
23
+ #error NitroModules cannot be found! Are you sure you installed NitroModules properly?
24
+ #endif
25
+
26
+ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
27
+
28
+ /**
29
+ * An enum which can be represented as a JavaScript union (VoiceAudioEncoding).
30
+ */
31
+ enum class VoiceAudioEncoding {
32
+ LINEAR16 SWIFT_NAME(linear16) = 0,
33
+ MULAW SWIFT_NAME(mulaw) = 1,
34
+ ALAW SWIFT_NAME(alaw) = 2,
35
+ } CLOSED_ENUM;
36
+
37
+ } // namespace margelo::nitro::swe::iternio::reactnativeautoplay
38
+
39
+ namespace margelo::nitro {
40
+
41
+ // C++ VoiceAudioEncoding <> JS VoiceAudioEncoding (union)
42
+ template <>
43
+ struct JSIConverter<margelo::nitro::swe::iternio::reactnativeautoplay::VoiceAudioEncoding> final {
44
+ static inline margelo::nitro::swe::iternio::reactnativeautoplay::VoiceAudioEncoding fromJSI(jsi::Runtime& runtime, const jsi::Value& arg) {
45
+ std::string unionValue = JSIConverter<std::string>::fromJSI(runtime, arg);
46
+ switch (hashString(unionValue.c_str(), unionValue.size())) {
47
+ case hashString("LINEAR16"): return margelo::nitro::swe::iternio::reactnativeautoplay::VoiceAudioEncoding::LINEAR16;
48
+ case hashString("MULAW"): return margelo::nitro::swe::iternio::reactnativeautoplay::VoiceAudioEncoding::MULAW;
49
+ case hashString("ALAW"): return margelo::nitro::swe::iternio::reactnativeautoplay::VoiceAudioEncoding::ALAW;
50
+ default: [[unlikely]]
51
+ throw std::invalid_argument("Cannot convert \"" + unionValue + "\" to enum VoiceAudioEncoding - invalid value!");
52
+ }
53
+ }
54
+ static inline jsi::Value toJSI(jsi::Runtime& runtime, margelo::nitro::swe::iternio::reactnativeautoplay::VoiceAudioEncoding arg) {
55
+ switch (arg) {
56
+ case margelo::nitro::swe::iternio::reactnativeautoplay::VoiceAudioEncoding::LINEAR16: return JSIConverter<std::string>::toJSI(runtime, "LINEAR16");
57
+ case margelo::nitro::swe::iternio::reactnativeautoplay::VoiceAudioEncoding::MULAW: return JSIConverter<std::string>::toJSI(runtime, "MULAW");
58
+ case margelo::nitro::swe::iternio::reactnativeautoplay::VoiceAudioEncoding::ALAW: return JSIConverter<std::string>::toJSI(runtime, "ALAW");
59
+ default: [[unlikely]]
60
+ throw std::invalid_argument("Cannot convert VoiceAudioEncoding to JS - invalid value: "
61
+ + std::to_string(static_cast<int>(arg)) + "!");
62
+ }
63
+ }
64
+ static inline bool canConvert(jsi::Runtime& runtime, const jsi::Value& value) {
65
+ if (!value.isString()) {
66
+ return false;
67
+ }
68
+ std::string unionValue = JSIConverter<std::string>::fromJSI(runtime, value);
69
+ switch (hashString(unionValue.c_str(), unionValue.size())) {
70
+ case hashString("LINEAR16"):
71
+ case hashString("MULAW"):
72
+ case hashString("ALAW"):
73
+ return true;
74
+ default:
75
+ return false;
76
+ }
77
+ }
78
+ };
79
+
80
+ } // namespace margelo::nitro
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@iternio/react-native-auto-play",
3
- "version": "0.5.10",
3
+ "version": "0.5.12",
4
4
  "description": "Android Auto and Apple CarPlay for react-native",
5
5
  "main": "lib/index",
6
6
  "module": "lib/index",
@@ -24,6 +24,7 @@ const startVoiceInput: StartVoiceInput = async (options?: VoiceInputOptions) =>
24
24
  language,
25
25
  startSound,
26
26
  endSound,
27
+ encoding,
27
28
  } = options ?? {};
28
29
 
29
30
  const listeningImageRepeats =
@@ -42,7 +43,8 @@ const startVoiceInput: StartVoiceInput = async (options?: VoiceInputOptions) =>
42
43
  onChunk,
43
44
  language,
44
45
  startSoundUri,
45
- endSoundUri
46
+ endSoundUri,
47
+ encoding
46
48
  );
47
49
  };
48
50
 
@@ -82,6 +84,7 @@ export const HybridVoice = {
82
84
  * @param preferSpeechToText request STT transcription instead of raw PCM
83
85
  * @param onChunk optional streaming callback
84
86
  * @param language specify the language for the SpeechRecognizer, falls back to system language if not set
87
+ * @param encoding PCM encoding for onChunk audio and the final result (default LINEAR16)
85
88
  */
86
89
  startVoiceInput,
87
90
  /**
@@ -1,5 +1,5 @@
1
1
  import type { HybridObject } from 'react-native-nitro-modules';
2
- import type { VoiceInputChunk, VoiceInputResult } from '../types/Voice';
2
+ import type { VoiceAudioEncoding, VoiceInputChunk, VoiceInputResult } from '../types/Voice';
3
3
  import type { NitroImage } from '../utils/NitroImage';
4
4
 
5
5
  export interface Voice extends HybridObject<{ android: 'kotlin'; ios: 'swift' }> {
@@ -15,7 +15,8 @@ export interface Voice extends HybridObject<{ android: 'kotlin'; ios: 'swift' }>
15
15
  onChunk?: (chunk: VoiceInputChunk) => void,
16
16
  language?: string,
17
17
  startSoundUri?: string,
18
- endSoundUri?: string
18
+ endSoundUri?: string,
19
+ encoding?: VoiceAudioEncoding
19
20
  ): Promise<VoiceInputResult>;
20
21
  stopVoiceInput(): void;
21
22
  }
@@ -1,5 +1,10 @@
1
1
  import type { VoiceInputImage } from './Image';
2
2
 
3
+ /** PCM sample encoding for the `audio` payload of chunks and the final result.
4
+ * `LINEAR16` is raw 16-bit signed PCM
5
+ * `MULAW`/`ALAW` are G.711 8-bit companded encodings, halving payload size */
6
+ export type VoiceAudioEncoding = 'LINEAR16' | 'MULAW' | 'ALAW';
7
+
3
8
  export interface VoiceInputChunk {
4
9
  partial?: string;
5
10
  audio?: ArrayBuffer;
@@ -21,6 +26,8 @@ export interface VoiceInputOptions {
21
26
  preferSpeechToText?: boolean;
22
27
  onChunk?: (chunk: VoiceInputChunk) => void;
23
28
  language?: string;
29
+ /** PCM encoding for onChunk audio and the final result. Defaults to LINEAR16. */
30
+ encoding?: VoiceAudioEncoding;
24
31
  /** Sound played just before recording starts. Pass a Metro asset: `require('./beep_start.wav')`. */
25
32
  startSound?: number;
26
33
  /** Sound played just after recording stops. Pass a Metro asset: `require('./beep_end.wav')`. */