@iternio/react-native-auto-play 0.5.4 → 0.5.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -14,7 +14,7 @@
14
14
  ## Features
15
15
 
16
16
  - **Cross-Platform:** Write once, run on both Apple CarPlay and Android Auto.
17
- - **Both Architectures:** Supports both the legacy and the new React Native architecture.
17
+ - **New Architecture:** Supports React Native new architecture only.
18
18
  - **Template-Based UI:** Utilize a rich set of templates like `MapTemplate`, `ListTemplate`, `GridTemplate`, and more to build UIs that comply with automotive design guidelines.
19
19
  - **Navigation APIs:** Build full-featured navigation experiences with APIs for trip management, maneuvers, and route guidance.
20
20
  - **Dashboard & Cluster Support:** Extend your app's presence to the CarPlay Dashboard (CarPlay only) and instrument cluster displays (CarPlay & Android Auto).
@@ -837,36 +837,95 @@ new ListTemplate({
837
837
 
838
838
  ### Voice Input
839
839
 
840
- The library provides a cross-platform in-app voice recording API built on top of the car microphone (when connected) or the device microphone (when no car is connected).
840
+ The library provides a cross-platform in-app voice recording API built on top of the car microphone (when connected) or the device microphone (when no car is connected). The voice API lives in `HybridVoice`.
841
841
 
842
842
  #### Permission
843
843
 
844
844
  ```ts
845
- // Check whether permission is already granted
846
- const granted = HybridAutoPlay.hasVoiceInputPermission();
845
+ import { HybridVoice } from '@iternio/react-native-auto-play';
846
+
847
+ // Check whether permission is already granted (synchronous)
848
+ const granted = HybridVoice.hasVoiceInputPermission();
847
849
 
848
850
  // Request permission if not yet granted
849
- const granted = await HybridAutoPlay.requestVoiceInputPermission();
851
+ const granted = await HybridVoice.requestVoiceInputPermission();
850
852
  ```
851
853
 
854
+ On **iOS**: checks/requests both microphone and speech recognition authorization.
855
+ On **Android**: checks/requests `RECORD_AUDIO` via the car context when connected, otherwise via the RN application context.
856
+
852
857
  #### Recording
853
858
 
854
859
  ```ts
855
- // Start recording — resolves with a raw PCM ArrayBuffer (16 kHz, 16-bit, mono)
856
- // Recording stops automatically on silence or when maxDurationMs is reached
857
- const pcmBuffer = await HybridAutoPlay.startVoiceInput(
858
- 1500, // silenceThresholdMs (default 1500)
859
- 10_000, // maxDurationMs (default 10 000)
860
- 'Listening...' // text shown on the car screen while recording (iOS CarPlay)
861
- );
860
+ import { HybridVoice, ErrorUtil } from '@iternio/react-native-auto-play';
861
+
862
+ try {
863
+ const result = await HybridVoice.startVoiceInput({
864
+ silenceThresholdMs: 1500, // ms of silence before auto-stop (default 1500)
865
+ maxDurationMs: 10_000, // hard cap on recording duration (default 10 000)
866
+ listeningText: 'Listening…', // iOS CarPlay: text shown on CPVoiceControlTemplate
867
+ preferSpeechToText: false, // true → STT transcription; false → raw PCM (default)
868
+ startSound: require('./beep_start.mp3'), // played just before recording starts
869
+ endSound: require('./beep_end.mp3'), // played just after recording stops
870
+ onChunk: (chunk) => {
871
+ // chunk.audio — raw PCM ArrayBuffer chunk (PCM mode)
872
+ // chunk.partial — partial transcription string (STT mode)
873
+ },
874
+ });
875
+
876
+ if (result.transcription) {
877
+ console.log('Transcription:', result.transcription);
878
+ } else if (result.audio) {
879
+ console.log(`PCM audio: ${result.audio.byteLength} bytes`);
880
+ }
881
+ } catch (e) {
882
+ if (ErrorUtil.isVoiceInputCanceledError(e)) {
883
+ // User pressed the cancel button on the car screen
884
+ console.log('Voice input cancelled');
885
+ } else {
886
+ console.error(e);
887
+ }
888
+ }
862
889
 
863
- // Stop recording early — resolves startVoiceInput with the audio captured so far
864
- HybridAutoPlay.stopVoiceInput();
890
+ // Stop recording early — resolves startVoiceInput with audio captured so far
891
+ HybridVoice.stopVoiceInput();
865
892
  ```
866
893
 
867
- On **Android**: uses `CarAudioRecord` when Android Auto is connected, otherwise falls back to standard `AudioRecord`.
894
+ | Option | Type | Default | Description |
895
+ |---|---|---|---|
896
+ | `silenceThresholdMs` | `number` | `1500` | Auto-stop after this many ms of silence |
897
+ | `maxDurationMs` | `number` | `10000` | Hard recording time limit |
898
+ | `listeningText` | `string` | — | iOS only — text shown on `CPVoiceControlTemplate` |
899
+ | `listeningImage` | `VoiceInputImage` | — | iOS only — animated image in the CarPlay overlay |
900
+ | `preferSpeechToText` | `boolean` | `false` | `true` → resolve with `{ transcription }`; `false` → resolve with `{ audio }` |
901
+ | `startSound` | `number` | — | Metro asset (`require('./beep.mp3')`) played before recording. Takes audio focus so other apps pause. |
902
+ | `endSound` | `number` | — | Metro asset played after recording stops |
903
+ | `onChunk` | `(chunk) => void` | — | Streaming callback: `chunk.audio` (PCM) or `chunk.partial` (STT) |
904
+ | `language` | `string` | system | BCP-47 language tag for the STT recognizer |
905
+
906
+ **PCM result** (`preferSpeechToText: false`, default): resolves with `{ audio: ArrayBuffer }` — raw 16 kHz, 16-bit, mono PCM.
907
+
908
+ **STT result** (`preferSpeechToText: true`): resolves with `{ transcription: string }` on success, or falls back to `{ audio }` if recognition is unavailable.
868
909
 
869
- On **iOS**: presents `CPVoiceControlTemplate` on the car screen when CarPlay is connected, and captures audio via `AVAudioEngine`.
910
+ On **Android**: uses `CarAudioRecord` when Android Auto is connected, otherwise falls back to standard `AudioRecord`. STT uses `SpeechRecognizer`.
911
+
912
+ On **iOS**: presents `CPVoiceControlTemplate` on the car screen when CarPlay is connected, and captures audio via `AVAudioEngine`. STT uses `SFSpeechRecognizer`.
913
+
914
+ #### Cancel detection
915
+
916
+ When the user presses the cancel button on the car screen, `startVoiceInput` rejects with a `voiceInputCancelled` error on both platforms. Use `ErrorUtil.isVoiceInputCanceledError` to distinguish it from other errors:
917
+
918
+ ```ts
919
+ import { ErrorUtil } from '@iternio/react-native-auto-play';
920
+
921
+ HybridVoice.startVoiceInput().catch((e) => {
922
+ if (ErrorUtil.isVoiceInputCanceledError(e)) {
923
+ // user dismissed — no action needed
924
+ } else {
925
+ throw e;
926
+ }
927
+ });
928
+ ```
870
929
 
871
930
  #### OS-triggered voice input (Android only)
872
931
 
@@ -1015,7 +1074,8 @@ CarPlayDashboard.setButtons([
1015
1074
 
1016
1075
  - **Broken exceptions with `react-native-skia`**: When using `react-native-skia` exceptions on iOS are not reported correctly. This is fixed since version `2.4.19` of `react-native-skia`. For more details, see this [pull request](https://github.com/Shopify/react-native-skia/pull/3595) and [issue](https://github.com/Shopify/react-native-skia/issues/3635).
1017
1076
  - **AppState on iOS**: The `AppState` module from React Native does not work correctly on iOS because this library uses scenes, which are not supported by the stock `AppState` module. This library provides a custom state listener that works for both Android and iOS. Use `HybridAutoPlay.addListenerRenderState` instead of `AppState`.
1018
- - **Timers stop on screen lock**: iOS stops all timers when the device's main screen is turned off. To ensure timers continue to run (which is often necessary for background tasks related to autoplay), a patch for `react-native` is required. A patch is included in the root `patches/` directory and can be applied using `patch-package`.
1077
+ - **Timers stop on screen lock**: iOS stops all timers when the device main screen is turned off. To ensure timers continue to run (which is often necessary for background tasks related to autoplay), a patch for `react-native` is required. A patch is included in the root `patches/` directory and can be applied using `patch-package`.
1078
+ In case you are using Expo SDK >= 56 make sure to set `buildReactNativeFromSource` to `true` in your app config for [expo-build-properties](https://docs.expo.dev/versions/latest/sdk/build-properties/#sharedbuildconfigfields), otherwise the patch can't be applied.
1019
1079
  - **expo-splash-screen stuck on iOS**: The `expo-splash-screen` module is broken on iOS because it does not support scenes, which are used by this library. This can cause the splash screen to be stuck on either the mobile device or on CarPlay. To fix this, a patch for `expo-splash-screen` is included in the root `patches/` directory and can be applied using `patch-package`. After applying the patch, you can hide the splash screen for a specific scene by passing the module name to the `hide` or `hideAsync` function. The module name can be one of the values from the `AutoPlayModules` enum or the UUID of a cluster screen.
1020
1080
  ```tsx
1021
1081
  import { hideAsync } from 'expo-splash-screen';
@@ -1,5 +1,10 @@
1
1
  package com.margelo.nitro.swe.iternio.reactnativeautoplay
2
2
 
3
- class TemplateNotFoundException(private val templateId: String): Exception("templateNotFound(\"$templateId\")") {
3
+ class TemplateNotFoundException(private val templateId: String) :
4
+ Exception("templateNotFound(\"$templateId\")") {
4
5
  override fun toString(): String = "templateNotFound(\"$templateId\")"
5
6
  }
7
+
8
+ class VoiceInputCancelledException : Exception("voiceInputCancelled") {
9
+ override fun toString(): String = "voiceInputCancelled"
10
+ }
@@ -50,7 +50,7 @@ class HybridVoice : HybridVoiceSpec() {
50
50
  }
51
51
  cont.resume(
52
52
  grantResults.isNotEmpty() &&
53
- grantResults.first() == PackageManager.PERMISSION_GRANTED
53
+ grantResults.first() == PackageManager.PERMISSION_GRANTED
54
54
  )
55
55
  true
56
56
  }
@@ -68,7 +68,9 @@ class HybridVoice : HybridVoiceSpec() {
68
68
  listeningImageRepeats: Boolean?,
69
69
  preferSpeechToText: Boolean?,
70
70
  onChunk: ((chunk: VoiceInputChunk) -> Unit)?,
71
- language: String?
71
+ language: String?,
72
+ startSoundUri: String?,
73
+ endSoundUri: String?
72
74
  ): Promise<VoiceInputResult> {
73
75
  return Promise.async {
74
76
  if (Build.VERSION.SDK_INT < Build.VERSION_CODES.O) {
@@ -84,7 +86,9 @@ class HybridVoice : HybridVoiceSpec() {
84
86
  maxDurationMs = maxDurationMs?.toLong() ?: 10_000L,
85
87
  preferSpeechToText = preferSpeechToText ?: false,
86
88
  onChunk = onChunk,
87
- language = language
89
+ language = language,
90
+ startSoundUri = startSoundUri,
91
+ endSoundUri = endSoundUri,
88
92
  )
89
93
  } finally {
90
94
  voiceInputManager = null
@@ -10,7 +10,9 @@ import android.media.AudioFocusRequest
10
10
  import android.media.AudioFormat
11
11
  import android.media.AudioManager
12
12
  import android.media.AudioRecord
13
+ import android.media.MediaPlayer
13
14
  import android.media.MediaRecorder
15
+ import android.net.Uri
14
16
  import android.os.Build
15
17
  import android.os.Bundle
16
18
  import android.os.ParcelFileDescriptor
@@ -62,6 +64,9 @@ class VoiceInputManager(
62
64
  @Volatile
63
65
  private var isRecording = false
64
66
 
67
+ @Volatile
68
+ private var cancelledByUser = false
69
+
65
70
  // STT state — only set when SpeechRecognizer owns the mic
66
71
  @Volatile
67
72
  private var activeSpeechRecognizer: SpeechRecognizer? = null
@@ -72,22 +77,42 @@ class VoiceInputManager(
72
77
  maxDurationMs: Long = 10_000,
73
78
  preferSpeechToText: Boolean = false,
74
79
  onChunk: ((chunk: VoiceInputChunk) -> Unit)? = null,
75
- language: String? = null
80
+ language: String? = null,
81
+ startSoundUri: String? = null,
82
+ endSoundUri: String? = null,
76
83
  ): VoiceInputResult {
77
- if (preferSpeechToText) {
78
- val context = NitroModules.applicationContext ?: throw IllegalArgumentException()
79
- if (SpeechRecognizer.isRecognitionAvailable(context)) {
80
- if (carContext != null) {
81
- if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.TIRAMISU) {
82
- return startSTTFromCarAudio(silenceThresholdMs, maxDurationMs, onChunk, language)
84
+ cancelledByUser = false
85
+ if (!requestAudioFocus()) {
86
+ throw IllegalStateException("Audio focus request denied")
87
+ }
88
+ try {
89
+ val startSoundJob = startSoundUri?.let { uri -> scope.launch { playSound(uri) } }
90
+ val result = if (preferSpeechToText) {
91
+ val context = NitroModules.applicationContext ?: throw IllegalArgumentException()
92
+ if (SpeechRecognizer.isRecognitionAvailable(context)) {
93
+ if (carContext != null) {
94
+ if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.TIRAMISU) {
95
+ startSTTFromCarAudio(silenceThresholdMs, maxDurationMs, onChunk, language)
96
+ } else {
97
+ // Car connected but API < 33: EXTRA_AUDIO_SOURCE unavailable, fall back to PCM
98
+ startPCM(silenceThresholdMs, maxDurationMs, onChunk)
99
+ }
100
+ } else {
101
+ ThreadUtil.postOnUiAndAwait { startSTT(context, onChunk, language) }.getOrThrow()
83
102
  }
84
- // Car connected but API < 33: EXTRA_AUDIO_SOURCE unavailable, fall back to PCM
85
- return startPCM(silenceThresholdMs, maxDurationMs, onChunk)
103
+ } else {
104
+ startPCM(silenceThresholdMs, maxDurationMs, onChunk)
86
105
  }
87
- return ThreadUtil.postOnUiAndAwait { startSTT(context, onChunk, language) }.getOrThrow()
106
+ } else {
107
+ startPCM(silenceThresholdMs, maxDurationMs, onChunk)
88
108
  }
109
+ startSoundJob?.join()
110
+ if (cancelledByUser) throw VoiceInputCancelledException()
111
+ endSoundUri?.let { playSound(it) }
112
+ return result
113
+ } finally {
114
+ abandonAudioFocus()
89
115
  }
90
- return startPCM(silenceThresholdMs, maxDurationMs, onChunk)
91
116
  }
92
117
 
93
118
  // MARK: - STT path (SpeechRecognizer owns the mic)
@@ -309,35 +334,8 @@ class VoiceInputManager(
309
334
  return@suspendCancellableCoroutine
310
335
  }
311
336
 
312
- val appContext = NitroModules.applicationContext ?: run {
313
- cont.resumeWithException(SecurityException("Missing application context"))
314
- return@suspendCancellableCoroutine
315
- }
316
-
317
337
  pcmContinuation = cont
318
338
 
319
- val audioManager = appContext.getSystemService(AudioManager::class.java)
320
-
321
- val audioAttributes =
322
- AudioAttributes.Builder().setContentType(AudioAttributes.CONTENT_TYPE_SPEECH)
323
- .setUsage(AudioAttributes.USAGE_ASSISTANCE_NAVIGATION_GUIDANCE).build()
324
-
325
- val focusRequest =
326
- AudioFocusRequest.Builder(AudioManager.AUDIOFOCUS_GAIN_TRANSIENT_EXCLUSIVE)
327
- .setAudioAttributes(audioAttributes).setOnAudioFocusChangeListener { state ->
328
- if (state == AudioManager.AUDIOFOCUS_LOSS) {
329
- stop()
330
- }
331
- }.build()
332
-
333
- if (audioManager.requestAudioFocus(focusRequest) != AudioManager.AUDIOFOCUS_REQUEST_GRANTED) {
334
- pcmContinuation = null
335
- cont.resumeWithException(IllegalStateException("Audio focus request denied"))
336
- return@suspendCancellableCoroutine
337
- }
338
-
339
- audioFocusRequest = focusRequest
340
-
341
339
  val bufferSize: Int
342
340
 
343
341
  if (carContext != null) {
@@ -381,6 +379,8 @@ class VoiceInputManager(
381
379
  ) ?: -1
382
380
 
383
381
  if (read < 0) {
382
+ // Whenever the user dismisses the microphone on the car screen, the next call to read will return -1
383
+ cancelledByUser = carAudioRecord != null && read == -1
384
384
  break
385
385
  }
386
386
 
@@ -450,6 +450,72 @@ class VoiceInputManager(
450
450
  audioRecord?.stop()
451
451
  }
452
452
 
453
+ @RequiresApi(Build.VERSION_CODES.O)
454
+ private fun requestAudioFocus(): Boolean {
455
+ val appContext = NitroModules.applicationContext ?: return false
456
+ val audioManager = appContext.getSystemService(AudioManager::class.java)
457
+ val audioAttributes = AudioAttributes.Builder()
458
+ .setContentType(AudioAttributes.CONTENT_TYPE_SPEECH)
459
+ .setUsage(AudioAttributes.USAGE_ASSISTANCE_NAVIGATION_GUIDANCE)
460
+ .build()
461
+ val focusRequest = AudioFocusRequest.Builder(AudioManager.AUDIOFOCUS_GAIN_TRANSIENT_EXCLUSIVE)
462
+ .setAudioAttributes(audioAttributes)
463
+ .setOnAudioFocusChangeListener { state ->
464
+ if (state == AudioManager.AUDIOFOCUS_LOSS) { stop() }
465
+ }
466
+ .build()
467
+ return if (audioManager.requestAudioFocus(focusRequest) == AudioManager.AUDIOFOCUS_REQUEST_GRANTED) {
468
+ audioFocusRequest = focusRequest
469
+ true
470
+ } else {
471
+ false
472
+ }
473
+ }
474
+
475
+ @RequiresApi(Build.VERSION_CODES.O)
476
+ private fun abandonAudioFocus() {
477
+ audioFocusRequest?.let {
478
+ val audioManager = (NitroModules.applicationContext ?: carContext)
479
+ ?.getSystemService(AudioManager::class.java)
480
+ audioManager?.abandonAudioFocusRequest(it)
481
+ }
482
+ audioFocusRequest = null
483
+ }
484
+
485
+ private suspend fun playSound(uri: String) = suspendCancellableCoroutine<Unit> { cont ->
486
+ val context = NitroModules.applicationContext ?: run {
487
+ cont.resume(Unit)
488
+ return@suspendCancellableCoroutine
489
+ }
490
+ val player = MediaPlayer()
491
+ try {
492
+ player.setAudioAttributes(
493
+ AudioAttributes.Builder()
494
+ .setContentType(AudioAttributes.CONTENT_TYPE_SONIFICATION)
495
+ .setUsage(AudioAttributes.USAGE_ASSISTANCE_NAVIGATION_GUIDANCE)
496
+ .build()
497
+ )
498
+ player.setDataSource(context, Uri.parse(uri))
499
+ player.setOnCompletionListener {
500
+ it.release()
501
+ if (cont.isActive) { cont.resume(Unit) }
502
+ }
503
+ player.setOnErrorListener { mp, _, _ ->
504
+ mp.release()
505
+ if (cont.isActive) { cont.resume(Unit) }
506
+ true
507
+ }
508
+ player.prepare()
509
+ player.start()
510
+ } catch (_: Exception) {
511
+ try { player.release() } catch (_: Exception) {}
512
+ if (cont.isActive) { cont.resume(Unit) }
513
+ }
514
+ cont.invokeOnCancellation {
515
+ try { player.release() } catch (_: Exception) {}
516
+ }
517
+ }
518
+
453
519
  @RequiresApi(Build.VERSION_CODES.O)
454
520
  private fun releaseResources() {
455
521
  carAudioRecord?.stopRecording()
@@ -458,13 +524,6 @@ class VoiceInputManager(
458
524
  audioRecord?.release()
459
525
  audioRecord = null
460
526
  recordingJob = null
461
- audioFocusRequest?.let {
462
- val audioManager = (NitroModules.applicationContext ?: carContext)?.getSystemService(
463
- AudioManager::class.java,
464
- )
465
- audioManager?.abandonAudioFocusRequest(it)
466
- }
467
- audioFocusRequest = null
468
527
  }
469
528
 
470
529
  fun dispose() {
package/ios/Types.swift CHANGED
@@ -10,7 +10,7 @@ struct TemplateEventPayload {
10
10
  let state: VisibilityState
11
11
  }
12
12
 
13
- enum AutoPlayError: Error {
13
+ enum AutoPlayError: LocalizedError {
14
14
  case templateNotFound(String)
15
15
  case interfaceControllerNotFound(String)
16
16
  case invalidTemplateError(String)
@@ -19,4 +19,5 @@ enum AutoPlayError: Error {
19
19
  case invalidTemplateType(String)
20
20
  case noUiWindow(String)
21
21
  case initReactRootViewFailed(String)
22
+ case voiceInputCancelled
22
23
  }
@@ -85,3 +85,10 @@ extension CPAlertTemplate {
85
85
  initTemplate(template: self, id: id)
86
86
  }
87
87
  }
88
+
89
+ extension CPVoiceControlTemplate {
90
+ convenience init(voiceControlStates: [CPVoiceControlState], id: String) {
91
+ self.init(voiceControlStates: voiceControlStates)
92
+ initTemplate(template: self, id: id)
93
+ }
94
+ }
@@ -36,7 +36,9 @@ class HybridVoice: HybridVoiceSpec {
36
36
  listeningImageRepeats: Bool?,
37
37
  preferSpeechToText: Bool?,
38
38
  onChunk: ((_ chunk: VoiceInputChunk) -> Void)?,
39
- language: String?
39
+ language: String?,
40
+ startSoundUri: String?,
41
+ endSoundUri: String?
40
42
  ) throws -> Promise<VoiceInputResult> {
41
43
  return Promise.async {
42
44
  let interfaceController = try? await RootModule.withInterfaceController { $0 }
@@ -55,7 +57,9 @@ class HybridVoice: HybridVoiceSpec {
55
57
  listeningImageRepeats: listeningImageRepeats,
56
58
  preferSpeechToText: preferSpeechToText ?? false,
57
59
  onChunk: onChunk,
58
- language: language
60
+ language: language,
61
+ startSoundUri: startSoundUri,
62
+ endSoundUri: endSoundUri
59
63
  )
60
64
  }
61
65
  }
@@ -0,0 +1,33 @@
1
+ import CarPlay
2
+
3
+ class VoiceInputTemplate: AutoPlayTemplate {
4
+ let template: CPVoiceControlTemplate
5
+ private let onDidDisappearCallback: () -> Void
6
+
7
+ override func getTemplate() -> CPTemplate {
8
+ return template
9
+ }
10
+
11
+ init(
12
+ voiceControlStates: [CPVoiceControlState],
13
+ id: String,
14
+ onDidDisappear: @escaping () -> Void
15
+ ) {
16
+ self.template = CPVoiceControlTemplate(voiceControlStates: voiceControlStates, id: id)
17
+ self.onDidDisappearCallback = onDidDisappear
18
+
19
+ super.init()
20
+
21
+ try? RootModule.withTemplateStore { templateStore in
22
+ templateStore.addTemplate(template: self, templateId: id)
23
+ }
24
+ }
25
+
26
+ override func onDidDisappear(animated: Bool) {
27
+ onDidDisappearCallback()
28
+
29
+ try? RootModule.withTemplateStore { templateStore in
30
+ templateStore.removeTemplate(templateId: self.template.id)
31
+ }
32
+ }
33
+ }
@@ -3,8 +3,30 @@ import CarPlay
3
3
  import NitroModules
4
4
  import Speech
5
5
 
6
- /// Wraps CheckedContinuation so it can only be resumed once even when
7
- /// shared between a stop() call and an async recognition task callback.
6
+ /// Retains the player and itself until playback finishes — AVAudioPlayer.delegate is weak.
7
+ private final class AudioPlayerDelegate: NSObject, AVAudioPlayerDelegate, @unchecked Sendable {
8
+ private let onFinish: () -> Void
9
+ private var keepAlive: AudioPlayerDelegate?
10
+ private var player: AVAudioPlayer?
11
+
12
+ init(player: AVAudioPlayer, _ onFinish: @escaping () -> Void) {
13
+ self.onFinish = onFinish
14
+ self.player = player
15
+ super.init()
16
+ keepAlive = self
17
+ }
18
+
19
+ private func finish() {
20
+ player = nil
21
+ keepAlive = nil
22
+ onFinish()
23
+ }
24
+
25
+ func audioPlayerDidFinishPlaying(_: AVAudioPlayer, successfully _: Bool) { finish() }
26
+ func audioPlayerDecodeErrorDidOccur(_: AVAudioPlayer, error _: Error?) { finish() }
27
+ }
28
+
29
+ /// CheckedContinuation wrapper that can only be resumed once, safe across concurrent stop() and recognition callbacks.
8
30
  private final class ResultBox: @unchecked Sendable {
9
31
  private var continuation: CheckedContinuation<VoiceInputResult, Error>?
10
32
  private let lock = NSLock()
@@ -28,14 +50,14 @@ private final class ResultBox: @unchecked Sendable {
28
50
  }
29
51
  }
30
52
 
31
- /// Captures audio from the car microphone and buffers raw 16 kHz / 16-bit / mono PCM,
32
- /// or transcribes it via SFSpeechRecognizer when preferSpeechToText is true.
53
+ /// Records 16 kHz / 16-bit mono PCM from the car mic, or transcribes via SFSpeechRecognizer.
33
54
  class VoiceInputManager {
34
55
  private var audioEngine: AVAudioEngine?
35
56
  private var voiceControlTemplate: CPVoiceControlTemplate?
36
57
  private var resultBox: ResultBox?
37
58
  private var samples: [Int16] = []
38
59
  private var isStopping = false
60
+ private var cancelledByUser = false
39
61
  private let stopLock = NSLock()
40
62
 
41
63
  // STT
@@ -45,6 +67,7 @@ class VoiceInputManager {
45
67
  // Timing
46
68
  private var recordingStart: Date?
47
69
  private var silenceStart: Date?
70
+ private var firstBufferContinuation: CheckedContinuation<Void, Never>?
48
71
 
49
72
  private static let sampleRate: Double = 16_000
50
73
  private static let tapBufferSize: AVAudioFrameCount = 4_096
@@ -69,9 +92,22 @@ class VoiceInputManager {
69
92
  listeningImageRepeats: Bool?,
70
93
  preferSpeechToText: Bool,
71
94
  onChunk: ((_ chunk: VoiceInputChunk) -> Void)?,
72
- language: String?
95
+ language: String?,
96
+ startSoundUri: String?,
97
+ endSoundUri: String?
73
98
  ) async throws -> VoiceInputResult {
74
- return try await withCheckedThrowingContinuation { cont in
99
+ stopLock.withLock {
100
+ cancelledByUser = false
101
+ }
102
+ // Single session for the full flow (start sound + recording + end sound); defer deactivates once at the end.
103
+ let session = AVAudioSession.sharedInstance()
104
+ try session.setCategory(.playAndRecord, mode: .measurement, options: [])
105
+ try session.setActive(true)
106
+ defer {
107
+ try? session.setActive(false, options: .notifyOthersOnDeactivation)
108
+ }
109
+
110
+ let result = try await withCheckedThrowingContinuation { cont in
75
111
  let box = ResultBox(cont)
76
112
  self.resultBox = box
77
113
  self.samples = []
@@ -83,9 +119,6 @@ class VoiceInputManager {
83
119
  interfaceController: interfaceController,
84
120
  silenceThresholdMs: silenceThresholdMs,
85
121
  maxDurationMs: maxDurationMs,
86
- listeningText: listeningText,
87
- listeningImage: listeningImage,
88
- listeningImageRepeats: listeningImageRepeats,
89
122
  preferSpeechToText: preferSpeechToText,
90
123
  onChunk: onChunk,
91
124
  box: box,
@@ -95,8 +128,61 @@ class VoiceInputManager {
95
128
  catch {
96
129
  self.cleanup(interfaceController: interfaceController)
97
130
  box.resume(throwing: error)
131
+ return
132
+ }
133
+
134
+ // Start sound fires immediately; template is deferred until the first tap buffer so the mic indicator is already on.
135
+ if let uri = startSoundUri {
136
+ Task { await self.playSound(uri: uri) }
137
+ }
138
+ if let interfaceController = interfaceController {
139
+ Task {
140
+ await withCheckedContinuation { (cont: CheckedContinuation<Void, Never>) in
141
+ self.stopLock.withLock { self.firstBufferContinuation = cont }
142
+ }
143
+ // Skip if stop() fired before the first buffer — cleanup already dismissed.
144
+ guard !self.stopLock.withLock({ self.isStopping }) else { return }
145
+ await self.presentVoiceTemplate(
146
+ interfaceController: interfaceController,
147
+ listeningText: listeningText,
148
+ listeningImage: listeningImage,
149
+ listeningImageRepeats: listeningImageRepeats
150
+ )
151
+ }
152
+ }
153
+ }
154
+
155
+ if let uri = endSoundUri {
156
+ await playSound(uri: uri)
157
+ }
158
+
159
+ return result
160
+ }
161
+
162
+ private func playSound(uri: String) async {
163
+ guard let url = URL(string: uri) else { return }
164
+ do {
165
+ // URLSession handles both http:// (Metro dev server) and file:// (release bundle)
166
+ let (data, _) = try await URLSession.shared.data(from: url)
167
+ await withCheckedContinuation { (cont: CheckedContinuation<Void, Never>) in
168
+ DispatchQueue.main.async {
169
+ do {
170
+ let player = try AVAudioPlayer(data: data)
171
+ let delegate = AudioPlayerDelegate(player: player) { cont.resume() }
172
+ player.delegate = delegate
173
+ player.prepareToPlay()
174
+ player.play()
175
+ }
176
+ catch {
177
+ cont.resume()
178
+ }
179
+ }
98
180
  }
99
181
  }
182
+ catch {
183
+ print(error)
184
+ // fail silently — a broken sound file must not block voice input
185
+ }
100
186
  }
101
187
 
102
188
  func stop(interfaceController: AutoPlayInterfaceController? = nil) {
@@ -106,6 +192,7 @@ class VoiceInputManager {
106
192
  return
107
193
  }
108
194
  isStopping = true
195
+ let wasCancelled = cancelledByUser
109
196
  let wasSTTMode = isSTTMode
110
197
  let capturedRequest = recognitionRequest
111
198
  let box = resultBox
@@ -115,13 +202,17 @@ class VoiceInputManager {
115
202
  stopLock.unlock()
116
203
 
117
204
  if wasSTTMode {
118
- // endAudio() causes the recognition task to fire its final result,
119
- // which resumes the box. Engine teardown happens there too.
205
+ // endAudio() triggers the final recognition result, which resumes the box and tears down the engine.
120
206
  capturedRequest?.endAudio()
121
207
  }
122
208
  else {
123
209
  cleanup(interfaceController: interfaceController)
124
- box?.resume(returning: makePCMResult(from: capturedSamples))
210
+ if wasCancelled {
211
+ box?.resume(throwing: AutoPlayError.voiceInputCancelled)
212
+ }
213
+ else {
214
+ box?.resume(returning: makePCMResult(from: capturedSamples))
215
+ }
125
216
  }
126
217
  }
127
218
 
@@ -131,9 +222,6 @@ class VoiceInputManager {
131
222
  interfaceController: AutoPlayInterfaceController?,
132
223
  silenceThresholdMs: Double,
133
224
  maxDurationMs: Double,
134
- listeningText: String,
135
- listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?,
136
- listeningImageRepeats: Bool?,
137
225
  preferSpeechToText: Bool,
138
226
  onChunk: ((_ chunk: VoiceInputChunk) -> Void)?,
139
227
  box: ResultBox,
@@ -143,19 +231,6 @@ class VoiceInputManager {
143
231
  throw VoiceInputError.microphonePermissionDenied
144
232
  }
145
233
 
146
- let session = AVAudioSession.sharedInstance()
147
- try session.setCategory(.playAndRecord, mode: .measurement, options: [])
148
- try session.setActive(true)
149
-
150
- if let interfaceController {
151
- presentVoiceTemplate(
152
- interfaceController: interfaceController,
153
- listeningText: listeningText,
154
- listeningImage: listeningImage,
155
- listeningImageRepeats: listeningImageRepeats
156
- )
157
- }
158
-
159
234
  var activeRecognitionRequest: SFSpeechAudioBufferRecognitionRequest? = nil
160
235
 
161
236
  if preferSpeechToText, SFSpeechRecognizer.authorizationStatus() == .authorized,
@@ -176,12 +251,18 @@ class VoiceInputManager {
176
251
  // STT failed — fall back to whatever PCM was accumulated
177
252
  self.stopLock.lock()
178
253
  self.isStopping = true
254
+ let wasCancelled = self.cancelledByUser
179
255
  let capturedSamples = self.samples
180
256
  self.samples = []
181
257
  self.stopLock.unlock()
182
258
 
183
259
  self.cleanup(interfaceController: interfaceController)
184
- box.resume(returning: self.makePCMResult(from: capturedSamples))
260
+ if wasCancelled {
261
+ box.resume(throwing: AutoPlayError.voiceInputCancelled)
262
+ }
263
+ else {
264
+ box.resume(returning: self.makePCMResult(from: capturedSamples))
265
+ }
185
266
  return
186
267
  }
187
268
 
@@ -190,16 +271,22 @@ class VoiceInputManager {
190
271
  if result.isFinal {
191
272
  self.stopLock.lock()
192
273
  self.isStopping = true
274
+ let wasCancelled = self.cancelledByUser
193
275
  self.samples = []
194
276
  self.stopLock.unlock()
195
277
 
196
278
  self.cleanup(interfaceController: interfaceController)
197
- box.resume(
198
- returning: VoiceInputResult(
199
- transcription: result.bestTranscription.formattedString,
200
- audio: nil
279
+ if wasCancelled {
280
+ box.resume(throwing: AutoPlayError.voiceInputCancelled)
281
+ }
282
+ else {
283
+ box.resume(
284
+ returning: VoiceInputResult(
285
+ transcription: result.bestTranscription.formattedString,
286
+ audio: nil
287
+ )
201
288
  )
202
- )
289
+ }
203
290
  }
204
291
  else {
205
292
  onChunk?(VoiceInputChunk(partial: result.bestTranscription.formattedString, audio: nil))
@@ -217,13 +304,25 @@ class VoiceInputManager {
217
304
 
218
305
  recordingStart = Date()
219
306
  silenceStart = nil
307
+ firstBufferContinuation = nil
220
308
 
221
309
  inputNode.installTap(
222
310
  onBus: 0,
223
311
  bufferSize: VoiceInputManager.tapBufferSize,
224
312
  format: nativeFormat
225
313
  ) { [weak self] buffer, _ in
226
- guard let self, !self.isStopping else { return }
314
+ guard let self else { return }
315
+
316
+ self.stopLock.lock()
317
+ let stopping = self.isStopping
318
+ let recordingStartSnapshot = self.recordingStart
319
+ let firstBufferCont = self.firstBufferContinuation
320
+ self.firstBufferContinuation = nil
321
+ self.stopLock.unlock()
322
+
323
+ firstBufferCont?.resume()
324
+
325
+ guard !stopping else { return }
227
326
 
228
327
  // Feed STT if active
229
328
  activeRecognitionRequest?.append(buffer)
@@ -266,16 +365,15 @@ class VoiceInputManager {
266
365
  let now = Date()
267
366
 
268
367
  // Max duration — applies in both modes
269
- if let start = self.recordingStart,
368
+ if let start = recordingStartSnapshot,
270
369
  now.timeIntervalSince(start) * 1000 >= maxDurationMs
271
370
  {
272
371
  self.triggerAutoStop(interfaceController: interfaceController)
273
372
  return
274
373
  }
275
374
 
276
- // Silence detection — skip during warm-up so the pipeline has time
277
- // to stabilise before we start measuring amplitude
278
- if let start = self.recordingStart,
375
+ // Silence detection — skip during warm-up to let the pipeline stabilise.
376
+ if let start = recordingStartSnapshot,
279
377
  now.timeIntervalSince(start) * 1000 >= VoiceInputManager.warmupMs
280
378
  {
281
379
  let peak = newSamples.reduce(0) { max($0, abs(Int($1))) }
@@ -312,7 +410,13 @@ class VoiceInputManager {
312
410
  recognitionRequest = nil
313
411
  recordingStart = nil
314
412
  silenceStart = nil
315
- try? AVAudioSession.sharedInstance().setActive(false, options: .notifyOthersOnDeactivation)
413
+ // Drain firstBufferContinuation so the template Task doesn't hang if stop() fired before the first buffer.
414
+ let pendingCont = stopLock.withLock { () -> CheckedContinuation<Void, Never>? in
415
+ let c = firstBufferContinuation
416
+ firstBufferContinuation = nil
417
+ return c
418
+ }
419
+ pendingCont?.resume()
316
420
  if let interfaceController {
317
421
  dismissVoiceTemplate(interfaceController: interfaceController)
318
422
  }
@@ -327,15 +431,10 @@ class VoiceInputManager {
327
431
  // CPVoiceControlState enforces a maximum image size of 150x150 points.
328
432
  private static let voiceImageMaxSize = CGSize(width: 150, height: 150)
329
433
 
330
- // CPVoiceControlState also enforces a 0.3s–5s animation cycle; the 0.3s floor is applied
331
- // by the system regardless of what we pass, so we only need to clamp our own ceiling.
434
+ // CPVoiceControlState enforces a 0.3s–5s animation cycle; the 0.3s floor is system-applied, clamp only the ceiling.
332
435
  private static let maxVoiceImageCycleDuration: TimeInterval = 5.0
333
436
 
334
- // Bypasses RCTConvert for asset images: it collapses animated UIImages to a single frame
335
- // via CGImage during scale adjustment. Parser.decodeImage preserves animation frames by
336
- // walking every frame in the source via ImageIO — UIImage(data:) never builds a multi-frame
337
- // .images array itself, for GIF, APNG, or WebP. Tinting is skipped for animated images since
338
- // frames cannot be tinted individually.
437
+ // Uses Parser.decodeImage instead of RCTConvert to preserve animation frames for GIF/APNG/WebP.
339
438
  private func loadVoiceImage(image: Variant_GlyphImage_AssetImage_RemoteImage?, traitCollection: UITraitCollection)
340
439
  -> UIImage?
341
440
  {
@@ -381,12 +480,13 @@ class VoiceInputManager {
381
480
  return nil
382
481
  }
383
482
 
483
+ @MainActor
384
484
  private func presentVoiceTemplate(
385
485
  interfaceController: AutoPlayInterfaceController,
386
486
  listeningText: String,
387
487
  listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?,
388
488
  listeningImageRepeats: Bool?
389
- ) {
489
+ ) async {
390
490
  let traitCollection = SceneStore.getRootTraitCollection() ?? UITraitCollection.current
391
491
  let image = loadVoiceImage(
392
492
  image: listeningImage,
@@ -400,14 +500,21 @@ class VoiceInputManager {
400
500
  image: image,
401
501
  repeats: repeats
402
502
  )
403
- let template = CPVoiceControlTemplate(voiceControlStates: [listeningState])
404
- initTemplate(template: template, id: "voice-input")
405
- voiceControlTemplate = template
406
503
 
407
- Task { @MainActor in
408
- try? await interfaceController.presentTemplate(template, animated: true)
409
- template.activateVoiceControlState(withIdentifier: "listening")
504
+ let voiceTemplate = VoiceInputTemplate(
505
+ voiceControlStates: [listeningState],
506
+ id: "voice-input"
507
+ ) { [weak self] in
508
+ guard let self else { return }
509
+ self.stopLock.withLock {
510
+ if !self.isStopping { self.cancelledByUser = true }
511
+ }
512
+ self.stop()
410
513
  }
514
+
515
+ voiceControlTemplate = voiceTemplate.template
516
+ try? await interfaceController.presentTemplate(voiceTemplate.template, animated: true)
517
+ voiceTemplate.template.activateVoiceControlState(withIdentifier: "listening")
411
518
  }
412
519
 
413
520
  private func dismissVoiceTemplate(interfaceController: AutoPlayInterfaceController) {
@@ -1,10 +1,13 @@
1
+ import { Image } from 'react-native';
1
2
  import { NitroModules } from 'react-native-nitro-modules';
2
3
  import { NitroImageUtil } from '../utils/NitroImage';
3
4
  const _native = NitroModules.createHybridObject('Voice');
4
5
  const startVoiceInput = async (options) => {
5
- const { onChunk, silenceThresholdMs, maxDurationMs, listeningText, listeningImage, preferSpeechToText, language, } = options ?? {};
6
+ const { onChunk, silenceThresholdMs, maxDurationMs, listeningText, listeningImage, preferSpeechToText, language, startSound, endSound, } = options ?? {};
6
7
  const listeningImageRepeats = listeningImage?.type === 'asset' ? listeningImage.repeats : undefined;
7
- return await _native.startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, NitroImageUtil.convert(listeningImage), listeningImageRepeats, preferSpeechToText, onChunk, language);
8
+ const startSoundUri = startSound != null ? Image.resolveAssetSource(startSound).uri : undefined;
9
+ const endSoundUri = endSound != null ? Image.resolveAssetSource(endSound).uri : undefined;
10
+ return await _native.startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, NitroImageUtil.convert(listeningImage), listeningImageRepeats, preferSpeechToText, onChunk, language, startSoundUri, endSoundUri);
8
11
  };
9
12
  export const HybridVoice = {
10
13
  /**
@@ -7,6 +7,6 @@ export interface Voice extends HybridObject<{
7
7
  }> {
8
8
  hasVoiceInputPermission(): boolean;
9
9
  requestVoiceInputPermission(): Promise<boolean>;
10
- startVoiceInput(silenceThresholdMs?: number, maxDurationMs?: number, listeningText?: string, listeningImage?: NitroImage, listeningImageRepeats?: boolean, preferSpeechToText?: boolean, onChunk?: (chunk: VoiceInputChunk) => void, language?: string): Promise<VoiceInputResult>;
10
+ startVoiceInput(silenceThresholdMs?: number, maxDurationMs?: number, listeningText?: string, listeningImage?: NitroImage, listeningImageRepeats?: boolean, preferSpeechToText?: boolean, onChunk?: (chunk: VoiceInputChunk) => void, language?: string, startSoundUri?: string, endSoundUri?: string): Promise<VoiceInputResult>;
11
11
  stopVoiceInput(): void;
12
12
  }
@@ -18,4 +18,8 @@ export interface VoiceInputOptions {
18
18
  preferSpeechToText?: boolean;
19
19
  onChunk?: (chunk: VoiceInputChunk) => void;
20
20
  language?: string;
21
+ /** Sound played just before recording starts. Pass a Metro asset: `require('./beep_start.wav')`. */
22
+ startSound?: number;
23
+ /** Sound played just after recording stops. Pass a Metro asset: `require('./beep_end.wav')`. */
24
+ endSound?: number;
21
25
  }
@@ -1,3 +1,4 @@
1
1
  export declare const ErrorUtil: {
2
2
  isTemplateNotFoundError: (error: unknown) => error is Error;
3
+ isVoiceInputCanceledError: (error: unknown) => error is Error;
3
4
  };
@@ -1,4 +1,4 @@
1
- const isTemplateNotFoundError = (error) => {
1
+ const isError = (error) => {
2
2
  if (error == null) {
3
3
  return false;
4
4
  }
@@ -14,6 +14,18 @@ const isTemplateNotFoundError = (error) => {
14
14
  if (typeof error.message !== 'string') {
15
15
  return false;
16
16
  }
17
- return error.message.startsWith('templateNotFound');
17
+ return true;
18
+ };
19
+ const isTemplateNotFoundError = (error) => {
20
+ if (isError(error)) {
21
+ return error.message.startsWith('templateNotFound');
22
+ }
23
+ return false;
24
+ };
25
+ const isVoiceInputCanceledError = (error) => {
26
+ if (isError(error)) {
27
+ return error.message.startsWith('voiceInputCancelled');
28
+ }
29
+ return false;
18
30
  };
19
- export const ErrorUtil = { isTemplateNotFoundError };
31
+ export const ErrorUtil = { isTemplateNotFoundError, isVoiceInputCanceledError };
@@ -98,9 +98,9 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
98
98
  return __promise;
99
99
  }();
100
100
  }
101
- std::shared_ptr<Promise<VoiceInputResult>> JHybridVoiceSpec::startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language) {
102
- static const auto method = _javaPart->javaClassStatic()->getMethod<jni::local_ref<JPromise::javaobject>(jni::alias_ref<jni::JDouble> /* silenceThresholdMs */, jni::alias_ref<jni::JDouble> /* maxDurationMs */, jni::alias_ref<jni::JString> /* listeningText */, jni::alias_ref<JVariant_GlyphImage_AssetImage_RemoteImage> /* listeningImage */, jni::alias_ref<jni::JBoolean> /* listeningImageRepeats */, jni::alias_ref<jni::JBoolean> /* preferSpeechToText */, jni::alias_ref<JFunc_void_VoiceInputChunk::javaobject> /* onChunk */, jni::alias_ref<jni::JString> /* language */)>("startVoiceInput_cxx");
103
- auto __result = method(_javaPart, silenceThresholdMs.has_value() ? jni::JDouble::valueOf(silenceThresholdMs.value()) : nullptr, maxDurationMs.has_value() ? jni::JDouble::valueOf(maxDurationMs.value()) : nullptr, listeningText.has_value() ? jni::make_jstring(listeningText.value()) : nullptr, listeningImage.has_value() ? JVariant_GlyphImage_AssetImage_RemoteImage::fromCpp(listeningImage.value()) : nullptr, listeningImageRepeats.has_value() ? jni::JBoolean::valueOf(listeningImageRepeats.value()) : nullptr, preferSpeechToText.has_value() ? jni::JBoolean::valueOf(preferSpeechToText.value()) : nullptr, onChunk.has_value() ? JFunc_void_VoiceInputChunk_cxx::fromCpp(onChunk.value()) : nullptr, language.has_value() ? jni::make_jstring(language.value()) : nullptr);
101
+ std::shared_ptr<Promise<VoiceInputResult>> JHybridVoiceSpec::startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri) {
102
+ static const auto method = _javaPart->javaClassStatic()->getMethod<jni::local_ref<JPromise::javaobject>(jni::alias_ref<jni::JDouble> /* silenceThresholdMs */, jni::alias_ref<jni::JDouble> /* maxDurationMs */, jni::alias_ref<jni::JString> /* listeningText */, jni::alias_ref<JVariant_GlyphImage_AssetImage_RemoteImage> /* listeningImage */, jni::alias_ref<jni::JBoolean> /* listeningImageRepeats */, jni::alias_ref<jni::JBoolean> /* preferSpeechToText */, jni::alias_ref<JFunc_void_VoiceInputChunk::javaobject> /* onChunk */, jni::alias_ref<jni::JString> /* language */, jni::alias_ref<jni::JString> /* startSoundUri */, jni::alias_ref<jni::JString> /* endSoundUri */)>("startVoiceInput_cxx");
103
+ auto __result = method(_javaPart, silenceThresholdMs.has_value() ? jni::JDouble::valueOf(silenceThresholdMs.value()) : nullptr, maxDurationMs.has_value() ? jni::JDouble::valueOf(maxDurationMs.value()) : nullptr, listeningText.has_value() ? jni::make_jstring(listeningText.value()) : nullptr, listeningImage.has_value() ? JVariant_GlyphImage_AssetImage_RemoteImage::fromCpp(listeningImage.value()) : nullptr, listeningImageRepeats.has_value() ? jni::JBoolean::valueOf(listeningImageRepeats.value()) : nullptr, preferSpeechToText.has_value() ? jni::JBoolean::valueOf(preferSpeechToText.value()) : nullptr, onChunk.has_value() ? JFunc_void_VoiceInputChunk_cxx::fromCpp(onChunk.value()) : nullptr, language.has_value() ? jni::make_jstring(language.value()) : nullptr, startSoundUri.has_value() ? jni::make_jstring(startSoundUri.value()) : nullptr, endSoundUri.has_value() ? jni::make_jstring(endSoundUri.value()) : nullptr);
104
104
  return [&]() {
105
105
  auto __promise = Promise<VoiceInputResult>::create();
106
106
  __result->cthis()->addOnResolvedListener([=](const jni::alias_ref<jni::JObject>& __boxedResult) {
@@ -56,7 +56,7 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
56
56
  // Methods
57
57
  bool hasVoiceInputPermission() override;
58
58
  std::shared_ptr<Promise<bool>> requestVoiceInputPermission() override;
59
- std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language) override;
59
+ std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri) override;
60
60
  void stopVoiceInput() override;
61
61
 
62
62
  private:
@@ -37,12 +37,12 @@ abstract class HybridVoiceSpec: HybridObject() {
37
37
  @Keep
38
38
  abstract fun requestVoiceInputPermission(): Promise<Boolean>
39
39
 
40
- abstract fun startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Boolean?, preferSpeechToText: Boolean?, onChunk: ((chunk: VoiceInputChunk) -> Unit)?, language: String?): Promise<VoiceInputResult>
40
+ abstract fun startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Boolean?, preferSpeechToText: Boolean?, onChunk: ((chunk: VoiceInputChunk) -> Unit)?, language: String?, startSoundUri: String?, endSoundUri: String?): Promise<VoiceInputResult>
41
41
 
42
42
  @DoNotStrip
43
43
  @Keep
44
- private fun startVoiceInput_cxx(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Boolean?, preferSpeechToText: Boolean?, onChunk: Func_void_VoiceInputChunk?, language: String?): Promise<VoiceInputResult> {
45
- val __result = startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, listeningImage, listeningImageRepeats, preferSpeechToText, onChunk?.let { it }, language)
44
+ private fun startVoiceInput_cxx(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Boolean?, preferSpeechToText: Boolean?, onChunk: Func_void_VoiceInputChunk?, language: String?, startSoundUri: String?, endSoundUri: String?): Promise<VoiceInputResult> {
45
+ val __result = startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, listeningImage, listeningImageRepeats, preferSpeechToText, onChunk?.let { it }, language, startSoundUri, endSoundUri)
46
46
  return __result
47
47
  }
48
48
 
@@ -107,8 +107,8 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
107
107
  auto __value = std::move(__result.value());
108
108
  return __value;
109
109
  }
110
- inline std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language) override {
111
- auto __result = _swiftPart.startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, listeningImage, listeningImageRepeats, preferSpeechToText, onChunk, language);
110
+ inline std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri) override {
111
+ auto __result = _swiftPart.startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, listeningImage, listeningImageRepeats, preferSpeechToText, onChunk, language, startSoundUri, endSoundUri);
112
112
  if (__result.hasError()) [[unlikely]] {
113
113
  std::rethrow_exception(__result.error());
114
114
  }
@@ -15,7 +15,7 @@ public protocol HybridVoiceSpec_protocol: HybridObject {
15
15
  // Methods
16
16
  func hasVoiceInputPermission() throws -> Bool
17
17
  func requestVoiceInputPermission() throws -> Promise<Bool>
18
- func startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Bool?, preferSpeechToText: Bool?, onChunk: ((_ chunk: VoiceInputChunk) -> Void)?, language: String?) throws -> Promise<VoiceInputResult>
18
+ func startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Bool?, preferSpeechToText: Bool?, onChunk: ((_ chunk: VoiceInputChunk) -> Void)?, language: String?, startSoundUri: String?, endSoundUri: String?) throws -> Promise<VoiceInputResult>
19
19
  func stopVoiceInput() throws -> Void
20
20
  }
21
21
 
@@ -156,7 +156,7 @@ open class HybridVoiceSpec_cxx {
156
156
  }
157
157
 
158
158
  @inline(__always)
159
- public final func startVoiceInput(silenceThresholdMs: bridge.std__optional_double_, maxDurationMs: bridge.std__optional_double_, listeningText: bridge.std__optional_std__string_, listeningImage: bridge.std__optional_std__variant_GlyphImage__AssetImage__RemoteImage__, listeningImageRepeats: bridge.std__optional_bool_, preferSpeechToText: bridge.std__optional_bool_, onChunk: bridge.std__optional_std__function_void_const_VoiceInputChunk_____chunk______, language: bridge.std__optional_std__string_) -> bridge.Result_std__shared_ptr_Promise_VoiceInputResult___ {
159
+ public final func startVoiceInput(silenceThresholdMs: bridge.std__optional_double_, maxDurationMs: bridge.std__optional_double_, listeningText: bridge.std__optional_std__string_, listeningImage: bridge.std__optional_std__variant_GlyphImage__AssetImage__RemoteImage__, listeningImageRepeats: bridge.std__optional_bool_, preferSpeechToText: bridge.std__optional_bool_, onChunk: bridge.std__optional_std__function_void_const_VoiceInputChunk_____chunk______, language: bridge.std__optional_std__string_, startSoundUri: bridge.std__optional_std__string_, endSoundUri: bridge.std__optional_std__string_) -> bridge.Result_std__shared_ptr_Promise_VoiceInputResult___ {
160
160
  do {
161
161
  let __result = try self.__implementation.startVoiceInput(silenceThresholdMs: { () -> Double? in
162
162
  if bridge.has_value_std__optional_double_(silenceThresholdMs) {
@@ -234,6 +234,20 @@ open class HybridVoiceSpec_cxx {
234
234
  } else {
235
235
  return nil
236
236
  }
237
+ }(), startSoundUri: { () -> String? in
238
+ if bridge.has_value_std__optional_std__string_(startSoundUri) {
239
+ let __unwrapped = bridge.get_std__optional_std__string_(startSoundUri)
240
+ return String(__unwrapped)
241
+ } else {
242
+ return nil
243
+ }
244
+ }(), endSoundUri: { () -> String? in
245
+ if bridge.has_value_std__optional_std__string_(endSoundUri) {
246
+ let __unwrapped = bridge.get_std__optional_std__string_(endSoundUri)
247
+ return String(__unwrapped)
248
+ } else {
249
+ return nil
250
+ }
237
251
  }())
238
252
  let __resultCpp = { () -> bridge.std__shared_ptr_Promise_VoiceInputResult__ in
239
253
  let __promise = bridge.create_std__shared_ptr_Promise_VoiceInputResult__()
@@ -68,7 +68,7 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
68
68
  // Methods
69
69
  virtual bool hasVoiceInputPermission() = 0;
70
70
  virtual std::shared_ptr<Promise<bool>> requestVoiceInputPermission() = 0;
71
- virtual std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language) = 0;
71
+ virtual std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri) = 0;
72
72
  virtual void stopVoiceInput() = 0;
73
73
 
74
74
  protected:
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@iternio/react-native-auto-play",
3
- "version": "0.5.4",
3
+ "version": "0.5.6",
4
4
  "description": "Android Auto and Apple CarPlay for react-native",
5
5
  "main": "lib/index",
6
6
  "module": "lib/index",
@@ -1,3 +1,4 @@
1
+ import { Image } from 'react-native';
1
2
  import { NitroModules } from 'react-native-nitro-modules';
2
3
  import type { Voice } from '../specs/Voice.nitro';
3
4
  import type { VoiceInputOptions, VoiceInputResult } from '../types/Voice';
@@ -21,11 +22,16 @@ const startVoiceInput: StartVoiceInput = async (options?: VoiceInputOptions) =>
21
22
  listeningImage,
22
23
  preferSpeechToText,
23
24
  language,
25
+ startSound,
26
+ endSound,
24
27
  } = options ?? {};
25
28
 
26
29
  const listeningImageRepeats =
27
30
  listeningImage?.type === 'asset' ? listeningImage.repeats : undefined;
28
31
 
32
+ const startSoundUri = startSound != null ? Image.resolveAssetSource(startSound).uri : undefined;
33
+ const endSoundUri = endSound != null ? Image.resolveAssetSource(endSound).uri : undefined;
34
+
29
35
  return await _native.startVoiceInput(
30
36
  silenceThresholdMs,
31
37
  maxDurationMs,
@@ -34,7 +40,9 @@ const startVoiceInput: StartVoiceInput = async (options?: VoiceInputOptions) =>
34
40
  listeningImageRepeats,
35
41
  preferSpeechToText,
36
42
  onChunk,
37
- language
43
+ language,
44
+ startSoundUri,
45
+ endSoundUri
38
46
  );
39
47
  };
40
48
 
@@ -13,7 +13,9 @@ export interface Voice extends HybridObject<{ android: 'kotlin'; ios: 'swift' }>
13
13
  listeningImageRepeats?: boolean,
14
14
  preferSpeechToText?: boolean,
15
15
  onChunk?: (chunk: VoiceInputChunk) => void,
16
- language?: string
16
+ language?: string,
17
+ startSoundUri?: string,
18
+ endSoundUri?: string
17
19
  ): Promise<VoiceInputResult>;
18
20
  stopVoiceInput(): void;
19
21
  }
@@ -21,4 +21,8 @@ export interface VoiceInputOptions {
21
21
  preferSpeechToText?: boolean;
22
22
  onChunk?: (chunk: VoiceInputChunk) => void;
23
23
  language?: string;
24
+ /** Sound played just before recording starts. Pass a Metro asset: `require('./beep_start.wav')`. */
25
+ startSound?: number;
26
+ /** Sound played just after recording stops. Pass a Metro asset: `require('./beep_end.wav')`. */
27
+ endSound?: number;
24
28
  }
@@ -1,4 +1,4 @@
1
- const isTemplateNotFoundError = (error: unknown): error is Error => {
1
+ const isError = (error: unknown): error is { message: string } => {
2
2
  if (error == null) {
3
3
  return false;
4
4
  }
@@ -19,7 +19,23 @@ const isTemplateNotFoundError = (error: unknown): error is Error => {
19
19
  return false;
20
20
  }
21
21
 
22
- return error.message.startsWith('templateNotFound');
22
+ return true;
23
+ };
24
+
25
+ const isTemplateNotFoundError = (error: unknown): error is Error => {
26
+ if (isError(error)) {
27
+ return error.message.startsWith('templateNotFound');
28
+ }
29
+
30
+ return false;
31
+ };
32
+
33
+ const isVoiceInputCanceledError = (error: unknown): error is Error => {
34
+ if (isError(error)) {
35
+ return error.message.startsWith('voiceInputCancelled');
36
+ }
37
+
38
+ return false;
23
39
  };
24
40
 
25
- export const ErrorUtil = { isTemplateNotFoundError };
41
+ export const ErrorUtil = { isTemplateNotFoundError, isVoiceInputCanceledError };