@iternio/react-native-auto-play 0.5.4 → 0.5.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -14,7 +14,7 @@
14
14
  ## Features
15
15
 
16
16
  - **Cross-Platform:** Write once, run on both Apple CarPlay and Android Auto.
17
- - **Both Architectures:** Supports both the legacy and the new React Native architecture.
17
+ - **New Architecture:** Supports React Native new architecture only.
18
18
  - **Template-Based UI:** Utilize a rich set of templates like `MapTemplate`, `ListTemplate`, `GridTemplate`, and more to build UIs that comply with automotive design guidelines.
19
19
  - **Navigation APIs:** Build full-featured navigation experiences with APIs for trip management, maneuvers, and route guidance.
20
20
  - **Dashboard & Cluster Support:** Extend your app's presence to the CarPlay Dashboard (CarPlay only) and instrument cluster displays (CarPlay & Android Auto).
@@ -837,36 +837,95 @@ new ListTemplate({
837
837
 
838
838
  ### Voice Input
839
839
 
840
- The library provides a cross-platform in-app voice recording API built on top of the car microphone (when connected) or the device microphone (when no car is connected).
840
+ The library provides a cross-platform in-app voice recording API built on top of the car microphone (when connected) or the device microphone (when no car is connected). The voice API lives in `HybridVoice`.
841
841
 
842
842
  #### Permission
843
843
 
844
844
  ```ts
845
- // Check whether permission is already granted
846
- const granted = HybridAutoPlay.hasVoiceInputPermission();
845
+ import { HybridVoice } from '@iternio/react-native-auto-play';
846
+
847
+ // Check whether permission is already granted (synchronous)
848
+ const granted = HybridVoice.hasVoiceInputPermission();
847
849
 
848
850
  // Request permission if not yet granted
849
- const granted = await HybridAutoPlay.requestVoiceInputPermission();
851
+ const granted = await HybridVoice.requestVoiceInputPermission();
850
852
  ```
851
853
 
854
+ On **iOS**: checks/requests both microphone and speech recognition authorization.
855
+ On **Android**: checks/requests `RECORD_AUDIO` via the car context when connected, otherwise via the RN application context.
856
+
852
857
  #### Recording
853
858
 
854
859
  ```ts
855
- // Start recording — resolves with a raw PCM ArrayBuffer (16 kHz, 16-bit, mono)
856
- // Recording stops automatically on silence or when maxDurationMs is reached
857
- const pcmBuffer = await HybridAutoPlay.startVoiceInput(
858
- 1500, // silenceThresholdMs (default 1500)
859
- 10_000, // maxDurationMs (default 10 000)
860
- 'Listening...' // text shown on the car screen while recording (iOS CarPlay)
861
- );
860
+ import { HybridVoice, ErrorUtil } from '@iternio/react-native-auto-play';
861
+
862
+ try {
863
+ const result = await HybridVoice.startVoiceInput({
864
+ silenceThresholdMs: 1500, // ms of silence before auto-stop (default 1500)
865
+ maxDurationMs: 10_000, // hard cap on recording duration (default 10 000)
866
+ listeningText: 'Listening…', // iOS CarPlay: text shown on CPVoiceControlTemplate
867
+ preferSpeechToText: false, // true → STT transcription; false → raw PCM (default)
868
+ startSound: require('./beep_start.mp3'), // played just before recording starts
869
+ endSound: require('./beep_end.mp3'), // played just after recording stops
870
+ onChunk: (chunk) => {
871
+ // chunk.audio — raw PCM ArrayBuffer chunk (PCM mode)
872
+ // chunk.partial — partial transcription string (STT mode)
873
+ },
874
+ });
875
+
876
+ if (result.transcription) {
877
+ console.log('Transcription:', result.transcription);
878
+ } else if (result.audio) {
879
+ console.log(`PCM audio: ${result.audio.byteLength} bytes`);
880
+ }
881
+ } catch (e) {
882
+ if (ErrorUtil.isVoiceInputCanceledError(e)) {
883
+ // User pressed the cancel button on the car screen
884
+ console.log('Voice input cancelled');
885
+ } else {
886
+ console.error(e);
887
+ }
888
+ }
862
889
 
863
- // Stop recording early — resolves startVoiceInput with the audio captured so far
864
- HybridAutoPlay.stopVoiceInput();
890
+ // Stop recording early — resolves startVoiceInput with audio captured so far
891
+ HybridVoice.stopVoiceInput();
865
892
  ```
866
893
 
867
- On **Android**: uses `CarAudioRecord` when Android Auto is connected, otherwise falls back to standard `AudioRecord`.
894
+ | Option | Type | Default | Description |
895
+ |---|---|---|---|
896
+ | `silenceThresholdMs` | `number` | `1500` | Auto-stop after this many ms of silence |
897
+ | `maxDurationMs` | `number` | `10000` | Hard recording time limit |
898
+ | `listeningText` | `string` | — | iOS only — text shown on `CPVoiceControlTemplate` |
899
+ | `listeningImage` | `VoiceInputImage` | — | iOS only — animated image in the CarPlay overlay |
900
+ | `preferSpeechToText` | `boolean` | `false` | `true` → resolve with `{ transcription }`; `false` → resolve with `{ audio }` |
901
+ | `startSound` | `number` | — | Metro asset (`require('./beep.mp3')`) played before recording. Takes audio focus so other apps pause. |
902
+ | `endSound` | `number` | — | Metro asset played after recording stops |
903
+ | `onChunk` | `(chunk) => void` | — | Streaming callback: `chunk.audio` (PCM) or `chunk.partial` (STT) |
904
+ | `language` | `string` | system | BCP-47 language tag for the STT recognizer |
905
+
906
+ **PCM result** (`preferSpeechToText: false`, default): resolves with `{ audio: ArrayBuffer }` — raw 16 kHz, 16-bit, mono PCM.
907
+
908
+ **STT result** (`preferSpeechToText: true`): resolves with `{ transcription: string }` on success, or falls back to `{ audio }` if recognition is unavailable.
868
909
 
869
- On **iOS**: presents `CPVoiceControlTemplate` on the car screen when CarPlay is connected, and captures audio via `AVAudioEngine`.
910
+ On **Android**: uses `CarAudioRecord` when Android Auto is connected, otherwise falls back to standard `AudioRecord`. STT uses `SpeechRecognizer`.
911
+
912
+ On **iOS**: presents `CPVoiceControlTemplate` on the car screen when CarPlay is connected, and captures audio via `AVAudioEngine`. STT uses `SFSpeechRecognizer`.
913
+
914
+ #### Cancel detection
915
+
916
+ When the user presses the cancel button on the car screen, `startVoiceInput` rejects with a `voiceInputCancelled` error on both platforms. Use `ErrorUtil.isVoiceInputCanceledError` to distinguish it from other errors:
917
+
918
+ ```ts
919
+ import { ErrorUtil } from '@iternio/react-native-auto-play';
920
+
921
+ HybridVoice.startVoiceInput().catch((e) => {
922
+ if (ErrorUtil.isVoiceInputCanceledError(e)) {
923
+ // user dismissed — no action needed
924
+ } else {
925
+ throw e;
926
+ }
927
+ });
928
+ ```
870
929
 
871
930
  #### OS-triggered voice input (Android only)
872
931
 
@@ -1,5 +1,10 @@
1
1
  package com.margelo.nitro.swe.iternio.reactnativeautoplay
2
2
 
3
- class TemplateNotFoundException(private val templateId: String): Exception("templateNotFound(\"$templateId\")") {
3
+ class TemplateNotFoundException(private val templateId: String) :
4
+ Exception("templateNotFound(\"$templateId\")") {
4
5
  override fun toString(): String = "templateNotFound(\"$templateId\")"
5
6
  }
7
+
8
+ class VoiceInputCancelledException : Exception("voiceInputCancelled") {
9
+ override fun toString(): String = "voiceInputCancelled"
10
+ }
@@ -50,7 +50,7 @@ class HybridVoice : HybridVoiceSpec() {
50
50
  }
51
51
  cont.resume(
52
52
  grantResults.isNotEmpty() &&
53
- grantResults.first() == PackageManager.PERMISSION_GRANTED
53
+ grantResults.first() == PackageManager.PERMISSION_GRANTED
54
54
  )
55
55
  true
56
56
  }
@@ -68,7 +68,9 @@ class HybridVoice : HybridVoiceSpec() {
68
68
  listeningImageRepeats: Boolean?,
69
69
  preferSpeechToText: Boolean?,
70
70
  onChunk: ((chunk: VoiceInputChunk) -> Unit)?,
71
- language: String?
71
+ language: String?,
72
+ startSoundUri: String?,
73
+ endSoundUri: String?
72
74
  ): Promise<VoiceInputResult> {
73
75
  return Promise.async {
74
76
  if (Build.VERSION.SDK_INT < Build.VERSION_CODES.O) {
@@ -84,7 +86,9 @@ class HybridVoice : HybridVoiceSpec() {
84
86
  maxDurationMs = maxDurationMs?.toLong() ?: 10_000L,
85
87
  preferSpeechToText = preferSpeechToText ?: false,
86
88
  onChunk = onChunk,
87
- language = language
89
+ language = language,
90
+ startSoundUri = startSoundUri,
91
+ endSoundUri = endSoundUri,
88
92
  )
89
93
  } finally {
90
94
  voiceInputManager = null
@@ -10,7 +10,9 @@ import android.media.AudioFocusRequest
10
10
  import android.media.AudioFormat
11
11
  import android.media.AudioManager
12
12
  import android.media.AudioRecord
13
+ import android.media.MediaPlayer
13
14
  import android.media.MediaRecorder
15
+ import android.net.Uri
14
16
  import android.os.Build
15
17
  import android.os.Bundle
16
18
  import android.os.ParcelFileDescriptor
@@ -62,6 +64,9 @@ class VoiceInputManager(
62
64
  @Volatile
63
65
  private var isRecording = false
64
66
 
67
+ @Volatile
68
+ private var cancelledByUser = false
69
+
65
70
  // STT state — only set when SpeechRecognizer owns the mic
66
71
  @Volatile
67
72
  private var activeSpeechRecognizer: SpeechRecognizer? = null
@@ -72,22 +77,41 @@ class VoiceInputManager(
72
77
  maxDurationMs: Long = 10_000,
73
78
  preferSpeechToText: Boolean = false,
74
79
  onChunk: ((chunk: VoiceInputChunk) -> Unit)? = null,
75
- language: String? = null
80
+ language: String? = null,
81
+ startSoundUri: String? = null,
82
+ endSoundUri: String? = null,
76
83
  ): VoiceInputResult {
77
- if (preferSpeechToText) {
78
- val context = NitroModules.applicationContext ?: throw IllegalArgumentException()
79
- if (SpeechRecognizer.isRecognitionAvailable(context)) {
80
- if (carContext != null) {
81
- if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.TIRAMISU) {
82
- return startSTTFromCarAudio(silenceThresholdMs, maxDurationMs, onChunk, language)
84
+ cancelledByUser = false
85
+ if (!requestAudioFocus()) {
86
+ throw IllegalStateException("Audio focus request denied")
87
+ }
88
+ try {
89
+ startSoundUri?.let { playSound(it) }
90
+ val result = if (preferSpeechToText) {
91
+ val context = NitroModules.applicationContext ?: throw IllegalArgumentException()
92
+ if (SpeechRecognizer.isRecognitionAvailable(context)) {
93
+ if (carContext != null) {
94
+ if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.TIRAMISU) {
95
+ startSTTFromCarAudio(silenceThresholdMs, maxDurationMs, onChunk, language)
96
+ } else {
97
+ // Car connected but API < 33: EXTRA_AUDIO_SOURCE unavailable, fall back to PCM
98
+ startPCM(silenceThresholdMs, maxDurationMs, onChunk)
99
+ }
100
+ } else {
101
+ ThreadUtil.postOnUiAndAwait { startSTT(context, onChunk, language) }.getOrThrow()
83
102
  }
84
- // Car connected but API < 33: EXTRA_AUDIO_SOURCE unavailable, fall back to PCM
85
- return startPCM(silenceThresholdMs, maxDurationMs, onChunk)
103
+ } else {
104
+ startPCM(silenceThresholdMs, maxDurationMs, onChunk)
86
105
  }
87
- return ThreadUtil.postOnUiAndAwait { startSTT(context, onChunk, language) }.getOrThrow()
106
+ } else {
107
+ startPCM(silenceThresholdMs, maxDurationMs, onChunk)
88
108
  }
109
+ if (cancelledByUser) throw VoiceInputCancelledException()
110
+ endSoundUri?.let { playSound(it) }
111
+ return result
112
+ } finally {
113
+ abandonAudioFocus()
89
114
  }
90
- return startPCM(silenceThresholdMs, maxDurationMs, onChunk)
91
115
  }
92
116
 
93
117
  // MARK: - STT path (SpeechRecognizer owns the mic)
@@ -309,35 +333,8 @@ class VoiceInputManager(
309
333
  return@suspendCancellableCoroutine
310
334
  }
311
335
 
312
- val appContext = NitroModules.applicationContext ?: run {
313
- cont.resumeWithException(SecurityException("Missing application context"))
314
- return@suspendCancellableCoroutine
315
- }
316
-
317
336
  pcmContinuation = cont
318
337
 
319
- val audioManager = appContext.getSystemService(AudioManager::class.java)
320
-
321
- val audioAttributes =
322
- AudioAttributes.Builder().setContentType(AudioAttributes.CONTENT_TYPE_SPEECH)
323
- .setUsage(AudioAttributes.USAGE_ASSISTANCE_NAVIGATION_GUIDANCE).build()
324
-
325
- val focusRequest =
326
- AudioFocusRequest.Builder(AudioManager.AUDIOFOCUS_GAIN_TRANSIENT_EXCLUSIVE)
327
- .setAudioAttributes(audioAttributes).setOnAudioFocusChangeListener { state ->
328
- if (state == AudioManager.AUDIOFOCUS_LOSS) {
329
- stop()
330
- }
331
- }.build()
332
-
333
- if (audioManager.requestAudioFocus(focusRequest) != AudioManager.AUDIOFOCUS_REQUEST_GRANTED) {
334
- pcmContinuation = null
335
- cont.resumeWithException(IllegalStateException("Audio focus request denied"))
336
- return@suspendCancellableCoroutine
337
- }
338
-
339
- audioFocusRequest = focusRequest
340
-
341
338
  val bufferSize: Int
342
339
 
343
340
  if (carContext != null) {
@@ -381,6 +378,8 @@ class VoiceInputManager(
381
378
  ) ?: -1
382
379
 
383
380
  if (read < 0) {
381
+ // Whenever the user dismisses the microphone on the car screen, the next call to read will return -1
382
+ cancelledByUser = carAudioRecord != null && read == -1
384
383
  break
385
384
  }
386
385
 
@@ -450,6 +449,72 @@ class VoiceInputManager(
450
449
  audioRecord?.stop()
451
450
  }
452
451
 
452
+ @RequiresApi(Build.VERSION_CODES.O)
453
+ private fun requestAudioFocus(): Boolean {
454
+ val appContext = NitroModules.applicationContext ?: return false
455
+ val audioManager = appContext.getSystemService(AudioManager::class.java)
456
+ val audioAttributes = AudioAttributes.Builder()
457
+ .setContentType(AudioAttributes.CONTENT_TYPE_SPEECH)
458
+ .setUsage(AudioAttributes.USAGE_ASSISTANCE_NAVIGATION_GUIDANCE)
459
+ .build()
460
+ val focusRequest = AudioFocusRequest.Builder(AudioManager.AUDIOFOCUS_GAIN_TRANSIENT_EXCLUSIVE)
461
+ .setAudioAttributes(audioAttributes)
462
+ .setOnAudioFocusChangeListener { state ->
463
+ if (state == AudioManager.AUDIOFOCUS_LOSS) { stop() }
464
+ }
465
+ .build()
466
+ return if (audioManager.requestAudioFocus(focusRequest) == AudioManager.AUDIOFOCUS_REQUEST_GRANTED) {
467
+ audioFocusRequest = focusRequest
468
+ true
469
+ } else {
470
+ false
471
+ }
472
+ }
473
+
474
+ @RequiresApi(Build.VERSION_CODES.O)
475
+ private fun abandonAudioFocus() {
476
+ audioFocusRequest?.let {
477
+ val audioManager = (NitroModules.applicationContext ?: carContext)
478
+ ?.getSystemService(AudioManager::class.java)
479
+ audioManager?.abandonAudioFocusRequest(it)
480
+ }
481
+ audioFocusRequest = null
482
+ }
483
+
484
+ private suspend fun playSound(uri: String) = suspendCancellableCoroutine<Unit> { cont ->
485
+ val context = NitroModules.applicationContext ?: run {
486
+ cont.resume(Unit)
487
+ return@suspendCancellableCoroutine
488
+ }
489
+ val player = MediaPlayer()
490
+ try {
491
+ player.setAudioAttributes(
492
+ AudioAttributes.Builder()
493
+ .setContentType(AudioAttributes.CONTENT_TYPE_SONIFICATION)
494
+ .setUsage(AudioAttributes.USAGE_ASSISTANCE_NAVIGATION_GUIDANCE)
495
+ .build()
496
+ )
497
+ player.setDataSource(context, Uri.parse(uri))
498
+ player.setOnCompletionListener {
499
+ it.release()
500
+ if (cont.isActive) { cont.resume(Unit) }
501
+ }
502
+ player.setOnErrorListener { mp, _, _ ->
503
+ mp.release()
504
+ if (cont.isActive) { cont.resume(Unit) }
505
+ true
506
+ }
507
+ player.prepare()
508
+ player.start()
509
+ } catch (_: Exception) {
510
+ try { player.release() } catch (_: Exception) {}
511
+ if (cont.isActive) { cont.resume(Unit) }
512
+ }
513
+ cont.invokeOnCancellation {
514
+ try { player.release() } catch (_: Exception) {}
515
+ }
516
+ }
517
+
453
518
  @RequiresApi(Build.VERSION_CODES.O)
454
519
  private fun releaseResources() {
455
520
  carAudioRecord?.stopRecording()
@@ -458,13 +523,6 @@ class VoiceInputManager(
458
523
  audioRecord?.release()
459
524
  audioRecord = null
460
525
  recordingJob = null
461
- audioFocusRequest?.let {
462
- val audioManager = (NitroModules.applicationContext ?: carContext)?.getSystemService(
463
- AudioManager::class.java,
464
- )
465
- audioManager?.abandonAudioFocusRequest(it)
466
- }
467
- audioFocusRequest = null
468
526
  }
469
527
 
470
528
  fun dispose() {
package/ios/Types.swift CHANGED
@@ -10,7 +10,7 @@ struct TemplateEventPayload {
10
10
  let state: VisibilityState
11
11
  }
12
12
 
13
- enum AutoPlayError: Error {
13
+ enum AutoPlayError: LocalizedError {
14
14
  case templateNotFound(String)
15
15
  case interfaceControllerNotFound(String)
16
16
  case invalidTemplateError(String)
@@ -19,4 +19,5 @@ enum AutoPlayError: Error {
19
19
  case invalidTemplateType(String)
20
20
  case noUiWindow(String)
21
21
  case initReactRootViewFailed(String)
22
+ case voiceInputCancelled
22
23
  }
@@ -85,3 +85,10 @@ extension CPAlertTemplate {
85
85
  initTemplate(template: self, id: id)
86
86
  }
87
87
  }
88
+
89
+ extension CPVoiceControlTemplate {
90
+ convenience init(voiceControlStates: [CPVoiceControlState], id: String) {
91
+ self.init(voiceControlStates: voiceControlStates)
92
+ initTemplate(template: self, id: id)
93
+ }
94
+ }
@@ -36,7 +36,9 @@ class HybridVoice: HybridVoiceSpec {
36
36
  listeningImageRepeats: Bool?,
37
37
  preferSpeechToText: Bool?,
38
38
  onChunk: ((_ chunk: VoiceInputChunk) -> Void)?,
39
- language: String?
39
+ language: String?,
40
+ startSoundUri: String?,
41
+ endSoundUri: String?
40
42
  ) throws -> Promise<VoiceInputResult> {
41
43
  return Promise.async {
42
44
  let interfaceController = try? await RootModule.withInterfaceController { $0 }
@@ -55,7 +57,9 @@ class HybridVoice: HybridVoiceSpec {
55
57
  listeningImageRepeats: listeningImageRepeats,
56
58
  preferSpeechToText: preferSpeechToText ?? false,
57
59
  onChunk: onChunk,
58
- language: language
60
+ language: language,
61
+ startSoundUri: startSoundUri,
62
+ endSoundUri: endSoundUri
59
63
  )
60
64
  }
61
65
  }
@@ -0,0 +1,33 @@
1
+ import CarPlay
2
+
3
+ class VoiceInputTemplate: AutoPlayTemplate {
4
+ let template: CPVoiceControlTemplate
5
+ private let onDidDisappearCallback: () -> Void
6
+
7
+ override func getTemplate() -> CPTemplate {
8
+ return template
9
+ }
10
+
11
+ init(
12
+ voiceControlStates: [CPVoiceControlState],
13
+ id: String,
14
+ onDidDisappear: @escaping () -> Void
15
+ ) {
16
+ self.template = CPVoiceControlTemplate(voiceControlStates: voiceControlStates, id: id)
17
+ self.onDidDisappearCallback = onDidDisappear
18
+
19
+ super.init()
20
+
21
+ try? RootModule.withTemplateStore { templateStore in
22
+ templateStore.addTemplate(template: self, templateId: id)
23
+ }
24
+ }
25
+
26
+ override func onDidDisappear(animated: Bool) {
27
+ onDidDisappearCallback()
28
+
29
+ try? RootModule.withTemplateStore { templateStore in
30
+ templateStore.removeTemplate(templateId: self.template.id)
31
+ }
32
+ }
33
+ }
@@ -3,6 +3,31 @@ import CarPlay
3
3
  import NitroModules
4
4
  import Speech
5
5
 
6
+ /// Keeps itself and the player alive until playback finishes.
7
+ /// Needed because AVAudioPlayer.delegate is weak, so without an external strong
8
+ /// reference the delegate (and player) would be released immediately after play().
9
+ private final class AudioPlayerDelegate: NSObject, AVAudioPlayerDelegate, @unchecked Sendable {
10
+ private let onFinish: () -> Void
11
+ private var keepAlive: AudioPlayerDelegate?
12
+ private var player: AVAudioPlayer?
13
+
14
+ init(player: AVAudioPlayer, _ onFinish: @escaping () -> Void) {
15
+ self.onFinish = onFinish
16
+ self.player = player
17
+ super.init()
18
+ keepAlive = self
19
+ }
20
+
21
+ private func finish() {
22
+ player = nil
23
+ keepAlive = nil
24
+ onFinish()
25
+ }
26
+
27
+ func audioPlayerDidFinishPlaying(_: AVAudioPlayer, successfully _: Bool) { finish() }
28
+ func audioPlayerDecodeErrorDidOccur(_: AVAudioPlayer, error _: Error?) { finish() }
29
+ }
30
+
6
31
  /// Wraps CheckedContinuation so it can only be resumed once even when
7
32
  /// shared between a stop() call and an async recognition task callback.
8
33
  private final class ResultBox: @unchecked Sendable {
@@ -36,6 +61,8 @@ class VoiceInputManager {
36
61
  private var resultBox: ResultBox?
37
62
  private var samples: [Int16] = []
38
63
  private var isStopping = false
64
+ private var cancelledByUser = false
65
+ private var isIgnoringSamples = false
39
66
  private let stopLock = NSLock()
40
67
 
41
68
  // STT
@@ -69,9 +96,15 @@ class VoiceInputManager {
69
96
  listeningImageRepeats: Bool?,
70
97
  preferSpeechToText: Bool,
71
98
  onChunk: ((_ chunk: VoiceInputChunk) -> Void)?,
72
- language: String?
99
+ language: String?,
100
+ startSoundUri: String?,
101
+ endSoundUri: String?
73
102
  ) async throws -> VoiceInputResult {
74
- return try await withCheckedThrowingContinuation { cont in
103
+ stopLock.withLock {
104
+ cancelledByUser = false
105
+ }
106
+
107
+ let result = try await withCheckedThrowingContinuation { cont in
75
108
  let box = ResultBox(cont)
76
109
  self.resultBox = box
77
110
  self.samples = []
@@ -83,9 +116,6 @@ class VoiceInputManager {
83
116
  interfaceController: interfaceController,
84
117
  silenceThresholdMs: silenceThresholdMs,
85
118
  maxDurationMs: maxDurationMs,
86
- listeningText: listeningText,
87
- listeningImage: listeningImage,
88
- listeningImageRepeats: listeningImageRepeats,
89
119
  preferSpeechToText: preferSpeechToText,
90
120
  onChunk: onChunk,
91
121
  box: box,
@@ -95,8 +125,68 @@ class VoiceInputManager {
95
125
  catch {
96
126
  self.cleanup(interfaceController: interfaceController)
97
127
  box.resume(throwing: error)
128
+ return
129
+ }
130
+
131
+ // Mic is open — present template then play start sound.
132
+ // isIgnoringSamples discards tap buffers during the sound so it isn't recorded.
133
+ // recordingStart is set after the sound so silence/max-duration timers are accurate.
134
+ self.stopLock.withLock { self.isIgnoringSamples = startSoundUri != nil }
135
+ Task {
136
+ if let interfaceController = interfaceController {
137
+ await self.presentVoiceTemplate(
138
+ interfaceController: interfaceController,
139
+ listeningText: listeningText,
140
+ listeningImage: listeningImage,
141
+ listeningImageRepeats: listeningImageRepeats
142
+ )
143
+ }
144
+ if let uri = startSoundUri {
145
+ await self.playSound(uri: uri, setupSession: false)
146
+ }
147
+ self.stopLock.withLock {
148
+ self.isIgnoringSamples = false
149
+ self.recordingStart = Date()
150
+ }
151
+ }
152
+ }
153
+
154
+ if let uri = endSoundUri {
155
+ await playSound(uri: uri)
156
+ }
157
+
158
+ return result
159
+ }
160
+
161
+ private func playSound(uri: String, setupSession: Bool = true) async {
162
+ guard let url = URL(string: uri) else { return }
163
+ do {
164
+ let session = AVAudioSession.sharedInstance()
165
+ if setupSession {
166
+ try session.setCategory(.playback, mode: .default)
167
+ try session.setActive(true)
168
+ }
169
+ // URLSession handles both http:// (Metro dev server) and file:// (release bundle)
170
+ let (data, _) = try await URLSession.shared.data(from: url)
171
+ await withCheckedContinuation { (cont: CheckedContinuation<Void, Never>) in
172
+ DispatchQueue.main.async {
173
+ do {
174
+ let player = try AVAudioPlayer(data: data)
175
+ let delegate = AudioPlayerDelegate(player: player) { cont.resume() }
176
+ player.delegate = delegate
177
+ player.prepareToPlay()
178
+ player.play()
179
+ }
180
+ catch {
181
+ cont.resume()
182
+ }
183
+ }
98
184
  }
99
185
  }
186
+ catch {
187
+ print(error)
188
+ // fail silently — a broken sound file must not block voice input
189
+ }
100
190
  }
101
191
 
102
192
  func stop(interfaceController: AutoPlayInterfaceController? = nil) {
@@ -106,6 +196,7 @@ class VoiceInputManager {
106
196
  return
107
197
  }
108
198
  isStopping = true
199
+ let wasCancelled = cancelledByUser
109
200
  let wasSTTMode = isSTTMode
110
201
  let capturedRequest = recognitionRequest
111
202
  let box = resultBox
@@ -121,7 +212,12 @@ class VoiceInputManager {
121
212
  }
122
213
  else {
123
214
  cleanup(interfaceController: interfaceController)
124
- box?.resume(returning: makePCMResult(from: capturedSamples))
215
+ if wasCancelled {
216
+ box?.resume(throwing: AutoPlayError.voiceInputCancelled)
217
+ }
218
+ else {
219
+ box?.resume(returning: makePCMResult(from: capturedSamples))
220
+ }
125
221
  }
126
222
  }
127
223
 
@@ -131,9 +227,6 @@ class VoiceInputManager {
131
227
  interfaceController: AutoPlayInterfaceController?,
132
228
  silenceThresholdMs: Double,
133
229
  maxDurationMs: Double,
134
- listeningText: String,
135
- listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?,
136
- listeningImageRepeats: Bool?,
137
230
  preferSpeechToText: Bool,
138
231
  onChunk: ((_ chunk: VoiceInputChunk) -> Void)?,
139
232
  box: ResultBox,
@@ -147,15 +240,6 @@ class VoiceInputManager {
147
240
  try session.setCategory(.playAndRecord, mode: .measurement, options: [])
148
241
  try session.setActive(true)
149
242
 
150
- if let interfaceController {
151
- presentVoiceTemplate(
152
- interfaceController: interfaceController,
153
- listeningText: listeningText,
154
- listeningImage: listeningImage,
155
- listeningImageRepeats: listeningImageRepeats
156
- )
157
- }
158
-
159
243
  var activeRecognitionRequest: SFSpeechAudioBufferRecognitionRequest? = nil
160
244
 
161
245
  if preferSpeechToText, SFSpeechRecognizer.authorizationStatus() == .authorized,
@@ -176,12 +260,18 @@ class VoiceInputManager {
176
260
  // STT failed — fall back to whatever PCM was accumulated
177
261
  self.stopLock.lock()
178
262
  self.isStopping = true
263
+ let wasCancelled = self.cancelledByUser
179
264
  let capturedSamples = self.samples
180
265
  self.samples = []
181
266
  self.stopLock.unlock()
182
267
 
183
268
  self.cleanup(interfaceController: interfaceController)
184
- box.resume(returning: self.makePCMResult(from: capturedSamples))
269
+ if wasCancelled {
270
+ box.resume(throwing: AutoPlayError.voiceInputCancelled)
271
+ }
272
+ else {
273
+ box.resume(returning: self.makePCMResult(from: capturedSamples))
274
+ }
185
275
  return
186
276
  }
187
277
 
@@ -190,16 +280,22 @@ class VoiceInputManager {
190
280
  if result.isFinal {
191
281
  self.stopLock.lock()
192
282
  self.isStopping = true
283
+ let wasCancelled = self.cancelledByUser
193
284
  self.samples = []
194
285
  self.stopLock.unlock()
195
286
 
196
287
  self.cleanup(interfaceController: interfaceController)
197
- box.resume(
198
- returning: VoiceInputResult(
199
- transcription: result.bestTranscription.formattedString,
200
- audio: nil
288
+ if wasCancelled {
289
+ box.resume(throwing: AutoPlayError.voiceInputCancelled)
290
+ }
291
+ else {
292
+ box.resume(
293
+ returning: VoiceInputResult(
294
+ transcription: result.bestTranscription.formattedString,
295
+ audio: nil
296
+ )
201
297
  )
202
- )
298
+ }
203
299
  }
204
300
  else {
205
301
  onChunk?(VoiceInputChunk(partial: result.bestTranscription.formattedString, audio: nil))
@@ -215,15 +311,24 @@ class VoiceInputManager {
215
311
  throw VoiceInputError.converterUnavailable
216
312
  }
217
313
 
218
- recordingStart = Date()
314
+ recordingStart = nil
219
315
  silenceStart = nil
316
+ isIgnoringSamples = false
220
317
 
221
318
  inputNode.installTap(
222
319
  onBus: 0,
223
320
  bufferSize: VoiceInputManager.tapBufferSize,
224
321
  format: nativeFormat
225
322
  ) { [weak self] buffer, _ in
226
- guard let self, !self.isStopping else { return }
323
+ guard let self else { return }
324
+
325
+ self.stopLock.lock()
326
+ let stopping = self.isStopping
327
+ let ignoringSamples = self.isIgnoringSamples
328
+ let recordingStartSnapshot = self.recordingStart
329
+ self.stopLock.unlock()
330
+
331
+ guard !stopping, !ignoringSamples else { return }
227
332
 
228
333
  // Feed STT if active
229
334
  activeRecognitionRequest?.append(buffer)
@@ -266,7 +371,7 @@ class VoiceInputManager {
266
371
  let now = Date()
267
372
 
268
373
  // Max duration — applies in both modes
269
- if let start = self.recordingStart,
374
+ if let start = recordingStartSnapshot,
270
375
  now.timeIntervalSince(start) * 1000 >= maxDurationMs
271
376
  {
272
377
  self.triggerAutoStop(interfaceController: interfaceController)
@@ -275,7 +380,7 @@ class VoiceInputManager {
275
380
 
276
381
  // Silence detection — skip during warm-up so the pipeline has time
277
382
  // to stabilise before we start measuring amplitude
278
- if let start = self.recordingStart,
383
+ if let start = recordingStartSnapshot,
279
384
  now.timeIntervalSince(start) * 1000 >= VoiceInputManager.warmupMs
280
385
  {
281
386
  let peak = newSamples.reduce(0) { max($0, abs(Int($1))) }
@@ -381,12 +486,13 @@ class VoiceInputManager {
381
486
  return nil
382
487
  }
383
488
 
489
+ @MainActor
384
490
  private func presentVoiceTemplate(
385
491
  interfaceController: AutoPlayInterfaceController,
386
492
  listeningText: String,
387
493
  listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?,
388
494
  listeningImageRepeats: Bool?
389
- ) {
495
+ ) async {
390
496
  let traitCollection = SceneStore.getRootTraitCollection() ?? UITraitCollection.current
391
497
  let image = loadVoiceImage(
392
498
  image: listeningImage,
@@ -400,14 +506,21 @@ class VoiceInputManager {
400
506
  image: image,
401
507
  repeats: repeats
402
508
  )
403
- let template = CPVoiceControlTemplate(voiceControlStates: [listeningState])
404
- initTemplate(template: template, id: "voice-input")
405
- voiceControlTemplate = template
406
509
 
407
- Task { @MainActor in
408
- try? await interfaceController.presentTemplate(template, animated: true)
409
- template.activateVoiceControlState(withIdentifier: "listening")
510
+ let voiceTemplate = VoiceInputTemplate(
511
+ voiceControlStates: [listeningState],
512
+ id: "voice-input"
513
+ ) { [weak self] in
514
+ guard let self else { return }
515
+ self.stopLock.withLock {
516
+ if !self.isStopping { self.cancelledByUser = true }
517
+ }
518
+ self.stop()
410
519
  }
520
+
521
+ voiceControlTemplate = voiceTemplate.template
522
+ try? await interfaceController.presentTemplate(voiceTemplate.template, animated: true)
523
+ voiceTemplate.template.activateVoiceControlState(withIdentifier: "listening")
411
524
  }
412
525
 
413
526
  private func dismissVoiceTemplate(interfaceController: AutoPlayInterfaceController) {
@@ -1,10 +1,13 @@
1
+ import { Image } from 'react-native';
1
2
  import { NitroModules } from 'react-native-nitro-modules';
2
3
  import { NitroImageUtil } from '../utils/NitroImage';
3
4
  const _native = NitroModules.createHybridObject('Voice');
4
5
  const startVoiceInput = async (options) => {
5
- const { onChunk, silenceThresholdMs, maxDurationMs, listeningText, listeningImage, preferSpeechToText, language, } = options ?? {};
6
+ const { onChunk, silenceThresholdMs, maxDurationMs, listeningText, listeningImage, preferSpeechToText, language, startSound, endSound, } = options ?? {};
6
7
  const listeningImageRepeats = listeningImage?.type === 'asset' ? listeningImage.repeats : undefined;
7
- return await _native.startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, NitroImageUtil.convert(listeningImage), listeningImageRepeats, preferSpeechToText, onChunk, language);
8
+ const startSoundUri = startSound != null ? Image.resolveAssetSource(startSound).uri : undefined;
9
+ const endSoundUri = endSound != null ? Image.resolveAssetSource(endSound).uri : undefined;
10
+ return await _native.startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, NitroImageUtil.convert(listeningImage), listeningImageRepeats, preferSpeechToText, onChunk, language, startSoundUri, endSoundUri);
8
11
  };
9
12
  export const HybridVoice = {
10
13
  /**
@@ -7,6 +7,6 @@ export interface Voice extends HybridObject<{
7
7
  }> {
8
8
  hasVoiceInputPermission(): boolean;
9
9
  requestVoiceInputPermission(): Promise<boolean>;
10
- startVoiceInput(silenceThresholdMs?: number, maxDurationMs?: number, listeningText?: string, listeningImage?: NitroImage, listeningImageRepeats?: boolean, preferSpeechToText?: boolean, onChunk?: (chunk: VoiceInputChunk) => void, language?: string): Promise<VoiceInputResult>;
10
+ startVoiceInput(silenceThresholdMs?: number, maxDurationMs?: number, listeningText?: string, listeningImage?: NitroImage, listeningImageRepeats?: boolean, preferSpeechToText?: boolean, onChunk?: (chunk: VoiceInputChunk) => void, language?: string, startSoundUri?: string, endSoundUri?: string): Promise<VoiceInputResult>;
11
11
  stopVoiceInput(): void;
12
12
  }
@@ -18,4 +18,8 @@ export interface VoiceInputOptions {
18
18
  preferSpeechToText?: boolean;
19
19
  onChunk?: (chunk: VoiceInputChunk) => void;
20
20
  language?: string;
21
+ /** Sound played just before recording starts. Pass a Metro asset: `require('./beep_start.wav')`. */
22
+ startSound?: number;
23
+ /** Sound played just after recording stops. Pass a Metro asset: `require('./beep_end.wav')`. */
24
+ endSound?: number;
21
25
  }
@@ -1,3 +1,4 @@
1
1
  export declare const ErrorUtil: {
2
2
  isTemplateNotFoundError: (error: unknown) => error is Error;
3
+ isVoiceInputCanceledError: (error: unknown) => error is Error;
3
4
  };
@@ -1,4 +1,4 @@
1
- const isTemplateNotFoundError = (error) => {
1
+ const isError = (error) => {
2
2
  if (error == null) {
3
3
  return false;
4
4
  }
@@ -14,6 +14,18 @@ const isTemplateNotFoundError = (error) => {
14
14
  if (typeof error.message !== 'string') {
15
15
  return false;
16
16
  }
17
- return error.message.startsWith('templateNotFound');
17
+ return true;
18
+ };
19
+ const isTemplateNotFoundError = (error) => {
20
+ if (isError(error)) {
21
+ return error.message.startsWith('templateNotFound');
22
+ }
23
+ return false;
24
+ };
25
+ const isVoiceInputCanceledError = (error) => {
26
+ if (isError(error)) {
27
+ return error.message.startsWith('voiceInputCancelled');
28
+ }
29
+ return false;
18
30
  };
19
- export const ErrorUtil = { isTemplateNotFoundError };
31
+ export const ErrorUtil = { isTemplateNotFoundError, isVoiceInputCanceledError };
@@ -98,9 +98,9 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
98
98
  return __promise;
99
99
  }();
100
100
  }
101
- std::shared_ptr<Promise<VoiceInputResult>> JHybridVoiceSpec::startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language) {
102
- static const auto method = _javaPart->javaClassStatic()->getMethod<jni::local_ref<JPromise::javaobject>(jni::alias_ref<jni::JDouble> /* silenceThresholdMs */, jni::alias_ref<jni::JDouble> /* maxDurationMs */, jni::alias_ref<jni::JString> /* listeningText */, jni::alias_ref<JVariant_GlyphImage_AssetImage_RemoteImage> /* listeningImage */, jni::alias_ref<jni::JBoolean> /* listeningImageRepeats */, jni::alias_ref<jni::JBoolean> /* preferSpeechToText */, jni::alias_ref<JFunc_void_VoiceInputChunk::javaobject> /* onChunk */, jni::alias_ref<jni::JString> /* language */)>("startVoiceInput_cxx");
103
- auto __result = method(_javaPart, silenceThresholdMs.has_value() ? jni::JDouble::valueOf(silenceThresholdMs.value()) : nullptr, maxDurationMs.has_value() ? jni::JDouble::valueOf(maxDurationMs.value()) : nullptr, listeningText.has_value() ? jni::make_jstring(listeningText.value()) : nullptr, listeningImage.has_value() ? JVariant_GlyphImage_AssetImage_RemoteImage::fromCpp(listeningImage.value()) : nullptr, listeningImageRepeats.has_value() ? jni::JBoolean::valueOf(listeningImageRepeats.value()) : nullptr, preferSpeechToText.has_value() ? jni::JBoolean::valueOf(preferSpeechToText.value()) : nullptr, onChunk.has_value() ? JFunc_void_VoiceInputChunk_cxx::fromCpp(onChunk.value()) : nullptr, language.has_value() ? jni::make_jstring(language.value()) : nullptr);
101
+ std::shared_ptr<Promise<VoiceInputResult>> JHybridVoiceSpec::startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri) {
102
+ static const auto method = _javaPart->javaClassStatic()->getMethod<jni::local_ref<JPromise::javaobject>(jni::alias_ref<jni::JDouble> /* silenceThresholdMs */, jni::alias_ref<jni::JDouble> /* maxDurationMs */, jni::alias_ref<jni::JString> /* listeningText */, jni::alias_ref<JVariant_GlyphImage_AssetImage_RemoteImage> /* listeningImage */, jni::alias_ref<jni::JBoolean> /* listeningImageRepeats */, jni::alias_ref<jni::JBoolean> /* preferSpeechToText */, jni::alias_ref<JFunc_void_VoiceInputChunk::javaobject> /* onChunk */, jni::alias_ref<jni::JString> /* language */, jni::alias_ref<jni::JString> /* startSoundUri */, jni::alias_ref<jni::JString> /* endSoundUri */)>("startVoiceInput_cxx");
103
+ auto __result = method(_javaPart, silenceThresholdMs.has_value() ? jni::JDouble::valueOf(silenceThresholdMs.value()) : nullptr, maxDurationMs.has_value() ? jni::JDouble::valueOf(maxDurationMs.value()) : nullptr, listeningText.has_value() ? jni::make_jstring(listeningText.value()) : nullptr, listeningImage.has_value() ? JVariant_GlyphImage_AssetImage_RemoteImage::fromCpp(listeningImage.value()) : nullptr, listeningImageRepeats.has_value() ? jni::JBoolean::valueOf(listeningImageRepeats.value()) : nullptr, preferSpeechToText.has_value() ? jni::JBoolean::valueOf(preferSpeechToText.value()) : nullptr, onChunk.has_value() ? JFunc_void_VoiceInputChunk_cxx::fromCpp(onChunk.value()) : nullptr, language.has_value() ? jni::make_jstring(language.value()) : nullptr, startSoundUri.has_value() ? jni::make_jstring(startSoundUri.value()) : nullptr, endSoundUri.has_value() ? jni::make_jstring(endSoundUri.value()) : nullptr);
104
104
  return [&]() {
105
105
  auto __promise = Promise<VoiceInputResult>::create();
106
106
  __result->cthis()->addOnResolvedListener([=](const jni::alias_ref<jni::JObject>& __boxedResult) {
@@ -56,7 +56,7 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
56
56
  // Methods
57
57
  bool hasVoiceInputPermission() override;
58
58
  std::shared_ptr<Promise<bool>> requestVoiceInputPermission() override;
59
- std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language) override;
59
+ std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri) override;
60
60
  void stopVoiceInput() override;
61
61
 
62
62
  private:
@@ -37,12 +37,12 @@ abstract class HybridVoiceSpec: HybridObject() {
37
37
  @Keep
38
38
  abstract fun requestVoiceInputPermission(): Promise<Boolean>
39
39
 
40
- abstract fun startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Boolean?, preferSpeechToText: Boolean?, onChunk: ((chunk: VoiceInputChunk) -> Unit)?, language: String?): Promise<VoiceInputResult>
40
+ abstract fun startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Boolean?, preferSpeechToText: Boolean?, onChunk: ((chunk: VoiceInputChunk) -> Unit)?, language: String?, startSoundUri: String?, endSoundUri: String?): Promise<VoiceInputResult>
41
41
 
42
42
  @DoNotStrip
43
43
  @Keep
44
- private fun startVoiceInput_cxx(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Boolean?, preferSpeechToText: Boolean?, onChunk: Func_void_VoiceInputChunk?, language: String?): Promise<VoiceInputResult> {
45
- val __result = startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, listeningImage, listeningImageRepeats, preferSpeechToText, onChunk?.let { it }, language)
44
+ private fun startVoiceInput_cxx(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Boolean?, preferSpeechToText: Boolean?, onChunk: Func_void_VoiceInputChunk?, language: String?, startSoundUri: String?, endSoundUri: String?): Promise<VoiceInputResult> {
45
+ val __result = startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, listeningImage, listeningImageRepeats, preferSpeechToText, onChunk?.let { it }, language, startSoundUri, endSoundUri)
46
46
  return __result
47
47
  }
48
48
 
@@ -107,8 +107,8 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
107
107
  auto __value = std::move(__result.value());
108
108
  return __value;
109
109
  }
110
- inline std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language) override {
111
- auto __result = _swiftPart.startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, listeningImage, listeningImageRepeats, preferSpeechToText, onChunk, language);
110
+ inline std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri) override {
111
+ auto __result = _swiftPart.startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, listeningImage, listeningImageRepeats, preferSpeechToText, onChunk, language, startSoundUri, endSoundUri);
112
112
  if (__result.hasError()) [[unlikely]] {
113
113
  std::rethrow_exception(__result.error());
114
114
  }
@@ -15,7 +15,7 @@ public protocol HybridVoiceSpec_protocol: HybridObject {
15
15
  // Methods
16
16
  func hasVoiceInputPermission() throws -> Bool
17
17
  func requestVoiceInputPermission() throws -> Promise<Bool>
18
- func startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Bool?, preferSpeechToText: Bool?, onChunk: ((_ chunk: VoiceInputChunk) -> Void)?, language: String?) throws -> Promise<VoiceInputResult>
18
+ func startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Bool?, preferSpeechToText: Bool?, onChunk: ((_ chunk: VoiceInputChunk) -> Void)?, language: String?, startSoundUri: String?, endSoundUri: String?) throws -> Promise<VoiceInputResult>
19
19
  func stopVoiceInput() throws -> Void
20
20
  }
21
21
 
@@ -156,7 +156,7 @@ open class HybridVoiceSpec_cxx {
156
156
  }
157
157
 
158
158
  @inline(__always)
159
- public final func startVoiceInput(silenceThresholdMs: bridge.std__optional_double_, maxDurationMs: bridge.std__optional_double_, listeningText: bridge.std__optional_std__string_, listeningImage: bridge.std__optional_std__variant_GlyphImage__AssetImage__RemoteImage__, listeningImageRepeats: bridge.std__optional_bool_, preferSpeechToText: bridge.std__optional_bool_, onChunk: bridge.std__optional_std__function_void_const_VoiceInputChunk_____chunk______, language: bridge.std__optional_std__string_) -> bridge.Result_std__shared_ptr_Promise_VoiceInputResult___ {
159
+ public final func startVoiceInput(silenceThresholdMs: bridge.std__optional_double_, maxDurationMs: bridge.std__optional_double_, listeningText: bridge.std__optional_std__string_, listeningImage: bridge.std__optional_std__variant_GlyphImage__AssetImage__RemoteImage__, listeningImageRepeats: bridge.std__optional_bool_, preferSpeechToText: bridge.std__optional_bool_, onChunk: bridge.std__optional_std__function_void_const_VoiceInputChunk_____chunk______, language: bridge.std__optional_std__string_, startSoundUri: bridge.std__optional_std__string_, endSoundUri: bridge.std__optional_std__string_) -> bridge.Result_std__shared_ptr_Promise_VoiceInputResult___ {
160
160
  do {
161
161
  let __result = try self.__implementation.startVoiceInput(silenceThresholdMs: { () -> Double? in
162
162
  if bridge.has_value_std__optional_double_(silenceThresholdMs) {
@@ -234,6 +234,20 @@ open class HybridVoiceSpec_cxx {
234
234
  } else {
235
235
  return nil
236
236
  }
237
+ }(), startSoundUri: { () -> String? in
238
+ if bridge.has_value_std__optional_std__string_(startSoundUri) {
239
+ let __unwrapped = bridge.get_std__optional_std__string_(startSoundUri)
240
+ return String(__unwrapped)
241
+ } else {
242
+ return nil
243
+ }
244
+ }(), endSoundUri: { () -> String? in
245
+ if bridge.has_value_std__optional_std__string_(endSoundUri) {
246
+ let __unwrapped = bridge.get_std__optional_std__string_(endSoundUri)
247
+ return String(__unwrapped)
248
+ } else {
249
+ return nil
250
+ }
237
251
  }())
238
252
  let __resultCpp = { () -> bridge.std__shared_ptr_Promise_VoiceInputResult__ in
239
253
  let __promise = bridge.create_std__shared_ptr_Promise_VoiceInputResult__()
@@ -68,7 +68,7 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
68
68
  // Methods
69
69
  virtual bool hasVoiceInputPermission() = 0;
70
70
  virtual std::shared_ptr<Promise<bool>> requestVoiceInputPermission() = 0;
71
- virtual std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language) = 0;
71
+ virtual std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri) = 0;
72
72
  virtual void stopVoiceInput() = 0;
73
73
 
74
74
  protected:
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@iternio/react-native-auto-play",
3
- "version": "0.5.4",
3
+ "version": "0.5.5",
4
4
  "description": "Android Auto and Apple CarPlay for react-native",
5
5
  "main": "lib/index",
6
6
  "module": "lib/index",
@@ -1,3 +1,4 @@
1
+ import { Image } from 'react-native';
1
2
  import { NitroModules } from 'react-native-nitro-modules';
2
3
  import type { Voice } from '../specs/Voice.nitro';
3
4
  import type { VoiceInputOptions, VoiceInputResult } from '../types/Voice';
@@ -21,11 +22,16 @@ const startVoiceInput: StartVoiceInput = async (options?: VoiceInputOptions) =>
21
22
  listeningImage,
22
23
  preferSpeechToText,
23
24
  language,
25
+ startSound,
26
+ endSound,
24
27
  } = options ?? {};
25
28
 
26
29
  const listeningImageRepeats =
27
30
  listeningImage?.type === 'asset' ? listeningImage.repeats : undefined;
28
31
 
32
+ const startSoundUri = startSound != null ? Image.resolveAssetSource(startSound).uri : undefined;
33
+ const endSoundUri = endSound != null ? Image.resolveAssetSource(endSound).uri : undefined;
34
+
29
35
  return await _native.startVoiceInput(
30
36
  silenceThresholdMs,
31
37
  maxDurationMs,
@@ -34,7 +40,9 @@ const startVoiceInput: StartVoiceInput = async (options?: VoiceInputOptions) =>
34
40
  listeningImageRepeats,
35
41
  preferSpeechToText,
36
42
  onChunk,
37
- language
43
+ language,
44
+ startSoundUri,
45
+ endSoundUri
38
46
  );
39
47
  };
40
48
 
@@ -13,7 +13,9 @@ export interface Voice extends HybridObject<{ android: 'kotlin'; ios: 'swift' }>
13
13
  listeningImageRepeats?: boolean,
14
14
  preferSpeechToText?: boolean,
15
15
  onChunk?: (chunk: VoiceInputChunk) => void,
16
- language?: string
16
+ language?: string,
17
+ startSoundUri?: string,
18
+ endSoundUri?: string
17
19
  ): Promise<VoiceInputResult>;
18
20
  stopVoiceInput(): void;
19
21
  }
@@ -21,4 +21,8 @@ export interface VoiceInputOptions {
21
21
  preferSpeechToText?: boolean;
22
22
  onChunk?: (chunk: VoiceInputChunk) => void;
23
23
  language?: string;
24
+ /** Sound played just before recording starts. Pass a Metro asset: `require('./beep_start.wav')`. */
25
+ startSound?: number;
26
+ /** Sound played just after recording stops. Pass a Metro asset: `require('./beep_end.wav')`. */
27
+ endSound?: number;
24
28
  }
@@ -1,4 +1,4 @@
1
- const isTemplateNotFoundError = (error: unknown): error is Error => {
1
+ const isError = (error: unknown): error is { message: string } => {
2
2
  if (error == null) {
3
3
  return false;
4
4
  }
@@ -19,7 +19,23 @@ const isTemplateNotFoundError = (error: unknown): error is Error => {
19
19
  return false;
20
20
  }
21
21
 
22
- return error.message.startsWith('templateNotFound');
22
+ return true;
23
+ };
24
+
25
+ const isTemplateNotFoundError = (error: unknown): error is Error => {
26
+ if (isError(error)) {
27
+ return error.message.startsWith('templateNotFound');
28
+ }
29
+
30
+ return false;
31
+ };
32
+
33
+ const isVoiceInputCanceledError = (error: unknown): error is Error => {
34
+ if (isError(error)) {
35
+ return error.message.startsWith('voiceInputCancelled');
36
+ }
37
+
38
+ return false;
23
39
  };
24
40
 
25
- export const ErrorUtil = { isTemplateNotFoundError };
41
+ export const ErrorUtil = { isTemplateNotFoundError, isVoiceInputCanceledError };