@iternio/react-native-auto-play 0.5.4 → 0.5.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +77 -17
- package/android/src/main/java/com/margelo/nitro/swe/iternio/reactnativeautoplay/AutoPlayError.kt +6 -1
- package/android/src/main/java/com/margelo/nitro/swe/iternio/reactnativeautoplay/HybridVoice.kt +7 -3
- package/android/src/main/java/com/margelo/nitro/swe/iternio/reactnativeautoplay/VoiceInputManager.kt +104 -45
- package/ios/Types.swift +2 -1
- package/ios/extensions/CarPlayTemplateExtensions.swift +7 -0
- package/ios/hybrid/HybridVoice.swift +6 -2
- package/ios/templates/VoiceInputTemplate.swift +33 -0
- package/ios/utils/VoiceInputManager.swift +161 -54
- package/lib/hybrid/HybridVoice.js +5 -2
- package/lib/specs/Voice.nitro.d.ts +1 -1
- package/lib/types/Voice.d.ts +4 -0
- package/lib/utils/ErrorUtil.d.ts +1 -0
- package/lib/utils/ErrorUtil.js +15 -3
- package/nitrogen/generated/android/c++/JHybridVoiceSpec.cpp +3 -3
- package/nitrogen/generated/android/c++/JHybridVoiceSpec.hpp +1 -1
- package/nitrogen/generated/android/kotlin/com/margelo/nitro/swe/iternio/reactnativeautoplay/HybridVoiceSpec.kt +3 -3
- package/nitrogen/generated/ios/c++/HybridVoiceSpecSwift.hpp +2 -2
- package/nitrogen/generated/ios/swift/HybridVoiceSpec.swift +1 -1
- package/nitrogen/generated/ios/swift/HybridVoiceSpec_cxx.swift +15 -1
- package/nitrogen/generated/shared/c++/HybridVoiceSpec.hpp +1 -1
- package/package.json +1 -1
- package/src/hybrid/HybridVoice.ts +9 -1
- package/src/specs/Voice.nitro.ts +3 -1
- package/src/types/Voice.ts +4 -0
- package/src/utils/ErrorUtil.ts +19 -3
package/README.md
CHANGED
|
@@ -14,7 +14,7 @@
|
|
|
14
14
|
## Features
|
|
15
15
|
|
|
16
16
|
- **Cross-Platform:** Write once, run on both Apple CarPlay and Android Auto.
|
|
17
|
-
-
|
|
17
|
+
- **New Architecture:** Supports React Native new architecture only.
|
|
18
18
|
- **Template-Based UI:** Utilize a rich set of templates like `MapTemplate`, `ListTemplate`, `GridTemplate`, and more to build UIs that comply with automotive design guidelines.
|
|
19
19
|
- **Navigation APIs:** Build full-featured navigation experiences with APIs for trip management, maneuvers, and route guidance.
|
|
20
20
|
- **Dashboard & Cluster Support:** Extend your app's presence to the CarPlay Dashboard (CarPlay only) and instrument cluster displays (CarPlay & Android Auto).
|
|
@@ -837,36 +837,95 @@ new ListTemplate({
|
|
|
837
837
|
|
|
838
838
|
### Voice Input
|
|
839
839
|
|
|
840
|
-
The library provides a cross-platform in-app voice recording API built on top of the car microphone (when connected) or the device microphone (when no car is connected).
|
|
840
|
+
The library provides a cross-platform in-app voice recording API built on top of the car microphone (when connected) or the device microphone (when no car is connected). The voice API lives in `HybridVoice`.
|
|
841
841
|
|
|
842
842
|
#### Permission
|
|
843
843
|
|
|
844
844
|
```ts
|
|
845
|
-
|
|
846
|
-
|
|
845
|
+
import { HybridVoice } from '@iternio/react-native-auto-play';
|
|
846
|
+
|
|
847
|
+
// Check whether permission is already granted (synchronous)
|
|
848
|
+
const granted = HybridVoice.hasVoiceInputPermission();
|
|
847
849
|
|
|
848
850
|
// Request permission if not yet granted
|
|
849
|
-
const granted = await
|
|
851
|
+
const granted = await HybridVoice.requestVoiceInputPermission();
|
|
850
852
|
```
|
|
851
853
|
|
|
854
|
+
On **iOS**: checks/requests both microphone and speech recognition authorization.
|
|
855
|
+
On **Android**: checks/requests `RECORD_AUDIO` via the car context when connected, otherwise via the RN application context.
|
|
856
|
+
|
|
852
857
|
#### Recording
|
|
853
858
|
|
|
854
859
|
```ts
|
|
855
|
-
|
|
856
|
-
|
|
857
|
-
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
|
|
861
|
-
|
|
860
|
+
import { HybridVoice, ErrorUtil } from '@iternio/react-native-auto-play';
|
|
861
|
+
|
|
862
|
+
try {
|
|
863
|
+
const result = await HybridVoice.startVoiceInput({
|
|
864
|
+
silenceThresholdMs: 1500, // ms of silence before auto-stop (default 1500)
|
|
865
|
+
maxDurationMs: 10_000, // hard cap on recording duration (default 10 000)
|
|
866
|
+
listeningText: 'Listening…', // iOS CarPlay: text shown on CPVoiceControlTemplate
|
|
867
|
+
preferSpeechToText: false, // true → STT transcription; false → raw PCM (default)
|
|
868
|
+
startSound: require('./beep_start.mp3'), // played just before recording starts
|
|
869
|
+
endSound: require('./beep_end.mp3'), // played just after recording stops
|
|
870
|
+
onChunk: (chunk) => {
|
|
871
|
+
// chunk.audio — raw PCM ArrayBuffer chunk (PCM mode)
|
|
872
|
+
// chunk.partial — partial transcription string (STT mode)
|
|
873
|
+
},
|
|
874
|
+
});
|
|
875
|
+
|
|
876
|
+
if (result.transcription) {
|
|
877
|
+
console.log('Transcription:', result.transcription);
|
|
878
|
+
} else if (result.audio) {
|
|
879
|
+
console.log(`PCM audio: ${result.audio.byteLength} bytes`);
|
|
880
|
+
}
|
|
881
|
+
} catch (e) {
|
|
882
|
+
if (ErrorUtil.isVoiceInputCanceledError(e)) {
|
|
883
|
+
// User pressed the cancel button on the car screen
|
|
884
|
+
console.log('Voice input cancelled');
|
|
885
|
+
} else {
|
|
886
|
+
console.error(e);
|
|
887
|
+
}
|
|
888
|
+
}
|
|
862
889
|
|
|
863
|
-
// Stop recording early — resolves startVoiceInput with
|
|
864
|
-
|
|
890
|
+
// Stop recording early — resolves startVoiceInput with audio captured so far
|
|
891
|
+
HybridVoice.stopVoiceInput();
|
|
865
892
|
```
|
|
866
893
|
|
|
867
|
-
|
|
894
|
+
| Option | Type | Default | Description |
|
|
895
|
+
|---|---|---|---|
|
|
896
|
+
| `silenceThresholdMs` | `number` | `1500` | Auto-stop after this many ms of silence |
|
|
897
|
+
| `maxDurationMs` | `number` | `10000` | Hard recording time limit |
|
|
898
|
+
| `listeningText` | `string` | — | iOS only — text shown on `CPVoiceControlTemplate` |
|
|
899
|
+
| `listeningImage` | `VoiceInputImage` | — | iOS only — animated image in the CarPlay overlay |
|
|
900
|
+
| `preferSpeechToText` | `boolean` | `false` | `true` → resolve with `{ transcription }`; `false` → resolve with `{ audio }` |
|
|
901
|
+
| `startSound` | `number` | — | Metro asset (`require('./beep.mp3')`) played before recording. Takes audio focus so other apps pause. |
|
|
902
|
+
| `endSound` | `number` | — | Metro asset played after recording stops |
|
|
903
|
+
| `onChunk` | `(chunk) => void` | — | Streaming callback: `chunk.audio` (PCM) or `chunk.partial` (STT) |
|
|
904
|
+
| `language` | `string` | system | BCP-47 language tag for the STT recognizer |
|
|
905
|
+
|
|
906
|
+
**PCM result** (`preferSpeechToText: false`, default): resolves with `{ audio: ArrayBuffer }` — raw 16 kHz, 16-bit, mono PCM.
|
|
907
|
+
|
|
908
|
+
**STT result** (`preferSpeechToText: true`): resolves with `{ transcription: string }` on success, or falls back to `{ audio }` if recognition is unavailable.
|
|
868
909
|
|
|
869
|
-
On **
|
|
910
|
+
On **Android**: uses `CarAudioRecord` when Android Auto is connected, otherwise falls back to standard `AudioRecord`. STT uses `SpeechRecognizer`.
|
|
911
|
+
|
|
912
|
+
On **iOS**: presents `CPVoiceControlTemplate` on the car screen when CarPlay is connected, and captures audio via `AVAudioEngine`. STT uses `SFSpeechRecognizer`.
|
|
913
|
+
|
|
914
|
+
#### Cancel detection
|
|
915
|
+
|
|
916
|
+
When the user presses the cancel button on the car screen, `startVoiceInput` rejects with a `voiceInputCancelled` error on both platforms. Use `ErrorUtil.isVoiceInputCanceledError` to distinguish it from other errors:
|
|
917
|
+
|
|
918
|
+
```ts
|
|
919
|
+
import { ErrorUtil } from '@iternio/react-native-auto-play';
|
|
920
|
+
|
|
921
|
+
HybridVoice.startVoiceInput().catch((e) => {
|
|
922
|
+
if (ErrorUtil.isVoiceInputCanceledError(e)) {
|
|
923
|
+
// user dismissed — no action needed
|
|
924
|
+
} else {
|
|
925
|
+
throw e;
|
|
926
|
+
}
|
|
927
|
+
});
|
|
928
|
+
```
|
|
870
929
|
|
|
871
930
|
#### OS-triggered voice input (Android only)
|
|
872
931
|
|
|
@@ -1015,7 +1074,8 @@ CarPlayDashboard.setButtons([
|
|
|
1015
1074
|
|
|
1016
1075
|
- **Broken exceptions with `react-native-skia`**: When using `react-native-skia` exceptions on iOS are not reported correctly. This is fixed since version `2.4.19` of `react-native-skia`. For more details, see this [pull request](https://github.com/Shopify/react-native-skia/pull/3595) and [issue](https://github.com/Shopify/react-native-skia/issues/3635).
|
|
1017
1076
|
- **AppState on iOS**: The `AppState` module from React Native does not work correctly on iOS because this library uses scenes, which are not supported by the stock `AppState` module. This library provides a custom state listener that works for both Android and iOS. Use `HybridAutoPlay.addListenerRenderState` instead of `AppState`.
|
|
1018
|
-
- **Timers stop on screen lock**: iOS stops all timers when the device
|
|
1077
|
+
- **Timers stop on screen lock**: iOS stops all timers when the device main screen is turned off. To ensure timers continue to run (which is often necessary for background tasks related to autoplay), a patch for `react-native` is required. A patch is included in the root `patches/` directory and can be applied using `patch-package`.
|
|
1078
|
+
In case you are using Expo SDK >= 56 make sure to set `buildReactNativeFromSource` to `true` in your app config for [expo-build-properties](https://docs.expo.dev/versions/latest/sdk/build-properties/#sharedbuildconfigfields), otherwise the patch can't be applied.
|
|
1019
1079
|
- **expo-splash-screen stuck on iOS**: The `expo-splash-screen` module is broken on iOS because it does not support scenes, which are used by this library. This can cause the splash screen to be stuck on either the mobile device or on CarPlay. To fix this, a patch for `expo-splash-screen` is included in the root `patches/` directory and can be applied using `patch-package`. After applying the patch, you can hide the splash screen for a specific scene by passing the module name to the `hide` or `hideAsync` function. The module name can be one of the values from the `AutoPlayModules` enum or the UUID of a cluster screen.
|
|
1020
1080
|
```tsx
|
|
1021
1081
|
import { hideAsync } from 'expo-splash-screen';
|
package/android/src/main/java/com/margelo/nitro/swe/iternio/reactnativeautoplay/AutoPlayError.kt
CHANGED
|
@@ -1,5 +1,10 @@
|
|
|
1
1
|
package com.margelo.nitro.swe.iternio.reactnativeautoplay
|
|
2
2
|
|
|
3
|
-
class TemplateNotFoundException(private val templateId: String):
|
|
3
|
+
class TemplateNotFoundException(private val templateId: String) :
|
|
4
|
+
Exception("templateNotFound(\"$templateId\")") {
|
|
4
5
|
override fun toString(): String = "templateNotFound(\"$templateId\")"
|
|
5
6
|
}
|
|
7
|
+
|
|
8
|
+
class VoiceInputCancelledException : Exception("voiceInputCancelled") {
|
|
9
|
+
override fun toString(): String = "voiceInputCancelled"
|
|
10
|
+
}
|
package/android/src/main/java/com/margelo/nitro/swe/iternio/reactnativeautoplay/HybridVoice.kt
CHANGED
|
@@ -50,7 +50,7 @@ class HybridVoice : HybridVoiceSpec() {
|
|
|
50
50
|
}
|
|
51
51
|
cont.resume(
|
|
52
52
|
grantResults.isNotEmpty() &&
|
|
53
|
-
|
|
53
|
+
grantResults.first() == PackageManager.PERMISSION_GRANTED
|
|
54
54
|
)
|
|
55
55
|
true
|
|
56
56
|
}
|
|
@@ -68,7 +68,9 @@ class HybridVoice : HybridVoiceSpec() {
|
|
|
68
68
|
listeningImageRepeats: Boolean?,
|
|
69
69
|
preferSpeechToText: Boolean?,
|
|
70
70
|
onChunk: ((chunk: VoiceInputChunk) -> Unit)?,
|
|
71
|
-
language: String
|
|
71
|
+
language: String?,
|
|
72
|
+
startSoundUri: String?,
|
|
73
|
+
endSoundUri: String?
|
|
72
74
|
): Promise<VoiceInputResult> {
|
|
73
75
|
return Promise.async {
|
|
74
76
|
if (Build.VERSION.SDK_INT < Build.VERSION_CODES.O) {
|
|
@@ -84,7 +86,9 @@ class HybridVoice : HybridVoiceSpec() {
|
|
|
84
86
|
maxDurationMs = maxDurationMs?.toLong() ?: 10_000L,
|
|
85
87
|
preferSpeechToText = preferSpeechToText ?: false,
|
|
86
88
|
onChunk = onChunk,
|
|
87
|
-
language = language
|
|
89
|
+
language = language,
|
|
90
|
+
startSoundUri = startSoundUri,
|
|
91
|
+
endSoundUri = endSoundUri,
|
|
88
92
|
)
|
|
89
93
|
} finally {
|
|
90
94
|
voiceInputManager = null
|
package/android/src/main/java/com/margelo/nitro/swe/iternio/reactnativeautoplay/VoiceInputManager.kt
CHANGED
|
@@ -10,7 +10,9 @@ import android.media.AudioFocusRequest
|
|
|
10
10
|
import android.media.AudioFormat
|
|
11
11
|
import android.media.AudioManager
|
|
12
12
|
import android.media.AudioRecord
|
|
13
|
+
import android.media.MediaPlayer
|
|
13
14
|
import android.media.MediaRecorder
|
|
15
|
+
import android.net.Uri
|
|
14
16
|
import android.os.Build
|
|
15
17
|
import android.os.Bundle
|
|
16
18
|
import android.os.ParcelFileDescriptor
|
|
@@ -62,6 +64,9 @@ class VoiceInputManager(
|
|
|
62
64
|
@Volatile
|
|
63
65
|
private var isRecording = false
|
|
64
66
|
|
|
67
|
+
@Volatile
|
|
68
|
+
private var cancelledByUser = false
|
|
69
|
+
|
|
65
70
|
// STT state — only set when SpeechRecognizer owns the mic
|
|
66
71
|
@Volatile
|
|
67
72
|
private var activeSpeechRecognizer: SpeechRecognizer? = null
|
|
@@ -72,22 +77,42 @@ class VoiceInputManager(
|
|
|
72
77
|
maxDurationMs: Long = 10_000,
|
|
73
78
|
preferSpeechToText: Boolean = false,
|
|
74
79
|
onChunk: ((chunk: VoiceInputChunk) -> Unit)? = null,
|
|
75
|
-
language: String? = null
|
|
80
|
+
language: String? = null,
|
|
81
|
+
startSoundUri: String? = null,
|
|
82
|
+
endSoundUri: String? = null,
|
|
76
83
|
): VoiceInputResult {
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
84
|
+
cancelledByUser = false
|
|
85
|
+
if (!requestAudioFocus()) {
|
|
86
|
+
throw IllegalStateException("Audio focus request denied")
|
|
87
|
+
}
|
|
88
|
+
try {
|
|
89
|
+
val startSoundJob = startSoundUri?.let { uri -> scope.launch { playSound(uri) } }
|
|
90
|
+
val result = if (preferSpeechToText) {
|
|
91
|
+
val context = NitroModules.applicationContext ?: throw IllegalArgumentException()
|
|
92
|
+
if (SpeechRecognizer.isRecognitionAvailable(context)) {
|
|
93
|
+
if (carContext != null) {
|
|
94
|
+
if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.TIRAMISU) {
|
|
95
|
+
startSTTFromCarAudio(silenceThresholdMs, maxDurationMs, onChunk, language)
|
|
96
|
+
} else {
|
|
97
|
+
// Car connected but API < 33: EXTRA_AUDIO_SOURCE unavailable, fall back to PCM
|
|
98
|
+
startPCM(silenceThresholdMs, maxDurationMs, onChunk)
|
|
99
|
+
}
|
|
100
|
+
} else {
|
|
101
|
+
ThreadUtil.postOnUiAndAwait { startSTT(context, onChunk, language) }.getOrThrow()
|
|
83
102
|
}
|
|
84
|
-
|
|
85
|
-
|
|
103
|
+
} else {
|
|
104
|
+
startPCM(silenceThresholdMs, maxDurationMs, onChunk)
|
|
86
105
|
}
|
|
87
|
-
|
|
106
|
+
} else {
|
|
107
|
+
startPCM(silenceThresholdMs, maxDurationMs, onChunk)
|
|
88
108
|
}
|
|
109
|
+
startSoundJob?.join()
|
|
110
|
+
if (cancelledByUser) throw VoiceInputCancelledException()
|
|
111
|
+
endSoundUri?.let { playSound(it) }
|
|
112
|
+
return result
|
|
113
|
+
} finally {
|
|
114
|
+
abandonAudioFocus()
|
|
89
115
|
}
|
|
90
|
-
return startPCM(silenceThresholdMs, maxDurationMs, onChunk)
|
|
91
116
|
}
|
|
92
117
|
|
|
93
118
|
// MARK: - STT path (SpeechRecognizer owns the mic)
|
|
@@ -309,35 +334,8 @@ class VoiceInputManager(
|
|
|
309
334
|
return@suspendCancellableCoroutine
|
|
310
335
|
}
|
|
311
336
|
|
|
312
|
-
val appContext = NitroModules.applicationContext ?: run {
|
|
313
|
-
cont.resumeWithException(SecurityException("Missing application context"))
|
|
314
|
-
return@suspendCancellableCoroutine
|
|
315
|
-
}
|
|
316
|
-
|
|
317
337
|
pcmContinuation = cont
|
|
318
338
|
|
|
319
|
-
val audioManager = appContext.getSystemService(AudioManager::class.java)
|
|
320
|
-
|
|
321
|
-
val audioAttributes =
|
|
322
|
-
AudioAttributes.Builder().setContentType(AudioAttributes.CONTENT_TYPE_SPEECH)
|
|
323
|
-
.setUsage(AudioAttributes.USAGE_ASSISTANCE_NAVIGATION_GUIDANCE).build()
|
|
324
|
-
|
|
325
|
-
val focusRequest =
|
|
326
|
-
AudioFocusRequest.Builder(AudioManager.AUDIOFOCUS_GAIN_TRANSIENT_EXCLUSIVE)
|
|
327
|
-
.setAudioAttributes(audioAttributes).setOnAudioFocusChangeListener { state ->
|
|
328
|
-
if (state == AudioManager.AUDIOFOCUS_LOSS) {
|
|
329
|
-
stop()
|
|
330
|
-
}
|
|
331
|
-
}.build()
|
|
332
|
-
|
|
333
|
-
if (audioManager.requestAudioFocus(focusRequest) != AudioManager.AUDIOFOCUS_REQUEST_GRANTED) {
|
|
334
|
-
pcmContinuation = null
|
|
335
|
-
cont.resumeWithException(IllegalStateException("Audio focus request denied"))
|
|
336
|
-
return@suspendCancellableCoroutine
|
|
337
|
-
}
|
|
338
|
-
|
|
339
|
-
audioFocusRequest = focusRequest
|
|
340
|
-
|
|
341
339
|
val bufferSize: Int
|
|
342
340
|
|
|
343
341
|
if (carContext != null) {
|
|
@@ -381,6 +379,8 @@ class VoiceInputManager(
|
|
|
381
379
|
) ?: -1
|
|
382
380
|
|
|
383
381
|
if (read < 0) {
|
|
382
|
+
// Whenever the user dismisses the microphone on the car screen, the next call to read will return -1
|
|
383
|
+
cancelledByUser = carAudioRecord != null && read == -1
|
|
384
384
|
break
|
|
385
385
|
}
|
|
386
386
|
|
|
@@ -450,6 +450,72 @@ class VoiceInputManager(
|
|
|
450
450
|
audioRecord?.stop()
|
|
451
451
|
}
|
|
452
452
|
|
|
453
|
+
@RequiresApi(Build.VERSION_CODES.O)
|
|
454
|
+
private fun requestAudioFocus(): Boolean {
|
|
455
|
+
val appContext = NitroModules.applicationContext ?: return false
|
|
456
|
+
val audioManager = appContext.getSystemService(AudioManager::class.java)
|
|
457
|
+
val audioAttributes = AudioAttributes.Builder()
|
|
458
|
+
.setContentType(AudioAttributes.CONTENT_TYPE_SPEECH)
|
|
459
|
+
.setUsage(AudioAttributes.USAGE_ASSISTANCE_NAVIGATION_GUIDANCE)
|
|
460
|
+
.build()
|
|
461
|
+
val focusRequest = AudioFocusRequest.Builder(AudioManager.AUDIOFOCUS_GAIN_TRANSIENT_EXCLUSIVE)
|
|
462
|
+
.setAudioAttributes(audioAttributes)
|
|
463
|
+
.setOnAudioFocusChangeListener { state ->
|
|
464
|
+
if (state == AudioManager.AUDIOFOCUS_LOSS) { stop() }
|
|
465
|
+
}
|
|
466
|
+
.build()
|
|
467
|
+
return if (audioManager.requestAudioFocus(focusRequest) == AudioManager.AUDIOFOCUS_REQUEST_GRANTED) {
|
|
468
|
+
audioFocusRequest = focusRequest
|
|
469
|
+
true
|
|
470
|
+
} else {
|
|
471
|
+
false
|
|
472
|
+
}
|
|
473
|
+
}
|
|
474
|
+
|
|
475
|
+
@RequiresApi(Build.VERSION_CODES.O)
|
|
476
|
+
private fun abandonAudioFocus() {
|
|
477
|
+
audioFocusRequest?.let {
|
|
478
|
+
val audioManager = (NitroModules.applicationContext ?: carContext)
|
|
479
|
+
?.getSystemService(AudioManager::class.java)
|
|
480
|
+
audioManager?.abandonAudioFocusRequest(it)
|
|
481
|
+
}
|
|
482
|
+
audioFocusRequest = null
|
|
483
|
+
}
|
|
484
|
+
|
|
485
|
+
private suspend fun playSound(uri: String) = suspendCancellableCoroutine<Unit> { cont ->
|
|
486
|
+
val context = NitroModules.applicationContext ?: run {
|
|
487
|
+
cont.resume(Unit)
|
|
488
|
+
return@suspendCancellableCoroutine
|
|
489
|
+
}
|
|
490
|
+
val player = MediaPlayer()
|
|
491
|
+
try {
|
|
492
|
+
player.setAudioAttributes(
|
|
493
|
+
AudioAttributes.Builder()
|
|
494
|
+
.setContentType(AudioAttributes.CONTENT_TYPE_SONIFICATION)
|
|
495
|
+
.setUsage(AudioAttributes.USAGE_ASSISTANCE_NAVIGATION_GUIDANCE)
|
|
496
|
+
.build()
|
|
497
|
+
)
|
|
498
|
+
player.setDataSource(context, Uri.parse(uri))
|
|
499
|
+
player.setOnCompletionListener {
|
|
500
|
+
it.release()
|
|
501
|
+
if (cont.isActive) { cont.resume(Unit) }
|
|
502
|
+
}
|
|
503
|
+
player.setOnErrorListener { mp, _, _ ->
|
|
504
|
+
mp.release()
|
|
505
|
+
if (cont.isActive) { cont.resume(Unit) }
|
|
506
|
+
true
|
|
507
|
+
}
|
|
508
|
+
player.prepare()
|
|
509
|
+
player.start()
|
|
510
|
+
} catch (_: Exception) {
|
|
511
|
+
try { player.release() } catch (_: Exception) {}
|
|
512
|
+
if (cont.isActive) { cont.resume(Unit) }
|
|
513
|
+
}
|
|
514
|
+
cont.invokeOnCancellation {
|
|
515
|
+
try { player.release() } catch (_: Exception) {}
|
|
516
|
+
}
|
|
517
|
+
}
|
|
518
|
+
|
|
453
519
|
@RequiresApi(Build.VERSION_CODES.O)
|
|
454
520
|
private fun releaseResources() {
|
|
455
521
|
carAudioRecord?.stopRecording()
|
|
@@ -458,13 +524,6 @@ class VoiceInputManager(
|
|
|
458
524
|
audioRecord?.release()
|
|
459
525
|
audioRecord = null
|
|
460
526
|
recordingJob = null
|
|
461
|
-
audioFocusRequest?.let {
|
|
462
|
-
val audioManager = (NitroModules.applicationContext ?: carContext)?.getSystemService(
|
|
463
|
-
AudioManager::class.java,
|
|
464
|
-
)
|
|
465
|
-
audioManager?.abandonAudioFocusRequest(it)
|
|
466
|
-
}
|
|
467
|
-
audioFocusRequest = null
|
|
468
527
|
}
|
|
469
528
|
|
|
470
529
|
fun dispose() {
|
package/ios/Types.swift
CHANGED
|
@@ -10,7 +10,7 @@ struct TemplateEventPayload {
|
|
|
10
10
|
let state: VisibilityState
|
|
11
11
|
}
|
|
12
12
|
|
|
13
|
-
enum AutoPlayError:
|
|
13
|
+
enum AutoPlayError: LocalizedError {
|
|
14
14
|
case templateNotFound(String)
|
|
15
15
|
case interfaceControllerNotFound(String)
|
|
16
16
|
case invalidTemplateError(String)
|
|
@@ -19,4 +19,5 @@ enum AutoPlayError: Error {
|
|
|
19
19
|
case invalidTemplateType(String)
|
|
20
20
|
case noUiWindow(String)
|
|
21
21
|
case initReactRootViewFailed(String)
|
|
22
|
+
case voiceInputCancelled
|
|
22
23
|
}
|
|
@@ -85,3 +85,10 @@ extension CPAlertTemplate {
|
|
|
85
85
|
initTemplate(template: self, id: id)
|
|
86
86
|
}
|
|
87
87
|
}
|
|
88
|
+
|
|
89
|
+
extension CPVoiceControlTemplate {
|
|
90
|
+
convenience init(voiceControlStates: [CPVoiceControlState], id: String) {
|
|
91
|
+
self.init(voiceControlStates: voiceControlStates)
|
|
92
|
+
initTemplate(template: self, id: id)
|
|
93
|
+
}
|
|
94
|
+
}
|
|
@@ -36,7 +36,9 @@ class HybridVoice: HybridVoiceSpec {
|
|
|
36
36
|
listeningImageRepeats: Bool?,
|
|
37
37
|
preferSpeechToText: Bool?,
|
|
38
38
|
onChunk: ((_ chunk: VoiceInputChunk) -> Void)?,
|
|
39
|
-
language: String
|
|
39
|
+
language: String?,
|
|
40
|
+
startSoundUri: String?,
|
|
41
|
+
endSoundUri: String?
|
|
40
42
|
) throws -> Promise<VoiceInputResult> {
|
|
41
43
|
return Promise.async {
|
|
42
44
|
let interfaceController = try? await RootModule.withInterfaceController { $0 }
|
|
@@ -55,7 +57,9 @@ class HybridVoice: HybridVoiceSpec {
|
|
|
55
57
|
listeningImageRepeats: listeningImageRepeats,
|
|
56
58
|
preferSpeechToText: preferSpeechToText ?? false,
|
|
57
59
|
onChunk: onChunk,
|
|
58
|
-
language: language
|
|
60
|
+
language: language,
|
|
61
|
+
startSoundUri: startSoundUri,
|
|
62
|
+
endSoundUri: endSoundUri
|
|
59
63
|
)
|
|
60
64
|
}
|
|
61
65
|
}
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
import CarPlay
|
|
2
|
+
|
|
3
|
+
class VoiceInputTemplate: AutoPlayTemplate {
|
|
4
|
+
let template: CPVoiceControlTemplate
|
|
5
|
+
private let onDidDisappearCallback: () -> Void
|
|
6
|
+
|
|
7
|
+
override func getTemplate() -> CPTemplate {
|
|
8
|
+
return template
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
init(
|
|
12
|
+
voiceControlStates: [CPVoiceControlState],
|
|
13
|
+
id: String,
|
|
14
|
+
onDidDisappear: @escaping () -> Void
|
|
15
|
+
) {
|
|
16
|
+
self.template = CPVoiceControlTemplate(voiceControlStates: voiceControlStates, id: id)
|
|
17
|
+
self.onDidDisappearCallback = onDidDisappear
|
|
18
|
+
|
|
19
|
+
super.init()
|
|
20
|
+
|
|
21
|
+
try? RootModule.withTemplateStore { templateStore in
|
|
22
|
+
templateStore.addTemplate(template: self, templateId: id)
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
override func onDidDisappear(animated: Bool) {
|
|
27
|
+
onDidDisappearCallback()
|
|
28
|
+
|
|
29
|
+
try? RootModule.withTemplateStore { templateStore in
|
|
30
|
+
templateStore.removeTemplate(templateId: self.template.id)
|
|
31
|
+
}
|
|
32
|
+
}
|
|
33
|
+
}
|
|
@@ -3,8 +3,30 @@ import CarPlay
|
|
|
3
3
|
import NitroModules
|
|
4
4
|
import Speech
|
|
5
5
|
|
|
6
|
-
///
|
|
7
|
-
|
|
6
|
+
/// Retains the player and itself until playback finishes — AVAudioPlayer.delegate is weak.
|
|
7
|
+
private final class AudioPlayerDelegate: NSObject, AVAudioPlayerDelegate, @unchecked Sendable {
|
|
8
|
+
private let onFinish: () -> Void
|
|
9
|
+
private var keepAlive: AudioPlayerDelegate?
|
|
10
|
+
private var player: AVAudioPlayer?
|
|
11
|
+
|
|
12
|
+
init(player: AVAudioPlayer, _ onFinish: @escaping () -> Void) {
|
|
13
|
+
self.onFinish = onFinish
|
|
14
|
+
self.player = player
|
|
15
|
+
super.init()
|
|
16
|
+
keepAlive = self
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
private func finish() {
|
|
20
|
+
player = nil
|
|
21
|
+
keepAlive = nil
|
|
22
|
+
onFinish()
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
func audioPlayerDidFinishPlaying(_: AVAudioPlayer, successfully _: Bool) { finish() }
|
|
26
|
+
func audioPlayerDecodeErrorDidOccur(_: AVAudioPlayer, error _: Error?) { finish() }
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/// CheckedContinuation wrapper that can only be resumed once, safe across concurrent stop() and recognition callbacks.
|
|
8
30
|
private final class ResultBox: @unchecked Sendable {
|
|
9
31
|
private var continuation: CheckedContinuation<VoiceInputResult, Error>?
|
|
10
32
|
private let lock = NSLock()
|
|
@@ -28,14 +50,14 @@ private final class ResultBox: @unchecked Sendable {
|
|
|
28
50
|
}
|
|
29
51
|
}
|
|
30
52
|
|
|
31
|
-
///
|
|
32
|
-
/// or transcribes it via SFSpeechRecognizer when preferSpeechToText is true.
|
|
53
|
+
/// Records 16 kHz / 16-bit mono PCM from the car mic, or transcribes via SFSpeechRecognizer.
|
|
33
54
|
class VoiceInputManager {
|
|
34
55
|
private var audioEngine: AVAudioEngine?
|
|
35
56
|
private var voiceControlTemplate: CPVoiceControlTemplate?
|
|
36
57
|
private var resultBox: ResultBox?
|
|
37
58
|
private var samples: [Int16] = []
|
|
38
59
|
private var isStopping = false
|
|
60
|
+
private var cancelledByUser = false
|
|
39
61
|
private let stopLock = NSLock()
|
|
40
62
|
|
|
41
63
|
// STT
|
|
@@ -45,6 +67,7 @@ class VoiceInputManager {
|
|
|
45
67
|
// Timing
|
|
46
68
|
private var recordingStart: Date?
|
|
47
69
|
private var silenceStart: Date?
|
|
70
|
+
private var firstBufferContinuation: CheckedContinuation<Void, Never>?
|
|
48
71
|
|
|
49
72
|
private static let sampleRate: Double = 16_000
|
|
50
73
|
private static let tapBufferSize: AVAudioFrameCount = 4_096
|
|
@@ -69,9 +92,22 @@ class VoiceInputManager {
|
|
|
69
92
|
listeningImageRepeats: Bool?,
|
|
70
93
|
preferSpeechToText: Bool,
|
|
71
94
|
onChunk: ((_ chunk: VoiceInputChunk) -> Void)?,
|
|
72
|
-
language: String
|
|
95
|
+
language: String?,
|
|
96
|
+
startSoundUri: String?,
|
|
97
|
+
endSoundUri: String?
|
|
73
98
|
) async throws -> VoiceInputResult {
|
|
74
|
-
|
|
99
|
+
stopLock.withLock {
|
|
100
|
+
cancelledByUser = false
|
|
101
|
+
}
|
|
102
|
+
// Single session for the full flow (start sound + recording + end sound); defer deactivates once at the end.
|
|
103
|
+
let session = AVAudioSession.sharedInstance()
|
|
104
|
+
try session.setCategory(.playAndRecord, mode: .measurement, options: [])
|
|
105
|
+
try session.setActive(true)
|
|
106
|
+
defer {
|
|
107
|
+
try? session.setActive(false, options: .notifyOthersOnDeactivation)
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
let result = try await withCheckedThrowingContinuation { cont in
|
|
75
111
|
let box = ResultBox(cont)
|
|
76
112
|
self.resultBox = box
|
|
77
113
|
self.samples = []
|
|
@@ -83,9 +119,6 @@ class VoiceInputManager {
|
|
|
83
119
|
interfaceController: interfaceController,
|
|
84
120
|
silenceThresholdMs: silenceThresholdMs,
|
|
85
121
|
maxDurationMs: maxDurationMs,
|
|
86
|
-
listeningText: listeningText,
|
|
87
|
-
listeningImage: listeningImage,
|
|
88
|
-
listeningImageRepeats: listeningImageRepeats,
|
|
89
122
|
preferSpeechToText: preferSpeechToText,
|
|
90
123
|
onChunk: onChunk,
|
|
91
124
|
box: box,
|
|
@@ -95,8 +128,61 @@ class VoiceInputManager {
|
|
|
95
128
|
catch {
|
|
96
129
|
self.cleanup(interfaceController: interfaceController)
|
|
97
130
|
box.resume(throwing: error)
|
|
131
|
+
return
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
// Start sound fires immediately; template is deferred until the first tap buffer so the mic indicator is already on.
|
|
135
|
+
if let uri = startSoundUri {
|
|
136
|
+
Task { await self.playSound(uri: uri) }
|
|
137
|
+
}
|
|
138
|
+
if let interfaceController = interfaceController {
|
|
139
|
+
Task {
|
|
140
|
+
await withCheckedContinuation { (cont: CheckedContinuation<Void, Never>) in
|
|
141
|
+
self.stopLock.withLock { self.firstBufferContinuation = cont }
|
|
142
|
+
}
|
|
143
|
+
// Skip if stop() fired before the first buffer — cleanup already dismissed.
|
|
144
|
+
guard !self.stopLock.withLock({ self.isStopping }) else { return }
|
|
145
|
+
await self.presentVoiceTemplate(
|
|
146
|
+
interfaceController: interfaceController,
|
|
147
|
+
listeningText: listeningText,
|
|
148
|
+
listeningImage: listeningImage,
|
|
149
|
+
listeningImageRepeats: listeningImageRepeats
|
|
150
|
+
)
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
if let uri = endSoundUri {
|
|
156
|
+
await playSound(uri: uri)
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
return result
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
private func playSound(uri: String) async {
|
|
163
|
+
guard let url = URL(string: uri) else { return }
|
|
164
|
+
do {
|
|
165
|
+
// URLSession handles both http:// (Metro dev server) and file:// (release bundle)
|
|
166
|
+
let (data, _) = try await URLSession.shared.data(from: url)
|
|
167
|
+
await withCheckedContinuation { (cont: CheckedContinuation<Void, Never>) in
|
|
168
|
+
DispatchQueue.main.async {
|
|
169
|
+
do {
|
|
170
|
+
let player = try AVAudioPlayer(data: data)
|
|
171
|
+
let delegate = AudioPlayerDelegate(player: player) { cont.resume() }
|
|
172
|
+
player.delegate = delegate
|
|
173
|
+
player.prepareToPlay()
|
|
174
|
+
player.play()
|
|
175
|
+
}
|
|
176
|
+
catch {
|
|
177
|
+
cont.resume()
|
|
178
|
+
}
|
|
179
|
+
}
|
|
98
180
|
}
|
|
99
181
|
}
|
|
182
|
+
catch {
|
|
183
|
+
print(error)
|
|
184
|
+
// fail silently — a broken sound file must not block voice input
|
|
185
|
+
}
|
|
100
186
|
}
|
|
101
187
|
|
|
102
188
|
func stop(interfaceController: AutoPlayInterfaceController? = nil) {
|
|
@@ -106,6 +192,7 @@ class VoiceInputManager {
|
|
|
106
192
|
return
|
|
107
193
|
}
|
|
108
194
|
isStopping = true
|
|
195
|
+
let wasCancelled = cancelledByUser
|
|
109
196
|
let wasSTTMode = isSTTMode
|
|
110
197
|
let capturedRequest = recognitionRequest
|
|
111
198
|
let box = resultBox
|
|
@@ -115,13 +202,17 @@ class VoiceInputManager {
|
|
|
115
202
|
stopLock.unlock()
|
|
116
203
|
|
|
117
204
|
if wasSTTMode {
|
|
118
|
-
// endAudio()
|
|
119
|
-
// which resumes the box. Engine teardown happens there too.
|
|
205
|
+
// endAudio() triggers the final recognition result, which resumes the box and tears down the engine.
|
|
120
206
|
capturedRequest?.endAudio()
|
|
121
207
|
}
|
|
122
208
|
else {
|
|
123
209
|
cleanup(interfaceController: interfaceController)
|
|
124
|
-
|
|
210
|
+
if wasCancelled {
|
|
211
|
+
box?.resume(throwing: AutoPlayError.voiceInputCancelled)
|
|
212
|
+
}
|
|
213
|
+
else {
|
|
214
|
+
box?.resume(returning: makePCMResult(from: capturedSamples))
|
|
215
|
+
}
|
|
125
216
|
}
|
|
126
217
|
}
|
|
127
218
|
|
|
@@ -131,9 +222,6 @@ class VoiceInputManager {
|
|
|
131
222
|
interfaceController: AutoPlayInterfaceController?,
|
|
132
223
|
silenceThresholdMs: Double,
|
|
133
224
|
maxDurationMs: Double,
|
|
134
|
-
listeningText: String,
|
|
135
|
-
listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?,
|
|
136
|
-
listeningImageRepeats: Bool?,
|
|
137
225
|
preferSpeechToText: Bool,
|
|
138
226
|
onChunk: ((_ chunk: VoiceInputChunk) -> Void)?,
|
|
139
227
|
box: ResultBox,
|
|
@@ -143,19 +231,6 @@ class VoiceInputManager {
|
|
|
143
231
|
throw VoiceInputError.microphonePermissionDenied
|
|
144
232
|
}
|
|
145
233
|
|
|
146
|
-
let session = AVAudioSession.sharedInstance()
|
|
147
|
-
try session.setCategory(.playAndRecord, mode: .measurement, options: [])
|
|
148
|
-
try session.setActive(true)
|
|
149
|
-
|
|
150
|
-
if let interfaceController {
|
|
151
|
-
presentVoiceTemplate(
|
|
152
|
-
interfaceController: interfaceController,
|
|
153
|
-
listeningText: listeningText,
|
|
154
|
-
listeningImage: listeningImage,
|
|
155
|
-
listeningImageRepeats: listeningImageRepeats
|
|
156
|
-
)
|
|
157
|
-
}
|
|
158
|
-
|
|
159
234
|
var activeRecognitionRequest: SFSpeechAudioBufferRecognitionRequest? = nil
|
|
160
235
|
|
|
161
236
|
if preferSpeechToText, SFSpeechRecognizer.authorizationStatus() == .authorized,
|
|
@@ -176,12 +251,18 @@ class VoiceInputManager {
|
|
|
176
251
|
// STT failed — fall back to whatever PCM was accumulated
|
|
177
252
|
self.stopLock.lock()
|
|
178
253
|
self.isStopping = true
|
|
254
|
+
let wasCancelled = self.cancelledByUser
|
|
179
255
|
let capturedSamples = self.samples
|
|
180
256
|
self.samples = []
|
|
181
257
|
self.stopLock.unlock()
|
|
182
258
|
|
|
183
259
|
self.cleanup(interfaceController: interfaceController)
|
|
184
|
-
|
|
260
|
+
if wasCancelled {
|
|
261
|
+
box.resume(throwing: AutoPlayError.voiceInputCancelled)
|
|
262
|
+
}
|
|
263
|
+
else {
|
|
264
|
+
box.resume(returning: self.makePCMResult(from: capturedSamples))
|
|
265
|
+
}
|
|
185
266
|
return
|
|
186
267
|
}
|
|
187
268
|
|
|
@@ -190,16 +271,22 @@ class VoiceInputManager {
|
|
|
190
271
|
if result.isFinal {
|
|
191
272
|
self.stopLock.lock()
|
|
192
273
|
self.isStopping = true
|
|
274
|
+
let wasCancelled = self.cancelledByUser
|
|
193
275
|
self.samples = []
|
|
194
276
|
self.stopLock.unlock()
|
|
195
277
|
|
|
196
278
|
self.cleanup(interfaceController: interfaceController)
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
279
|
+
if wasCancelled {
|
|
280
|
+
box.resume(throwing: AutoPlayError.voiceInputCancelled)
|
|
281
|
+
}
|
|
282
|
+
else {
|
|
283
|
+
box.resume(
|
|
284
|
+
returning: VoiceInputResult(
|
|
285
|
+
transcription: result.bestTranscription.formattedString,
|
|
286
|
+
audio: nil
|
|
287
|
+
)
|
|
201
288
|
)
|
|
202
|
-
|
|
289
|
+
}
|
|
203
290
|
}
|
|
204
291
|
else {
|
|
205
292
|
onChunk?(VoiceInputChunk(partial: result.bestTranscription.formattedString, audio: nil))
|
|
@@ -217,13 +304,25 @@ class VoiceInputManager {
|
|
|
217
304
|
|
|
218
305
|
recordingStart = Date()
|
|
219
306
|
silenceStart = nil
|
|
307
|
+
firstBufferContinuation = nil
|
|
220
308
|
|
|
221
309
|
inputNode.installTap(
|
|
222
310
|
onBus: 0,
|
|
223
311
|
bufferSize: VoiceInputManager.tapBufferSize,
|
|
224
312
|
format: nativeFormat
|
|
225
313
|
) { [weak self] buffer, _ in
|
|
226
|
-
guard let self
|
|
314
|
+
guard let self else { return }
|
|
315
|
+
|
|
316
|
+
self.stopLock.lock()
|
|
317
|
+
let stopping = self.isStopping
|
|
318
|
+
let recordingStartSnapshot = self.recordingStart
|
|
319
|
+
let firstBufferCont = self.firstBufferContinuation
|
|
320
|
+
self.firstBufferContinuation = nil
|
|
321
|
+
self.stopLock.unlock()
|
|
322
|
+
|
|
323
|
+
firstBufferCont?.resume()
|
|
324
|
+
|
|
325
|
+
guard !stopping else { return }
|
|
227
326
|
|
|
228
327
|
// Feed STT if active
|
|
229
328
|
activeRecognitionRequest?.append(buffer)
|
|
@@ -266,16 +365,15 @@ class VoiceInputManager {
|
|
|
266
365
|
let now = Date()
|
|
267
366
|
|
|
268
367
|
// Max duration — applies in both modes
|
|
269
|
-
if let start =
|
|
368
|
+
if let start = recordingStartSnapshot,
|
|
270
369
|
now.timeIntervalSince(start) * 1000 >= maxDurationMs
|
|
271
370
|
{
|
|
272
371
|
self.triggerAutoStop(interfaceController: interfaceController)
|
|
273
372
|
return
|
|
274
373
|
}
|
|
275
374
|
|
|
276
|
-
// Silence detection — skip during warm-up
|
|
277
|
-
|
|
278
|
-
if let start = self.recordingStart,
|
|
375
|
+
// Silence detection — skip during warm-up to let the pipeline stabilise.
|
|
376
|
+
if let start = recordingStartSnapshot,
|
|
279
377
|
now.timeIntervalSince(start) * 1000 >= VoiceInputManager.warmupMs
|
|
280
378
|
{
|
|
281
379
|
let peak = newSamples.reduce(0) { max($0, abs(Int($1))) }
|
|
@@ -312,7 +410,13 @@ class VoiceInputManager {
|
|
|
312
410
|
recognitionRequest = nil
|
|
313
411
|
recordingStart = nil
|
|
314
412
|
silenceStart = nil
|
|
315
|
-
|
|
413
|
+
// Drain firstBufferContinuation so the template Task doesn't hang if stop() fired before the first buffer.
|
|
414
|
+
let pendingCont = stopLock.withLock { () -> CheckedContinuation<Void, Never>? in
|
|
415
|
+
let c = firstBufferContinuation
|
|
416
|
+
firstBufferContinuation = nil
|
|
417
|
+
return c
|
|
418
|
+
}
|
|
419
|
+
pendingCont?.resume()
|
|
316
420
|
if let interfaceController {
|
|
317
421
|
dismissVoiceTemplate(interfaceController: interfaceController)
|
|
318
422
|
}
|
|
@@ -327,15 +431,10 @@ class VoiceInputManager {
|
|
|
327
431
|
// CPVoiceControlState enforces a maximum image size of 150x150 points.
|
|
328
432
|
private static let voiceImageMaxSize = CGSize(width: 150, height: 150)
|
|
329
433
|
|
|
330
|
-
// CPVoiceControlState
|
|
331
|
-
// by the system regardless of what we pass, so we only need to clamp our own ceiling.
|
|
434
|
+
// CPVoiceControlState enforces a 0.3s–5s animation cycle; the 0.3s floor is system-applied, clamp only the ceiling.
|
|
332
435
|
private static let maxVoiceImageCycleDuration: TimeInterval = 5.0
|
|
333
436
|
|
|
334
|
-
//
|
|
335
|
-
// via CGImage during scale adjustment. Parser.decodeImage preserves animation frames by
|
|
336
|
-
// walking every frame in the source via ImageIO — UIImage(data:) never builds a multi-frame
|
|
337
|
-
// .images array itself, for GIF, APNG, or WebP. Tinting is skipped for animated images since
|
|
338
|
-
// frames cannot be tinted individually.
|
|
437
|
+
// Uses Parser.decodeImage instead of RCTConvert to preserve animation frames for GIF/APNG/WebP.
|
|
339
438
|
private func loadVoiceImage(image: Variant_GlyphImage_AssetImage_RemoteImage?, traitCollection: UITraitCollection)
|
|
340
439
|
-> UIImage?
|
|
341
440
|
{
|
|
@@ -381,12 +480,13 @@ class VoiceInputManager {
|
|
|
381
480
|
return nil
|
|
382
481
|
}
|
|
383
482
|
|
|
483
|
+
@MainActor
|
|
384
484
|
private func presentVoiceTemplate(
|
|
385
485
|
interfaceController: AutoPlayInterfaceController,
|
|
386
486
|
listeningText: String,
|
|
387
487
|
listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?,
|
|
388
488
|
listeningImageRepeats: Bool?
|
|
389
|
-
) {
|
|
489
|
+
) async {
|
|
390
490
|
let traitCollection = SceneStore.getRootTraitCollection() ?? UITraitCollection.current
|
|
391
491
|
let image = loadVoiceImage(
|
|
392
492
|
image: listeningImage,
|
|
@@ -400,14 +500,21 @@ class VoiceInputManager {
|
|
|
400
500
|
image: image,
|
|
401
501
|
repeats: repeats
|
|
402
502
|
)
|
|
403
|
-
let template = CPVoiceControlTemplate(voiceControlStates: [listeningState])
|
|
404
|
-
initTemplate(template: template, id: "voice-input")
|
|
405
|
-
voiceControlTemplate = template
|
|
406
503
|
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
504
|
+
let voiceTemplate = VoiceInputTemplate(
|
|
505
|
+
voiceControlStates: [listeningState],
|
|
506
|
+
id: "voice-input"
|
|
507
|
+
) { [weak self] in
|
|
508
|
+
guard let self else { return }
|
|
509
|
+
self.stopLock.withLock {
|
|
510
|
+
if !self.isStopping { self.cancelledByUser = true }
|
|
511
|
+
}
|
|
512
|
+
self.stop()
|
|
410
513
|
}
|
|
514
|
+
|
|
515
|
+
voiceControlTemplate = voiceTemplate.template
|
|
516
|
+
try? await interfaceController.presentTemplate(voiceTemplate.template, animated: true)
|
|
517
|
+
voiceTemplate.template.activateVoiceControlState(withIdentifier: "listening")
|
|
411
518
|
}
|
|
412
519
|
|
|
413
520
|
private func dismissVoiceTemplate(interfaceController: AutoPlayInterfaceController) {
|
|
@@ -1,10 +1,13 @@
|
|
|
1
|
+
import { Image } from 'react-native';
|
|
1
2
|
import { NitroModules } from 'react-native-nitro-modules';
|
|
2
3
|
import { NitroImageUtil } from '../utils/NitroImage';
|
|
3
4
|
const _native = NitroModules.createHybridObject('Voice');
|
|
4
5
|
const startVoiceInput = async (options) => {
|
|
5
|
-
const { onChunk, silenceThresholdMs, maxDurationMs, listeningText, listeningImage, preferSpeechToText, language, } = options ?? {};
|
|
6
|
+
const { onChunk, silenceThresholdMs, maxDurationMs, listeningText, listeningImage, preferSpeechToText, language, startSound, endSound, } = options ?? {};
|
|
6
7
|
const listeningImageRepeats = listeningImage?.type === 'asset' ? listeningImage.repeats : undefined;
|
|
7
|
-
|
|
8
|
+
const startSoundUri = startSound != null ? Image.resolveAssetSource(startSound).uri : undefined;
|
|
9
|
+
const endSoundUri = endSound != null ? Image.resolveAssetSource(endSound).uri : undefined;
|
|
10
|
+
return await _native.startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, NitroImageUtil.convert(listeningImage), listeningImageRepeats, preferSpeechToText, onChunk, language, startSoundUri, endSoundUri);
|
|
8
11
|
};
|
|
9
12
|
export const HybridVoice = {
|
|
10
13
|
/**
|
|
@@ -7,6 +7,6 @@ export interface Voice extends HybridObject<{
|
|
|
7
7
|
}> {
|
|
8
8
|
hasVoiceInputPermission(): boolean;
|
|
9
9
|
requestVoiceInputPermission(): Promise<boolean>;
|
|
10
|
-
startVoiceInput(silenceThresholdMs?: number, maxDurationMs?: number, listeningText?: string, listeningImage?: NitroImage, listeningImageRepeats?: boolean, preferSpeechToText?: boolean, onChunk?: (chunk: VoiceInputChunk) => void, language?: string): Promise<VoiceInputResult>;
|
|
10
|
+
startVoiceInput(silenceThresholdMs?: number, maxDurationMs?: number, listeningText?: string, listeningImage?: NitroImage, listeningImageRepeats?: boolean, preferSpeechToText?: boolean, onChunk?: (chunk: VoiceInputChunk) => void, language?: string, startSoundUri?: string, endSoundUri?: string): Promise<VoiceInputResult>;
|
|
11
11
|
stopVoiceInput(): void;
|
|
12
12
|
}
|
package/lib/types/Voice.d.ts
CHANGED
|
@@ -18,4 +18,8 @@ export interface VoiceInputOptions {
|
|
|
18
18
|
preferSpeechToText?: boolean;
|
|
19
19
|
onChunk?: (chunk: VoiceInputChunk) => void;
|
|
20
20
|
language?: string;
|
|
21
|
+
/** Sound played just before recording starts. Pass a Metro asset: `require('./beep_start.wav')`. */
|
|
22
|
+
startSound?: number;
|
|
23
|
+
/** Sound played just after recording stops. Pass a Metro asset: `require('./beep_end.wav')`. */
|
|
24
|
+
endSound?: number;
|
|
21
25
|
}
|
package/lib/utils/ErrorUtil.d.ts
CHANGED
package/lib/utils/ErrorUtil.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
const
|
|
1
|
+
const isError = (error) => {
|
|
2
2
|
if (error == null) {
|
|
3
3
|
return false;
|
|
4
4
|
}
|
|
@@ -14,6 +14,18 @@ const isTemplateNotFoundError = (error) => {
|
|
|
14
14
|
if (typeof error.message !== 'string') {
|
|
15
15
|
return false;
|
|
16
16
|
}
|
|
17
|
-
return
|
|
17
|
+
return true;
|
|
18
|
+
};
|
|
19
|
+
const isTemplateNotFoundError = (error) => {
|
|
20
|
+
if (isError(error)) {
|
|
21
|
+
return error.message.startsWith('templateNotFound');
|
|
22
|
+
}
|
|
23
|
+
return false;
|
|
24
|
+
};
|
|
25
|
+
const isVoiceInputCanceledError = (error) => {
|
|
26
|
+
if (isError(error)) {
|
|
27
|
+
return error.message.startsWith('voiceInputCancelled');
|
|
28
|
+
}
|
|
29
|
+
return false;
|
|
18
30
|
};
|
|
19
|
-
export const ErrorUtil = { isTemplateNotFoundError };
|
|
31
|
+
export const ErrorUtil = { isTemplateNotFoundError, isVoiceInputCanceledError };
|
|
@@ -98,9 +98,9 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
|
|
|
98
98
|
return __promise;
|
|
99
99
|
}();
|
|
100
100
|
}
|
|
101
|
-
std::shared_ptr<Promise<VoiceInputResult>> JHybridVoiceSpec::startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language) {
|
|
102
|
-
static const auto method = _javaPart->javaClassStatic()->getMethod<jni::local_ref<JPromise::javaobject>(jni::alias_ref<jni::JDouble> /* silenceThresholdMs */, jni::alias_ref<jni::JDouble> /* maxDurationMs */, jni::alias_ref<jni::JString> /* listeningText */, jni::alias_ref<JVariant_GlyphImage_AssetImage_RemoteImage> /* listeningImage */, jni::alias_ref<jni::JBoolean> /* listeningImageRepeats */, jni::alias_ref<jni::JBoolean> /* preferSpeechToText */, jni::alias_ref<JFunc_void_VoiceInputChunk::javaobject> /* onChunk */, jni::alias_ref<jni::JString> /* language */)>("startVoiceInput_cxx");
|
|
103
|
-
auto __result = method(_javaPart, silenceThresholdMs.has_value() ? jni::JDouble::valueOf(silenceThresholdMs.value()) : nullptr, maxDurationMs.has_value() ? jni::JDouble::valueOf(maxDurationMs.value()) : nullptr, listeningText.has_value() ? jni::make_jstring(listeningText.value()) : nullptr, listeningImage.has_value() ? JVariant_GlyphImage_AssetImage_RemoteImage::fromCpp(listeningImage.value()) : nullptr, listeningImageRepeats.has_value() ? jni::JBoolean::valueOf(listeningImageRepeats.value()) : nullptr, preferSpeechToText.has_value() ? jni::JBoolean::valueOf(preferSpeechToText.value()) : nullptr, onChunk.has_value() ? JFunc_void_VoiceInputChunk_cxx::fromCpp(onChunk.value()) : nullptr, language.has_value() ? jni::make_jstring(language.value()) : nullptr);
|
|
101
|
+
std::shared_ptr<Promise<VoiceInputResult>> JHybridVoiceSpec::startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri) {
|
|
102
|
+
static const auto method = _javaPart->javaClassStatic()->getMethod<jni::local_ref<JPromise::javaobject>(jni::alias_ref<jni::JDouble> /* silenceThresholdMs */, jni::alias_ref<jni::JDouble> /* maxDurationMs */, jni::alias_ref<jni::JString> /* listeningText */, jni::alias_ref<JVariant_GlyphImage_AssetImage_RemoteImage> /* listeningImage */, jni::alias_ref<jni::JBoolean> /* listeningImageRepeats */, jni::alias_ref<jni::JBoolean> /* preferSpeechToText */, jni::alias_ref<JFunc_void_VoiceInputChunk::javaobject> /* onChunk */, jni::alias_ref<jni::JString> /* language */, jni::alias_ref<jni::JString> /* startSoundUri */, jni::alias_ref<jni::JString> /* endSoundUri */)>("startVoiceInput_cxx");
|
|
103
|
+
auto __result = method(_javaPart, silenceThresholdMs.has_value() ? jni::JDouble::valueOf(silenceThresholdMs.value()) : nullptr, maxDurationMs.has_value() ? jni::JDouble::valueOf(maxDurationMs.value()) : nullptr, listeningText.has_value() ? jni::make_jstring(listeningText.value()) : nullptr, listeningImage.has_value() ? JVariant_GlyphImage_AssetImage_RemoteImage::fromCpp(listeningImage.value()) : nullptr, listeningImageRepeats.has_value() ? jni::JBoolean::valueOf(listeningImageRepeats.value()) : nullptr, preferSpeechToText.has_value() ? jni::JBoolean::valueOf(preferSpeechToText.value()) : nullptr, onChunk.has_value() ? JFunc_void_VoiceInputChunk_cxx::fromCpp(onChunk.value()) : nullptr, language.has_value() ? jni::make_jstring(language.value()) : nullptr, startSoundUri.has_value() ? jni::make_jstring(startSoundUri.value()) : nullptr, endSoundUri.has_value() ? jni::make_jstring(endSoundUri.value()) : nullptr);
|
|
104
104
|
return [&]() {
|
|
105
105
|
auto __promise = Promise<VoiceInputResult>::create();
|
|
106
106
|
__result->cthis()->addOnResolvedListener([=](const jni::alias_ref<jni::JObject>& __boxedResult) {
|
|
@@ -56,7 +56,7 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
|
|
|
56
56
|
// Methods
|
|
57
57
|
bool hasVoiceInputPermission() override;
|
|
58
58
|
std::shared_ptr<Promise<bool>> requestVoiceInputPermission() override;
|
|
59
|
-
std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language) override;
|
|
59
|
+
std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri) override;
|
|
60
60
|
void stopVoiceInput() override;
|
|
61
61
|
|
|
62
62
|
private:
|
|
@@ -37,12 +37,12 @@ abstract class HybridVoiceSpec: HybridObject() {
|
|
|
37
37
|
@Keep
|
|
38
38
|
abstract fun requestVoiceInputPermission(): Promise<Boolean>
|
|
39
39
|
|
|
40
|
-
abstract fun startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Boolean?, preferSpeechToText: Boolean?, onChunk: ((chunk: VoiceInputChunk) -> Unit)?, language: String?): Promise<VoiceInputResult>
|
|
40
|
+
abstract fun startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Boolean?, preferSpeechToText: Boolean?, onChunk: ((chunk: VoiceInputChunk) -> Unit)?, language: String?, startSoundUri: String?, endSoundUri: String?): Promise<VoiceInputResult>
|
|
41
41
|
|
|
42
42
|
@DoNotStrip
|
|
43
43
|
@Keep
|
|
44
|
-
private fun startVoiceInput_cxx(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Boolean?, preferSpeechToText: Boolean?, onChunk: Func_void_VoiceInputChunk?, language: String?): Promise<VoiceInputResult> {
|
|
45
|
-
val __result = startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, listeningImage, listeningImageRepeats, preferSpeechToText, onChunk?.let { it }, language)
|
|
44
|
+
private fun startVoiceInput_cxx(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Boolean?, preferSpeechToText: Boolean?, onChunk: Func_void_VoiceInputChunk?, language: String?, startSoundUri: String?, endSoundUri: String?): Promise<VoiceInputResult> {
|
|
45
|
+
val __result = startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, listeningImage, listeningImageRepeats, preferSpeechToText, onChunk?.let { it }, language, startSoundUri, endSoundUri)
|
|
46
46
|
return __result
|
|
47
47
|
}
|
|
48
48
|
|
|
@@ -107,8 +107,8 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
|
|
|
107
107
|
auto __value = std::move(__result.value());
|
|
108
108
|
return __value;
|
|
109
109
|
}
|
|
110
|
-
inline std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language) override {
|
|
111
|
-
auto __result = _swiftPart.startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, listeningImage, listeningImageRepeats, preferSpeechToText, onChunk, language);
|
|
110
|
+
inline std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri) override {
|
|
111
|
+
auto __result = _swiftPart.startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, listeningImage, listeningImageRepeats, preferSpeechToText, onChunk, language, startSoundUri, endSoundUri);
|
|
112
112
|
if (__result.hasError()) [[unlikely]] {
|
|
113
113
|
std::rethrow_exception(__result.error());
|
|
114
114
|
}
|
|
@@ -15,7 +15,7 @@ public protocol HybridVoiceSpec_protocol: HybridObject {
|
|
|
15
15
|
// Methods
|
|
16
16
|
func hasVoiceInputPermission() throws -> Bool
|
|
17
17
|
func requestVoiceInputPermission() throws -> Promise<Bool>
|
|
18
|
-
func startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Bool?, preferSpeechToText: Bool?, onChunk: ((_ chunk: VoiceInputChunk) -> Void)?, language: String?) throws -> Promise<VoiceInputResult>
|
|
18
|
+
func startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Bool?, preferSpeechToText: Bool?, onChunk: ((_ chunk: VoiceInputChunk) -> Void)?, language: String?, startSoundUri: String?, endSoundUri: String?) throws -> Promise<VoiceInputResult>
|
|
19
19
|
func stopVoiceInput() throws -> Void
|
|
20
20
|
}
|
|
21
21
|
|
|
@@ -156,7 +156,7 @@ open class HybridVoiceSpec_cxx {
|
|
|
156
156
|
}
|
|
157
157
|
|
|
158
158
|
@inline(__always)
|
|
159
|
-
public final func startVoiceInput(silenceThresholdMs: bridge.std__optional_double_, maxDurationMs: bridge.std__optional_double_, listeningText: bridge.std__optional_std__string_, listeningImage: bridge.std__optional_std__variant_GlyphImage__AssetImage__RemoteImage__, listeningImageRepeats: bridge.std__optional_bool_, preferSpeechToText: bridge.std__optional_bool_, onChunk: bridge.std__optional_std__function_void_const_VoiceInputChunk_____chunk______, language: bridge.std__optional_std__string_) -> bridge.Result_std__shared_ptr_Promise_VoiceInputResult___ {
|
|
159
|
+
public final func startVoiceInput(silenceThresholdMs: bridge.std__optional_double_, maxDurationMs: bridge.std__optional_double_, listeningText: bridge.std__optional_std__string_, listeningImage: bridge.std__optional_std__variant_GlyphImage__AssetImage__RemoteImage__, listeningImageRepeats: bridge.std__optional_bool_, preferSpeechToText: bridge.std__optional_bool_, onChunk: bridge.std__optional_std__function_void_const_VoiceInputChunk_____chunk______, language: bridge.std__optional_std__string_, startSoundUri: bridge.std__optional_std__string_, endSoundUri: bridge.std__optional_std__string_) -> bridge.Result_std__shared_ptr_Promise_VoiceInputResult___ {
|
|
160
160
|
do {
|
|
161
161
|
let __result = try self.__implementation.startVoiceInput(silenceThresholdMs: { () -> Double? in
|
|
162
162
|
if bridge.has_value_std__optional_double_(silenceThresholdMs) {
|
|
@@ -234,6 +234,20 @@ open class HybridVoiceSpec_cxx {
|
|
|
234
234
|
} else {
|
|
235
235
|
return nil
|
|
236
236
|
}
|
|
237
|
+
}(), startSoundUri: { () -> String? in
|
|
238
|
+
if bridge.has_value_std__optional_std__string_(startSoundUri) {
|
|
239
|
+
let __unwrapped = bridge.get_std__optional_std__string_(startSoundUri)
|
|
240
|
+
return String(__unwrapped)
|
|
241
|
+
} else {
|
|
242
|
+
return nil
|
|
243
|
+
}
|
|
244
|
+
}(), endSoundUri: { () -> String? in
|
|
245
|
+
if bridge.has_value_std__optional_std__string_(endSoundUri) {
|
|
246
|
+
let __unwrapped = bridge.get_std__optional_std__string_(endSoundUri)
|
|
247
|
+
return String(__unwrapped)
|
|
248
|
+
} else {
|
|
249
|
+
return nil
|
|
250
|
+
}
|
|
237
251
|
}())
|
|
238
252
|
let __resultCpp = { () -> bridge.std__shared_ptr_Promise_VoiceInputResult__ in
|
|
239
253
|
let __promise = bridge.create_std__shared_ptr_Promise_VoiceInputResult__()
|
|
@@ -68,7 +68,7 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
|
|
|
68
68
|
// Methods
|
|
69
69
|
virtual bool hasVoiceInputPermission() = 0;
|
|
70
70
|
virtual std::shared_ptr<Promise<bool>> requestVoiceInputPermission() = 0;
|
|
71
|
-
virtual std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language) = 0;
|
|
71
|
+
virtual std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri) = 0;
|
|
72
72
|
virtual void stopVoiceInput() = 0;
|
|
73
73
|
|
|
74
74
|
protected:
|
package/package.json
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { Image } from 'react-native';
|
|
1
2
|
import { NitroModules } from 'react-native-nitro-modules';
|
|
2
3
|
import type { Voice } from '../specs/Voice.nitro';
|
|
3
4
|
import type { VoiceInputOptions, VoiceInputResult } from '../types/Voice';
|
|
@@ -21,11 +22,16 @@ const startVoiceInput: StartVoiceInput = async (options?: VoiceInputOptions) =>
|
|
|
21
22
|
listeningImage,
|
|
22
23
|
preferSpeechToText,
|
|
23
24
|
language,
|
|
25
|
+
startSound,
|
|
26
|
+
endSound,
|
|
24
27
|
} = options ?? {};
|
|
25
28
|
|
|
26
29
|
const listeningImageRepeats =
|
|
27
30
|
listeningImage?.type === 'asset' ? listeningImage.repeats : undefined;
|
|
28
31
|
|
|
32
|
+
const startSoundUri = startSound != null ? Image.resolveAssetSource(startSound).uri : undefined;
|
|
33
|
+
const endSoundUri = endSound != null ? Image.resolveAssetSource(endSound).uri : undefined;
|
|
34
|
+
|
|
29
35
|
return await _native.startVoiceInput(
|
|
30
36
|
silenceThresholdMs,
|
|
31
37
|
maxDurationMs,
|
|
@@ -34,7 +40,9 @@ const startVoiceInput: StartVoiceInput = async (options?: VoiceInputOptions) =>
|
|
|
34
40
|
listeningImageRepeats,
|
|
35
41
|
preferSpeechToText,
|
|
36
42
|
onChunk,
|
|
37
|
-
language
|
|
43
|
+
language,
|
|
44
|
+
startSoundUri,
|
|
45
|
+
endSoundUri
|
|
38
46
|
);
|
|
39
47
|
};
|
|
40
48
|
|
package/src/specs/Voice.nitro.ts
CHANGED
|
@@ -13,7 +13,9 @@ export interface Voice extends HybridObject<{ android: 'kotlin'; ios: 'swift' }>
|
|
|
13
13
|
listeningImageRepeats?: boolean,
|
|
14
14
|
preferSpeechToText?: boolean,
|
|
15
15
|
onChunk?: (chunk: VoiceInputChunk) => void,
|
|
16
|
-
language?: string
|
|
16
|
+
language?: string,
|
|
17
|
+
startSoundUri?: string,
|
|
18
|
+
endSoundUri?: string
|
|
17
19
|
): Promise<VoiceInputResult>;
|
|
18
20
|
stopVoiceInput(): void;
|
|
19
21
|
}
|
package/src/types/Voice.ts
CHANGED
|
@@ -21,4 +21,8 @@ export interface VoiceInputOptions {
|
|
|
21
21
|
preferSpeechToText?: boolean;
|
|
22
22
|
onChunk?: (chunk: VoiceInputChunk) => void;
|
|
23
23
|
language?: string;
|
|
24
|
+
/** Sound played just before recording starts. Pass a Metro asset: `require('./beep_start.wav')`. */
|
|
25
|
+
startSound?: number;
|
|
26
|
+
/** Sound played just after recording stops. Pass a Metro asset: `require('./beep_end.wav')`. */
|
|
27
|
+
endSound?: number;
|
|
24
28
|
}
|
package/src/utils/ErrorUtil.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
const
|
|
1
|
+
const isError = (error: unknown): error is { message: string } => {
|
|
2
2
|
if (error == null) {
|
|
3
3
|
return false;
|
|
4
4
|
}
|
|
@@ -19,7 +19,23 @@ const isTemplateNotFoundError = (error: unknown): error is Error => {
|
|
|
19
19
|
return false;
|
|
20
20
|
}
|
|
21
21
|
|
|
22
|
-
return
|
|
22
|
+
return true;
|
|
23
|
+
};
|
|
24
|
+
|
|
25
|
+
const isTemplateNotFoundError = (error: unknown): error is Error => {
|
|
26
|
+
if (isError(error)) {
|
|
27
|
+
return error.message.startsWith('templateNotFound');
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
return false;
|
|
31
|
+
};
|
|
32
|
+
|
|
33
|
+
const isVoiceInputCanceledError = (error: unknown): error is Error => {
|
|
34
|
+
if (isError(error)) {
|
|
35
|
+
return error.message.startsWith('voiceInputCancelled');
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
return false;
|
|
23
39
|
};
|
|
24
40
|
|
|
25
|
-
export const ErrorUtil = { isTemplateNotFoundError };
|
|
41
|
+
export const ErrorUtil = { isTemplateNotFoundError, isVoiceInputCanceledError };
|