@iternio/react-native-auto-play 0.5.4 → 0.5.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +75 -16
- package/android/src/main/java/com/margelo/nitro/swe/iternio/reactnativeautoplay/AutoPlayError.kt +6 -1
- package/android/src/main/java/com/margelo/nitro/swe/iternio/reactnativeautoplay/HybridVoice.kt +7 -3
- package/android/src/main/java/com/margelo/nitro/swe/iternio/reactnativeautoplay/VoiceInputManager.kt +103 -45
- package/ios/Types.swift +2 -1
- package/ios/extensions/CarPlayTemplateExtensions.swift +7 -0
- package/ios/hybrid/HybridVoice.swift +6 -2
- package/ios/templates/VoiceInputTemplate.swift +33 -0
- package/ios/utils/VoiceInputManager.swift +148 -35
- package/lib/hybrid/HybridVoice.js +5 -2
- package/lib/specs/Voice.nitro.d.ts +1 -1
- package/lib/types/Voice.d.ts +4 -0
- package/lib/utils/ErrorUtil.d.ts +1 -0
- package/lib/utils/ErrorUtil.js +15 -3
- package/nitrogen/generated/android/c++/JHybridVoiceSpec.cpp +3 -3
- package/nitrogen/generated/android/c++/JHybridVoiceSpec.hpp +1 -1
- package/nitrogen/generated/android/kotlin/com/margelo/nitro/swe/iternio/reactnativeautoplay/HybridVoiceSpec.kt +3 -3
- package/nitrogen/generated/ios/c++/HybridVoiceSpecSwift.hpp +2 -2
- package/nitrogen/generated/ios/swift/HybridVoiceSpec.swift +1 -1
- package/nitrogen/generated/ios/swift/HybridVoiceSpec_cxx.swift +15 -1
- package/nitrogen/generated/shared/c++/HybridVoiceSpec.hpp +1 -1
- package/package.json +1 -1
- package/src/hybrid/HybridVoice.ts +9 -1
- package/src/specs/Voice.nitro.ts +3 -1
- package/src/types/Voice.ts +4 -0
- package/src/utils/ErrorUtil.ts +19 -3
package/README.md
CHANGED
|
@@ -14,7 +14,7 @@
|
|
|
14
14
|
## Features
|
|
15
15
|
|
|
16
16
|
- **Cross-Platform:** Write once, run on both Apple CarPlay and Android Auto.
|
|
17
|
-
-
|
|
17
|
+
- **New Architecture:** Supports React Native new architecture only.
|
|
18
18
|
- **Template-Based UI:** Utilize a rich set of templates like `MapTemplate`, `ListTemplate`, `GridTemplate`, and more to build UIs that comply with automotive design guidelines.
|
|
19
19
|
- **Navigation APIs:** Build full-featured navigation experiences with APIs for trip management, maneuvers, and route guidance.
|
|
20
20
|
- **Dashboard & Cluster Support:** Extend your app's presence to the CarPlay Dashboard (CarPlay only) and instrument cluster displays (CarPlay & Android Auto).
|
|
@@ -837,36 +837,95 @@ new ListTemplate({
|
|
|
837
837
|
|
|
838
838
|
### Voice Input
|
|
839
839
|
|
|
840
|
-
The library provides a cross-platform in-app voice recording API built on top of the car microphone (when connected) or the device microphone (when no car is connected).
|
|
840
|
+
The library provides a cross-platform in-app voice recording API built on top of the car microphone (when connected) or the device microphone (when no car is connected). The voice API lives in `HybridVoice`.
|
|
841
841
|
|
|
842
842
|
#### Permission
|
|
843
843
|
|
|
844
844
|
```ts
|
|
845
|
-
|
|
846
|
-
|
|
845
|
+
import { HybridVoice } from '@iternio/react-native-auto-play';
|
|
846
|
+
|
|
847
|
+
// Check whether permission is already granted (synchronous)
|
|
848
|
+
const granted = HybridVoice.hasVoiceInputPermission();
|
|
847
849
|
|
|
848
850
|
// Request permission if not yet granted
|
|
849
|
-
const granted = await
|
|
851
|
+
const granted = await HybridVoice.requestVoiceInputPermission();
|
|
850
852
|
```
|
|
851
853
|
|
|
854
|
+
On **iOS**: checks/requests both microphone and speech recognition authorization.
|
|
855
|
+
On **Android**: checks/requests `RECORD_AUDIO` via the car context when connected, otherwise via the RN application context.
|
|
856
|
+
|
|
852
857
|
#### Recording
|
|
853
858
|
|
|
854
859
|
```ts
|
|
855
|
-
|
|
856
|
-
|
|
857
|
-
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
|
|
861
|
-
|
|
860
|
+
import { HybridVoice, ErrorUtil } from '@iternio/react-native-auto-play';
|
|
861
|
+
|
|
862
|
+
try {
|
|
863
|
+
const result = await HybridVoice.startVoiceInput({
|
|
864
|
+
silenceThresholdMs: 1500, // ms of silence before auto-stop (default 1500)
|
|
865
|
+
maxDurationMs: 10_000, // hard cap on recording duration (default 10 000)
|
|
866
|
+
listeningText: 'Listening…', // iOS CarPlay: text shown on CPVoiceControlTemplate
|
|
867
|
+
preferSpeechToText: false, // true → STT transcription; false → raw PCM (default)
|
|
868
|
+
startSound: require('./beep_start.mp3'), // played just before recording starts
|
|
869
|
+
endSound: require('./beep_end.mp3'), // played just after recording stops
|
|
870
|
+
onChunk: (chunk) => {
|
|
871
|
+
// chunk.audio — raw PCM ArrayBuffer chunk (PCM mode)
|
|
872
|
+
// chunk.partial — partial transcription string (STT mode)
|
|
873
|
+
},
|
|
874
|
+
});
|
|
875
|
+
|
|
876
|
+
if (result.transcription) {
|
|
877
|
+
console.log('Transcription:', result.transcription);
|
|
878
|
+
} else if (result.audio) {
|
|
879
|
+
console.log(`PCM audio: ${result.audio.byteLength} bytes`);
|
|
880
|
+
}
|
|
881
|
+
} catch (e) {
|
|
882
|
+
if (ErrorUtil.isVoiceInputCanceledError(e)) {
|
|
883
|
+
// User pressed the cancel button on the car screen
|
|
884
|
+
console.log('Voice input cancelled');
|
|
885
|
+
} else {
|
|
886
|
+
console.error(e);
|
|
887
|
+
}
|
|
888
|
+
}
|
|
862
889
|
|
|
863
|
-
// Stop recording early — resolves startVoiceInput with
|
|
864
|
-
|
|
890
|
+
// Stop recording early — resolves startVoiceInput with audio captured so far
|
|
891
|
+
HybridVoice.stopVoiceInput();
|
|
865
892
|
```
|
|
866
893
|
|
|
867
|
-
|
|
894
|
+
| Option | Type | Default | Description |
|
|
895
|
+
|---|---|---|---|
|
|
896
|
+
| `silenceThresholdMs` | `number` | `1500` | Auto-stop after this many ms of silence |
|
|
897
|
+
| `maxDurationMs` | `number` | `10000` | Hard recording time limit |
|
|
898
|
+
| `listeningText` | `string` | — | iOS only — text shown on `CPVoiceControlTemplate` |
|
|
899
|
+
| `listeningImage` | `VoiceInputImage` | — | iOS only — animated image in the CarPlay overlay |
|
|
900
|
+
| `preferSpeechToText` | `boolean` | `false` | `true` → resolve with `{ transcription }`; `false` → resolve with `{ audio }` |
|
|
901
|
+
| `startSound` | `number` | — | Metro asset (`require('./beep.mp3')`) played before recording. Takes audio focus so other apps pause. |
|
|
902
|
+
| `endSound` | `number` | — | Metro asset played after recording stops |
|
|
903
|
+
| `onChunk` | `(chunk) => void` | — | Streaming callback: `chunk.audio` (PCM) or `chunk.partial` (STT) |
|
|
904
|
+
| `language` | `string` | system | BCP-47 language tag for the STT recognizer |
|
|
905
|
+
|
|
906
|
+
**PCM result** (`preferSpeechToText: false`, default): resolves with `{ audio: ArrayBuffer }` — raw 16 kHz, 16-bit, mono PCM.
|
|
907
|
+
|
|
908
|
+
**STT result** (`preferSpeechToText: true`): resolves with `{ transcription: string }` on success, or falls back to `{ audio }` if recognition is unavailable.
|
|
868
909
|
|
|
869
|
-
On **
|
|
910
|
+
On **Android**: uses `CarAudioRecord` when Android Auto is connected, otherwise falls back to standard `AudioRecord`. STT uses `SpeechRecognizer`.
|
|
911
|
+
|
|
912
|
+
On **iOS**: presents `CPVoiceControlTemplate` on the car screen when CarPlay is connected, and captures audio via `AVAudioEngine`. STT uses `SFSpeechRecognizer`.
|
|
913
|
+
|
|
914
|
+
#### Cancel detection
|
|
915
|
+
|
|
916
|
+
When the user presses the cancel button on the car screen, `startVoiceInput` rejects with a `voiceInputCancelled` error on both platforms. Use `ErrorUtil.isVoiceInputCanceledError` to distinguish it from other errors:
|
|
917
|
+
|
|
918
|
+
```ts
|
|
919
|
+
import { ErrorUtil } from '@iternio/react-native-auto-play';
|
|
920
|
+
|
|
921
|
+
HybridVoice.startVoiceInput().catch((e) => {
|
|
922
|
+
if (ErrorUtil.isVoiceInputCanceledError(e)) {
|
|
923
|
+
// user dismissed — no action needed
|
|
924
|
+
} else {
|
|
925
|
+
throw e;
|
|
926
|
+
}
|
|
927
|
+
});
|
|
928
|
+
```
|
|
870
929
|
|
|
871
930
|
#### OS-triggered voice input (Android only)
|
|
872
931
|
|
package/android/src/main/java/com/margelo/nitro/swe/iternio/reactnativeautoplay/AutoPlayError.kt
CHANGED
|
@@ -1,5 +1,10 @@
|
|
|
1
1
|
package com.margelo.nitro.swe.iternio.reactnativeautoplay
|
|
2
2
|
|
|
3
|
-
class TemplateNotFoundException(private val templateId: String):
|
|
3
|
+
class TemplateNotFoundException(private val templateId: String) :
|
|
4
|
+
Exception("templateNotFound(\"$templateId\")") {
|
|
4
5
|
override fun toString(): String = "templateNotFound(\"$templateId\")"
|
|
5
6
|
}
|
|
7
|
+
|
|
8
|
+
class VoiceInputCancelledException : Exception("voiceInputCancelled") {
|
|
9
|
+
override fun toString(): String = "voiceInputCancelled"
|
|
10
|
+
}
|
package/android/src/main/java/com/margelo/nitro/swe/iternio/reactnativeautoplay/HybridVoice.kt
CHANGED
|
@@ -50,7 +50,7 @@ class HybridVoice : HybridVoiceSpec() {
|
|
|
50
50
|
}
|
|
51
51
|
cont.resume(
|
|
52
52
|
grantResults.isNotEmpty() &&
|
|
53
|
-
|
|
53
|
+
grantResults.first() == PackageManager.PERMISSION_GRANTED
|
|
54
54
|
)
|
|
55
55
|
true
|
|
56
56
|
}
|
|
@@ -68,7 +68,9 @@ class HybridVoice : HybridVoiceSpec() {
|
|
|
68
68
|
listeningImageRepeats: Boolean?,
|
|
69
69
|
preferSpeechToText: Boolean?,
|
|
70
70
|
onChunk: ((chunk: VoiceInputChunk) -> Unit)?,
|
|
71
|
-
language: String
|
|
71
|
+
language: String?,
|
|
72
|
+
startSoundUri: String?,
|
|
73
|
+
endSoundUri: String?
|
|
72
74
|
): Promise<VoiceInputResult> {
|
|
73
75
|
return Promise.async {
|
|
74
76
|
if (Build.VERSION.SDK_INT < Build.VERSION_CODES.O) {
|
|
@@ -84,7 +86,9 @@ class HybridVoice : HybridVoiceSpec() {
|
|
|
84
86
|
maxDurationMs = maxDurationMs?.toLong() ?: 10_000L,
|
|
85
87
|
preferSpeechToText = preferSpeechToText ?: false,
|
|
86
88
|
onChunk = onChunk,
|
|
87
|
-
language = language
|
|
89
|
+
language = language,
|
|
90
|
+
startSoundUri = startSoundUri,
|
|
91
|
+
endSoundUri = endSoundUri,
|
|
88
92
|
)
|
|
89
93
|
} finally {
|
|
90
94
|
voiceInputManager = null
|
package/android/src/main/java/com/margelo/nitro/swe/iternio/reactnativeautoplay/VoiceInputManager.kt
CHANGED
|
@@ -10,7 +10,9 @@ import android.media.AudioFocusRequest
|
|
|
10
10
|
import android.media.AudioFormat
|
|
11
11
|
import android.media.AudioManager
|
|
12
12
|
import android.media.AudioRecord
|
|
13
|
+
import android.media.MediaPlayer
|
|
13
14
|
import android.media.MediaRecorder
|
|
15
|
+
import android.net.Uri
|
|
14
16
|
import android.os.Build
|
|
15
17
|
import android.os.Bundle
|
|
16
18
|
import android.os.ParcelFileDescriptor
|
|
@@ -62,6 +64,9 @@ class VoiceInputManager(
|
|
|
62
64
|
@Volatile
|
|
63
65
|
private var isRecording = false
|
|
64
66
|
|
|
67
|
+
@Volatile
|
|
68
|
+
private var cancelledByUser = false
|
|
69
|
+
|
|
65
70
|
// STT state — only set when SpeechRecognizer owns the mic
|
|
66
71
|
@Volatile
|
|
67
72
|
private var activeSpeechRecognizer: SpeechRecognizer? = null
|
|
@@ -72,22 +77,41 @@ class VoiceInputManager(
|
|
|
72
77
|
maxDurationMs: Long = 10_000,
|
|
73
78
|
preferSpeechToText: Boolean = false,
|
|
74
79
|
onChunk: ((chunk: VoiceInputChunk) -> Unit)? = null,
|
|
75
|
-
language: String? = null
|
|
80
|
+
language: String? = null,
|
|
81
|
+
startSoundUri: String? = null,
|
|
82
|
+
endSoundUri: String? = null,
|
|
76
83
|
): VoiceInputResult {
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
84
|
+
cancelledByUser = false
|
|
85
|
+
if (!requestAudioFocus()) {
|
|
86
|
+
throw IllegalStateException("Audio focus request denied")
|
|
87
|
+
}
|
|
88
|
+
try {
|
|
89
|
+
startSoundUri?.let { playSound(it) }
|
|
90
|
+
val result = if (preferSpeechToText) {
|
|
91
|
+
val context = NitroModules.applicationContext ?: throw IllegalArgumentException()
|
|
92
|
+
if (SpeechRecognizer.isRecognitionAvailable(context)) {
|
|
93
|
+
if (carContext != null) {
|
|
94
|
+
if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.TIRAMISU) {
|
|
95
|
+
startSTTFromCarAudio(silenceThresholdMs, maxDurationMs, onChunk, language)
|
|
96
|
+
} else {
|
|
97
|
+
// Car connected but API < 33: EXTRA_AUDIO_SOURCE unavailable, fall back to PCM
|
|
98
|
+
startPCM(silenceThresholdMs, maxDurationMs, onChunk)
|
|
99
|
+
}
|
|
100
|
+
} else {
|
|
101
|
+
ThreadUtil.postOnUiAndAwait { startSTT(context, onChunk, language) }.getOrThrow()
|
|
83
102
|
}
|
|
84
|
-
|
|
85
|
-
|
|
103
|
+
} else {
|
|
104
|
+
startPCM(silenceThresholdMs, maxDurationMs, onChunk)
|
|
86
105
|
}
|
|
87
|
-
|
|
106
|
+
} else {
|
|
107
|
+
startPCM(silenceThresholdMs, maxDurationMs, onChunk)
|
|
88
108
|
}
|
|
109
|
+
if (cancelledByUser) throw VoiceInputCancelledException()
|
|
110
|
+
endSoundUri?.let { playSound(it) }
|
|
111
|
+
return result
|
|
112
|
+
} finally {
|
|
113
|
+
abandonAudioFocus()
|
|
89
114
|
}
|
|
90
|
-
return startPCM(silenceThresholdMs, maxDurationMs, onChunk)
|
|
91
115
|
}
|
|
92
116
|
|
|
93
117
|
// MARK: - STT path (SpeechRecognizer owns the mic)
|
|
@@ -309,35 +333,8 @@ class VoiceInputManager(
|
|
|
309
333
|
return@suspendCancellableCoroutine
|
|
310
334
|
}
|
|
311
335
|
|
|
312
|
-
val appContext = NitroModules.applicationContext ?: run {
|
|
313
|
-
cont.resumeWithException(SecurityException("Missing application context"))
|
|
314
|
-
return@suspendCancellableCoroutine
|
|
315
|
-
}
|
|
316
|
-
|
|
317
336
|
pcmContinuation = cont
|
|
318
337
|
|
|
319
|
-
val audioManager = appContext.getSystemService(AudioManager::class.java)
|
|
320
|
-
|
|
321
|
-
val audioAttributes =
|
|
322
|
-
AudioAttributes.Builder().setContentType(AudioAttributes.CONTENT_TYPE_SPEECH)
|
|
323
|
-
.setUsage(AudioAttributes.USAGE_ASSISTANCE_NAVIGATION_GUIDANCE).build()
|
|
324
|
-
|
|
325
|
-
val focusRequest =
|
|
326
|
-
AudioFocusRequest.Builder(AudioManager.AUDIOFOCUS_GAIN_TRANSIENT_EXCLUSIVE)
|
|
327
|
-
.setAudioAttributes(audioAttributes).setOnAudioFocusChangeListener { state ->
|
|
328
|
-
if (state == AudioManager.AUDIOFOCUS_LOSS) {
|
|
329
|
-
stop()
|
|
330
|
-
}
|
|
331
|
-
}.build()
|
|
332
|
-
|
|
333
|
-
if (audioManager.requestAudioFocus(focusRequest) != AudioManager.AUDIOFOCUS_REQUEST_GRANTED) {
|
|
334
|
-
pcmContinuation = null
|
|
335
|
-
cont.resumeWithException(IllegalStateException("Audio focus request denied"))
|
|
336
|
-
return@suspendCancellableCoroutine
|
|
337
|
-
}
|
|
338
|
-
|
|
339
|
-
audioFocusRequest = focusRequest
|
|
340
|
-
|
|
341
338
|
val bufferSize: Int
|
|
342
339
|
|
|
343
340
|
if (carContext != null) {
|
|
@@ -381,6 +378,8 @@ class VoiceInputManager(
|
|
|
381
378
|
) ?: -1
|
|
382
379
|
|
|
383
380
|
if (read < 0) {
|
|
381
|
+
// Whenever the user dismisses the microphone on the car screen, the next call to read will return -1
|
|
382
|
+
cancelledByUser = carAudioRecord != null && read == -1
|
|
384
383
|
break
|
|
385
384
|
}
|
|
386
385
|
|
|
@@ -450,6 +449,72 @@ class VoiceInputManager(
|
|
|
450
449
|
audioRecord?.stop()
|
|
451
450
|
}
|
|
452
451
|
|
|
452
|
+
@RequiresApi(Build.VERSION_CODES.O)
|
|
453
|
+
private fun requestAudioFocus(): Boolean {
|
|
454
|
+
val appContext = NitroModules.applicationContext ?: return false
|
|
455
|
+
val audioManager = appContext.getSystemService(AudioManager::class.java)
|
|
456
|
+
val audioAttributes = AudioAttributes.Builder()
|
|
457
|
+
.setContentType(AudioAttributes.CONTENT_TYPE_SPEECH)
|
|
458
|
+
.setUsage(AudioAttributes.USAGE_ASSISTANCE_NAVIGATION_GUIDANCE)
|
|
459
|
+
.build()
|
|
460
|
+
val focusRequest = AudioFocusRequest.Builder(AudioManager.AUDIOFOCUS_GAIN_TRANSIENT_EXCLUSIVE)
|
|
461
|
+
.setAudioAttributes(audioAttributes)
|
|
462
|
+
.setOnAudioFocusChangeListener { state ->
|
|
463
|
+
if (state == AudioManager.AUDIOFOCUS_LOSS) { stop() }
|
|
464
|
+
}
|
|
465
|
+
.build()
|
|
466
|
+
return if (audioManager.requestAudioFocus(focusRequest) == AudioManager.AUDIOFOCUS_REQUEST_GRANTED) {
|
|
467
|
+
audioFocusRequest = focusRequest
|
|
468
|
+
true
|
|
469
|
+
} else {
|
|
470
|
+
false
|
|
471
|
+
}
|
|
472
|
+
}
|
|
473
|
+
|
|
474
|
+
@RequiresApi(Build.VERSION_CODES.O)
|
|
475
|
+
private fun abandonAudioFocus() {
|
|
476
|
+
audioFocusRequest?.let {
|
|
477
|
+
val audioManager = (NitroModules.applicationContext ?: carContext)
|
|
478
|
+
?.getSystemService(AudioManager::class.java)
|
|
479
|
+
audioManager?.abandonAudioFocusRequest(it)
|
|
480
|
+
}
|
|
481
|
+
audioFocusRequest = null
|
|
482
|
+
}
|
|
483
|
+
|
|
484
|
+
private suspend fun playSound(uri: String) = suspendCancellableCoroutine<Unit> { cont ->
|
|
485
|
+
val context = NitroModules.applicationContext ?: run {
|
|
486
|
+
cont.resume(Unit)
|
|
487
|
+
return@suspendCancellableCoroutine
|
|
488
|
+
}
|
|
489
|
+
val player = MediaPlayer()
|
|
490
|
+
try {
|
|
491
|
+
player.setAudioAttributes(
|
|
492
|
+
AudioAttributes.Builder()
|
|
493
|
+
.setContentType(AudioAttributes.CONTENT_TYPE_SONIFICATION)
|
|
494
|
+
.setUsage(AudioAttributes.USAGE_ASSISTANCE_NAVIGATION_GUIDANCE)
|
|
495
|
+
.build()
|
|
496
|
+
)
|
|
497
|
+
player.setDataSource(context, Uri.parse(uri))
|
|
498
|
+
player.setOnCompletionListener {
|
|
499
|
+
it.release()
|
|
500
|
+
if (cont.isActive) { cont.resume(Unit) }
|
|
501
|
+
}
|
|
502
|
+
player.setOnErrorListener { mp, _, _ ->
|
|
503
|
+
mp.release()
|
|
504
|
+
if (cont.isActive) { cont.resume(Unit) }
|
|
505
|
+
true
|
|
506
|
+
}
|
|
507
|
+
player.prepare()
|
|
508
|
+
player.start()
|
|
509
|
+
} catch (_: Exception) {
|
|
510
|
+
try { player.release() } catch (_: Exception) {}
|
|
511
|
+
if (cont.isActive) { cont.resume(Unit) }
|
|
512
|
+
}
|
|
513
|
+
cont.invokeOnCancellation {
|
|
514
|
+
try { player.release() } catch (_: Exception) {}
|
|
515
|
+
}
|
|
516
|
+
}
|
|
517
|
+
|
|
453
518
|
@RequiresApi(Build.VERSION_CODES.O)
|
|
454
519
|
private fun releaseResources() {
|
|
455
520
|
carAudioRecord?.stopRecording()
|
|
@@ -458,13 +523,6 @@ class VoiceInputManager(
|
|
|
458
523
|
audioRecord?.release()
|
|
459
524
|
audioRecord = null
|
|
460
525
|
recordingJob = null
|
|
461
|
-
audioFocusRequest?.let {
|
|
462
|
-
val audioManager = (NitroModules.applicationContext ?: carContext)?.getSystemService(
|
|
463
|
-
AudioManager::class.java,
|
|
464
|
-
)
|
|
465
|
-
audioManager?.abandonAudioFocusRequest(it)
|
|
466
|
-
}
|
|
467
|
-
audioFocusRequest = null
|
|
468
526
|
}
|
|
469
527
|
|
|
470
528
|
fun dispose() {
|
package/ios/Types.swift
CHANGED
|
@@ -10,7 +10,7 @@ struct TemplateEventPayload {
|
|
|
10
10
|
let state: VisibilityState
|
|
11
11
|
}
|
|
12
12
|
|
|
13
|
-
enum AutoPlayError:
|
|
13
|
+
enum AutoPlayError: LocalizedError {
|
|
14
14
|
case templateNotFound(String)
|
|
15
15
|
case interfaceControllerNotFound(String)
|
|
16
16
|
case invalidTemplateError(String)
|
|
@@ -19,4 +19,5 @@ enum AutoPlayError: Error {
|
|
|
19
19
|
case invalidTemplateType(String)
|
|
20
20
|
case noUiWindow(String)
|
|
21
21
|
case initReactRootViewFailed(String)
|
|
22
|
+
case voiceInputCancelled
|
|
22
23
|
}
|
|
@@ -85,3 +85,10 @@ extension CPAlertTemplate {
|
|
|
85
85
|
initTemplate(template: self, id: id)
|
|
86
86
|
}
|
|
87
87
|
}
|
|
88
|
+
|
|
89
|
+
extension CPVoiceControlTemplate {
|
|
90
|
+
convenience init(voiceControlStates: [CPVoiceControlState], id: String) {
|
|
91
|
+
self.init(voiceControlStates: voiceControlStates)
|
|
92
|
+
initTemplate(template: self, id: id)
|
|
93
|
+
}
|
|
94
|
+
}
|
|
@@ -36,7 +36,9 @@ class HybridVoice: HybridVoiceSpec {
|
|
|
36
36
|
listeningImageRepeats: Bool?,
|
|
37
37
|
preferSpeechToText: Bool?,
|
|
38
38
|
onChunk: ((_ chunk: VoiceInputChunk) -> Void)?,
|
|
39
|
-
language: String
|
|
39
|
+
language: String?,
|
|
40
|
+
startSoundUri: String?,
|
|
41
|
+
endSoundUri: String?
|
|
40
42
|
) throws -> Promise<VoiceInputResult> {
|
|
41
43
|
return Promise.async {
|
|
42
44
|
let interfaceController = try? await RootModule.withInterfaceController { $0 }
|
|
@@ -55,7 +57,9 @@ class HybridVoice: HybridVoiceSpec {
|
|
|
55
57
|
listeningImageRepeats: listeningImageRepeats,
|
|
56
58
|
preferSpeechToText: preferSpeechToText ?? false,
|
|
57
59
|
onChunk: onChunk,
|
|
58
|
-
language: language
|
|
60
|
+
language: language,
|
|
61
|
+
startSoundUri: startSoundUri,
|
|
62
|
+
endSoundUri: endSoundUri
|
|
59
63
|
)
|
|
60
64
|
}
|
|
61
65
|
}
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
import CarPlay
|
|
2
|
+
|
|
3
|
+
class VoiceInputTemplate: AutoPlayTemplate {
|
|
4
|
+
let template: CPVoiceControlTemplate
|
|
5
|
+
private let onDidDisappearCallback: () -> Void
|
|
6
|
+
|
|
7
|
+
override func getTemplate() -> CPTemplate {
|
|
8
|
+
return template
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
init(
|
|
12
|
+
voiceControlStates: [CPVoiceControlState],
|
|
13
|
+
id: String,
|
|
14
|
+
onDidDisappear: @escaping () -> Void
|
|
15
|
+
) {
|
|
16
|
+
self.template = CPVoiceControlTemplate(voiceControlStates: voiceControlStates, id: id)
|
|
17
|
+
self.onDidDisappearCallback = onDidDisappear
|
|
18
|
+
|
|
19
|
+
super.init()
|
|
20
|
+
|
|
21
|
+
try? RootModule.withTemplateStore { templateStore in
|
|
22
|
+
templateStore.addTemplate(template: self, templateId: id)
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
override func onDidDisappear(animated: Bool) {
|
|
27
|
+
onDidDisappearCallback()
|
|
28
|
+
|
|
29
|
+
try? RootModule.withTemplateStore { templateStore in
|
|
30
|
+
templateStore.removeTemplate(templateId: self.template.id)
|
|
31
|
+
}
|
|
32
|
+
}
|
|
33
|
+
}
|
|
@@ -3,6 +3,31 @@ import CarPlay
|
|
|
3
3
|
import NitroModules
|
|
4
4
|
import Speech
|
|
5
5
|
|
|
6
|
+
/// Keeps itself and the player alive until playback finishes.
|
|
7
|
+
/// Needed because AVAudioPlayer.delegate is weak, so without an external strong
|
|
8
|
+
/// reference the delegate (and player) would be released immediately after play().
|
|
9
|
+
private final class AudioPlayerDelegate: NSObject, AVAudioPlayerDelegate, @unchecked Sendable {
|
|
10
|
+
private let onFinish: () -> Void
|
|
11
|
+
private var keepAlive: AudioPlayerDelegate?
|
|
12
|
+
private var player: AVAudioPlayer?
|
|
13
|
+
|
|
14
|
+
init(player: AVAudioPlayer, _ onFinish: @escaping () -> Void) {
|
|
15
|
+
self.onFinish = onFinish
|
|
16
|
+
self.player = player
|
|
17
|
+
super.init()
|
|
18
|
+
keepAlive = self
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
private func finish() {
|
|
22
|
+
player = nil
|
|
23
|
+
keepAlive = nil
|
|
24
|
+
onFinish()
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
func audioPlayerDidFinishPlaying(_: AVAudioPlayer, successfully _: Bool) { finish() }
|
|
28
|
+
func audioPlayerDecodeErrorDidOccur(_: AVAudioPlayer, error _: Error?) { finish() }
|
|
29
|
+
}
|
|
30
|
+
|
|
6
31
|
/// Wraps CheckedContinuation so it can only be resumed once even when
|
|
7
32
|
/// shared between a stop() call and an async recognition task callback.
|
|
8
33
|
private final class ResultBox: @unchecked Sendable {
|
|
@@ -36,6 +61,8 @@ class VoiceInputManager {
|
|
|
36
61
|
private var resultBox: ResultBox?
|
|
37
62
|
private var samples: [Int16] = []
|
|
38
63
|
private var isStopping = false
|
|
64
|
+
private var cancelledByUser = false
|
|
65
|
+
private var isIgnoringSamples = false
|
|
39
66
|
private let stopLock = NSLock()
|
|
40
67
|
|
|
41
68
|
// STT
|
|
@@ -69,9 +96,15 @@ class VoiceInputManager {
|
|
|
69
96
|
listeningImageRepeats: Bool?,
|
|
70
97
|
preferSpeechToText: Bool,
|
|
71
98
|
onChunk: ((_ chunk: VoiceInputChunk) -> Void)?,
|
|
72
|
-
language: String
|
|
99
|
+
language: String?,
|
|
100
|
+
startSoundUri: String?,
|
|
101
|
+
endSoundUri: String?
|
|
73
102
|
) async throws -> VoiceInputResult {
|
|
74
|
-
|
|
103
|
+
stopLock.withLock {
|
|
104
|
+
cancelledByUser = false
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
let result = try await withCheckedThrowingContinuation { cont in
|
|
75
108
|
let box = ResultBox(cont)
|
|
76
109
|
self.resultBox = box
|
|
77
110
|
self.samples = []
|
|
@@ -83,9 +116,6 @@ class VoiceInputManager {
|
|
|
83
116
|
interfaceController: interfaceController,
|
|
84
117
|
silenceThresholdMs: silenceThresholdMs,
|
|
85
118
|
maxDurationMs: maxDurationMs,
|
|
86
|
-
listeningText: listeningText,
|
|
87
|
-
listeningImage: listeningImage,
|
|
88
|
-
listeningImageRepeats: listeningImageRepeats,
|
|
89
119
|
preferSpeechToText: preferSpeechToText,
|
|
90
120
|
onChunk: onChunk,
|
|
91
121
|
box: box,
|
|
@@ -95,8 +125,68 @@ class VoiceInputManager {
|
|
|
95
125
|
catch {
|
|
96
126
|
self.cleanup(interfaceController: interfaceController)
|
|
97
127
|
box.resume(throwing: error)
|
|
128
|
+
return
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
// Mic is open — present template then play start sound.
|
|
132
|
+
// isIgnoringSamples discards tap buffers during the sound so it isn't recorded.
|
|
133
|
+
// recordingStart is set after the sound so silence/max-duration timers are accurate.
|
|
134
|
+
self.stopLock.withLock { self.isIgnoringSamples = startSoundUri != nil }
|
|
135
|
+
Task {
|
|
136
|
+
if let interfaceController = interfaceController {
|
|
137
|
+
await self.presentVoiceTemplate(
|
|
138
|
+
interfaceController: interfaceController,
|
|
139
|
+
listeningText: listeningText,
|
|
140
|
+
listeningImage: listeningImage,
|
|
141
|
+
listeningImageRepeats: listeningImageRepeats
|
|
142
|
+
)
|
|
143
|
+
}
|
|
144
|
+
if let uri = startSoundUri {
|
|
145
|
+
await self.playSound(uri: uri, setupSession: false)
|
|
146
|
+
}
|
|
147
|
+
self.stopLock.withLock {
|
|
148
|
+
self.isIgnoringSamples = false
|
|
149
|
+
self.recordingStart = Date()
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
if let uri = endSoundUri {
|
|
155
|
+
await playSound(uri: uri)
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
return result
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
private func playSound(uri: String, setupSession: Bool = true) async {
|
|
162
|
+
guard let url = URL(string: uri) else { return }
|
|
163
|
+
do {
|
|
164
|
+
let session = AVAudioSession.sharedInstance()
|
|
165
|
+
if setupSession {
|
|
166
|
+
try session.setCategory(.playback, mode: .default)
|
|
167
|
+
try session.setActive(true)
|
|
168
|
+
}
|
|
169
|
+
// URLSession handles both http:// (Metro dev server) and file:// (release bundle)
|
|
170
|
+
let (data, _) = try await URLSession.shared.data(from: url)
|
|
171
|
+
await withCheckedContinuation { (cont: CheckedContinuation<Void, Never>) in
|
|
172
|
+
DispatchQueue.main.async {
|
|
173
|
+
do {
|
|
174
|
+
let player = try AVAudioPlayer(data: data)
|
|
175
|
+
let delegate = AudioPlayerDelegate(player: player) { cont.resume() }
|
|
176
|
+
player.delegate = delegate
|
|
177
|
+
player.prepareToPlay()
|
|
178
|
+
player.play()
|
|
179
|
+
}
|
|
180
|
+
catch {
|
|
181
|
+
cont.resume()
|
|
182
|
+
}
|
|
183
|
+
}
|
|
98
184
|
}
|
|
99
185
|
}
|
|
186
|
+
catch {
|
|
187
|
+
print(error)
|
|
188
|
+
// fail silently — a broken sound file must not block voice input
|
|
189
|
+
}
|
|
100
190
|
}
|
|
101
191
|
|
|
102
192
|
func stop(interfaceController: AutoPlayInterfaceController? = nil) {
|
|
@@ -106,6 +196,7 @@ class VoiceInputManager {
|
|
|
106
196
|
return
|
|
107
197
|
}
|
|
108
198
|
isStopping = true
|
|
199
|
+
let wasCancelled = cancelledByUser
|
|
109
200
|
let wasSTTMode = isSTTMode
|
|
110
201
|
let capturedRequest = recognitionRequest
|
|
111
202
|
let box = resultBox
|
|
@@ -121,7 +212,12 @@ class VoiceInputManager {
|
|
|
121
212
|
}
|
|
122
213
|
else {
|
|
123
214
|
cleanup(interfaceController: interfaceController)
|
|
124
|
-
|
|
215
|
+
if wasCancelled {
|
|
216
|
+
box?.resume(throwing: AutoPlayError.voiceInputCancelled)
|
|
217
|
+
}
|
|
218
|
+
else {
|
|
219
|
+
box?.resume(returning: makePCMResult(from: capturedSamples))
|
|
220
|
+
}
|
|
125
221
|
}
|
|
126
222
|
}
|
|
127
223
|
|
|
@@ -131,9 +227,6 @@ class VoiceInputManager {
|
|
|
131
227
|
interfaceController: AutoPlayInterfaceController?,
|
|
132
228
|
silenceThresholdMs: Double,
|
|
133
229
|
maxDurationMs: Double,
|
|
134
|
-
listeningText: String,
|
|
135
|
-
listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?,
|
|
136
|
-
listeningImageRepeats: Bool?,
|
|
137
230
|
preferSpeechToText: Bool,
|
|
138
231
|
onChunk: ((_ chunk: VoiceInputChunk) -> Void)?,
|
|
139
232
|
box: ResultBox,
|
|
@@ -147,15 +240,6 @@ class VoiceInputManager {
|
|
|
147
240
|
try session.setCategory(.playAndRecord, mode: .measurement, options: [])
|
|
148
241
|
try session.setActive(true)
|
|
149
242
|
|
|
150
|
-
if let interfaceController {
|
|
151
|
-
presentVoiceTemplate(
|
|
152
|
-
interfaceController: interfaceController,
|
|
153
|
-
listeningText: listeningText,
|
|
154
|
-
listeningImage: listeningImage,
|
|
155
|
-
listeningImageRepeats: listeningImageRepeats
|
|
156
|
-
)
|
|
157
|
-
}
|
|
158
|
-
|
|
159
243
|
var activeRecognitionRequest: SFSpeechAudioBufferRecognitionRequest? = nil
|
|
160
244
|
|
|
161
245
|
if preferSpeechToText, SFSpeechRecognizer.authorizationStatus() == .authorized,
|
|
@@ -176,12 +260,18 @@ class VoiceInputManager {
|
|
|
176
260
|
// STT failed — fall back to whatever PCM was accumulated
|
|
177
261
|
self.stopLock.lock()
|
|
178
262
|
self.isStopping = true
|
|
263
|
+
let wasCancelled = self.cancelledByUser
|
|
179
264
|
let capturedSamples = self.samples
|
|
180
265
|
self.samples = []
|
|
181
266
|
self.stopLock.unlock()
|
|
182
267
|
|
|
183
268
|
self.cleanup(interfaceController: interfaceController)
|
|
184
|
-
|
|
269
|
+
if wasCancelled {
|
|
270
|
+
box.resume(throwing: AutoPlayError.voiceInputCancelled)
|
|
271
|
+
}
|
|
272
|
+
else {
|
|
273
|
+
box.resume(returning: self.makePCMResult(from: capturedSamples))
|
|
274
|
+
}
|
|
185
275
|
return
|
|
186
276
|
}
|
|
187
277
|
|
|
@@ -190,16 +280,22 @@ class VoiceInputManager {
|
|
|
190
280
|
if result.isFinal {
|
|
191
281
|
self.stopLock.lock()
|
|
192
282
|
self.isStopping = true
|
|
283
|
+
let wasCancelled = self.cancelledByUser
|
|
193
284
|
self.samples = []
|
|
194
285
|
self.stopLock.unlock()
|
|
195
286
|
|
|
196
287
|
self.cleanup(interfaceController: interfaceController)
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
288
|
+
if wasCancelled {
|
|
289
|
+
box.resume(throwing: AutoPlayError.voiceInputCancelled)
|
|
290
|
+
}
|
|
291
|
+
else {
|
|
292
|
+
box.resume(
|
|
293
|
+
returning: VoiceInputResult(
|
|
294
|
+
transcription: result.bestTranscription.formattedString,
|
|
295
|
+
audio: nil
|
|
296
|
+
)
|
|
201
297
|
)
|
|
202
|
-
|
|
298
|
+
}
|
|
203
299
|
}
|
|
204
300
|
else {
|
|
205
301
|
onChunk?(VoiceInputChunk(partial: result.bestTranscription.formattedString, audio: nil))
|
|
@@ -215,15 +311,24 @@ class VoiceInputManager {
|
|
|
215
311
|
throw VoiceInputError.converterUnavailable
|
|
216
312
|
}
|
|
217
313
|
|
|
218
|
-
recordingStart =
|
|
314
|
+
recordingStart = nil
|
|
219
315
|
silenceStart = nil
|
|
316
|
+
isIgnoringSamples = false
|
|
220
317
|
|
|
221
318
|
inputNode.installTap(
|
|
222
319
|
onBus: 0,
|
|
223
320
|
bufferSize: VoiceInputManager.tapBufferSize,
|
|
224
321
|
format: nativeFormat
|
|
225
322
|
) { [weak self] buffer, _ in
|
|
226
|
-
guard let self
|
|
323
|
+
guard let self else { return }
|
|
324
|
+
|
|
325
|
+
self.stopLock.lock()
|
|
326
|
+
let stopping = self.isStopping
|
|
327
|
+
let ignoringSamples = self.isIgnoringSamples
|
|
328
|
+
let recordingStartSnapshot = self.recordingStart
|
|
329
|
+
self.stopLock.unlock()
|
|
330
|
+
|
|
331
|
+
guard !stopping, !ignoringSamples else { return }
|
|
227
332
|
|
|
228
333
|
// Feed STT if active
|
|
229
334
|
activeRecognitionRequest?.append(buffer)
|
|
@@ -266,7 +371,7 @@ class VoiceInputManager {
|
|
|
266
371
|
let now = Date()
|
|
267
372
|
|
|
268
373
|
// Max duration — applies in both modes
|
|
269
|
-
if let start =
|
|
374
|
+
if let start = recordingStartSnapshot,
|
|
270
375
|
now.timeIntervalSince(start) * 1000 >= maxDurationMs
|
|
271
376
|
{
|
|
272
377
|
self.triggerAutoStop(interfaceController: interfaceController)
|
|
@@ -275,7 +380,7 @@ class VoiceInputManager {
|
|
|
275
380
|
|
|
276
381
|
// Silence detection — skip during warm-up so the pipeline has time
|
|
277
382
|
// to stabilise before we start measuring amplitude
|
|
278
|
-
if let start =
|
|
383
|
+
if let start = recordingStartSnapshot,
|
|
279
384
|
now.timeIntervalSince(start) * 1000 >= VoiceInputManager.warmupMs
|
|
280
385
|
{
|
|
281
386
|
let peak = newSamples.reduce(0) { max($0, abs(Int($1))) }
|
|
@@ -381,12 +486,13 @@ class VoiceInputManager {
|
|
|
381
486
|
return nil
|
|
382
487
|
}
|
|
383
488
|
|
|
489
|
+
@MainActor
|
|
384
490
|
private func presentVoiceTemplate(
|
|
385
491
|
interfaceController: AutoPlayInterfaceController,
|
|
386
492
|
listeningText: String,
|
|
387
493
|
listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?,
|
|
388
494
|
listeningImageRepeats: Bool?
|
|
389
|
-
) {
|
|
495
|
+
) async {
|
|
390
496
|
let traitCollection = SceneStore.getRootTraitCollection() ?? UITraitCollection.current
|
|
391
497
|
let image = loadVoiceImage(
|
|
392
498
|
image: listeningImage,
|
|
@@ -400,14 +506,21 @@ class VoiceInputManager {
|
|
|
400
506
|
image: image,
|
|
401
507
|
repeats: repeats
|
|
402
508
|
)
|
|
403
|
-
let template = CPVoiceControlTemplate(voiceControlStates: [listeningState])
|
|
404
|
-
initTemplate(template: template, id: "voice-input")
|
|
405
|
-
voiceControlTemplate = template
|
|
406
509
|
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
510
|
+
let voiceTemplate = VoiceInputTemplate(
|
|
511
|
+
voiceControlStates: [listeningState],
|
|
512
|
+
id: "voice-input"
|
|
513
|
+
) { [weak self] in
|
|
514
|
+
guard let self else { return }
|
|
515
|
+
self.stopLock.withLock {
|
|
516
|
+
if !self.isStopping { self.cancelledByUser = true }
|
|
517
|
+
}
|
|
518
|
+
self.stop()
|
|
410
519
|
}
|
|
520
|
+
|
|
521
|
+
voiceControlTemplate = voiceTemplate.template
|
|
522
|
+
try? await interfaceController.presentTemplate(voiceTemplate.template, animated: true)
|
|
523
|
+
voiceTemplate.template.activateVoiceControlState(withIdentifier: "listening")
|
|
411
524
|
}
|
|
412
525
|
|
|
413
526
|
private func dismissVoiceTemplate(interfaceController: AutoPlayInterfaceController) {
|
|
@@ -1,10 +1,13 @@
|
|
|
1
|
+
import { Image } from 'react-native';
|
|
1
2
|
import { NitroModules } from 'react-native-nitro-modules';
|
|
2
3
|
import { NitroImageUtil } from '../utils/NitroImage';
|
|
3
4
|
const _native = NitroModules.createHybridObject('Voice');
|
|
4
5
|
const startVoiceInput = async (options) => {
|
|
5
|
-
const { onChunk, silenceThresholdMs, maxDurationMs, listeningText, listeningImage, preferSpeechToText, language, } = options ?? {};
|
|
6
|
+
const { onChunk, silenceThresholdMs, maxDurationMs, listeningText, listeningImage, preferSpeechToText, language, startSound, endSound, } = options ?? {};
|
|
6
7
|
const listeningImageRepeats = listeningImage?.type === 'asset' ? listeningImage.repeats : undefined;
|
|
7
|
-
|
|
8
|
+
const startSoundUri = startSound != null ? Image.resolveAssetSource(startSound).uri : undefined;
|
|
9
|
+
const endSoundUri = endSound != null ? Image.resolveAssetSource(endSound).uri : undefined;
|
|
10
|
+
return await _native.startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, NitroImageUtil.convert(listeningImage), listeningImageRepeats, preferSpeechToText, onChunk, language, startSoundUri, endSoundUri);
|
|
8
11
|
};
|
|
9
12
|
export const HybridVoice = {
|
|
10
13
|
/**
|
|
@@ -7,6 +7,6 @@ export interface Voice extends HybridObject<{
|
|
|
7
7
|
}> {
|
|
8
8
|
hasVoiceInputPermission(): boolean;
|
|
9
9
|
requestVoiceInputPermission(): Promise<boolean>;
|
|
10
|
-
startVoiceInput(silenceThresholdMs?: number, maxDurationMs?: number, listeningText?: string, listeningImage?: NitroImage, listeningImageRepeats?: boolean, preferSpeechToText?: boolean, onChunk?: (chunk: VoiceInputChunk) => void, language?: string): Promise<VoiceInputResult>;
|
|
10
|
+
startVoiceInput(silenceThresholdMs?: number, maxDurationMs?: number, listeningText?: string, listeningImage?: NitroImage, listeningImageRepeats?: boolean, preferSpeechToText?: boolean, onChunk?: (chunk: VoiceInputChunk) => void, language?: string, startSoundUri?: string, endSoundUri?: string): Promise<VoiceInputResult>;
|
|
11
11
|
stopVoiceInput(): void;
|
|
12
12
|
}
|
package/lib/types/Voice.d.ts
CHANGED
|
@@ -18,4 +18,8 @@ export interface VoiceInputOptions {
|
|
|
18
18
|
preferSpeechToText?: boolean;
|
|
19
19
|
onChunk?: (chunk: VoiceInputChunk) => void;
|
|
20
20
|
language?: string;
|
|
21
|
+
/** Sound played just before recording starts. Pass a Metro asset: `require('./beep_start.wav')`. */
|
|
22
|
+
startSound?: number;
|
|
23
|
+
/** Sound played just after recording stops. Pass a Metro asset: `require('./beep_end.wav')`. */
|
|
24
|
+
endSound?: number;
|
|
21
25
|
}
|
package/lib/utils/ErrorUtil.d.ts
CHANGED
package/lib/utils/ErrorUtil.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
const
|
|
1
|
+
const isError = (error) => {
|
|
2
2
|
if (error == null) {
|
|
3
3
|
return false;
|
|
4
4
|
}
|
|
@@ -14,6 +14,18 @@ const isTemplateNotFoundError = (error) => {
|
|
|
14
14
|
if (typeof error.message !== 'string') {
|
|
15
15
|
return false;
|
|
16
16
|
}
|
|
17
|
-
return
|
|
17
|
+
return true;
|
|
18
|
+
};
|
|
19
|
+
const isTemplateNotFoundError = (error) => {
|
|
20
|
+
if (isError(error)) {
|
|
21
|
+
return error.message.startsWith('templateNotFound');
|
|
22
|
+
}
|
|
23
|
+
return false;
|
|
24
|
+
};
|
|
25
|
+
const isVoiceInputCanceledError = (error) => {
|
|
26
|
+
if (isError(error)) {
|
|
27
|
+
return error.message.startsWith('voiceInputCancelled');
|
|
28
|
+
}
|
|
29
|
+
return false;
|
|
18
30
|
};
|
|
19
|
-
export const ErrorUtil = { isTemplateNotFoundError };
|
|
31
|
+
export const ErrorUtil = { isTemplateNotFoundError, isVoiceInputCanceledError };
|
|
@@ -98,9 +98,9 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
|
|
|
98
98
|
return __promise;
|
|
99
99
|
}();
|
|
100
100
|
}
|
|
101
|
-
std::shared_ptr<Promise<VoiceInputResult>> JHybridVoiceSpec::startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language) {
|
|
102
|
-
static const auto method = _javaPart->javaClassStatic()->getMethod<jni::local_ref<JPromise::javaobject>(jni::alias_ref<jni::JDouble> /* silenceThresholdMs */, jni::alias_ref<jni::JDouble> /* maxDurationMs */, jni::alias_ref<jni::JString> /* listeningText */, jni::alias_ref<JVariant_GlyphImage_AssetImage_RemoteImage> /* listeningImage */, jni::alias_ref<jni::JBoolean> /* listeningImageRepeats */, jni::alias_ref<jni::JBoolean> /* preferSpeechToText */, jni::alias_ref<JFunc_void_VoiceInputChunk::javaobject> /* onChunk */, jni::alias_ref<jni::JString> /* language */)>("startVoiceInput_cxx");
|
|
103
|
-
auto __result = method(_javaPart, silenceThresholdMs.has_value() ? jni::JDouble::valueOf(silenceThresholdMs.value()) : nullptr, maxDurationMs.has_value() ? jni::JDouble::valueOf(maxDurationMs.value()) : nullptr, listeningText.has_value() ? jni::make_jstring(listeningText.value()) : nullptr, listeningImage.has_value() ? JVariant_GlyphImage_AssetImage_RemoteImage::fromCpp(listeningImage.value()) : nullptr, listeningImageRepeats.has_value() ? jni::JBoolean::valueOf(listeningImageRepeats.value()) : nullptr, preferSpeechToText.has_value() ? jni::JBoolean::valueOf(preferSpeechToText.value()) : nullptr, onChunk.has_value() ? JFunc_void_VoiceInputChunk_cxx::fromCpp(onChunk.value()) : nullptr, language.has_value() ? jni::make_jstring(language.value()) : nullptr);
|
|
101
|
+
std::shared_ptr<Promise<VoiceInputResult>> JHybridVoiceSpec::startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri) {
|
|
102
|
+
static const auto method = _javaPart->javaClassStatic()->getMethod<jni::local_ref<JPromise::javaobject>(jni::alias_ref<jni::JDouble> /* silenceThresholdMs */, jni::alias_ref<jni::JDouble> /* maxDurationMs */, jni::alias_ref<jni::JString> /* listeningText */, jni::alias_ref<JVariant_GlyphImage_AssetImage_RemoteImage> /* listeningImage */, jni::alias_ref<jni::JBoolean> /* listeningImageRepeats */, jni::alias_ref<jni::JBoolean> /* preferSpeechToText */, jni::alias_ref<JFunc_void_VoiceInputChunk::javaobject> /* onChunk */, jni::alias_ref<jni::JString> /* language */, jni::alias_ref<jni::JString> /* startSoundUri */, jni::alias_ref<jni::JString> /* endSoundUri */)>("startVoiceInput_cxx");
|
|
103
|
+
auto __result = method(_javaPart, silenceThresholdMs.has_value() ? jni::JDouble::valueOf(silenceThresholdMs.value()) : nullptr, maxDurationMs.has_value() ? jni::JDouble::valueOf(maxDurationMs.value()) : nullptr, listeningText.has_value() ? jni::make_jstring(listeningText.value()) : nullptr, listeningImage.has_value() ? JVariant_GlyphImage_AssetImage_RemoteImage::fromCpp(listeningImage.value()) : nullptr, listeningImageRepeats.has_value() ? jni::JBoolean::valueOf(listeningImageRepeats.value()) : nullptr, preferSpeechToText.has_value() ? jni::JBoolean::valueOf(preferSpeechToText.value()) : nullptr, onChunk.has_value() ? JFunc_void_VoiceInputChunk_cxx::fromCpp(onChunk.value()) : nullptr, language.has_value() ? jni::make_jstring(language.value()) : nullptr, startSoundUri.has_value() ? jni::make_jstring(startSoundUri.value()) : nullptr, endSoundUri.has_value() ? jni::make_jstring(endSoundUri.value()) : nullptr);
|
|
104
104
|
return [&]() {
|
|
105
105
|
auto __promise = Promise<VoiceInputResult>::create();
|
|
106
106
|
__result->cthis()->addOnResolvedListener([=](const jni::alias_ref<jni::JObject>& __boxedResult) {
|
|
@@ -56,7 +56,7 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
|
|
|
56
56
|
// Methods
|
|
57
57
|
bool hasVoiceInputPermission() override;
|
|
58
58
|
std::shared_ptr<Promise<bool>> requestVoiceInputPermission() override;
|
|
59
|
-
std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language) override;
|
|
59
|
+
std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri) override;
|
|
60
60
|
void stopVoiceInput() override;
|
|
61
61
|
|
|
62
62
|
private:
|
|
@@ -37,12 +37,12 @@ abstract class HybridVoiceSpec: HybridObject() {
|
|
|
37
37
|
@Keep
|
|
38
38
|
abstract fun requestVoiceInputPermission(): Promise<Boolean>
|
|
39
39
|
|
|
40
|
-
abstract fun startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Boolean?, preferSpeechToText: Boolean?, onChunk: ((chunk: VoiceInputChunk) -> Unit)?, language: String?): Promise<VoiceInputResult>
|
|
40
|
+
abstract fun startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Boolean?, preferSpeechToText: Boolean?, onChunk: ((chunk: VoiceInputChunk) -> Unit)?, language: String?, startSoundUri: String?, endSoundUri: String?): Promise<VoiceInputResult>
|
|
41
41
|
|
|
42
42
|
@DoNotStrip
|
|
43
43
|
@Keep
|
|
44
|
-
private fun startVoiceInput_cxx(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Boolean?, preferSpeechToText: Boolean?, onChunk: Func_void_VoiceInputChunk?, language: String?): Promise<VoiceInputResult> {
|
|
45
|
-
val __result = startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, listeningImage, listeningImageRepeats, preferSpeechToText, onChunk?.let { it }, language)
|
|
44
|
+
private fun startVoiceInput_cxx(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Boolean?, preferSpeechToText: Boolean?, onChunk: Func_void_VoiceInputChunk?, language: String?, startSoundUri: String?, endSoundUri: String?): Promise<VoiceInputResult> {
|
|
45
|
+
val __result = startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, listeningImage, listeningImageRepeats, preferSpeechToText, onChunk?.let { it }, language, startSoundUri, endSoundUri)
|
|
46
46
|
return __result
|
|
47
47
|
}
|
|
48
48
|
|
|
@@ -107,8 +107,8 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
|
|
|
107
107
|
auto __value = std::move(__result.value());
|
|
108
108
|
return __value;
|
|
109
109
|
}
|
|
110
|
-
inline std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language) override {
|
|
111
|
-
auto __result = _swiftPart.startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, listeningImage, listeningImageRepeats, preferSpeechToText, onChunk, language);
|
|
110
|
+
inline std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri) override {
|
|
111
|
+
auto __result = _swiftPart.startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, listeningImage, listeningImageRepeats, preferSpeechToText, onChunk, language, startSoundUri, endSoundUri);
|
|
112
112
|
if (__result.hasError()) [[unlikely]] {
|
|
113
113
|
std::rethrow_exception(__result.error());
|
|
114
114
|
}
|
|
@@ -15,7 +15,7 @@ public protocol HybridVoiceSpec_protocol: HybridObject {
|
|
|
15
15
|
// Methods
|
|
16
16
|
func hasVoiceInputPermission() throws -> Bool
|
|
17
17
|
func requestVoiceInputPermission() throws -> Promise<Bool>
|
|
18
|
-
func startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Bool?, preferSpeechToText: Bool?, onChunk: ((_ chunk: VoiceInputChunk) -> Void)?, language: String?) throws -> Promise<VoiceInputResult>
|
|
18
|
+
func startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Bool?, preferSpeechToText: Bool?, onChunk: ((_ chunk: VoiceInputChunk) -> Void)?, language: String?, startSoundUri: String?, endSoundUri: String?) throws -> Promise<VoiceInputResult>
|
|
19
19
|
func stopVoiceInput() throws -> Void
|
|
20
20
|
}
|
|
21
21
|
|
|
@@ -156,7 +156,7 @@ open class HybridVoiceSpec_cxx {
|
|
|
156
156
|
}
|
|
157
157
|
|
|
158
158
|
@inline(__always)
|
|
159
|
-
public final func startVoiceInput(silenceThresholdMs: bridge.std__optional_double_, maxDurationMs: bridge.std__optional_double_, listeningText: bridge.std__optional_std__string_, listeningImage: bridge.std__optional_std__variant_GlyphImage__AssetImage__RemoteImage__, listeningImageRepeats: bridge.std__optional_bool_, preferSpeechToText: bridge.std__optional_bool_, onChunk: bridge.std__optional_std__function_void_const_VoiceInputChunk_____chunk______, language: bridge.std__optional_std__string_) -> bridge.Result_std__shared_ptr_Promise_VoiceInputResult___ {
|
|
159
|
+
public final func startVoiceInput(silenceThresholdMs: bridge.std__optional_double_, maxDurationMs: bridge.std__optional_double_, listeningText: bridge.std__optional_std__string_, listeningImage: bridge.std__optional_std__variant_GlyphImage__AssetImage__RemoteImage__, listeningImageRepeats: bridge.std__optional_bool_, preferSpeechToText: bridge.std__optional_bool_, onChunk: bridge.std__optional_std__function_void_const_VoiceInputChunk_____chunk______, language: bridge.std__optional_std__string_, startSoundUri: bridge.std__optional_std__string_, endSoundUri: bridge.std__optional_std__string_) -> bridge.Result_std__shared_ptr_Promise_VoiceInputResult___ {
|
|
160
160
|
do {
|
|
161
161
|
let __result = try self.__implementation.startVoiceInput(silenceThresholdMs: { () -> Double? in
|
|
162
162
|
if bridge.has_value_std__optional_double_(silenceThresholdMs) {
|
|
@@ -234,6 +234,20 @@ open class HybridVoiceSpec_cxx {
|
|
|
234
234
|
} else {
|
|
235
235
|
return nil
|
|
236
236
|
}
|
|
237
|
+
}(), startSoundUri: { () -> String? in
|
|
238
|
+
if bridge.has_value_std__optional_std__string_(startSoundUri) {
|
|
239
|
+
let __unwrapped = bridge.get_std__optional_std__string_(startSoundUri)
|
|
240
|
+
return String(__unwrapped)
|
|
241
|
+
} else {
|
|
242
|
+
return nil
|
|
243
|
+
}
|
|
244
|
+
}(), endSoundUri: { () -> String? in
|
|
245
|
+
if bridge.has_value_std__optional_std__string_(endSoundUri) {
|
|
246
|
+
let __unwrapped = bridge.get_std__optional_std__string_(endSoundUri)
|
|
247
|
+
return String(__unwrapped)
|
|
248
|
+
} else {
|
|
249
|
+
return nil
|
|
250
|
+
}
|
|
237
251
|
}())
|
|
238
252
|
let __resultCpp = { () -> bridge.std__shared_ptr_Promise_VoiceInputResult__ in
|
|
239
253
|
let __promise = bridge.create_std__shared_ptr_Promise_VoiceInputResult__()
|
|
@@ -68,7 +68,7 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
|
|
|
68
68
|
// Methods
|
|
69
69
|
virtual bool hasVoiceInputPermission() = 0;
|
|
70
70
|
virtual std::shared_ptr<Promise<bool>> requestVoiceInputPermission() = 0;
|
|
71
|
-
virtual std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language) = 0;
|
|
71
|
+
virtual std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri) = 0;
|
|
72
72
|
virtual void stopVoiceInput() = 0;
|
|
73
73
|
|
|
74
74
|
protected:
|
package/package.json
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { Image } from 'react-native';
|
|
1
2
|
import { NitroModules } from 'react-native-nitro-modules';
|
|
2
3
|
import type { Voice } from '../specs/Voice.nitro';
|
|
3
4
|
import type { VoiceInputOptions, VoiceInputResult } from '../types/Voice';
|
|
@@ -21,11 +22,16 @@ const startVoiceInput: StartVoiceInput = async (options?: VoiceInputOptions) =>
|
|
|
21
22
|
listeningImage,
|
|
22
23
|
preferSpeechToText,
|
|
23
24
|
language,
|
|
25
|
+
startSound,
|
|
26
|
+
endSound,
|
|
24
27
|
} = options ?? {};
|
|
25
28
|
|
|
26
29
|
const listeningImageRepeats =
|
|
27
30
|
listeningImage?.type === 'asset' ? listeningImage.repeats : undefined;
|
|
28
31
|
|
|
32
|
+
const startSoundUri = startSound != null ? Image.resolveAssetSource(startSound).uri : undefined;
|
|
33
|
+
const endSoundUri = endSound != null ? Image.resolveAssetSource(endSound).uri : undefined;
|
|
34
|
+
|
|
29
35
|
return await _native.startVoiceInput(
|
|
30
36
|
silenceThresholdMs,
|
|
31
37
|
maxDurationMs,
|
|
@@ -34,7 +40,9 @@ const startVoiceInput: StartVoiceInput = async (options?: VoiceInputOptions) =>
|
|
|
34
40
|
listeningImageRepeats,
|
|
35
41
|
preferSpeechToText,
|
|
36
42
|
onChunk,
|
|
37
|
-
language
|
|
43
|
+
language,
|
|
44
|
+
startSoundUri,
|
|
45
|
+
endSoundUri
|
|
38
46
|
);
|
|
39
47
|
};
|
|
40
48
|
|
package/src/specs/Voice.nitro.ts
CHANGED
|
@@ -13,7 +13,9 @@ export interface Voice extends HybridObject<{ android: 'kotlin'; ios: 'swift' }>
|
|
|
13
13
|
listeningImageRepeats?: boolean,
|
|
14
14
|
preferSpeechToText?: boolean,
|
|
15
15
|
onChunk?: (chunk: VoiceInputChunk) => void,
|
|
16
|
-
language?: string
|
|
16
|
+
language?: string,
|
|
17
|
+
startSoundUri?: string,
|
|
18
|
+
endSoundUri?: string
|
|
17
19
|
): Promise<VoiceInputResult>;
|
|
18
20
|
stopVoiceInput(): void;
|
|
19
21
|
}
|
package/src/types/Voice.ts
CHANGED
|
@@ -21,4 +21,8 @@ export interface VoiceInputOptions {
|
|
|
21
21
|
preferSpeechToText?: boolean;
|
|
22
22
|
onChunk?: (chunk: VoiceInputChunk) => void;
|
|
23
23
|
language?: string;
|
|
24
|
+
/** Sound played just before recording starts. Pass a Metro asset: `require('./beep_start.wav')`. */
|
|
25
|
+
startSound?: number;
|
|
26
|
+
/** Sound played just after recording stops. Pass a Metro asset: `require('./beep_end.wav')`. */
|
|
27
|
+
endSound?: number;
|
|
24
28
|
}
|
package/src/utils/ErrorUtil.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
const
|
|
1
|
+
const isError = (error: unknown): error is { message: string } => {
|
|
2
2
|
if (error == null) {
|
|
3
3
|
return false;
|
|
4
4
|
}
|
|
@@ -19,7 +19,23 @@ const isTemplateNotFoundError = (error: unknown): error is Error => {
|
|
|
19
19
|
return false;
|
|
20
20
|
}
|
|
21
21
|
|
|
22
|
-
return
|
|
22
|
+
return true;
|
|
23
|
+
};
|
|
24
|
+
|
|
25
|
+
const isTemplateNotFoundError = (error: unknown): error is Error => {
|
|
26
|
+
if (isError(error)) {
|
|
27
|
+
return error.message.startsWith('templateNotFound');
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
return false;
|
|
31
|
+
};
|
|
32
|
+
|
|
33
|
+
const isVoiceInputCanceledError = (error: unknown): error is Error => {
|
|
34
|
+
if (isError(error)) {
|
|
35
|
+
return error.message.startsWith('voiceInputCancelled');
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
return false;
|
|
23
39
|
};
|
|
24
40
|
|
|
25
|
-
export const ErrorUtil = { isTemplateNotFoundError };
|
|
41
|
+
export const ErrorUtil = { isTemplateNotFoundError, isVoiceInputCanceledError };
|