@iternio/react-native-auto-play 0.5.3 → 0.5.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +75 -16
- package/android/src/main/java/com/margelo/nitro/swe/iternio/reactnativeautoplay/AutoPlayError.kt +6 -1
- package/android/src/main/java/com/margelo/nitro/swe/iternio/reactnativeautoplay/HybridVoice.kt +9 -3
- package/android/src/main/java/com/margelo/nitro/swe/iternio/reactnativeautoplay/VoiceInputManager.kt +103 -45
- package/ios/Types.swift +2 -1
- package/ios/extensions/CarPlayTemplateExtensions.swift +7 -0
- package/ios/hybrid/HybridAutoPlay.swift +0 -1
- package/ios/hybrid/HybridVoice.swift +10 -2
- package/ios/templates/Parser.swift +95 -0
- package/ios/templates/VoiceInputTemplate.swift +33 -0
- package/ios/utils/VoiceInputManager.swift +224 -29
- package/lib/hybrid/HybridVoice.js +7 -2
- package/lib/specs/Voice.nitro.d.ts +2 -1
- package/lib/templates/Template.d.ts +1 -1
- package/lib/types/Image.d.ts +16 -0
- package/lib/types/Voice.d.ts +9 -0
- package/lib/utils/ErrorUtil.d.ts +1 -0
- package/lib/utils/ErrorUtil.js +15 -3
- package/nitrogen/generated/android/c++/JHybridVoiceSpec.cpp +21 -3
- package/nitrogen/generated/android/c++/JHybridVoiceSpec.hpp +1 -1
- package/nitrogen/generated/android/kotlin/com/margelo/nitro/swe/iternio/reactnativeautoplay/HybridVoiceSpec.kt +3 -3
- package/nitrogen/generated/ios/c++/HybridVoiceSpecSwift.hpp +15 -2
- package/nitrogen/generated/ios/swift/HybridVoiceSpec.swift +1 -1
- package/nitrogen/generated/ios/swift/HybridVoiceSpec_cxx.swift +44 -1
- package/nitrogen/generated/shared/c++/HybridVoiceSpec.hpp +11 -1
- package/package.json +1 -1
- package/src/hybrid/HybridVoice.ts +16 -1
- package/src/specs/Voice.nitro.ts +6 -1
- package/src/templates/Template.ts +1 -1
- package/src/types/Image.ts +19 -0
- package/src/types/Voice.ts +10 -0
- package/src/utils/ErrorUtil.ts +19 -3
|
@@ -3,6 +3,31 @@ import CarPlay
|
|
|
3
3
|
import NitroModules
|
|
4
4
|
import Speech
|
|
5
5
|
|
|
6
|
+
/// Keeps itself and the player alive until playback finishes.
|
|
7
|
+
/// Needed because AVAudioPlayer.delegate is weak, so without an external strong
|
|
8
|
+
/// reference the delegate (and player) would be released immediately after play().
|
|
9
|
+
private final class AudioPlayerDelegate: NSObject, AVAudioPlayerDelegate, @unchecked Sendable {
|
|
10
|
+
private let onFinish: () -> Void
|
|
11
|
+
private var keepAlive: AudioPlayerDelegate?
|
|
12
|
+
private var player: AVAudioPlayer?
|
|
13
|
+
|
|
14
|
+
init(player: AVAudioPlayer, _ onFinish: @escaping () -> Void) {
|
|
15
|
+
self.onFinish = onFinish
|
|
16
|
+
self.player = player
|
|
17
|
+
super.init()
|
|
18
|
+
keepAlive = self
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
private func finish() {
|
|
22
|
+
player = nil
|
|
23
|
+
keepAlive = nil
|
|
24
|
+
onFinish()
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
func audioPlayerDidFinishPlaying(_: AVAudioPlayer, successfully _: Bool) { finish() }
|
|
28
|
+
func audioPlayerDecodeErrorDidOccur(_: AVAudioPlayer, error _: Error?) { finish() }
|
|
29
|
+
}
|
|
30
|
+
|
|
6
31
|
/// Wraps CheckedContinuation so it can only be resumed once even when
|
|
7
32
|
/// shared between a stop() call and an async recognition task callback.
|
|
8
33
|
private final class ResultBox: @unchecked Sendable {
|
|
@@ -36,6 +61,8 @@ class VoiceInputManager {
|
|
|
36
61
|
private var resultBox: ResultBox?
|
|
37
62
|
private var samples: [Int16] = []
|
|
38
63
|
private var isStopping = false
|
|
64
|
+
private var cancelledByUser = false
|
|
65
|
+
private var isIgnoringSamples = false
|
|
39
66
|
private let stopLock = NSLock()
|
|
40
67
|
|
|
41
68
|
// STT
|
|
@@ -65,11 +92,19 @@ class VoiceInputManager {
|
|
|
65
92
|
silenceThresholdMs: Double,
|
|
66
93
|
maxDurationMs: Double,
|
|
67
94
|
listeningText: String,
|
|
95
|
+
listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?,
|
|
96
|
+
listeningImageRepeats: Bool?,
|
|
68
97
|
preferSpeechToText: Bool,
|
|
69
98
|
onChunk: ((_ chunk: VoiceInputChunk) -> Void)?,
|
|
70
|
-
language: String
|
|
99
|
+
language: String?,
|
|
100
|
+
startSoundUri: String?,
|
|
101
|
+
endSoundUri: String?
|
|
71
102
|
) async throws -> VoiceInputResult {
|
|
72
|
-
|
|
103
|
+
stopLock.withLock {
|
|
104
|
+
cancelledByUser = false
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
let result = try await withCheckedThrowingContinuation { cont in
|
|
73
108
|
let box = ResultBox(cont)
|
|
74
109
|
self.resultBox = box
|
|
75
110
|
self.samples = []
|
|
@@ -81,7 +116,6 @@ class VoiceInputManager {
|
|
|
81
116
|
interfaceController: interfaceController,
|
|
82
117
|
silenceThresholdMs: silenceThresholdMs,
|
|
83
118
|
maxDurationMs: maxDurationMs,
|
|
84
|
-
listeningText: listeningText,
|
|
85
119
|
preferSpeechToText: preferSpeechToText,
|
|
86
120
|
onChunk: onChunk,
|
|
87
121
|
box: box,
|
|
@@ -91,8 +125,68 @@ class VoiceInputManager {
|
|
|
91
125
|
catch {
|
|
92
126
|
self.cleanup(interfaceController: interfaceController)
|
|
93
127
|
box.resume(throwing: error)
|
|
128
|
+
return
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
// Mic is open — present template then play start sound.
|
|
132
|
+
// isIgnoringSamples discards tap buffers during the sound so it isn't recorded.
|
|
133
|
+
// recordingStart is set after the sound so silence/max-duration timers are accurate.
|
|
134
|
+
self.stopLock.withLock { self.isIgnoringSamples = startSoundUri != nil }
|
|
135
|
+
Task {
|
|
136
|
+
if let interfaceController = interfaceController {
|
|
137
|
+
await self.presentVoiceTemplate(
|
|
138
|
+
interfaceController: interfaceController,
|
|
139
|
+
listeningText: listeningText,
|
|
140
|
+
listeningImage: listeningImage,
|
|
141
|
+
listeningImageRepeats: listeningImageRepeats
|
|
142
|
+
)
|
|
143
|
+
}
|
|
144
|
+
if let uri = startSoundUri {
|
|
145
|
+
await self.playSound(uri: uri, setupSession: false)
|
|
146
|
+
}
|
|
147
|
+
self.stopLock.withLock {
|
|
148
|
+
self.isIgnoringSamples = false
|
|
149
|
+
self.recordingStart = Date()
|
|
150
|
+
}
|
|
94
151
|
}
|
|
95
152
|
}
|
|
153
|
+
|
|
154
|
+
if let uri = endSoundUri {
|
|
155
|
+
await playSound(uri: uri)
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
return result
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
private func playSound(uri: String, setupSession: Bool = true) async {
|
|
162
|
+
guard let url = URL(string: uri) else { return }
|
|
163
|
+
do {
|
|
164
|
+
let session = AVAudioSession.sharedInstance()
|
|
165
|
+
if setupSession {
|
|
166
|
+
try session.setCategory(.playback, mode: .default)
|
|
167
|
+
try session.setActive(true)
|
|
168
|
+
}
|
|
169
|
+
// URLSession handles both http:// (Metro dev server) and file:// (release bundle)
|
|
170
|
+
let (data, _) = try await URLSession.shared.data(from: url)
|
|
171
|
+
await withCheckedContinuation { (cont: CheckedContinuation<Void, Never>) in
|
|
172
|
+
DispatchQueue.main.async {
|
|
173
|
+
do {
|
|
174
|
+
let player = try AVAudioPlayer(data: data)
|
|
175
|
+
let delegate = AudioPlayerDelegate(player: player) { cont.resume() }
|
|
176
|
+
player.delegate = delegate
|
|
177
|
+
player.prepareToPlay()
|
|
178
|
+
player.play()
|
|
179
|
+
}
|
|
180
|
+
catch {
|
|
181
|
+
cont.resume()
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
catch {
|
|
187
|
+
print(error)
|
|
188
|
+
// fail silently — a broken sound file must not block voice input
|
|
189
|
+
}
|
|
96
190
|
}
|
|
97
191
|
|
|
98
192
|
func stop(interfaceController: AutoPlayInterfaceController? = nil) {
|
|
@@ -102,6 +196,7 @@ class VoiceInputManager {
|
|
|
102
196
|
return
|
|
103
197
|
}
|
|
104
198
|
isStopping = true
|
|
199
|
+
let wasCancelled = cancelledByUser
|
|
105
200
|
let wasSTTMode = isSTTMode
|
|
106
201
|
let capturedRequest = recognitionRequest
|
|
107
202
|
let box = resultBox
|
|
@@ -117,7 +212,12 @@ class VoiceInputManager {
|
|
|
117
212
|
}
|
|
118
213
|
else {
|
|
119
214
|
cleanup(interfaceController: interfaceController)
|
|
120
|
-
|
|
215
|
+
if wasCancelled {
|
|
216
|
+
box?.resume(throwing: AutoPlayError.voiceInputCancelled)
|
|
217
|
+
}
|
|
218
|
+
else {
|
|
219
|
+
box?.resume(returning: makePCMResult(from: capturedSamples))
|
|
220
|
+
}
|
|
121
221
|
}
|
|
122
222
|
}
|
|
123
223
|
|
|
@@ -127,7 +227,6 @@ class VoiceInputManager {
|
|
|
127
227
|
interfaceController: AutoPlayInterfaceController?,
|
|
128
228
|
silenceThresholdMs: Double,
|
|
129
229
|
maxDurationMs: Double,
|
|
130
|
-
listeningText: String,
|
|
131
230
|
preferSpeechToText: Bool,
|
|
132
231
|
onChunk: ((_ chunk: VoiceInputChunk) -> Void)?,
|
|
133
232
|
box: ResultBox,
|
|
@@ -141,14 +240,12 @@ class VoiceInputManager {
|
|
|
141
240
|
try session.setCategory(.playAndRecord, mode: .measurement, options: [])
|
|
142
241
|
try session.setActive(true)
|
|
143
242
|
|
|
144
|
-
if let interfaceController {
|
|
145
|
-
presentVoiceTemplate(interfaceController: interfaceController, listeningText: listeningText)
|
|
146
|
-
}
|
|
147
|
-
|
|
148
243
|
var activeRecognitionRequest: SFSpeechAudioBufferRecognitionRequest? = nil
|
|
149
244
|
|
|
150
245
|
if preferSpeechToText, SFSpeechRecognizer.authorizationStatus() == .authorized,
|
|
151
|
-
let recognizer = language != nil
|
|
246
|
+
let recognizer = language != nil
|
|
247
|
+
? SFSpeechRecognizer(locale: Locale(identifier: language!))
|
|
248
|
+
: SFSpeechRecognizer(locale: Locale.current),
|
|
152
249
|
recognizer.isAvailable
|
|
153
250
|
{
|
|
154
251
|
let request = SFSpeechAudioBufferRecognitionRequest()
|
|
@@ -163,12 +260,18 @@ class VoiceInputManager {
|
|
|
163
260
|
// STT failed — fall back to whatever PCM was accumulated
|
|
164
261
|
self.stopLock.lock()
|
|
165
262
|
self.isStopping = true
|
|
263
|
+
let wasCancelled = self.cancelledByUser
|
|
166
264
|
let capturedSamples = self.samples
|
|
167
265
|
self.samples = []
|
|
168
266
|
self.stopLock.unlock()
|
|
169
267
|
|
|
170
268
|
self.cleanup(interfaceController: interfaceController)
|
|
171
|
-
|
|
269
|
+
if wasCancelled {
|
|
270
|
+
box.resume(throwing: AutoPlayError.voiceInputCancelled)
|
|
271
|
+
}
|
|
272
|
+
else {
|
|
273
|
+
box.resume(returning: self.makePCMResult(from: capturedSamples))
|
|
274
|
+
}
|
|
172
275
|
return
|
|
173
276
|
}
|
|
174
277
|
|
|
@@ -177,16 +280,22 @@ class VoiceInputManager {
|
|
|
177
280
|
if result.isFinal {
|
|
178
281
|
self.stopLock.lock()
|
|
179
282
|
self.isStopping = true
|
|
283
|
+
let wasCancelled = self.cancelledByUser
|
|
180
284
|
self.samples = []
|
|
181
285
|
self.stopLock.unlock()
|
|
182
286
|
|
|
183
287
|
self.cleanup(interfaceController: interfaceController)
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
288
|
+
if wasCancelled {
|
|
289
|
+
box.resume(throwing: AutoPlayError.voiceInputCancelled)
|
|
290
|
+
}
|
|
291
|
+
else {
|
|
292
|
+
box.resume(
|
|
293
|
+
returning: VoiceInputResult(
|
|
294
|
+
transcription: result.bestTranscription.formattedString,
|
|
295
|
+
audio: nil
|
|
296
|
+
)
|
|
188
297
|
)
|
|
189
|
-
|
|
298
|
+
}
|
|
190
299
|
}
|
|
191
300
|
else {
|
|
192
301
|
onChunk?(VoiceInputChunk(partial: result.bestTranscription.formattedString, audio: nil))
|
|
@@ -202,15 +311,24 @@ class VoiceInputManager {
|
|
|
202
311
|
throw VoiceInputError.converterUnavailable
|
|
203
312
|
}
|
|
204
313
|
|
|
205
|
-
recordingStart =
|
|
314
|
+
recordingStart = nil
|
|
206
315
|
silenceStart = nil
|
|
316
|
+
isIgnoringSamples = false
|
|
207
317
|
|
|
208
318
|
inputNode.installTap(
|
|
209
319
|
onBus: 0,
|
|
210
320
|
bufferSize: VoiceInputManager.tapBufferSize,
|
|
211
321
|
format: nativeFormat
|
|
212
322
|
) { [weak self] buffer, _ in
|
|
213
|
-
guard let self
|
|
323
|
+
guard let self else { return }
|
|
324
|
+
|
|
325
|
+
self.stopLock.lock()
|
|
326
|
+
let stopping = self.isStopping
|
|
327
|
+
let ignoringSamples = self.isIgnoringSamples
|
|
328
|
+
let recordingStartSnapshot = self.recordingStart
|
|
329
|
+
self.stopLock.unlock()
|
|
330
|
+
|
|
331
|
+
guard !stopping, !ignoringSamples else { return }
|
|
214
332
|
|
|
215
333
|
// Feed STT if active
|
|
216
334
|
activeRecognitionRequest?.append(buffer)
|
|
@@ -253,7 +371,7 @@ class VoiceInputManager {
|
|
|
253
371
|
let now = Date()
|
|
254
372
|
|
|
255
373
|
// Max duration — applies in both modes
|
|
256
|
-
if let start =
|
|
374
|
+
if let start = recordingStartSnapshot,
|
|
257
375
|
now.timeIntervalSince(start) * 1000 >= maxDurationMs
|
|
258
376
|
{
|
|
259
377
|
self.triggerAutoStop(interfaceController: interfaceController)
|
|
@@ -262,7 +380,7 @@ class VoiceInputManager {
|
|
|
262
380
|
|
|
263
381
|
// Silence detection — skip during warm-up so the pipeline has time
|
|
264
382
|
// to stabilise before we start measuring amplitude
|
|
265
|
-
if let start =
|
|
383
|
+
if let start = recordingStartSnapshot,
|
|
266
384
|
now.timeIntervalSince(start) * 1000 >= VoiceInputManager.warmupMs
|
|
267
385
|
{
|
|
268
386
|
let peak = newSamples.reduce(0) { max($0, abs(Int($1))) }
|
|
@@ -311,21 +429,98 @@ class VoiceInputManager {
|
|
|
311
429
|
return VoiceInputResult(transcription: nil, audio: buffer)
|
|
312
430
|
}
|
|
313
431
|
|
|
314
|
-
|
|
432
|
+
// CPVoiceControlState enforces a maximum image size of 150x150 points.
|
|
433
|
+
private static let voiceImageMaxSize = CGSize(width: 150, height: 150)
|
|
434
|
+
|
|
435
|
+
// CPVoiceControlState also enforces a 0.3s–5s animation cycle; the 0.3s floor is applied
|
|
436
|
+
// by the system regardless of what we pass, so we only need to clamp our own ceiling.
|
|
437
|
+
private static let maxVoiceImageCycleDuration: TimeInterval = 5.0
|
|
438
|
+
|
|
439
|
+
// Bypasses RCTConvert for asset images: it collapses animated UIImages to a single frame
|
|
440
|
+
// via CGImage during scale adjustment. Parser.decodeImage preserves animation frames by
|
|
441
|
+
// walking every frame in the source via ImageIO — UIImage(data:) never builds a multi-frame
|
|
442
|
+
// .images array itself, for GIF, APNG, or WebP. Tinting is skipped for animated images since
|
|
443
|
+
// frames cannot be tinted individually.
|
|
444
|
+
private func loadVoiceImage(image: Variant_GlyphImage_AssetImage_RemoteImage?, traitCollection: UITraitCollection)
|
|
445
|
+
-> UIImage?
|
|
446
|
+
{
|
|
447
|
+
guard let image else { return nil }
|
|
448
|
+
|
|
449
|
+
if let assetImage = image.assetImage {
|
|
450
|
+
guard let url = URL(string: assetImage.uri),
|
|
451
|
+
let data = try? Data(contentsOf: url),
|
|
452
|
+
let uiImage = Parser.decodeImage(
|
|
453
|
+
data: data,
|
|
454
|
+
scale: CGFloat(assetImage.scale),
|
|
455
|
+
maxDuration: VoiceInputManager.maxVoiceImageCycleDuration
|
|
456
|
+
)
|
|
457
|
+
else { return nil }
|
|
458
|
+
|
|
459
|
+
if uiImage.images != nil {
|
|
460
|
+
return Parser.resizeAnimated(uiImage, max: VoiceInputManager.voiceImageMaxSize)
|
|
461
|
+
}
|
|
462
|
+
|
|
463
|
+
if assetImage.color == nil {
|
|
464
|
+
return Parser.resize(uiImage, max: VoiceInputManager.voiceImageMaxSize)
|
|
465
|
+
}
|
|
466
|
+
|
|
467
|
+
guard let tinted = Parser.parseAssetImage(assetImage: assetImage, traitCollection: traitCollection) else {
|
|
468
|
+
return nil
|
|
469
|
+
}
|
|
470
|
+
return Parser.resize(tinted, max: VoiceInputManager.voiceImageMaxSize)
|
|
471
|
+
}
|
|
472
|
+
|
|
473
|
+
if let glyphImage = image.glyphImage {
|
|
474
|
+
return
|
|
475
|
+
SymbolFont
|
|
476
|
+
.imageFromGlyph(
|
|
477
|
+
glyphImage: glyphImage,
|
|
478
|
+
size: 150, // according to docs on CPVoiceControlState.image
|
|
479
|
+
foregroundColor: glyphImage.color,
|
|
480
|
+
backgroundColor: glyphImage.backgroundColor,
|
|
481
|
+
fontScale: glyphImage.fontScale ?? 1.0,
|
|
482
|
+
traitCollection: traitCollection
|
|
483
|
+
)
|
|
484
|
+
}
|
|
485
|
+
|
|
486
|
+
return nil
|
|
487
|
+
}
|
|
488
|
+
|
|
489
|
+
@MainActor
|
|
490
|
+
private func presentVoiceTemplate(
|
|
491
|
+
interfaceController: AutoPlayInterfaceController,
|
|
492
|
+
listeningText: String,
|
|
493
|
+
listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?,
|
|
494
|
+
listeningImageRepeats: Bool?
|
|
495
|
+
) async {
|
|
496
|
+
let traitCollection = SceneStore.getRootTraitCollection() ?? UITraitCollection.current
|
|
497
|
+
let image = loadVoiceImage(
|
|
498
|
+
image: listeningImage,
|
|
499
|
+
traitCollection: traitCollection
|
|
500
|
+
)
|
|
501
|
+
|
|
502
|
+
let repeats = listeningImageRepeats ?? (image?.images != nil)
|
|
315
503
|
let listeningState = CPVoiceControlState(
|
|
316
504
|
identifier: "listening",
|
|
317
505
|
titleVariants: [listeningText],
|
|
318
|
-
image:
|
|
319
|
-
repeats:
|
|
506
|
+
image: image,
|
|
507
|
+
repeats: repeats
|
|
320
508
|
)
|
|
321
|
-
let template = CPVoiceControlTemplate(voiceControlStates: [listeningState])
|
|
322
|
-
initTemplate(template: template, id: "voice-input")
|
|
323
|
-
voiceControlTemplate = template
|
|
324
509
|
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
510
|
+
let voiceTemplate = VoiceInputTemplate(
|
|
511
|
+
voiceControlStates: [listeningState],
|
|
512
|
+
id: "voice-input"
|
|
513
|
+
) { [weak self] in
|
|
514
|
+
guard let self else { return }
|
|
515
|
+
self.stopLock.withLock {
|
|
516
|
+
if !self.isStopping { self.cancelledByUser = true }
|
|
517
|
+
}
|
|
518
|
+
self.stop()
|
|
328
519
|
}
|
|
520
|
+
|
|
521
|
+
voiceControlTemplate = voiceTemplate.template
|
|
522
|
+
try? await interfaceController.presentTemplate(voiceTemplate.template, animated: true)
|
|
523
|
+
voiceTemplate.template.activateVoiceControlState(withIdentifier: "listening")
|
|
329
524
|
}
|
|
330
525
|
|
|
331
526
|
private func dismissVoiceTemplate(interfaceController: AutoPlayInterfaceController) {
|
|
@@ -1,8 +1,13 @@
|
|
|
1
|
+
import { Image } from 'react-native';
|
|
1
2
|
import { NitroModules } from 'react-native-nitro-modules';
|
|
3
|
+
import { NitroImageUtil } from '../utils/NitroImage';
|
|
2
4
|
const _native = NitroModules.createHybridObject('Voice');
|
|
3
5
|
const startVoiceInput = async (options) => {
|
|
4
|
-
const { onChunk, silenceThresholdMs, maxDurationMs, listeningText, preferSpeechToText, language, } = options ?? {};
|
|
5
|
-
|
|
6
|
+
const { onChunk, silenceThresholdMs, maxDurationMs, listeningText, listeningImage, preferSpeechToText, language, startSound, endSound, } = options ?? {};
|
|
7
|
+
const listeningImageRepeats = listeningImage?.type === 'asset' ? listeningImage.repeats : undefined;
|
|
8
|
+
const startSoundUri = startSound != null ? Image.resolveAssetSource(startSound).uri : undefined;
|
|
9
|
+
const endSoundUri = endSound != null ? Image.resolveAssetSource(endSound).uri : undefined;
|
|
10
|
+
return await _native.startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, NitroImageUtil.convert(listeningImage), listeningImageRepeats, preferSpeechToText, onChunk, language, startSoundUri, endSoundUri);
|
|
6
11
|
};
|
|
7
12
|
export const HybridVoice = {
|
|
8
13
|
/**
|
|
@@ -1,11 +1,12 @@
|
|
|
1
1
|
import type { HybridObject } from 'react-native-nitro-modules';
|
|
2
2
|
import type { VoiceInputChunk, VoiceInputResult } from '../types/Voice';
|
|
3
|
+
import type { NitroImage } from '../utils/NitroImage';
|
|
3
4
|
export interface Voice extends HybridObject<{
|
|
4
5
|
android: 'kotlin';
|
|
5
6
|
ios: 'swift';
|
|
6
7
|
}> {
|
|
7
8
|
hasVoiceInputPermission(): boolean;
|
|
8
9
|
requestVoiceInputPermission(): Promise<boolean>;
|
|
9
|
-
startVoiceInput(silenceThresholdMs?: number, maxDurationMs?: number, listeningText?: string, preferSpeechToText?: boolean, onChunk?: (chunk: VoiceInputChunk) => void, language?: string): Promise<VoiceInputResult>;
|
|
10
|
+
startVoiceInput(silenceThresholdMs?: number, maxDurationMs?: number, listeningText?: string, listeningImage?: NitroImage, listeningImageRepeats?: boolean, preferSpeechToText?: boolean, onChunk?: (chunk: VoiceInputChunk) => void, language?: string, startSoundUri?: string, endSoundUri?: string): Promise<VoiceInputResult>;
|
|
10
11
|
stopVoiceInput(): void;
|
|
11
12
|
}
|
|
@@ -4,7 +4,7 @@ import type { NitroMapButton } from '../utils/NitroMapButton';
|
|
|
4
4
|
export type ActionButton<T> = ActionButtonAndroid<T> | ActionButtonIos<T>;
|
|
5
5
|
export type HeaderActionsIos<T> = {
|
|
6
6
|
/**
|
|
7
|
-
* @
|
|
7
|
+
* @namespace iOS - the back button can not be hidden or disabled, if you don't define it iOS will provide a default back button popping the template
|
|
8
8
|
*/
|
|
9
9
|
backButton?: BackButton<T>;
|
|
10
10
|
leadingNavigationBarButtons?: [ActionButtonIos<T>, ActionButtonIos<T>] | [ActionButtonIos<T>];
|
package/lib/types/Image.d.ts
CHANGED
|
@@ -66,4 +66,20 @@ export type AutoImage = AutoGlyph | {
|
|
|
66
66
|
timeoutMs?: number;
|
|
67
67
|
type: 'remote';
|
|
68
68
|
};
|
|
69
|
+
/**
|
|
70
|
+
* Image for the CarPlay voice control overlay. Animated GIF, APNG, and WebP all animate.
|
|
71
|
+
* `color` is ignored for animated assets — frames cannot be tinted individually. Static
|
|
72
|
+
* assets and glyphs are resized/capped to fit within 150×150pt (`fontScale` controls a
|
|
73
|
+
* glyph's relative size). CarPlay enforces a 0.3s–5s animation cycle duration: shorter
|
|
74
|
+
* source animations are stretched to 0.3s by the system, longer ones are capped to 5s.
|
|
75
|
+
* @namespace ios
|
|
76
|
+
*/
|
|
77
|
+
export type VoiceInputImage = AutoGlyph | {
|
|
78
|
+
type: 'asset';
|
|
79
|
+
image: ImageSourcePropType;
|
|
80
|
+
/** Tints the image. Ignored for animated assets — frames cannot be tinted individually. */
|
|
81
|
+
color?: ThemedColor | string;
|
|
82
|
+
/** Whether the animation loops. Defaults to `true` for animated images, `false` for static. */
|
|
83
|
+
repeats?: boolean;
|
|
84
|
+
};
|
|
69
85
|
export {};
|
package/lib/types/Voice.d.ts
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import type { VoiceInputImage } from './Image';
|
|
1
2
|
export interface VoiceInputChunk {
|
|
2
3
|
partial?: string;
|
|
3
4
|
audio?: ArrayBuffer;
|
|
@@ -10,7 +11,15 @@ export interface VoiceInputOptions {
|
|
|
10
11
|
silenceThresholdMs?: number;
|
|
11
12
|
maxDurationMs?: number;
|
|
12
13
|
listeningText?: string;
|
|
14
|
+
/** Image displayed in the CarPlay voice control overlay. See {@link VoiceInputImage}.
|
|
15
|
+
* @namespace ios
|
|
16
|
+
*/
|
|
17
|
+
listeningImage?: VoiceInputImage;
|
|
13
18
|
preferSpeechToText?: boolean;
|
|
14
19
|
onChunk?: (chunk: VoiceInputChunk) => void;
|
|
15
20
|
language?: string;
|
|
21
|
+
/** Sound played just before recording starts. Pass a Metro asset: `require('./beep_start.wav')`. */
|
|
22
|
+
startSound?: number;
|
|
23
|
+
/** Sound played just after recording stops. Pass a Metro asset: `require('./beep_end.wav')`. */
|
|
24
|
+
endSound?: number;
|
|
16
25
|
}
|
package/lib/utils/ErrorUtil.d.ts
CHANGED
package/lib/utils/ErrorUtil.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
const
|
|
1
|
+
const isError = (error) => {
|
|
2
2
|
if (error == null) {
|
|
3
3
|
return false;
|
|
4
4
|
}
|
|
@@ -14,6 +14,18 @@ const isTemplateNotFoundError = (error) => {
|
|
|
14
14
|
if (typeof error.message !== 'string') {
|
|
15
15
|
return false;
|
|
16
16
|
}
|
|
17
|
-
return
|
|
17
|
+
return true;
|
|
18
|
+
};
|
|
19
|
+
const isTemplateNotFoundError = (error) => {
|
|
20
|
+
if (isError(error)) {
|
|
21
|
+
return error.message.startsWith('templateNotFound');
|
|
22
|
+
}
|
|
23
|
+
return false;
|
|
24
|
+
};
|
|
25
|
+
const isVoiceInputCanceledError = (error) => {
|
|
26
|
+
if (isError(error)) {
|
|
27
|
+
return error.message.startsWith('voiceInputCancelled');
|
|
28
|
+
}
|
|
29
|
+
return false;
|
|
18
30
|
};
|
|
19
|
-
export const ErrorUtil = { isTemplateNotFoundError };
|
|
31
|
+
export const ErrorUtil = { isTemplateNotFoundError, isVoiceInputCanceledError };
|
|
@@ -9,6 +9,14 @@
|
|
|
9
9
|
|
|
10
10
|
// Forward declaration of `VoiceInputResult` to properly resolve imports.
|
|
11
11
|
namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct VoiceInputResult; }
|
|
12
|
+
// Forward declaration of `GlyphImage` to properly resolve imports.
|
|
13
|
+
namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct GlyphImage; }
|
|
14
|
+
// Forward declaration of `AssetImage` to properly resolve imports.
|
|
15
|
+
namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct AssetImage; }
|
|
16
|
+
// Forward declaration of `RemoteImage` to properly resolve imports.
|
|
17
|
+
namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct RemoteImage; }
|
|
18
|
+
// Forward declaration of `NitroColor` to properly resolve imports.
|
|
19
|
+
namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct NitroColor; }
|
|
12
20
|
// Forward declaration of `VoiceInputChunk` to properly resolve imports.
|
|
13
21
|
namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct VoiceInputChunk; }
|
|
14
22
|
|
|
@@ -20,6 +28,16 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct VoiceInputC
|
|
|
20
28
|
#include <optional>
|
|
21
29
|
#include <NitroModules/ArrayBuffer.hpp>
|
|
22
30
|
#include <NitroModules/JArrayBuffer.hpp>
|
|
31
|
+
#include "GlyphImage.hpp"
|
|
32
|
+
#include "AssetImage.hpp"
|
|
33
|
+
#include "RemoteImage.hpp"
|
|
34
|
+
#include <variant>
|
|
35
|
+
#include "JVariant_GlyphImage_AssetImage_RemoteImage.hpp"
|
|
36
|
+
#include "JGlyphImage.hpp"
|
|
37
|
+
#include "NitroColor.hpp"
|
|
38
|
+
#include "JNitroColor.hpp"
|
|
39
|
+
#include "JAssetImage.hpp"
|
|
40
|
+
#include "JRemoteImage.hpp"
|
|
23
41
|
#include "VoiceInputChunk.hpp"
|
|
24
42
|
#include <functional>
|
|
25
43
|
#include "JFunc_void_VoiceInputChunk.hpp"
|
|
@@ -80,9 +98,9 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
|
|
|
80
98
|
return __promise;
|
|
81
99
|
}();
|
|
82
100
|
}
|
|
83
|
-
std::shared_ptr<Promise<VoiceInputResult>> JHybridVoiceSpec::startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language) {
|
|
84
|
-
static const auto method = _javaPart->javaClassStatic()->getMethod<jni::local_ref<JPromise::javaobject>(jni::alias_ref<jni::JDouble> /* silenceThresholdMs */, jni::alias_ref<jni::JDouble> /* maxDurationMs */, jni::alias_ref<jni::JString> /* listeningText */, jni::alias_ref<jni::JBoolean> /* preferSpeechToText */, jni::alias_ref<JFunc_void_VoiceInputChunk::javaobject> /* onChunk */, jni::alias_ref<jni::JString> /* language */)>("startVoiceInput_cxx");
|
|
85
|
-
auto __result = method(_javaPart, silenceThresholdMs.has_value() ? jni::JDouble::valueOf(silenceThresholdMs.value()) : nullptr, maxDurationMs.has_value() ? jni::JDouble::valueOf(maxDurationMs.value()) : nullptr, listeningText.has_value() ? jni::make_jstring(listeningText.value()) : nullptr, preferSpeechToText.has_value() ? jni::JBoolean::valueOf(preferSpeechToText.value()) : nullptr, onChunk.has_value() ? JFunc_void_VoiceInputChunk_cxx::fromCpp(onChunk.value()) : nullptr, language.has_value() ? jni::make_jstring(language.value()) : nullptr);
|
|
101
|
+
std::shared_ptr<Promise<VoiceInputResult>> JHybridVoiceSpec::startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri) {
|
|
102
|
+
static const auto method = _javaPart->javaClassStatic()->getMethod<jni::local_ref<JPromise::javaobject>(jni::alias_ref<jni::JDouble> /* silenceThresholdMs */, jni::alias_ref<jni::JDouble> /* maxDurationMs */, jni::alias_ref<jni::JString> /* listeningText */, jni::alias_ref<JVariant_GlyphImage_AssetImage_RemoteImage> /* listeningImage */, jni::alias_ref<jni::JBoolean> /* listeningImageRepeats */, jni::alias_ref<jni::JBoolean> /* preferSpeechToText */, jni::alias_ref<JFunc_void_VoiceInputChunk::javaobject> /* onChunk */, jni::alias_ref<jni::JString> /* language */, jni::alias_ref<jni::JString> /* startSoundUri */, jni::alias_ref<jni::JString> /* endSoundUri */)>("startVoiceInput_cxx");
|
|
103
|
+
auto __result = method(_javaPart, silenceThresholdMs.has_value() ? jni::JDouble::valueOf(silenceThresholdMs.value()) : nullptr, maxDurationMs.has_value() ? jni::JDouble::valueOf(maxDurationMs.value()) : nullptr, listeningText.has_value() ? jni::make_jstring(listeningText.value()) : nullptr, listeningImage.has_value() ? JVariant_GlyphImage_AssetImage_RemoteImage::fromCpp(listeningImage.value()) : nullptr, listeningImageRepeats.has_value() ? jni::JBoolean::valueOf(listeningImageRepeats.value()) : nullptr, preferSpeechToText.has_value() ? jni::JBoolean::valueOf(preferSpeechToText.value()) : nullptr, onChunk.has_value() ? JFunc_void_VoiceInputChunk_cxx::fromCpp(onChunk.value()) : nullptr, language.has_value() ? jni::make_jstring(language.value()) : nullptr, startSoundUri.has_value() ? jni::make_jstring(startSoundUri.value()) : nullptr, endSoundUri.has_value() ? jni::make_jstring(endSoundUri.value()) : nullptr);
|
|
86
104
|
return [&]() {
|
|
87
105
|
auto __promise = Promise<VoiceInputResult>::create();
|
|
88
106
|
__result->cthis()->addOnResolvedListener([=](const jni::alias_ref<jni::JObject>& __boxedResult) {
|
|
@@ -56,7 +56,7 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
|
|
|
56
56
|
// Methods
|
|
57
57
|
bool hasVoiceInputPermission() override;
|
|
58
58
|
std::shared_ptr<Promise<bool>> requestVoiceInputPermission() override;
|
|
59
|
-
std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language) override;
|
|
59
|
+
std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri) override;
|
|
60
60
|
void stopVoiceInput() override;
|
|
61
61
|
|
|
62
62
|
private:
|
|
@@ -37,12 +37,12 @@ abstract class HybridVoiceSpec: HybridObject() {
|
|
|
37
37
|
@Keep
|
|
38
38
|
abstract fun requestVoiceInputPermission(): Promise<Boolean>
|
|
39
39
|
|
|
40
|
-
abstract fun startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, preferSpeechToText: Boolean?, onChunk: ((chunk: VoiceInputChunk) -> Unit)?, language: String?): Promise<VoiceInputResult>
|
|
40
|
+
abstract fun startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Boolean?, preferSpeechToText: Boolean?, onChunk: ((chunk: VoiceInputChunk) -> Unit)?, language: String?, startSoundUri: String?, endSoundUri: String?): Promise<VoiceInputResult>
|
|
41
41
|
|
|
42
42
|
@DoNotStrip
|
|
43
43
|
@Keep
|
|
44
|
-
private fun startVoiceInput_cxx(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, preferSpeechToText: Boolean?, onChunk: Func_void_VoiceInputChunk?, language: String?): Promise<VoiceInputResult> {
|
|
45
|
-
val __result = startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, preferSpeechToText, onChunk?.let { it }, language)
|
|
44
|
+
private fun startVoiceInput_cxx(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Boolean?, preferSpeechToText: Boolean?, onChunk: Func_void_VoiceInputChunk?, language: String?, startSoundUri: String?, endSoundUri: String?): Promise<VoiceInputResult> {
|
|
45
|
+
val __result = startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, listeningImage, listeningImageRepeats, preferSpeechToText, onChunk?.let { it }, language, startSoundUri, endSoundUri)
|
|
46
46
|
return __result
|
|
47
47
|
}
|
|
48
48
|
|
|
@@ -16,6 +16,14 @@ namespace ReactNativeAutoPlay { class HybridVoiceSpec_cxx; }
|
|
|
16
16
|
namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct VoiceInputResult; }
|
|
17
17
|
// Forward declaration of `ArrayBufferHolder` to properly resolve imports.
|
|
18
18
|
namespace NitroModules { class ArrayBufferHolder; }
|
|
19
|
+
// Forward declaration of `GlyphImage` to properly resolve imports.
|
|
20
|
+
namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct GlyphImage; }
|
|
21
|
+
// Forward declaration of `AssetImage` to properly resolve imports.
|
|
22
|
+
namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct AssetImage; }
|
|
23
|
+
// Forward declaration of `RemoteImage` to properly resolve imports.
|
|
24
|
+
namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct RemoteImage; }
|
|
25
|
+
// Forward declaration of `NitroColor` to properly resolve imports.
|
|
26
|
+
namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct NitroColor; }
|
|
19
27
|
// Forward declaration of `VoiceInputChunk` to properly resolve imports.
|
|
20
28
|
namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct VoiceInputChunk; }
|
|
21
29
|
|
|
@@ -25,6 +33,11 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay { struct VoiceInputC
|
|
|
25
33
|
#include <optional>
|
|
26
34
|
#include <NitroModules/ArrayBuffer.hpp>
|
|
27
35
|
#include <NitroModules/ArrayBufferHolder.hpp>
|
|
36
|
+
#include "GlyphImage.hpp"
|
|
37
|
+
#include "AssetImage.hpp"
|
|
38
|
+
#include "RemoteImage.hpp"
|
|
39
|
+
#include <variant>
|
|
40
|
+
#include "NitroColor.hpp"
|
|
28
41
|
#include "VoiceInputChunk.hpp"
|
|
29
42
|
#include <functional>
|
|
30
43
|
|
|
@@ -94,8 +107,8 @@ namespace margelo::nitro::swe::iternio::reactnativeautoplay {
|
|
|
94
107
|
auto __value = std::move(__result.value());
|
|
95
108
|
return __value;
|
|
96
109
|
}
|
|
97
|
-
inline std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language) override {
|
|
98
|
-
auto __result = _swiftPart.startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, preferSpeechToText, onChunk, language);
|
|
110
|
+
inline std::shared_ptr<Promise<VoiceInputResult>> startVoiceInput(std::optional<double> silenceThresholdMs, std::optional<double> maxDurationMs, const std::optional<std::string>& listeningText, const std::optional<std::variant<GlyphImage, AssetImage, RemoteImage>>& listeningImage, std::optional<bool> listeningImageRepeats, std::optional<bool> preferSpeechToText, const std::optional<std::function<void(const VoiceInputChunk& /* chunk */)>>& onChunk, const std::optional<std::string>& language, const std::optional<std::string>& startSoundUri, const std::optional<std::string>& endSoundUri) override {
|
|
111
|
+
auto __result = _swiftPart.startVoiceInput(silenceThresholdMs, maxDurationMs, listeningText, listeningImage, listeningImageRepeats, preferSpeechToText, onChunk, language, startSoundUri, endSoundUri);
|
|
99
112
|
if (__result.hasError()) [[unlikely]] {
|
|
100
113
|
std::rethrow_exception(__result.error());
|
|
101
114
|
}
|
|
@@ -15,7 +15,7 @@ public protocol HybridVoiceSpec_protocol: HybridObject {
|
|
|
15
15
|
// Methods
|
|
16
16
|
func hasVoiceInputPermission() throws -> Bool
|
|
17
17
|
func requestVoiceInputPermission() throws -> Promise<Bool>
|
|
18
|
-
func startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, preferSpeechToText: Bool?, onChunk: ((_ chunk: VoiceInputChunk) -> Void)?, language: String?) throws -> Promise<VoiceInputResult>
|
|
18
|
+
func startVoiceInput(silenceThresholdMs: Double?, maxDurationMs: Double?, listeningText: String?, listeningImage: Variant_GlyphImage_AssetImage_RemoteImage?, listeningImageRepeats: Bool?, preferSpeechToText: Bool?, onChunk: ((_ chunk: VoiceInputChunk) -> Void)?, language: String?, startSoundUri: String?, endSoundUri: String?) throws -> Promise<VoiceInputResult>
|
|
19
19
|
func stopVoiceInput() throws -> Void
|
|
20
20
|
}
|
|
21
21
|
|