@capgo/capacitor-speech-recognition 8.0.10 → 8.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +318 -21
- package/android/src/main/java/app/capgo/speechrecognition/Constants.java +2 -0
- package/android/src/main/java/app/capgo/speechrecognition/SpeechRecognitionPlugin.java +974 -164
- package/dist/docs.json +392 -21
- package/dist/esm/definitions.d.ts +178 -3
- package/dist/esm/definitions.js.map +1 -1
- package/dist/esm/web.d.ts +5 -1
- package/dist/esm/web.js +12 -0
- package/dist/esm/web.js.map +1 -1
- package/dist/plugin.cjs.js +12 -0
- package/dist/plugin.cjs.js.map +1 -1
- package/dist/plugin.js +12 -0
- package/dist/plugin.js.map +1 -1
- package/ios/Sources/SpeechRecognitionPlugin/SpeechAnalyzerRecognitionSession.swift +437 -0
- package/ios/Sources/SpeechRecognitionPlugin/SpeechRecognitionPlugin.swift +622 -70
- package/package.json +2 -1
|
@@ -0,0 +1,437 @@
|
|
|
1
|
+
import Foundation
|
|
2
|
+
@preconcurrency import AVFoundation
|
|
3
|
+
import Speech
|
|
4
|
+
|
|
5
|
+
#if compiler(>=6.2)
|
|
6
|
+
|
|
7
|
+
@available(iOS 26.0, *)
|
|
8
|
+
enum SpeechAnalyzerRecognitionSupport {
|
|
9
|
+
static func supports(locale: Locale) async -> Bool {
|
|
10
|
+
guard SpeechTranscriber.isAvailable else {
|
|
11
|
+
return false
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
let target = normalizedIdentifier(for: locale)
|
|
15
|
+
let supportedLocales = await SpeechTranscriber.supportedLocales
|
|
16
|
+
return supportedLocales.contains { normalizedIdentifier(for: $0) == target }
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
static func supportedLanguageIdentifiers() async -> [String] {
|
|
20
|
+
let modernLocales = await SpeechTranscriber.supportedLocales
|
|
21
|
+
return modernLocales
|
|
22
|
+
.map(normalizedIdentifier(for:))
|
|
23
|
+
.sorted()
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
static func normalizedIdentifier(for locale: Locale) -> String {
|
|
27
|
+
locale.identifier(.bcp47)
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
@available(iOS 26.0, *)
|
|
32
|
+
enum SpeechAnalyzerRecognitionError: LocalizedError {
|
|
33
|
+
case unavailable
|
|
34
|
+
case unsupportedLocale(String)
|
|
35
|
+
case setupFailed(String)
|
|
36
|
+
|
|
37
|
+
var errorDescription: String? {
|
|
38
|
+
switch self {
|
|
39
|
+
case .unavailable:
|
|
40
|
+
return "Speech transcriber is not available on this device."
|
|
41
|
+
case .unsupportedLocale(let language):
|
|
42
|
+
return "Unsupported locale: \(language)"
|
|
43
|
+
case .setupFailed(let message):
|
|
44
|
+
return message
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
@available(iOS 26.0, *)
|
|
50
|
+
@MainActor
|
|
51
|
+
final class SpeechAnalyzerRecognitionSession {
|
|
52
|
+
typealias ResultHandler = @MainActor ([String], Bool) -> Void
|
|
53
|
+
typealias VoidHandler = @MainActor () -> Void
|
|
54
|
+
typealias ErrorHandler = @MainActor (Error) -> Void
|
|
55
|
+
|
|
56
|
+
private static let microphoneTapBufferSize: AVAudioFrameCount = 2048
|
|
57
|
+
|
|
58
|
+
private let locale: Locale
|
|
59
|
+
private let maxResults: Int
|
|
60
|
+
private let includePartialResults: Bool
|
|
61
|
+
private let processingActor = SpeechAnalyzerAudioProcessingActor()
|
|
62
|
+
private let modelManager = SpeechAnalyzerModelManager()
|
|
63
|
+
|
|
64
|
+
private lazy var audioEngine = AVAudioEngine()
|
|
65
|
+
private var transcriber: SpeechTranscriber?
|
|
66
|
+
private var analyzer: SpeechAnalyzer?
|
|
67
|
+
private var analyzerInputContinuation: AsyncStream<AnalyzerInput>.Continuation?
|
|
68
|
+
private var analyzerFormat: AVAudioFormat?
|
|
69
|
+
private var resultTask: Task<Void, Never>?
|
|
70
|
+
private var hasInstalledTap = false
|
|
71
|
+
private var isTearingDown = false
|
|
72
|
+
private var isAudioSessionActive = false
|
|
73
|
+
|
|
74
|
+
var onListeningStarted: VoidHandler?
|
|
75
|
+
var onListeningStopped: VoidHandler?
|
|
76
|
+
var onResult: ResultHandler?
|
|
77
|
+
var onError: ErrorHandler?
|
|
78
|
+
|
|
79
|
+
var isRunning: Bool {
|
|
80
|
+
audioEngine.isRunning || resultTask != nil || isTearingDown
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
init(locale: Locale, maxResults: Int, includePartialResults: Bool) {
|
|
84
|
+
self.locale = locale
|
|
85
|
+
self.maxResults = maxResults
|
|
86
|
+
self.includePartialResults = includePartialResults
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
func start() async throws {
|
|
90
|
+
guard SpeechTranscriber.isAvailable else {
|
|
91
|
+
throw SpeechAnalyzerRecognitionError.unavailable
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
guard await SpeechAnalyzerRecognitionSupport.supports(locale: locale) else {
|
|
95
|
+
throw SpeechAnalyzerRecognitionError.unsupportedLocale(
|
|
96
|
+
SpeechAnalyzerRecognitionSupport.normalizedIdentifier(for: locale)
|
|
97
|
+
)
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
let reportingOptions: Set<SpeechTranscriber.ReportingOption> = includePartialResults ? [.volatileResults] : []
|
|
101
|
+
let transcriber = SpeechTranscriber(
|
|
102
|
+
locale: locale,
|
|
103
|
+
transcriptionOptions: [],
|
|
104
|
+
reportingOptions: reportingOptions,
|
|
105
|
+
attributeOptions: []
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
self.transcriber = transcriber
|
|
109
|
+
try await modelManager.ensureModel(for: transcriber, locale: locale)
|
|
110
|
+
|
|
111
|
+
let modules: [any SpeechModule] = [transcriber]
|
|
112
|
+
guard let bestFormat = await SpeechAnalyzer.bestAvailableAudioFormat(compatibleWith: modules) else {
|
|
113
|
+
throw SpeechAnalyzerRecognitionError.setupFailed("Unable to determine a compatible analyzer format.")
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
analyzerFormat = bestFormat
|
|
117
|
+
|
|
118
|
+
let analyzer = SpeechAnalyzer(modules: modules)
|
|
119
|
+
self.analyzer = analyzer
|
|
120
|
+
|
|
121
|
+
let (inputSequence, inputContinuation) = AsyncStream<AnalyzerInput>.makeStream()
|
|
122
|
+
analyzerInputContinuation = inputContinuation
|
|
123
|
+
|
|
124
|
+
try await analyzer.start(inputSequence: inputSequence)
|
|
125
|
+
startResultTask(for: transcriber)
|
|
126
|
+
try configureAudioSession()
|
|
127
|
+
try startAudioStreaming()
|
|
128
|
+
onListeningStarted?()
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
func stop() async {
|
|
132
|
+
await teardown(notifyStopped: true)
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
private func startResultTask(for transcriber: SpeechTranscriber) {
|
|
136
|
+
resultTask = Task { [weak self] in
|
|
137
|
+
guard let self else { return }
|
|
138
|
+
|
|
139
|
+
do {
|
|
140
|
+
for try await result in transcriber.results {
|
|
141
|
+
let transcript = String(result.text.characters).trimmingCharacters(in: .whitespacesAndNewlines)
|
|
142
|
+
guard !transcript.isEmpty else {
|
|
143
|
+
continue
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
onResult?(Array([transcript].prefix(maxResults)), result.isFinal)
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
await teardown(notifyStopped: true)
|
|
150
|
+
} catch is CancellationError {
|
|
151
|
+
await teardown(notifyStopped: false)
|
|
152
|
+
} catch {
|
|
153
|
+
await teardown(notifyStopped: false)
|
|
154
|
+
onError?(error)
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
private func configureAudioSession() throws {
|
|
160
|
+
let session = AVAudioSession.sharedInstance()
|
|
161
|
+
try session.setCategory(.playAndRecord, mode: .spokenAudio, options: [.duckOthers, .allowBluetoothHFP])
|
|
162
|
+
try session.setActive(true, options: .notifyOthersOnDeactivation)
|
|
163
|
+
isAudioSessionActive = true
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
private func startAudioStreaming() throws {
|
|
167
|
+
let inputNode = audioEngine.inputNode
|
|
168
|
+
let inputFormat = inputNode.outputFormat(forBus: 0)
|
|
169
|
+
|
|
170
|
+
inputNode.removeTap(onBus: 0)
|
|
171
|
+
inputNode.installTap(
|
|
172
|
+
onBus: 0,
|
|
173
|
+
bufferSize: Self.microphoneTapBufferSize,
|
|
174
|
+
format: inputFormat
|
|
175
|
+
) { [weak self] buffer, _ in
|
|
176
|
+
guard let self, let bufferCopy = buffer.copy() as? AVAudioPCMBuffer else {
|
|
177
|
+
return
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
let sendableBuffer = SpeechAnalyzerSendablePCMBuffer(buffer: bufferCopy)
|
|
181
|
+
Task {
|
|
182
|
+
do {
|
|
183
|
+
try await self.processAudioBuffer(sendableBuffer)
|
|
184
|
+
} catch {
|
|
185
|
+
self.onError?(error)
|
|
186
|
+
await self.teardown(notifyStopped: false)
|
|
187
|
+
}
|
|
188
|
+
}
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
hasInstalledTap = true
|
|
192
|
+
audioEngine.prepare()
|
|
193
|
+
try audioEngine.start()
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
private func processAudioBuffer(_ buffer: SpeechAnalyzerSendablePCMBuffer) async throws {
|
|
197
|
+
guard let analyzerInputContinuation, let analyzerFormat else {
|
|
198
|
+
throw SpeechAnalyzerRecognitionError.setupFailed("Analyzer input is not ready.")
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
let analyzerInput = try await processingActor.makeAnalyzerInput(
|
|
202
|
+
from: buffer,
|
|
203
|
+
analyzerFormat: analyzerFormat
|
|
204
|
+
)
|
|
205
|
+
analyzerInputContinuation.yield(analyzerInput)
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
private func teardown(notifyStopped: Bool) async {
|
|
209
|
+
if isTearingDown {
|
|
210
|
+
return
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
isTearingDown = true
|
|
214
|
+
|
|
215
|
+
let currentTask = resultTask
|
|
216
|
+
resultTask = nil
|
|
217
|
+
currentTask?.cancel()
|
|
218
|
+
|
|
219
|
+
if audioEngine.isRunning {
|
|
220
|
+
audioEngine.stop()
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
if hasInstalledTap {
|
|
224
|
+
audioEngine.inputNode.removeTap(onBus: 0)
|
|
225
|
+
hasInstalledTap = false
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
analyzerInputContinuation?.finish()
|
|
229
|
+
analyzerInputContinuation = nil
|
|
230
|
+
|
|
231
|
+
do {
|
|
232
|
+
try await analyzer?.finalizeAndFinishThroughEndOfInput()
|
|
233
|
+
} catch {
|
|
234
|
+
// Best-effort shutdown. The plugin still needs native cleanup to succeed.
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
analyzer = nil
|
|
238
|
+
analyzerFormat = nil
|
|
239
|
+
transcriber = nil
|
|
240
|
+
|
|
241
|
+
await modelManager.releaseLocales()
|
|
242
|
+
deactivateAudioSessionIfNeeded()
|
|
243
|
+
|
|
244
|
+
isTearingDown = false
|
|
245
|
+
|
|
246
|
+
if notifyStopped {
|
|
247
|
+
onListeningStopped?()
|
|
248
|
+
}
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
private func deactivateAudioSessionIfNeeded() {
|
|
252
|
+
guard isAudioSessionActive else {
|
|
253
|
+
return
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
do {
|
|
257
|
+
try AVAudioSession.sharedInstance().setActive(false, options: [.notifyOthersOnDeactivation])
|
|
258
|
+
} catch {
|
|
259
|
+
// Ignore deactivation failures during cleanup.
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
isAudioSessionActive = false
|
|
263
|
+
}
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
@available(iOS 26.0, *)
|
|
267
|
+
private actor SpeechAnalyzerModelManager {
|
|
268
|
+
private var reservedLocales = Set<String>()
|
|
269
|
+
|
|
270
|
+
func ensureModel(for transcriber: SpeechTranscriber, locale: Locale) async throws {
|
|
271
|
+
guard await SpeechAnalyzerRecognitionSupport.supports(locale: locale) else {
|
|
272
|
+
throw SpeechAnalyzerRecognitionError.unsupportedLocale(
|
|
273
|
+
SpeechAnalyzerRecognitionSupport.normalizedIdentifier(for: locale)
|
|
274
|
+
)
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
let installedLocales = await SpeechTranscriber.installedLocales
|
|
278
|
+
let target = SpeechAnalyzerRecognitionSupport.normalizedIdentifier(for: locale)
|
|
279
|
+
let hasInstalledModel = installedLocales.contains {
|
|
280
|
+
SpeechAnalyzerRecognitionSupport.normalizedIdentifier(for: $0) == target
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
if !hasInstalledModel,
|
|
284
|
+
let installer = try await AssetInventory.assetInstallationRequest(supporting: [transcriber]) {
|
|
285
|
+
try await installer.downloadAndInstall()
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
try await reserveLocaleIfNeeded(locale)
|
|
289
|
+
}
|
|
290
|
+
|
|
291
|
+
func releaseLocales() async {
|
|
292
|
+
let allReservedLocales = await AssetInventory.reservedLocales
|
|
293
|
+
for locale in allReservedLocales {
|
|
294
|
+
let normalized = SpeechAnalyzerRecognitionSupport.normalizedIdentifier(for: locale)
|
|
295
|
+
if reservedLocales.contains(normalized) {
|
|
296
|
+
await AssetInventory.release(reservedLocale: locale)
|
|
297
|
+
reservedLocales.remove(normalized)
|
|
298
|
+
}
|
|
299
|
+
}
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
private func reserveLocaleIfNeeded(_ locale: Locale) async throws {
|
|
303
|
+
let target = SpeechAnalyzerRecognitionSupport.normalizedIdentifier(for: locale)
|
|
304
|
+
let existingReservations = await AssetInventory.reservedLocales
|
|
305
|
+
let hasReservation = existingReservations.contains {
|
|
306
|
+
SpeechAnalyzerRecognitionSupport.normalizedIdentifier(for: $0) == target
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
if hasReservation {
|
|
310
|
+
return
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
try await AssetInventory.reserve(locale: locale)
|
|
314
|
+
reservedLocales.insert(target)
|
|
315
|
+
}
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
@available(iOS 26.0, *)
|
|
319
|
+
private struct SpeechAnalyzerSendablePCMBuffer: @unchecked Sendable {
|
|
320
|
+
let buffer: AVAudioPCMBuffer
|
|
321
|
+
}
|
|
322
|
+
|
|
323
|
+
@available(iOS 26.0, *)
|
|
324
|
+
private actor SpeechAnalyzerAudioProcessingActor {
|
|
325
|
+
private let converter = SpeechAnalyzerBufferConverter()
|
|
326
|
+
|
|
327
|
+
func makeAnalyzerInput(
|
|
328
|
+
from buffer: SpeechAnalyzerSendablePCMBuffer,
|
|
329
|
+
analyzerFormat: AVAudioFormat
|
|
330
|
+
) throws -> AnalyzerInput {
|
|
331
|
+
let convertedBuffer = try converter.convertBuffer(buffer.buffer, to: analyzerFormat)
|
|
332
|
+
return AnalyzerInput(buffer: convertedBuffer)
|
|
333
|
+
}
|
|
334
|
+
}
|
|
335
|
+
|
|
336
|
+
@available(iOS 26.0, *)
|
|
337
|
+
private final class SpeechAnalyzerBufferConverter: @unchecked Sendable {
|
|
338
|
+
private var converter: AVAudioConverter?
|
|
339
|
+
|
|
340
|
+
func convertBuffer(_ buffer: AVAudioPCMBuffer, to format: AVAudioFormat) throws -> AVAudioPCMBuffer {
|
|
341
|
+
let inputFormat = buffer.format
|
|
342
|
+
guard inputFormat != format else {
|
|
343
|
+
return buffer
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
if converter == nil || converter?.outputFormat != format {
|
|
347
|
+
converter = AVAudioConverter(from: inputFormat, to: format)
|
|
348
|
+
converter?.primeMethod = .none
|
|
349
|
+
}
|
|
350
|
+
|
|
351
|
+
guard let converter else {
|
|
352
|
+
throw SpeechAnalyzerRecognitionError.setupFailed("Failed to create audio converter.")
|
|
353
|
+
}
|
|
354
|
+
|
|
355
|
+
let sampleRateRatio = converter.outputFormat.sampleRate / converter.inputFormat.sampleRate
|
|
356
|
+
let scaledFrameLength = Double(buffer.frameLength) * sampleRateRatio
|
|
357
|
+
let frameCapacity = AVAudioFrameCount(scaledFrameLength.rounded(.up))
|
|
358
|
+
|
|
359
|
+
guard let conversionBuffer = AVAudioPCMBuffer(
|
|
360
|
+
pcmFormat: converter.outputFormat,
|
|
361
|
+
frameCapacity: frameCapacity
|
|
362
|
+
) else {
|
|
363
|
+
throw SpeechAnalyzerRecognitionError.setupFailed("Failed to create conversion buffer.")
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
var conversionError: NSError?
|
|
367
|
+
|
|
368
|
+
final class BufferState: @unchecked Sendable {
|
|
369
|
+
var hasSuppliedBuffer = false
|
|
370
|
+
}
|
|
371
|
+
|
|
372
|
+
let bufferState = BufferState()
|
|
373
|
+
let status = converter.convert(to: conversionBuffer, error: &conversionError) { _, statusPointer in
|
|
374
|
+
defer {
|
|
375
|
+
bufferState.hasSuppliedBuffer = true
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
statusPointer.pointee = bufferState.hasSuppliedBuffer ? .noDataNow : .haveData
|
|
379
|
+
return bufferState.hasSuppliedBuffer ? nil : buffer
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
if status == .error {
|
|
383
|
+
throw conversionError ?? SpeechAnalyzerRecognitionError.setupFailed("Audio conversion failed.")
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
return conversionBuffer
|
|
387
|
+
}
|
|
388
|
+
}
|
|
389
|
+
|
|
390
|
+
#else
|
|
391
|
+
|
|
392
|
+
enum SpeechAnalyzerRecognitionSupport {
|
|
393
|
+
static func supports(locale _: Locale) async -> Bool {
|
|
394
|
+
false
|
|
395
|
+
}
|
|
396
|
+
|
|
397
|
+
static func supportedLanguageIdentifiers() async -> [String] {
|
|
398
|
+
[]
|
|
399
|
+
}
|
|
400
|
+
|
|
401
|
+
static func normalizedIdentifier(for locale: Locale) -> String {
|
|
402
|
+
locale.identifier
|
|
403
|
+
}
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
enum SpeechAnalyzerRecognitionError: LocalizedError {
|
|
407
|
+
case unavailable
|
|
408
|
+
|
|
409
|
+
var errorDescription: String? {
|
|
410
|
+
"Speech analyzer requires a newer Apple SDK."
|
|
411
|
+
}
|
|
412
|
+
}
|
|
413
|
+
|
|
414
|
+
@MainActor
|
|
415
|
+
final class SpeechAnalyzerRecognitionSession: NSObject {
|
|
416
|
+
typealias ResultHandler = @MainActor ([String], Bool) -> Void
|
|
417
|
+
typealias VoidHandler = @MainActor () -> Void
|
|
418
|
+
typealias ErrorHandler = @MainActor (Error) -> Void
|
|
419
|
+
|
|
420
|
+
var isRunning = false
|
|
421
|
+
var onListeningStarted: VoidHandler?
|
|
422
|
+
var onListeningStopped: VoidHandler?
|
|
423
|
+
var onResult: ResultHandler?
|
|
424
|
+
var onError: ErrorHandler?
|
|
425
|
+
|
|
426
|
+
init(locale _: Locale, maxResults _: Int, includePartialResults _: Bool) {}
|
|
427
|
+
|
|
428
|
+
func start() async throws {
|
|
429
|
+
throw SpeechAnalyzerRecognitionError.unavailable
|
|
430
|
+
}
|
|
431
|
+
|
|
432
|
+
func stop() async {
|
|
433
|
+
isRunning = false
|
|
434
|
+
}
|
|
435
|
+
}
|
|
436
|
+
|
|
437
|
+
#endif
|