@framers/agentos-ext-google-cloud-stt 0.2.1 → 0.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/SKILL.md +5 -3
- package/dist/GoogleCloudSTTProvider.d.ts +55 -19
- package/dist/GoogleCloudSTTProvider.d.ts.map +1 -1
- package/dist/GoogleCloudSTTProvider.js +82 -21
- package/dist/GoogleCloudSTTProvider.js.map +1 -1
- package/dist/index.d.ts +1 -1
- package/dist/index.d.ts.map +1 -1
- package/manifest.json +1 -1
- package/package.json +2 -2
- package/src/GoogleCloudSTTProvider.ts +128 -35
- package/src/index.ts +1 -0
package/SKILL.md
CHANGED
|
@@ -14,12 +14,14 @@ Provide credentials via the `GOOGLE_CLOUD_STT_CREDENTIALS` secret. Accepts eithe
|
|
|
14
14
|
- An absolute path to a service-account JSON key file (contains `/` or `\`)
|
|
15
15
|
- A raw JSON string with the service-account credentials
|
|
16
16
|
|
|
17
|
+
Leave the secret unset to use Google's Application Default Credentials (`GOOGLE_APPLICATION_CREDENTIALS`, `gcloud auth application-default login`, or the metadata server on Google Cloud).
|
|
18
|
+
|
|
17
19
|
## Features
|
|
18
20
|
|
|
19
|
-
- LINEAR16 PCM
|
|
21
|
+
- WAV and FLAC files (detected from the bytes; Google reads the encoding and sample rate from the header) and raw LINEAR16 PCM, whatever MIME type the caller declares
|
|
20
22
|
- Configurable language code (BCP-47)
|
|
21
|
-
-
|
|
22
|
-
-
|
|
23
|
+
- Returns the AgentOS `SpeechTranscriptionResult`: `text` (each stretch's most likely transcript, in order), mean `confidence`, `isFinal`, and `segments` with timing when Google reports end times
|
|
24
|
+
- Batch only (`supportsStreaming: false`)
|
|
23
25
|
|
|
24
26
|
## Configuration
|
|
25
27
|
|
|
@@ -10,18 +10,39 @@
|
|
|
10
10
|
* @module google-cloud-stt
|
|
11
11
|
*/
|
|
12
12
|
/**
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
|
|
16
|
-
|
|
13
|
+
* One stretch of recognised speech: Google returns a result per consecutive
|
|
14
|
+
* portion of the audio.
|
|
15
|
+
*/
|
|
16
|
+
export interface SpeechTranscriptionSegment {
|
|
17
|
+
/** The most likely transcript of this stretch. */
|
|
18
|
+
text: string;
|
|
19
|
+
/** Start in seconds from the beginning of the audio. */
|
|
20
|
+
startTime: number;
|
|
21
|
+
/** End in seconds from the beginning of the audio. */
|
|
22
|
+
endTime: number;
|
|
23
|
+
/** Confidence score in [0, 1], when Google reports one. */
|
|
24
|
+
confidence?: number;
|
|
25
|
+
}
|
|
26
|
+
/**
|
|
27
|
+
* The transcription result of the AgentOS `SpeechToTextProvider` contract
|
|
28
|
+
* (`SpeechTranscriptionResult` in `@framers/agentos`), mirrored here so the
|
|
29
|
+
* pack carries no type import from agentos.
|
|
17
30
|
*/
|
|
18
31
|
export interface SpeechTranscriptionResult {
|
|
19
|
-
/** The recognised text. */
|
|
20
|
-
|
|
21
|
-
/**
|
|
22
|
-
|
|
23
|
-
/**
|
|
32
|
+
/** The recognised text: every stretch's most likely transcript, in order. */
|
|
33
|
+
text: string;
|
|
34
|
+
/** BCP-47 language of the transcript. */
|
|
35
|
+
language?: string;
|
|
36
|
+
/** Mean confidence of the stretches, in [0, 1], when Google reports any. */
|
|
37
|
+
confidence?: number;
|
|
38
|
+
/** Always `true`: batch recognition returns final results only. */
|
|
24
39
|
isFinal: boolean;
|
|
40
|
+
/** Always 0; cost is tracked above the provider, as for AgentOS's core batch providers. */
|
|
41
|
+
cost: number;
|
|
42
|
+
/** Per-stretch transcripts with their timing, when Google reports end times. */
|
|
43
|
+
segments?: SpeechTranscriptionSegment[];
|
|
44
|
+
/** The raw `RecognizeResponse`. */
|
|
45
|
+
providerResponse?: unknown;
|
|
25
46
|
}
|
|
26
47
|
/**
|
|
27
48
|
* Per-call transcription options forwarded to the Google Cloud API.
|
|
@@ -31,13 +52,18 @@ export interface GoogleCloudSTTOptions {
|
|
|
31
52
|
language?: string;
|
|
32
53
|
}
|
|
33
54
|
/**
|
|
34
|
-
* Audio
|
|
55
|
+
* Audio passed to {@link GoogleCloudSTTProvider.transcribe}: the fields of the
|
|
56
|
+
* AgentOS `SpeechAudioInput` this provider reads.
|
|
35
57
|
*/
|
|
36
58
|
export interface AudioData {
|
|
37
|
-
/**
|
|
59
|
+
/** The audio bytes: a WAV or FLAC file, or raw LINEAR16 PCM. */
|
|
38
60
|
data: Buffer;
|
|
39
|
-
/** Sample rate in Hz.
|
|
61
|
+
/** Sample rate in Hz. Raw PCM defaults to 16000; a WAV or FLAC header supplies its own. */
|
|
40
62
|
sampleRate?: number;
|
|
63
|
+
/** MIME type, such as `'audio/wav'`. Informational: the bytes decide the encoding. */
|
|
64
|
+
mimeType?: string;
|
|
65
|
+
/** Container format, such as `'wav'`. Informational: the bytes decide the encoding. */
|
|
66
|
+
format?: string;
|
|
41
67
|
}
|
|
42
68
|
/**
|
|
43
69
|
* Google Cloud Speech-to-Text batch provider.
|
|
@@ -49,6 +75,10 @@ export interface AudioData {
|
|
|
49
75
|
export declare class GoogleCloudSTTProvider {
|
|
50
76
|
/** Stable provider identifier used by the AgentOS extension registry. */
|
|
51
77
|
readonly id = "google-cloud-stt";
|
|
78
|
+
/** Human-readable provider name. */
|
|
79
|
+
readonly displayName = "Google Cloud Speech-to-Text";
|
|
80
|
+
/** Batch recognition only: this provider does not stream. */
|
|
81
|
+
readonly supportsStreaming = false;
|
|
52
82
|
/** Lazily initialised Speech client. */
|
|
53
83
|
private _client;
|
|
54
84
|
/** Resolved client constructor options (set in constructor, used in {@link _getClient}). */
|
|
@@ -70,16 +100,22 @@ export declare class GoogleCloudSTTProvider {
|
|
|
70
100
|
*/
|
|
71
101
|
private _getClient;
|
|
72
102
|
/**
|
|
73
|
-
*
|
|
103
|
+
* The provider's display name, as the AgentOS speech contract requires.
|
|
104
|
+
*
|
|
105
|
+
* @returns `'Google Cloud Speech-to-Text'`.
|
|
106
|
+
*/
|
|
107
|
+
getProviderName(): string;
|
|
108
|
+
/**
|
|
109
|
+
* Transcribe an audio file or raw PCM buffer using Google Cloud Speech-to-Text.
|
|
74
110
|
*
|
|
75
|
-
*
|
|
76
|
-
*
|
|
77
|
-
*
|
|
111
|
+
* Google returns one result per consecutive stretch of the audio, each with
|
|
112
|
+
* its alternatives ordered by likelihood. The transcript is every stretch's
|
|
113
|
+
* first alternative, in order.
|
|
78
114
|
*
|
|
79
|
-
* @param audio -
|
|
115
|
+
* @param audio - WAV or FLAC file bytes, or raw LINEAR16 PCM with its sample rate.
|
|
80
116
|
* @param options - Optional per-call parameters (language code).
|
|
81
|
-
* @returns
|
|
117
|
+
* @returns The transcription in the AgentOS `SpeechTranscriptionResult` shape.
|
|
82
118
|
*/
|
|
83
|
-
transcribe(audio: AudioData, options?: GoogleCloudSTTOptions): Promise<SpeechTranscriptionResult
|
|
119
|
+
transcribe(audio: AudioData, options?: GoogleCloudSTTOptions): Promise<SpeechTranscriptionResult>;
|
|
84
120
|
}
|
|
85
121
|
//# sourceMappingURL=GoogleCloudSTTProvider.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"GoogleCloudSTTProvider.d.ts","sourceRoot":"","sources":["../src/GoogleCloudSTTProvider.ts"],"names":[],"mappings":"AACA;;;;;;;;;;GAUG;AAOH
|
|
1
|
+
{"version":3,"file":"GoogleCloudSTTProvider.d.ts","sourceRoot":"","sources":["../src/GoogleCloudSTTProvider.ts"],"names":[],"mappings":"AACA;;;;;;;;;;GAUG;AAOH;;;GAGG;AACH,MAAM,WAAW,0BAA0B;IACzC,kDAAkD;IAClD,IAAI,EAAE,MAAM,CAAC;IACb,wDAAwD;IACxD,SAAS,EAAE,MAAM,CAAC;IAClB,sDAAsD;IACtD,OAAO,EAAE,MAAM,CAAC;IAChB,2DAA2D;IAC3D,UAAU,CAAC,EAAE,MAAM,CAAC;CACrB;AAED;;;;GAIG;AACH,MAAM,WAAW,yBAAyB;IACxC,6EAA6E;IAC7E,IAAI,EAAE,MAAM,CAAC;IACb,yCAAyC;IACzC,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,4EAA4E;IAC5E,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,mEAAmE;IACnE,OAAO,EAAE,OAAO,CAAC;IACjB,2FAA2F;IAC3F,IAAI,EAAE,MAAM,CAAC;IACb,gFAAgF;IAChF,QAAQ,CAAC,EAAE,0BAA0B,EAAE,CAAC;IACxC,mCAAmC;IACnC,gBAAgB,CAAC,EAAE,OAAO,CAAC;CAC5B;AAED;;GAEG;AACH,MAAM,WAAW,qBAAqB;IACpC,gFAAgF;IAChF,QAAQ,CAAC,EAAE,MAAM,CAAC;CACnB;AAED;;;GAGG;AACH,MAAM,WAAW,SAAS;IACxB,gEAAgE;IAChE,IAAI,EAAE,MAAM,CAAC;IACb,2FAA2F;IAC3F,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,sFAAsF;IACtF,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,uFAAuF;IACvF,MAAM,CAAC,EAAE,MAAM,CAAC;CACjB;AAsCD;;;;;;GAMG;AACH,qBAAa,sBAAsB;IACjC,yEAAyE;IACzE,QAAQ,CAAC,EAAE,sBAAsB;IAEjC,oCAAoC;IACpC,QAAQ,CAAC,WAAW,iCAAiC;IAErD,6DAA6D;IAC7D,QAAQ,CAAC,iBAAiB,SAAS;IAEnC,wCAAwC;IACxC,OAAO,CAAC,OAAO,CAA6B;IAE5C,4FAA4F;IAC5F,OAAO,CAAC,QAAQ,CAAC,cAAc,CAA0B;IAEzD;;;;;;;OAOG;gBACS,WAAW,EAAE,MAAM;IAmB/B;;;;;OAKG;YACW,UAAU;IAaxB;;;;OAIG;IACH,eAAe,IAAI,MAAM;IAIzB;;;;;;;;;;OAUG;IACG,UAAU,CACd,KAAK,EAAE,SAAS,EAChB,OAAO,CAAC,EAAE,qBAAqB,GAC9B,OAAO,CAAC,yBAAyB,CAAC;CA0CtC"}
|
|
@@ -10,6 +10,39 @@
|
|
|
10
10
|
*
|
|
11
11
|
* @module google-cloud-stt
|
|
12
12
|
*/
|
|
13
|
+
/** True when the bytes start with a RIFF/WAVE header. */
|
|
14
|
+
function hasWavHeader(data) {
|
|
15
|
+
return data.length >= 12 && data.toString('latin1', 0, 4) === 'RIFF' && data.toString('latin1', 8, 12) === 'WAVE';
|
|
16
|
+
}
|
|
17
|
+
/** True when the bytes start with the FLAC stream marker. */
|
|
18
|
+
function hasFlacHeader(data) {
|
|
19
|
+
return data.length >= 4 && data.toString('latin1', 0, 4) === 'fLaC';
|
|
20
|
+
}
|
|
21
|
+
/**
|
|
22
|
+
* The encoding fields of the recognition config for this audio.
|
|
23
|
+
*
|
|
24
|
+
* WAV and FLAC files carry a header that states the encoding and sample rate.
|
|
25
|
+
* Google reads both from it and rejects a request whose stated values disagree
|
|
26
|
+
* (google.cloud.speech.v1 `RecognitionConfig`), so for those files the
|
|
27
|
+
* encoding is left out and the sample rate is sent only when the caller gives
|
|
28
|
+
* one. Anything else is sent as raw LINEAR16 PCM.
|
|
29
|
+
*
|
|
30
|
+
* The bytes decide, not the declared type: AgentOS's speech adapter labels
|
|
31
|
+
* every buffer `audio/wav`, headerless PCM included.
|
|
32
|
+
*/
|
|
33
|
+
function encodingFor(audio) {
|
|
34
|
+
if (hasWavHeader(audio.data) || hasFlacHeader(audio.data)) {
|
|
35
|
+
return audio.sampleRate ? { sampleRateHertz: audio.sampleRate } : {};
|
|
36
|
+
}
|
|
37
|
+
return { encoding: 'LINEAR16', sampleRateHertz: audio.sampleRate ?? 16000 };
|
|
38
|
+
}
|
|
39
|
+
/** Seconds in a protobuf `Duration` (`{ seconds, nanos }`, seconds possibly a string). */
|
|
40
|
+
function durationSeconds(duration) {
|
|
41
|
+
if (!duration)
|
|
42
|
+
return undefined;
|
|
43
|
+
const seconds = Number(duration.seconds ?? 0) + Number(duration.nanos ?? 0) / 1e9;
|
|
44
|
+
return Number.isFinite(seconds) ? seconds : undefined;
|
|
45
|
+
}
|
|
13
46
|
/**
|
|
14
47
|
* Google Cloud Speech-to-Text batch provider.
|
|
15
48
|
*
|
|
@@ -20,6 +53,10 @@
|
|
|
20
53
|
export class GoogleCloudSTTProvider {
|
|
21
54
|
/** Stable provider identifier used by the AgentOS extension registry. */
|
|
22
55
|
id = 'google-cloud-stt';
|
|
56
|
+
/** Human-readable provider name. */
|
|
57
|
+
displayName = 'Google Cloud Speech-to-Text';
|
|
58
|
+
/** Batch recognition only: this provider does not stream. */
|
|
59
|
+
supportsStreaming = false;
|
|
23
60
|
/** Lazily initialised Speech client. */
|
|
24
61
|
_client = null;
|
|
25
62
|
/** Resolved client constructor options (set in constructor, used in {@link _getClient}). */
|
|
@@ -69,38 +106,62 @@ export class GoogleCloudSTTProvider {
|
|
|
69
106
|
// Public API
|
|
70
107
|
// ---------------------------------------------------------------------------
|
|
71
108
|
/**
|
|
72
|
-
*
|
|
109
|
+
* The provider's display name, as the AgentOS speech contract requires.
|
|
110
|
+
*
|
|
111
|
+
* @returns `'Google Cloud Speech-to-Text'`.
|
|
112
|
+
*/
|
|
113
|
+
getProviderName() {
|
|
114
|
+
return this.displayName;
|
|
115
|
+
}
|
|
116
|
+
/**
|
|
117
|
+
* Transcribe an audio file or raw PCM buffer using Google Cloud Speech-to-Text.
|
|
73
118
|
*
|
|
74
|
-
*
|
|
75
|
-
*
|
|
76
|
-
*
|
|
119
|
+
* Google returns one result per consecutive stretch of the audio, each with
|
|
120
|
+
* its alternatives ordered by likelihood. The transcript is every stretch's
|
|
121
|
+
* first alternative, in order.
|
|
77
122
|
*
|
|
78
|
-
* @param audio -
|
|
123
|
+
* @param audio - WAV or FLAC file bytes, or raw LINEAR16 PCM with its sample rate.
|
|
79
124
|
* @param options - Optional per-call parameters (language code).
|
|
80
|
-
* @returns
|
|
125
|
+
* @returns The transcription in the AgentOS `SpeechTranscriptionResult` shape.
|
|
81
126
|
*/
|
|
82
127
|
async transcribe(audio, options) {
|
|
83
128
|
const client = await this._getClient();
|
|
129
|
+
const languageCode = options?.language ?? 'en-US';
|
|
84
130
|
const response = await client.recognize({
|
|
85
131
|
audio: { content: audio.data.toString('base64') },
|
|
86
|
-
config: {
|
|
87
|
-
encoding: 'LINEAR16',
|
|
88
|
-
sampleRateHertz: audio.sampleRate ?? 16000,
|
|
89
|
-
languageCode: options?.language ?? 'en-US',
|
|
90
|
-
},
|
|
132
|
+
config: { ...encodingFor(audio), languageCode },
|
|
91
133
|
});
|
|
92
|
-
const
|
|
93
|
-
|
|
134
|
+
const recognized = response[0];
|
|
135
|
+
const stretches = [];
|
|
136
|
+
for (const result of recognized?.results ?? []) {
|
|
94
137
|
const alt = result?.alternatives?.[0];
|
|
95
|
-
if (alt)
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
}
|
|
138
|
+
if (!alt)
|
|
139
|
+
continue;
|
|
140
|
+
stretches.push({
|
|
141
|
+
text: (alt.transcript ?? '').trim(),
|
|
142
|
+
confidence: typeof alt.confidence === 'number' ? alt.confidence : undefined,
|
|
143
|
+
endTime: durationSeconds(result.resultEndTime),
|
|
144
|
+
});
|
|
102
145
|
}
|
|
103
|
-
|
|
146
|
+
const confidences = stretches.map((s) => s.confidence).filter((c) => c !== undefined);
|
|
147
|
+
// Timing is reported only when Google gives every stretch an end time.
|
|
148
|
+
let start = 0;
|
|
149
|
+
const segments = stretches.length > 0 && stretches.every((s) => s.endTime !== undefined)
|
|
150
|
+
? stretches.map((s) => {
|
|
151
|
+
const segment = { text: s.text, startTime: start, endTime: s.endTime, confidence: s.confidence };
|
|
152
|
+
start = s.endTime;
|
|
153
|
+
return segment;
|
|
154
|
+
})
|
|
155
|
+
: undefined;
|
|
156
|
+
return {
|
|
157
|
+
text: stretches.map((s) => s.text).filter((t) => t.length > 0).join(' '),
|
|
158
|
+
language: recognized?.results?.[0]?.languageCode || languageCode,
|
|
159
|
+
confidence: confidences.length > 0 ? confidences.reduce((sum, c) => sum + c, 0) / confidences.length : undefined,
|
|
160
|
+
isFinal: true,
|
|
161
|
+
cost: 0,
|
|
162
|
+
segments,
|
|
163
|
+
providerResponse: recognized,
|
|
164
|
+
};
|
|
104
165
|
}
|
|
105
166
|
}
|
|
106
167
|
//# sourceMappingURL=GoogleCloudSTTProvider.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"GoogleCloudSTTProvider.js","sourceRoot":"","sources":["../src/GoogleCloudSTTProvider.ts"],"names":[],"mappings":"AAAA,cAAc;AACd;;;;;;;;;;GAUG;
|
|
1
|
+
{"version":3,"file":"GoogleCloudSTTProvider.js","sourceRoot":"","sources":["../src/GoogleCloudSTTProvider.ts"],"names":[],"mappings":"AAAA,cAAc;AACd;;;;;;;;;;GAUG;AAmEH,yDAAyD;AACzD,SAAS,YAAY,CAAC,IAAY;IAChC,OAAO,IAAI,CAAC,MAAM,IAAI,EAAE,IAAI,IAAI,CAAC,QAAQ,CAAC,QAAQ,EAAE,CAAC,EAAE,CAAC,CAAC,KAAK,MAAM,IAAI,IAAI,CAAC,QAAQ,CAAC,QAAQ,EAAE,CAAC,EAAE,EAAE,CAAC,KAAK,MAAM,CAAC;AACpH,CAAC;AAED,6DAA6D;AAC7D,SAAS,aAAa,CAAC,IAAY;IACjC,OAAO,IAAI,CAAC,MAAM,IAAI,CAAC,IAAI,IAAI,CAAC,QAAQ,CAAC,QAAQ,EAAE,CAAC,EAAE,CAAC,CAAC,KAAK,MAAM,CAAC;AACtE,CAAC;AAED;;;;;;;;;;;GAWG;AACH,SAAS,WAAW,CAAC,KAAgB;IACnC,IAAI,YAAY,CAAC,KAAK,CAAC,IAAI,CAAC,IAAI,aAAa,CAAC,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC;QAC1D,OAAO,KAAK,CAAC,UAAU,CAAC,CAAC,CAAC,EAAE,eAAe,EAAE,KAAK,CAAC,UAAU,EAAE,CAAC,CAAC,CAAC,EAAE,CAAC;IACvE,CAAC;IACD,OAAO,EAAE,QAAQ,EAAE,UAAU,EAAE,eAAe,EAAE,KAAK,CAAC,UAAU,IAAI,KAAK,EAAE,CAAC;AAC9E,CAAC;AAED,0FAA0F;AAC1F,SAAS,eAAe,CAAC,QAAmE;IAC1F,IAAI,CAAC,QAAQ;QAAE,OAAO,SAAS,CAAC;IAChC,MAAM,OAAO,GAAG,MAAM,CAAC,QAAQ,CAAC,OAAO,IAAI,CAAC,CAAC,GAAG,MAAM,CAAC,QAAQ,CAAC,KAAK,IAAI,CAAC,CAAC,GAAG,GAAG,CAAC;IAClF,OAAO,MAAM,CAAC,QAAQ,CAAC,OAAO,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,SAAS,CAAC;AACxD,CAAC;AAED;;;;;;GAMG;AACH,MAAM,OAAO,sBAAsB;IACjC,yEAAyE;IAChE,EAAE,GAAG,kBAAkB,CAAC;IAEjC,oCAAoC;IAC3B,WAAW,GAAG,6BAA6B,CAAC;IAErD,6DAA6D;IACpD,iBAAiB,GAAG,KAAK,CAAC;IAEnC,wCAAwC;IAChC,OAAO,GAAwB,IAAI,CAAC;IAE5C,4FAA4F;IAC3E,cAAc,CAA0B;IAEzD;;;;;;;OAOG;IACH,YAAY,WAAmB;QAC7B,IAAI,CAAC,WAAW,CAAC,IAAI,EAAE,EAAE,CAAC;YACxB,wEAAwE;YACxE,wEAAwE;YACxE,yEAAyE;YACzE,IAAI,CAAC,cAAc,GAAG,EAAE,CAAC;QAC3B,CAAC;aAAM,IAAI,WAAW,CAAC,QAAQ,CAAC,GAAG,CAAC,IAAI,WAAW,CAAC,QAAQ,CAAC,IAAI,CAAC,EAAE,CAAC;YACnE,wBAAwB;YACxB,IAAI,CAAC,cAAc,GAAG,EAAE,WAAW,EAAE,WAAW,EAAE,CAAC;QACrD,CAAC;aAAM,CAAC;YACN,8CAA8C;YAC9C,IAAI,CAAC,cAAc,GAAG,EAAE,WAAW,EAAE,IAAI,CAAC,KAAK,CAAC,WAAW,CAA4B,EAAE,CAAC;QAC5F,CAAC;IACH,CAAC;IAED,8EAA8E;IAC9E,kBAAkB;IAClB,8EAA8E;IAE9E;;;;;OAKG;IACK,KAAK,CAAC,UAAU;QACtB,IAAI,CAAC,IAAI,CAAC,OAAO,EAAE,CAAC;YAClB,wEAAwE;YACxE,MAAM,EAAE,YAAY,EAAE,GAAG,MAAM,MAAM,CAAC,sBAAsB,CAAC,CAAC;YAC9D,IAAI,CAAC,OAAO,GAAG,IAAI,YAAY,CAAC,IAAI,CAAC,cAAc,CAAC,CAAC;QACvD,CAAC;QACD,OAAO,IAAI,CAAC,OAAO,CAAC;IACtB,CAAC;IAED,8EAA8E;IAC9E,aAAa;IACb,8EAA8E;IAE9E;;;;OAIG;IACH,eAAe;QACb,OAAO,IAAI,CAAC,WAAW,CAAC;IAC1B,CAAC;IAED;;;;;;;;;;OAUG;IACH,KAAK,CAAC,UAAU,CACd,KAAgB,EAChB,OAA+B;QAE/B,MAAM,MAAM,GAAG,MAAM,IAAI,CAAC,UAAU,EAAE,CAAC;QACvC,MAAM,YAAY,GAAG,OAAO,EAAE,QAAQ,IAAI,OAAO,CAAC;QAElD,MAAM,QAAQ,GAAG,MAAM,MAAM,CAAC,SAAS,CAAC;YACtC,KAAK,EAAE,EAAE,OAAO,EAAE,KAAK,CAAC,IAAI,CAAC,QAAQ,CAAC,QAAQ,CAAC,EAAE;YACjD,MAAM,EAAE,EAAE,GAAG,WAAW,CAAC,KAAK,CAAC,EAAE,YAAY,EAAE;SAChD,CAAC,CAAC;QACH,MAAM,UAAU,GAAG,QAAQ,CAAC,CAAC,CAAC,CAAC;QAE/B,MAAM,SAAS,GAAmE,EAAE,CAAC;QACrF,KAAK,MAAM,MAAM,IAAI,UAAU,EAAE,OAAO,IAAI,EAAE,EAAE,CAAC;YAC/C,MAAM,GAAG,GAAG,MAAM,EAAE,YAAY,EAAE,CAAC,CAAC,CAAC,CAAC;YACtC,IAAI,CAAC,GAAG;gBAAE,SAAS;YACnB,SAAS,CAAC,IAAI,CAAC;gBACb,IAAI,EAAE,CAAC,GAAG,CAAC,UAAU,IAAI,EAAE,CAAC,CAAC,IAAI,EAAE;gBACnC,UAAU,EAAE,OAAO,GAAG,CAAC,UAAU,KAAK,QAAQ,CAAC,CAAC,CAAC,GAAG,CAAC,UAAU,CAAC,CAAC,CAAC,SAAS;gBAC3E,OAAO,EAAE,eAAe,CAAC,MAAM,CAAC,aAAa,CAAC;aAC/C,CAAC,CAAC;QACL,CAAC;QAED,MAAM,WAAW,GAAG,SAAS,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,UAAU,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,EAAe,EAAE,CAAC,CAAC,KAAK,SAAS,CAAC,CAAC;QACnG,uEAAuE;QACvE,IAAI,KAAK,GAAG,CAAC,CAAC;QACd,MAAM,QAAQ,GAAG,SAAS,CAAC,MAAM,GAAG,CAAC,IAAI,SAAS,CAAC,KAAK,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,OAAO,KAAK,SAAS,CAAC;YACtF,CAAC,CAAC,SAAS,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE;gBAClB,MAAM,OAAO,GAAG,EAAE,IAAI,EAAE,CAAC,CAAC,IAAI,EAAE,SAAS,EAAE,KAAK,EAAE,OAAO,EAAE,CAAC,CAAC,OAAiB,EAAE,UAAU,EAAE,CAAC,CAAC,UAAU,EAAE,CAAC;gBAC3G,KAAK,GAAG,CAAC,CAAC,OAAiB,CAAC;gBAC5B,OAAO,OAAO,CAAC;YACjB,CAAC,CAAC;YACJ,CAAC,CAAC,SAAS,CAAC;QAEd,OAAO;YACL,IAAI,EAAE,SAAS,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC,IAAI,CAAC,GAAG,CAAC;YACxE,QAAQ,EAAE,UAAU,EAAE,OAAO,EAAE,CAAC,CAAC,CAAC,EAAE,YAAY,IAAI,YAAY;YAChE,UAAU,EAAE,WAAW,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC,CAAC,WAAW,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,CAAC,EAAE,EAAE,CAAC,GAAG,GAAG,CAAC,EAAE,CAAC,CAAC,GAAG,WAAW,CAAC,MAAM,CAAC,CAAC,CAAC,SAAS;YAChH,OAAO,EAAE,IAAI;YACb,IAAI,EAAE,CAAC;YACP,QAAQ;YACR,gBAAgB,EAAE,UAAU;SAC7B,CAAC;IACJ,CAAC;CACF"}
|
package/dist/index.d.ts
CHANGED
|
@@ -62,5 +62,5 @@ export declare function createGoogleCloudSTT(credentials: string): GoogleCloudST
|
|
|
62
62
|
*/
|
|
63
63
|
export declare function createExtensionPack(context: ExtensionPackContext): ExtensionPack;
|
|
64
64
|
export { GoogleCloudSTTProvider } from './GoogleCloudSTTProvider.js';
|
|
65
|
-
export type { SpeechTranscriptionResult, GoogleCloudSTTOptions, AudioData, } from './GoogleCloudSTTProvider.js';
|
|
65
|
+
export type { SpeechTranscriptionResult, SpeechTranscriptionSegment, GoogleCloudSTTOptions, AudioData, } from './GoogleCloudSTTProvider.js';
|
|
66
66
|
//# sourceMappingURL=index.d.ts.map
|
package/dist/index.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AACA;;;;;;;;;;;;;;;;;;;;;;GAsBG;AAEH,OAAO,EAAE,sBAAsB,EAAE,MAAM,6BAA6B,CAAC;AAMrE,2DAA2D;AAC3D,UAAU,mBAAmB;IAC3B,EAAE,EAAE,MAAM,CAAC;IACX,IAAI,EAAE,MAAM,CAAC;IACb,OAAO,EAAE,OAAO,CAAC;IACjB,eAAe,CAAC,EAAE,OAAO,CAAC;IAC1B,QAAQ,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;CACpC;AAED,qDAAqD;AACrD,UAAU,aAAa;IACrB,EAAE,EAAE,MAAM,CAAC;IACX,WAAW,EAAE,mBAAmB,EAAE,CAAC;CACpC;AAED,4DAA4D;AAC5D,UAAU,oBAAoB;IAC5B,SAAS,CAAC,EAAE,CAAC,EAAE,EAAE,MAAM,KAAK,MAAM,GAAG,SAAS,CAAC;IAC/C,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;CACnC;AASD;;;;;;;;GAQG;AACH,wBAAgB,oBAAoB,CAAC,WAAW,EAAE,MAAM,GAAG,sBAAsB,CAEhF;AAED;;;;;;;;;GASG;AACH,wBAAgB,mBAAmB,CAAC,OAAO,EAAE,oBAAoB,GAAG,aAAa,CAgBhF;AAMD,OAAO,EAAE,sBAAsB,EAAE,MAAM,6BAA6B,CAAC;AACrE,YAAY,EACV,yBAAyB,EACzB,qBAAqB,EACrB,SAAS,GACV,MAAM,6BAA6B,CAAC"}
|
|
1
|
+
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AACA;;;;;;;;;;;;;;;;;;;;;;GAsBG;AAEH,OAAO,EAAE,sBAAsB,EAAE,MAAM,6BAA6B,CAAC;AAMrE,2DAA2D;AAC3D,UAAU,mBAAmB;IAC3B,EAAE,EAAE,MAAM,CAAC;IACX,IAAI,EAAE,MAAM,CAAC;IACb,OAAO,EAAE,OAAO,CAAC;IACjB,eAAe,CAAC,EAAE,OAAO,CAAC;IAC1B,QAAQ,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;CACpC;AAED,qDAAqD;AACrD,UAAU,aAAa;IACrB,EAAE,EAAE,MAAM,CAAC;IACX,WAAW,EAAE,mBAAmB,EAAE,CAAC;CACpC;AAED,4DAA4D;AAC5D,UAAU,oBAAoB;IAC5B,SAAS,CAAC,EAAE,CAAC,EAAE,EAAE,MAAM,KAAK,MAAM,GAAG,SAAS,CAAC;IAC/C,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;CACnC;AASD;;;;;;;;GAQG;AACH,wBAAgB,oBAAoB,CAAC,WAAW,EAAE,MAAM,GAAG,sBAAsB,CAEhF;AAED;;;;;;;;;GASG;AACH,wBAAgB,mBAAmB,CAAC,OAAO,EAAE,oBAAoB,GAAG,aAAa,CAgBhF;AAMD,OAAO,EAAE,sBAAsB,EAAE,MAAM,6BAA6B,CAAC;AACrE,YAAY,EACV,yBAAyB,EACzB,0BAA0B,EAC1B,qBAAqB,EACrB,SAAS,GACV,MAAM,6BAA6B,CAAC"}
|
package/manifest.json
CHANGED
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@framers/agentos-ext-google-cloud-stt",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.3.1",
|
|
4
4
|
"description": "Batch speech-to-text via Google Cloud Speech-to-Text API for AgentOS voice pipeline",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "./dist/index.js",
|
|
@@ -22,7 +22,7 @@
|
|
|
22
22
|
"@google-cloud/speech": "^6.0.0"
|
|
23
23
|
},
|
|
24
24
|
"devDependencies": {
|
|
25
|
-
"@framers/agentos": "^0.
|
|
25
|
+
"@framers/agentos": "^0.12.8",
|
|
26
26
|
"@google-cloud/speech": "^6.0.0",
|
|
27
27
|
"typescript": "^5.5.0",
|
|
28
28
|
"vitest": "^3.2.7"
|
|
@@ -17,18 +17,40 @@
|
|
|
17
17
|
type SpeechClient = any;
|
|
18
18
|
|
|
19
19
|
/**
|
|
20
|
-
*
|
|
21
|
-
*
|
|
22
|
-
|
|
23
|
-
|
|
20
|
+
* One stretch of recognised speech: Google returns a result per consecutive
|
|
21
|
+
* portion of the audio.
|
|
22
|
+
*/
|
|
23
|
+
export interface SpeechTranscriptionSegment {
|
|
24
|
+
/** The most likely transcript of this stretch. */
|
|
25
|
+
text: string;
|
|
26
|
+
/** Start in seconds from the beginning of the audio. */
|
|
27
|
+
startTime: number;
|
|
28
|
+
/** End in seconds from the beginning of the audio. */
|
|
29
|
+
endTime: number;
|
|
30
|
+
/** Confidence score in [0, 1], when Google reports one. */
|
|
31
|
+
confidence?: number;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* The transcription result of the AgentOS `SpeechToTextProvider` contract
|
|
36
|
+
* (`SpeechTranscriptionResult` in `@framers/agentos`), mirrored here so the
|
|
37
|
+
* pack carries no type import from agentos.
|
|
24
38
|
*/
|
|
25
39
|
export interface SpeechTranscriptionResult {
|
|
26
|
-
/** The recognised text. */
|
|
27
|
-
|
|
28
|
-
/**
|
|
29
|
-
|
|
30
|
-
/**
|
|
40
|
+
/** The recognised text: every stretch's most likely transcript, in order. */
|
|
41
|
+
text: string;
|
|
42
|
+
/** BCP-47 language of the transcript. */
|
|
43
|
+
language?: string;
|
|
44
|
+
/** Mean confidence of the stretches, in [0, 1], when Google reports any. */
|
|
45
|
+
confidence?: number;
|
|
46
|
+
/** Always `true`: batch recognition returns final results only. */
|
|
31
47
|
isFinal: boolean;
|
|
48
|
+
/** Always 0; cost is tracked above the provider, as for AgentOS's core batch providers. */
|
|
49
|
+
cost: number;
|
|
50
|
+
/** Per-stretch transcripts with their timing, when Google reports end times. */
|
|
51
|
+
segments?: SpeechTranscriptionSegment[];
|
|
52
|
+
/** The raw `RecognizeResponse`. */
|
|
53
|
+
providerResponse?: unknown;
|
|
32
54
|
}
|
|
33
55
|
|
|
34
56
|
/**
|
|
@@ -40,13 +62,54 @@ export interface GoogleCloudSTTOptions {
|
|
|
40
62
|
}
|
|
41
63
|
|
|
42
64
|
/**
|
|
43
|
-
* Audio
|
|
65
|
+
* Audio passed to {@link GoogleCloudSTTProvider.transcribe}: the fields of the
|
|
66
|
+
* AgentOS `SpeechAudioInput` this provider reads.
|
|
44
67
|
*/
|
|
45
68
|
export interface AudioData {
|
|
46
|
-
/**
|
|
69
|
+
/** The audio bytes: a WAV or FLAC file, or raw LINEAR16 PCM. */
|
|
47
70
|
data: Buffer;
|
|
48
|
-
/** Sample rate in Hz.
|
|
71
|
+
/** Sample rate in Hz. Raw PCM defaults to 16000; a WAV or FLAC header supplies its own. */
|
|
49
72
|
sampleRate?: number;
|
|
73
|
+
/** MIME type, such as `'audio/wav'`. Informational: the bytes decide the encoding. */
|
|
74
|
+
mimeType?: string;
|
|
75
|
+
/** Container format, such as `'wav'`. Informational: the bytes decide the encoding. */
|
|
76
|
+
format?: string;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/** True when the bytes start with a RIFF/WAVE header. */
|
|
80
|
+
function hasWavHeader(data: Buffer): boolean {
|
|
81
|
+
return data.length >= 12 && data.toString('latin1', 0, 4) === 'RIFF' && data.toString('latin1', 8, 12) === 'WAVE';
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/** True when the bytes start with the FLAC stream marker. */
|
|
85
|
+
function hasFlacHeader(data: Buffer): boolean {
|
|
86
|
+
return data.length >= 4 && data.toString('latin1', 0, 4) === 'fLaC';
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/**
|
|
90
|
+
* The encoding fields of the recognition config for this audio.
|
|
91
|
+
*
|
|
92
|
+
* WAV and FLAC files carry a header that states the encoding and sample rate.
|
|
93
|
+
* Google reads both from it and rejects a request whose stated values disagree
|
|
94
|
+
* (google.cloud.speech.v1 `RecognitionConfig`), so for those files the
|
|
95
|
+
* encoding is left out and the sample rate is sent only when the caller gives
|
|
96
|
+
* one. Anything else is sent as raw LINEAR16 PCM.
|
|
97
|
+
*
|
|
98
|
+
* The bytes decide, not the declared type: AgentOS's speech adapter labels
|
|
99
|
+
* every buffer `audio/wav`, headerless PCM included.
|
|
100
|
+
*/
|
|
101
|
+
function encodingFor(audio: AudioData): { encoding?: string; sampleRateHertz?: number } {
|
|
102
|
+
if (hasWavHeader(audio.data) || hasFlacHeader(audio.data)) {
|
|
103
|
+
return audio.sampleRate ? { sampleRateHertz: audio.sampleRate } : {};
|
|
104
|
+
}
|
|
105
|
+
return { encoding: 'LINEAR16', sampleRateHertz: audio.sampleRate ?? 16000 };
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
/** Seconds in a protobuf `Duration` (`{ seconds, nanos }`, seconds possibly a string). */
|
|
109
|
+
function durationSeconds(duration: { seconds?: unknown; nanos?: unknown } | null | undefined): number | undefined {
|
|
110
|
+
if (!duration) return undefined;
|
|
111
|
+
const seconds = Number(duration.seconds ?? 0) + Number(duration.nanos ?? 0) / 1e9;
|
|
112
|
+
return Number.isFinite(seconds) ? seconds : undefined;
|
|
50
113
|
}
|
|
51
114
|
|
|
52
115
|
/**
|
|
@@ -60,6 +123,12 @@ export class GoogleCloudSTTProvider {
|
|
|
60
123
|
/** Stable provider identifier used by the AgentOS extension registry. */
|
|
61
124
|
readonly id = 'google-cloud-stt';
|
|
62
125
|
|
|
126
|
+
/** Human-readable provider name. */
|
|
127
|
+
readonly displayName = 'Google Cloud Speech-to-Text';
|
|
128
|
+
|
|
129
|
+
/** Batch recognition only: this provider does not stream. */
|
|
130
|
+
readonly supportsStreaming = false;
|
|
131
|
+
|
|
63
132
|
/** Lazily initialised Speech client. */
|
|
64
133
|
private _client: SpeechClient | null = null;
|
|
65
134
|
|
|
@@ -113,44 +182,68 @@ export class GoogleCloudSTTProvider {
|
|
|
113
182
|
// ---------------------------------------------------------------------------
|
|
114
183
|
|
|
115
184
|
/**
|
|
116
|
-
*
|
|
185
|
+
* The provider's display name, as the AgentOS speech contract requires.
|
|
117
186
|
*
|
|
118
|
-
*
|
|
119
|
-
|
|
120
|
-
|
|
187
|
+
* @returns `'Google Cloud Speech-to-Text'`.
|
|
188
|
+
*/
|
|
189
|
+
getProviderName(): string {
|
|
190
|
+
return this.displayName;
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
/**
|
|
194
|
+
* Transcribe an audio file or raw PCM buffer using Google Cloud Speech-to-Text.
|
|
195
|
+
*
|
|
196
|
+
* Google returns one result per consecutive stretch of the audio, each with
|
|
197
|
+
* its alternatives ordered by likelihood. The transcript is every stretch's
|
|
198
|
+
* first alternative, in order.
|
|
121
199
|
*
|
|
122
|
-
* @param audio -
|
|
200
|
+
* @param audio - WAV or FLAC file bytes, or raw LINEAR16 PCM with its sample rate.
|
|
123
201
|
* @param options - Optional per-call parameters (language code).
|
|
124
|
-
* @returns
|
|
202
|
+
* @returns The transcription in the AgentOS `SpeechTranscriptionResult` shape.
|
|
125
203
|
*/
|
|
126
204
|
async transcribe(
|
|
127
205
|
audio: AudioData,
|
|
128
206
|
options?: GoogleCloudSTTOptions,
|
|
129
|
-
): Promise<SpeechTranscriptionResult
|
|
207
|
+
): Promise<SpeechTranscriptionResult> {
|
|
130
208
|
const client = await this._getClient();
|
|
209
|
+
const languageCode = options?.language ?? 'en-US';
|
|
131
210
|
|
|
132
211
|
const response = await client.recognize({
|
|
133
212
|
audio: { content: audio.data.toString('base64') },
|
|
134
|
-
config: {
|
|
135
|
-
encoding: 'LINEAR16',
|
|
136
|
-
sampleRateHertz: audio.sampleRate ?? 16000,
|
|
137
|
-
languageCode: options?.language ?? 'en-US',
|
|
138
|
-
},
|
|
213
|
+
config: { ...encodingFor(audio), languageCode },
|
|
139
214
|
});
|
|
215
|
+
const recognized = response[0];
|
|
140
216
|
|
|
141
|
-
const
|
|
142
|
-
|
|
143
|
-
for (const result of response[0]?.results ?? []) {
|
|
217
|
+
const stretches: Array<{ text: string; confidence?: number; endTime?: number }> = [];
|
|
218
|
+
for (const result of recognized?.results ?? []) {
|
|
144
219
|
const alt = result?.alternatives?.[0];
|
|
145
|
-
if (alt)
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
}
|
|
220
|
+
if (!alt) continue;
|
|
221
|
+
stretches.push({
|
|
222
|
+
text: (alt.transcript ?? '').trim(),
|
|
223
|
+
confidence: typeof alt.confidence === 'number' ? alt.confidence : undefined,
|
|
224
|
+
endTime: durationSeconds(result.resultEndTime),
|
|
225
|
+
});
|
|
152
226
|
}
|
|
153
227
|
|
|
154
|
-
|
|
228
|
+
const confidences = stretches.map((s) => s.confidence).filter((c): c is number => c !== undefined);
|
|
229
|
+
// Timing is reported only when Google gives every stretch an end time.
|
|
230
|
+
let start = 0;
|
|
231
|
+
const segments = stretches.length > 0 && stretches.every((s) => s.endTime !== undefined)
|
|
232
|
+
? stretches.map((s) => {
|
|
233
|
+
const segment = { text: s.text, startTime: start, endTime: s.endTime as number, confidence: s.confidence };
|
|
234
|
+
start = s.endTime as number;
|
|
235
|
+
return segment;
|
|
236
|
+
})
|
|
237
|
+
: undefined;
|
|
238
|
+
|
|
239
|
+
return {
|
|
240
|
+
text: stretches.map((s) => s.text).filter((t) => t.length > 0).join(' '),
|
|
241
|
+
language: recognized?.results?.[0]?.languageCode || languageCode,
|
|
242
|
+
confidence: confidences.length > 0 ? confidences.reduce((sum, c) => sum + c, 0) / confidences.length : undefined,
|
|
243
|
+
isFinal: true,
|
|
244
|
+
cost: 0,
|
|
245
|
+
segments,
|
|
246
|
+
providerResponse: recognized,
|
|
247
|
+
};
|
|
155
248
|
}
|
|
156
249
|
}
|
package/src/index.ts
CHANGED
|
@@ -105,6 +105,7 @@ export function createExtensionPack(context: ExtensionPackContext): ExtensionPac
|
|
|
105
105
|
export { GoogleCloudSTTProvider } from './GoogleCloudSTTProvider.js';
|
|
106
106
|
export type {
|
|
107
107
|
SpeechTranscriptionResult,
|
|
108
|
+
SpeechTranscriptionSegment,
|
|
108
109
|
GoogleCloudSTTOptions,
|
|
109
110
|
AudioData,
|
|
110
111
|
} from './GoogleCloudSTTProvider.js';
|