@framers/agentos-ext-streaming-stt-whisper 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +23 -0
- package/SKILL.md +62 -0
- package/dist/SlidingWindowBuffer.d.ts +93 -0
- package/dist/SlidingWindowBuffer.d.ts.map +1 -0
- package/dist/SlidingWindowBuffer.js +144 -0
- package/dist/SlidingWindowBuffer.js.map +1 -0
- package/dist/WhisperChunkSession.d.ts +123 -0
- package/dist/WhisperChunkSession.d.ts.map +1 -0
- package/dist/WhisperChunkSession.js +371 -0
- package/dist/WhisperChunkSession.js.map +1 -0
- package/dist/WhisperChunkedSTT.d.ts +72 -0
- package/dist/WhisperChunkedSTT.d.ts.map +1 -0
- package/dist/WhisperChunkedSTT.js +89 -0
- package/dist/WhisperChunkedSTT.js.map +1 -0
- package/dist/index.d.ts +70 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +78 -0
- package/dist/index.js.map +1 -0
- package/dist/types.d.ts +102 -0
- package/dist/types.d.ts.map +1 -0
- package/dist/types.js +11 -0
- package/dist/types.js.map +1 -0
- package/manifest.json +8 -0
- package/package.json +43 -0
- package/src/SlidingWindowBuffer.ts +176 -0
- package/src/WhisperChunkSession.ts +436 -0
- package/src/WhisperChunkedSTT.ts +111 -0
- package/src/index.ts +115 -0
- package/src/types.ts +122 -0
package/src/types.ts
ADDED
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @file types.ts
|
|
3
|
+
* @description Whisper-specific configuration types for the chunked streaming STT extension pack.
|
|
4
|
+
*
|
|
5
|
+
* These types define the configuration for the sliding-window Whisper adapter that
|
|
6
|
+
* accumulates audio into 1-second chunks and sends them to the Whisper HTTP API.
|
|
7
|
+
*
|
|
8
|
+
* @module streaming-stt-whisper/types
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* Configuration for the Whisper chunked streaming STT session.
|
|
13
|
+
*
|
|
14
|
+
* All fields except `apiKey` are optional — sensible defaults are applied.
|
|
15
|
+
*/
|
|
16
|
+
export interface WhisperChunkedConfig {
|
|
17
|
+
/**
|
|
18
|
+
* OpenAI API key (or compatible provider key).
|
|
19
|
+
* Read from `OPENAI_API_KEY` when constructed via {@link createExtensionPack}.
|
|
20
|
+
*/
|
|
21
|
+
apiKey: string;
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* Base URL for the Whisper API endpoint.
|
|
25
|
+
* Override to use a self-hosted or compatible provider (e.g. Groq, local Faster-Whisper).
|
|
26
|
+
* @defaultValue 'https://api.openai.com'
|
|
27
|
+
*/
|
|
28
|
+
baseUrl?: string;
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* Whisper model name.
|
|
32
|
+
* @defaultValue 'whisper-1'
|
|
33
|
+
* @see {@link https://platform.openai.com/docs/models/whisper}
|
|
34
|
+
*/
|
|
35
|
+
model?: string;
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* BCP-47 language hint (e.g. `'en'`, `'fr'`, `'de'`).
|
|
39
|
+
* When omitted Whisper auto-detects the language.
|
|
40
|
+
*/
|
|
41
|
+
language?: string;
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Optional initial prompt to bias the first chunk's transcription.
|
|
45
|
+
* Subsequent chunks automatically receive the previous chunk's transcript as prompt.
|
|
46
|
+
* @see {@link https://platform.openai.com/docs/guides/speech-to-text/prompting}
|
|
47
|
+
*/
|
|
48
|
+
prompt?: string;
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* Size of each audio chunk in samples (at 16 kHz).
|
|
52
|
+
* @defaultValue 16000 (1 second at 16 kHz)
|
|
53
|
+
*/
|
|
54
|
+
chunkSizeSamples?: number;
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Number of samples to carry forward from each chunk as overlap.
|
|
58
|
+
* Prevents words at chunk boundaries from being silently dropped.
|
|
59
|
+
* @defaultValue 3200 (200 ms at 16 kHz)
|
|
60
|
+
*/
|
|
61
|
+
overlapSamples?: number;
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
// ---------------------------------------------------------------------------
|
|
65
|
+
// Voice pipeline shape mirrors (no runtime dep on @framers/agentos at test time)
|
|
66
|
+
// ---------------------------------------------------------------------------
|
|
67
|
+
|
|
68
|
+
/** Minimal AudioFrame shape — mirrors packages/agentos/src/voice-pipeline/types.ts */
|
|
69
|
+
export interface AudioFrame {
|
|
70
|
+
samples: Float32Array;
|
|
71
|
+
sampleRate: number;
|
|
72
|
+
timestamp: number;
|
|
73
|
+
speakerHint?: string;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/** A single recognised word with timing metadata. */
|
|
77
|
+
export interface TranscriptWord {
|
|
78
|
+
word: string;
|
|
79
|
+
start: number;
|
|
80
|
+
end: number;
|
|
81
|
+
confidence: number;
|
|
82
|
+
speaker?: string;
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/** A transcription result emitted by the session. */
|
|
86
|
+
export interface TranscriptEvent {
|
|
87
|
+
text: string;
|
|
88
|
+
confidence: number;
|
|
89
|
+
words: TranscriptWord[];
|
|
90
|
+
isFinal: boolean;
|
|
91
|
+
durationMs?: number;
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
// ---------------------------------------------------------------------------
|
|
95
|
+
// Whisper verbose_json response shape
|
|
96
|
+
// ---------------------------------------------------------------------------
|
|
97
|
+
|
|
98
|
+
/** A single segment from Whisper's verbose_json response. */
|
|
99
|
+
export interface WhisperSegment {
|
|
100
|
+
id: number;
|
|
101
|
+
start: number;
|
|
102
|
+
end: number;
|
|
103
|
+
text: string;
|
|
104
|
+
avg_logprob?: number;
|
|
105
|
+
words?: WhisperWord[];
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
/** A word-level entry from Whisper's verbose_json response (requires word timestamps). */
|
|
109
|
+
export interface WhisperWord {
|
|
110
|
+
word: string;
|
|
111
|
+
start: number;
|
|
112
|
+
end: number;
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/** Top-level shape of the Whisper verbose_json transcription response. */
|
|
116
|
+
export interface WhisperTranscriptionResponse {
|
|
117
|
+
task?: string;
|
|
118
|
+
language?: string;
|
|
119
|
+
duration?: number;
|
|
120
|
+
text: string;
|
|
121
|
+
segments?: WhisperSegment[];
|
|
122
|
+
}
|