@framers/agentos-ext-streaming-stt-whisper 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/LICENSE ADDED
@@ -0,0 +1,23 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2025 Framers
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
22
+
23
+
package/SKILL.md ADDED
@@ -0,0 +1,62 @@
1
+ ---
2
+ name: streaming-stt-whisper
3
+ description: Chunked sliding-window streaming speech-to-text via OpenAI Whisper HTTP API
4
+ category: voice
5
+ ---
6
+
7
+ # Whisper Chunked Streaming STT
8
+
9
+ Streaming speech-to-text using OpenAI's Whisper model via the `/v1/audio/transcriptions` HTTP API.
10
+ Audio is accumulated in a sliding-window ring buffer and sent as 1-second WAV chunks with 200 ms
11
+ overlap for continuity.
12
+
13
+ ## Setup
14
+
15
+ Set `OPENAI_API_KEY` in your environment or agent secrets store.
16
+
17
+ ## Features
18
+
19
+ - Works with any OpenAI-compatible Whisper endpoint (e.g. local Faster-Whisper, Groq, OpenRouter)
20
+ - Sliding-window ring buffer: 1 s chunks, 200 ms overlap to prevent word boundary clipping
21
+ - Inline WAV encoding — zero runtime dependencies (no `ws`, no native binaries)
22
+ - Previous chunk transcript forwarded as `prompt` for cross-chunk continuity
23
+ - Simple RMS energy detector emits `speech_start` / `speech_end` events
24
+ - On fetch failure, emits `error` and continues processing — no session crash
25
+
26
+ ## Configuration
27
+
28
+ In `agent.config.json`:
29
+
30
+ ```json
31
+ {
32
+ "voice": {
33
+ "stt": "whisper"
34
+ }
35
+ }
36
+ ```
37
+
38
+ Provider-specific options via `providerOptions`:
39
+
40
+ ```json
41
+ {
42
+ "voice": {
43
+ "stt": "whisper",
44
+ "providerOptions": {
45
+ "model": "whisper-1",
46
+ "language": "en",
47
+ "baseUrl": "https://api.openai.com"
48
+ }
49
+ }
50
+ }
51
+ ```
52
+
53
+ ## Events
54
+
55
+ | Event | Payload | Description |
56
+ |------------------------|-------------------|-------------------------------------------------|
57
+ | `interim_transcript` | `TranscriptEvent` | Emitted after each chunk is transcribed |
58
+ | `final_transcript` | `TranscriptEvent` | Emitted after flush() completes |
59
+ | `speech_start` | — | RMS energy crossed threshold (0.01) |
60
+ | `speech_end` | — | RMS energy dropped below threshold |
61
+ | `error` | `Error` | Fetch failure (session continues) |
62
+ | `close` | — | Session fully terminated |
@@ -0,0 +1,93 @@
1
+ /**
2
+ * @file SlidingWindowBuffer.ts
3
+ * @description Ring buffer that accumulates Float32 audio frames into fixed-size chunks
4
+ * with configurable overlap between consecutive chunks.
5
+ *
6
+ * When {@link pushSamples} fills the internal buffer to {@link chunkSizeSamples}, it
7
+ * emits a `'chunk_ready'` event carrying the complete `Float32Array` chunk, copies the
8
+ * last {@link overlapSamples} samples to the head of the buffer as overlap context for
9
+ * the next chunk, and resets the write cursor accordingly.
10
+ *
11
+ * The overlap strategy prevents words straddling chunk boundaries from being silently
12
+ * dropped by the Whisper model.
13
+ *
14
+ * @module streaming-stt-whisper/SlidingWindowBuffer
15
+ */
16
+ import { EventEmitter } from 'node:events';
17
+ /**
18
+ * Default chunk size in samples.
19
+ * At 16 kHz this corresponds to exactly 1 second of mono audio.
20
+ */
21
+ export declare const DEFAULT_CHUNK_SIZE_SAMPLES = 16000;
22
+ /**
23
+ * Default overlap in samples carried forward to the next chunk.
24
+ * At 16 kHz this corresponds to 200 ms of audio context.
25
+ */
26
+ export declare const DEFAULT_OVERLAP_SAMPLES = 3200;
27
+ /** Event map for {@link SlidingWindowBuffer}. */
28
+ export interface SlidingWindowBufferEvents {
29
+ /** Emitted when a complete chunk of {@link chunkSizeSamples} is ready. */
30
+ chunk_ready: [chunk: Float32Array];
31
+ }
32
+ /**
33
+ * Ring-buffer that accumulates raw PCM samples and emits fixed-size audio
34
+ * chunks with configurable overlap.
35
+ *
36
+ * @example
37
+ * ```ts
38
+ * const buf = new SlidingWindowBuffer(16_000, 3_200);
39
+ * buf.on('chunk_ready', (chunk) => sendToWhisper(chunk));
40
+ *
41
+ * microphone.on('frame', (f) => buf.pushSamples(f.samples));
42
+ * await buf.flush(); // emit any remaining samples
43
+ * ```
44
+ */
45
+ export declare class SlidingWindowBuffer extends EventEmitter {
46
+ private readonly chunkSizeSamples;
47
+ private readonly overlapSamples;
48
+ /**
49
+ * Internal sample store. Sized to {@link chunkSizeSamples} so a single
50
+ * allocation is reused for the lifetime of the session.
51
+ */
52
+ private buffer;
53
+ /**
54
+ * Current write position within {@link buffer}.
55
+ * Always in the range `[0, chunkSizeSamples)`.
56
+ */
57
+ private writePos;
58
+ /**
59
+ * @param chunkSizeSamples - Number of samples per emitted chunk.
60
+ * Defaults to {@link DEFAULT_CHUNK_SIZE_SAMPLES} (1 s at 16 kHz).
61
+ * @param overlapSamples - Number of samples carried forward from each chunk
62
+ * to the start of the next. Must be less than `chunkSizeSamples`.
63
+ * Defaults to {@link DEFAULT_OVERLAP_SAMPLES} (200 ms at 16 kHz).
64
+ */
65
+ constructor(chunkSizeSamples?: number, overlapSamples?: number);
66
+ /**
67
+ * Append audio samples to the internal buffer.
68
+ *
69
+ * If the incoming batch causes the buffer to reach or exceed
70
+ * {@link chunkSizeSamples}, one or more `'chunk_ready'` events are emitted
71
+ * before the remainder is retained for the next chunk. Each chunk includes
72
+ * an overlap region copied from the tail of the previous chunk.
73
+ *
74
+ * @param samples - Float32 PCM samples to append.
75
+ */
76
+ pushSamples(samples: Float32Array): void;
77
+ /**
78
+ * Emit any samples currently held in the buffer as a final partial chunk.
79
+ *
80
+ * If the buffer contains no samples (`writePos === 0`), this is a no-op.
81
+ * After flushing, the buffer is reset to an empty state.
82
+ */
83
+ flush(): void;
84
+ /**
85
+ * Clear all buffered samples and reset the write cursor to zero.
86
+ *
87
+ * Does NOT emit a `'chunk_ready'` event — use {@link flush} for that.
88
+ */
89
+ reset(): void;
90
+ /** Number of samples currently held in the buffer. */
91
+ get bufferedSamples(): number;
92
+ }
93
+ //# sourceMappingURL=SlidingWindowBuffer.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"SlidingWindowBuffer.d.ts","sourceRoot":"","sources":["../src/SlidingWindowBuffer.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;GAcG;AAEH,OAAO,EAAE,YAAY,EAAE,MAAM,aAAa,CAAC;AAM3C;;;GAGG;AACH,eAAO,MAAM,0BAA0B,QAAS,CAAC;AAEjD;;;GAGG;AACH,eAAO,MAAM,uBAAuB,OAAQ,CAAC;AAM7C,iDAAiD;AACjD,MAAM,WAAW,yBAAyB;IACxC,0EAA0E;IAC1E,WAAW,EAAE,CAAC,KAAK,EAAE,YAAY,CAAC,CAAC;CACpC;AAMD;;;;;;;;;;;;GAYG;AACH,qBAAa,mBAAoB,SAAQ,YAAY;IA6BjD,OAAO,CAAC,QAAQ,CAAC,gBAAgB;IACjC,OAAO,CAAC,QAAQ,CAAC,cAAc;IAzBjC;;;OAGG;IACH,OAAO,CAAC,MAAM,CAAe;IAE7B;;;OAGG;IACH,OAAO,CAAC,QAAQ,CAAK;IAMrB;;;;;;OAMG;gBAEgB,gBAAgB,GAAE,MAAmC,EACrD,cAAc,GAAE,MAAgC;IAiBnE;;;;;;;;;OASG;IACH,WAAW,CAAC,OAAO,EAAE,YAAY,GAAG,IAAI;IAyBxC;;;;;OAKG;IACH,KAAK,IAAI,IAAI;IAQb;;;;OAIG;IACH,KAAK,IAAI,IAAI;IASb,sDAAsD;IACtD,IAAI,eAAe,IAAI,MAAM,CAE5B;CACF"}
@@ -0,0 +1,144 @@
1
+ /**
2
+ * @file SlidingWindowBuffer.ts
3
+ * @description Ring buffer that accumulates Float32 audio frames into fixed-size chunks
4
+ * with configurable overlap between consecutive chunks.
5
+ *
6
+ * When {@link pushSamples} fills the internal buffer to {@link chunkSizeSamples}, it
7
+ * emits a `'chunk_ready'` event carrying the complete `Float32Array` chunk, copies the
8
+ * last {@link overlapSamples} samples to the head of the buffer as overlap context for
9
+ * the next chunk, and resets the write cursor accordingly.
10
+ *
11
+ * The overlap strategy prevents words straddling chunk boundaries from being silently
12
+ * dropped by the Whisper model.
13
+ *
14
+ * @module streaming-stt-whisper/SlidingWindowBuffer
15
+ */
16
+ import { EventEmitter } from 'node:events';
17
+ // ---------------------------------------------------------------------------
18
+ // Constants
19
+ // ---------------------------------------------------------------------------
20
+ /**
21
+ * Default chunk size in samples.
22
+ * At 16 kHz this corresponds to exactly 1 second of mono audio.
23
+ */
24
+ export const DEFAULT_CHUNK_SIZE_SAMPLES = 16_000;
25
+ /**
26
+ * Default overlap in samples carried forward to the next chunk.
27
+ * At 16 kHz this corresponds to 200 ms of audio context.
28
+ */
29
+ export const DEFAULT_OVERLAP_SAMPLES = 3_200;
30
+ // ---------------------------------------------------------------------------
31
+ // Main class
32
+ // ---------------------------------------------------------------------------
33
+ /**
34
+ * Ring-buffer that accumulates raw PCM samples and emits fixed-size audio
35
+ * chunks with configurable overlap.
36
+ *
37
+ * @example
38
+ * ```ts
39
+ * const buf = new SlidingWindowBuffer(16_000, 3_200);
40
+ * buf.on('chunk_ready', (chunk) => sendToWhisper(chunk));
41
+ *
42
+ * microphone.on('frame', (f) => buf.pushSamples(f.samples));
43
+ * await buf.flush(); // emit any remaining samples
44
+ * ```
45
+ */
46
+ export class SlidingWindowBuffer extends EventEmitter {
47
+ chunkSizeSamples;
48
+ overlapSamples;
49
+ // -------------------------------------------------------------------------
50
+ // Private state
51
+ // -------------------------------------------------------------------------
52
+ /**
53
+ * Internal sample store. Sized to {@link chunkSizeSamples} so a single
54
+ * allocation is reused for the lifetime of the session.
55
+ */
56
+ buffer;
57
+ /**
58
+ * Current write position within {@link buffer}.
59
+ * Always in the range `[0, chunkSizeSamples)`.
60
+ */
61
+ writePos = 0;
62
+ // -------------------------------------------------------------------------
63
+ // Constructor
64
+ // -------------------------------------------------------------------------
65
+ /**
66
+ * @param chunkSizeSamples - Number of samples per emitted chunk.
67
+ * Defaults to {@link DEFAULT_CHUNK_SIZE_SAMPLES} (1 s at 16 kHz).
68
+ * @param overlapSamples - Number of samples carried forward from each chunk
69
+ * to the start of the next. Must be less than `chunkSizeSamples`.
70
+ * Defaults to {@link DEFAULT_OVERLAP_SAMPLES} (200 ms at 16 kHz).
71
+ */
72
+ constructor(chunkSizeSamples = DEFAULT_CHUNK_SIZE_SAMPLES, overlapSamples = DEFAULT_OVERLAP_SAMPLES) {
73
+ super();
74
+ this.chunkSizeSamples = chunkSizeSamples;
75
+ this.overlapSamples = overlapSamples;
76
+ if (overlapSamples >= chunkSizeSamples) {
77
+ throw new RangeError(`overlapSamples (${overlapSamples}) must be less than chunkSizeSamples (${chunkSizeSamples})`);
78
+ }
79
+ this.buffer = new Float32Array(chunkSizeSamples);
80
+ }
81
+ // -------------------------------------------------------------------------
82
+ // Public API
83
+ // -------------------------------------------------------------------------
84
+ /**
85
+ * Append audio samples to the internal buffer.
86
+ *
87
+ * If the incoming batch causes the buffer to reach or exceed
88
+ * {@link chunkSizeSamples}, one or more `'chunk_ready'` events are emitted
89
+ * before the remainder is retained for the next chunk. Each chunk includes
90
+ * an overlap region copied from the tail of the previous chunk.
91
+ *
92
+ * @param samples - Float32 PCM samples to append.
93
+ */
94
+ pushSamples(samples) {
95
+ let srcOffset = 0;
96
+ while (srcOffset < samples.length) {
97
+ // How many samples can we copy into the current chunk before it is full?
98
+ const spaceLeft = this.chunkSizeSamples - this.writePos;
99
+ const copyCount = Math.min(spaceLeft, samples.length - srcOffset);
100
+ this.buffer.set(samples.subarray(srcOffset, srcOffset + copyCount), this.writePos);
101
+ this.writePos += copyCount;
102
+ srcOffset += copyCount;
103
+ if (this.writePos >= this.chunkSizeSamples) {
104
+ // Chunk is full — emit a copy (not a reference to the internal buffer).
105
+ this.emit('chunk_ready', this.buffer.slice());
106
+ // Copy the last `overlapSamples` to the beginning of the buffer so that
107
+ // the next chunk begins with audio context from the previous boundary.
108
+ const overlapStart = this.chunkSizeSamples - this.overlapSamples;
109
+ this.buffer.copyWithin(0, overlapStart, this.chunkSizeSamples);
110
+ this.writePos = this.overlapSamples;
111
+ }
112
+ }
113
+ }
114
+ /**
115
+ * Emit any samples currently held in the buffer as a final partial chunk.
116
+ *
117
+ * If the buffer contains no samples (`writePos === 0`), this is a no-op.
118
+ * After flushing, the buffer is reset to an empty state.
119
+ */
120
+ flush() {
121
+ if (this.writePos === 0)
122
+ return;
123
+ // Emit only the samples that were actually written (not the whole buffer).
124
+ this.emit('chunk_ready', this.buffer.slice(0, this.writePos));
125
+ this.reset();
126
+ }
127
+ /**
128
+ * Clear all buffered samples and reset the write cursor to zero.
129
+ *
130
+ * Does NOT emit a `'chunk_ready'` event — use {@link flush} for that.
131
+ */
132
+ reset() {
133
+ this.buffer = new Float32Array(this.chunkSizeSamples);
134
+ this.writePos = 0;
135
+ }
136
+ // -------------------------------------------------------------------------
137
+ // Accessors (useful for testing)
138
+ // -------------------------------------------------------------------------
139
+ /** Number of samples currently held in the buffer. */
140
+ get bufferedSamples() {
141
+ return this.writePos;
142
+ }
143
+ }
144
+ //# sourceMappingURL=SlidingWindowBuffer.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"SlidingWindowBuffer.js","sourceRoot":"","sources":["../src/SlidingWindowBuffer.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;GAcG;AAEH,OAAO,EAAE,YAAY,EAAE,MAAM,aAAa,CAAC;AAE3C,8EAA8E;AAC9E,YAAY;AACZ,8EAA8E;AAE9E;;;GAGG;AACH,MAAM,CAAC,MAAM,0BAA0B,GAAG,MAAM,CAAC;AAEjD;;;GAGG;AACH,MAAM,CAAC,MAAM,uBAAuB,GAAG,KAAK,CAAC;AAY7C,8EAA8E;AAC9E,aAAa;AACb,8EAA8E;AAE9E;;;;;;;;;;;;GAYG;AACH,MAAM,OAAO,mBAAoB,SAAQ,YAAY;IA6BhC;IACA;IA7BnB,4EAA4E;IAC5E,gBAAgB;IAChB,4EAA4E;IAE5E;;;OAGG;IACK,MAAM,CAAe;IAE7B;;;OAGG;IACK,QAAQ,GAAG,CAAC,CAAC;IAErB,4EAA4E;IAC5E,cAAc;IACd,4EAA4E;IAE5E;;;;;;OAMG;IACH,YACmB,mBAA2B,0BAA0B,EACrD,iBAAyB,uBAAuB;QAEjE,KAAK,EAAE,CAAC;QAHS,qBAAgB,GAAhB,gBAAgB,CAAqC;QACrD,mBAAc,GAAd,cAAc,CAAkC;QAIjE,IAAI,cAAc,IAAI,gBAAgB,EAAE,CAAC;YACvC,MAAM,IAAI,UAAU,CAClB,mBAAmB,cAAc,yCAAyC,gBAAgB,GAAG,CAC9F,CAAC;QACJ,CAAC;QAED,IAAI,CAAC,MAAM,GAAG,IAAI,YAAY,CAAC,gBAAgB,CAAC,CAAC;IACnD,CAAC;IAED,4EAA4E;IAC5E,aAAa;IACb,4EAA4E;IAE5E;;;;;;;;;OASG;IACH,WAAW,CAAC,OAAqB;QAC/B,IAAI,SAAS,GAAG,CAAC,CAAC;QAElB,OAAO,SAAS,GAAG,OAAO,CAAC,MAAM,EAAE,CAAC;YAClC,yEAAyE;YACzE,MAAM,SAAS,GAAG,IAAI,CAAC,gBAAgB,GAAG,IAAI,CAAC,QAAQ,CAAC;YACxD,MAAM,SAAS,GAAG,IAAI,CAAC,GAAG,CAAC,SAAS,EAAE,OAAO,CAAC,MAAM,GAAG,SAAS,CAAC,CAAC;YAElE,IAAI,CAAC,MAAM,CAAC,GAAG,CAAC,OAAO,CAAC,QAAQ,CAAC,SAAS,EAAE,SAAS,GAAG,SAAS,CAAC,EAAE,IAAI,CAAC,QAAQ,CAAC,CAAC;YACnF,IAAI,CAAC,QAAQ,IAAI,SAAS,CAAC;YAC3B,SAAS,IAAI,SAAS,CAAC;YAEvB,IAAI,IAAI,CAAC,QAAQ,IAAI,IAAI,CAAC,gBAAgB,EAAE,CAAC;gBAC3C,wEAAwE;gBACxE,IAAI,CAAC,IAAI,CAAC,aAAa,EAAE,IAAI,CAAC,MAAM,CAAC,KAAK,EAAE,CAAC,CAAC;gBAE9C,wEAAwE;gBACxE,uEAAuE;gBACvE,MAAM,YAAY,GAAG,IAAI,CAAC,gBAAgB,GAAG,IAAI,CAAC,cAAc,CAAC;gBACjE,IAAI,CAAC,MAAM,CAAC,UAAU,CAAC,CAAC,EAAE,YAAY,EAAE,IAAI,CAAC,gBAAgB,CAAC,CAAC;gBAC/D,IAAI,CAAC,QAAQ,GAAG,IAAI,CAAC,cAAc,CAAC;YACtC,CAAC;QACH,CAAC;IACH,CAAC;IAED;;;;;OAKG;IACH,KAAK;QACH,IAAI,IAAI,CAAC,QAAQ,KAAK,CAAC;YAAE,OAAO;QAEhC,2EAA2E;QAC3E,IAAI,CAAC,IAAI,CAAC,aAAa,EAAE,IAAI,CAAC,MAAM,CAAC,KAAK,CAAC,CAAC,EAAE,IAAI,CAAC,QAAQ,CAAC,CAAC,CAAC;QAC9D,IAAI,CAAC,KAAK,EAAE,CAAC;IACf,CAAC;IAED;;;;OAIG;IACH,KAAK;QACH,IAAI,CAAC,MAAM,GAAG,IAAI,YAAY,CAAC,IAAI,CAAC,gBAAgB,CAAC,CAAC;QACtD,IAAI,CAAC,QAAQ,GAAG,CAAC,CAAC;IACpB,CAAC;IAED,4EAA4E;IAC5E,iCAAiC;IACjC,4EAA4E;IAE5E,sDAAsD;IACtD,IAAI,eAAe;QACjB,OAAO,IAAI,CAAC,QAAQ,CAAC;IACvB,CAAC;CACF"}
@@ -0,0 +1,123 @@
1
+ /**
2
+ * @file WhisperChunkSession.ts
3
+ * @description Active streaming STT session backed by the OpenAI Whisper HTTP API.
4
+ *
5
+ * {@link WhisperChunkSession} implements the `StreamingSTTSession` interface
6
+ * (EventEmitter-based) using a sliding-window ring buffer to accumulate audio
7
+ * into fixed-size chunks. Each chunk is encoded as a RIFF/WAV file and posted
8
+ * to the Whisper `/v1/audio/transcriptions` endpoint.
9
+ *
10
+ * ### Chunk lifecycle
11
+ * 1. {@link pushAudio} feeds PCM frames into the internal {@link SlidingWindowBuffer}.
12
+ * 2. When a full chunk is ready, {@link onChunkReady} is invoked.
13
+ * 3. The chunk is WAV-encoded and POST-ed to Whisper with multipart/form-data.
14
+ * 4. The parsed response is emitted as `'interim_transcript'`.
15
+ * 5. The response text becomes the `prompt` for the next API call (continuity).
16
+ *
17
+ * ### Speech detection
18
+ * A simple RMS energy threshold (`RMS_THRESHOLD = 0.01`) gates `speech_start`
19
+ * and `speech_end` events. This is not VAD — it is a lightweight proxy that
20
+ * avoids emitting events on pure-silence frames.
21
+ *
22
+ * ### Error resilience
23
+ * On fetch failure the error is emitted as an `'error'` event and the session
24
+ * continues processing subsequent chunks rather than terminating.
25
+ *
26
+ * @module streaming-stt-whisper/WhisperChunkSession
27
+ */
28
+ import { EventEmitter } from 'node:events';
29
+ import type { WhisperChunkedConfig, AudioFrame } from './types.js';
30
+ /**
31
+ * Active chunked Whisper STT session.
32
+ *
33
+ * Construct via {@link WhisperChunkedSTT.startSession} rather than directly.
34
+ *
35
+ * @example
36
+ * ```ts
37
+ * const session = new WhisperChunkSession({ apiKey: process.env.OPENAI_API_KEY! });
38
+ *
39
+ * session.on('interim_transcript', (evt) => console.log('chunk:', evt.text));
40
+ * session.on('final_transcript', (evt) => console.log('done:', evt.text));
41
+ *
42
+ * microphone.on('frame', (f) => session.pushAudio(f));
43
+ * await session.flush();
44
+ * session.close();
45
+ * ```
46
+ */
47
+ export declare class WhisperChunkSession extends EventEmitter {
48
+ /** Resolved Whisper API configuration. */
49
+ private readonly cfg;
50
+ /** Sliding-window ring buffer feeding audio chunks. */
51
+ private readonly slidingBuffer;
52
+ /** Whether {@link close} has been called. */
53
+ private closed;
54
+ /** Whether the session is currently in a speech segment (above RMS threshold). */
55
+ private inSpeech;
56
+ /**
57
+ * Transcript text from the most recently completed chunk.
58
+ * Forwarded as `prompt` to the next Whisper API call for cross-chunk
59
+ * lexical continuity.
60
+ */
61
+ private previousPrompt;
62
+ /**
63
+ * @param config - Whisper session configuration. `apiKey` is required.
64
+ */
65
+ constructor(config: WhisperChunkedConfig);
66
+ /**
67
+ * Feed a raw audio frame into the session.
68
+ *
69
+ * The frame's samples are appended to the sliding-window buffer. When the
70
+ * buffer accumulates a full chunk, {@link onChunkReady} is invoked.
71
+ *
72
+ * Speech detection is performed on every frame: if the RMS energy crosses
73
+ * {@link RMS_THRESHOLD}, `'speech_start'` is emitted on the first such frame
74
+ * and `'speech_end'` when energy falls back below the threshold.
75
+ *
76
+ * @param frame - Audio frame with normalised Float32 samples.
77
+ */
78
+ pushAudio(frame: AudioFrame): void;
79
+ /**
80
+ * Flush any remaining buffered samples, transcribe the final partial chunk,
81
+ * and emit `'final_transcript'`.
82
+ *
83
+ * Must be called when the audio stream ends to ensure the tail of the
84
+ * recording is not silently discarded.
85
+ *
86
+ * @returns Promise that resolves when the final Whisper request completes.
87
+ */
88
+ flush(): Promise<void>;
89
+ /**
90
+ * Immediately terminate the session.
91
+ *
92
+ * No further events are emitted after `close()`.
93
+ */
94
+ close(): void;
95
+ /**
96
+ * Transcribe a ready audio chunk by POST-ing it to the Whisper API.
97
+ *
98
+ * Steps:
99
+ * 1. Encode the Float32 chunk as a RIFF/WAV `ArrayBuffer`.
100
+ * 2. Build a multipart/form-data body with the WAV blob.
101
+ * 3. POST to `${baseUrl}/v1/audio/transcriptions`.
102
+ * 4. Parse the `verbose_json` response into a {@link TranscriptEvent}.
103
+ * 5. Emit `'interim_transcript'` and save the text as `previousPrompt`.
104
+ *
105
+ * On any fetch error, `'error'` is emitted and the method returns normally
106
+ * so that subsequent chunks are still processed.
107
+ *
108
+ * @param chunk - Float32 PCM samples for one audio chunk.
109
+ */
110
+ private onChunkReady;
111
+ /**
112
+ * Convert a Whisper `verbose_json` response into a {@link TranscriptEvent}.
113
+ *
114
+ * Word-level timestamps are sourced from the first segment's `words` array
115
+ * when available. Confidence is approximated from the segment `avg_logprob`
116
+ * (clamped to [0, 1]) when present; falls back to 1 otherwise.
117
+ *
118
+ * @param response - Parsed Whisper verbose_json response.
119
+ * @returns A `TranscriptEvent` suitable for emission.
120
+ */
121
+ private parseWhisperResponse;
122
+ }
123
+ //# sourceMappingURL=WhisperChunkSession.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"WhisperChunkSession.d.ts","sourceRoot":"","sources":["../src/WhisperChunkSession.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;GA0BG;AAEH,OAAO,EAAE,YAAY,EAAE,MAAM,aAAa,CAAC;AAE3C,OAAO,KAAK,EACV,oBAAoB,EACpB,UAAU,EAKX,MAAM,YAAY,CAAC;AA6IpB;;;;;;;;;;;;;;;;GAgBG;AACH,qBAAa,mBAAoB,SAAQ,YAAY;IAKnD,0CAA0C;IAC1C,OAAO,CAAC,QAAQ,CAAC,GAAG,CACqB;IAEzC,uDAAuD;IACvD,OAAO,CAAC,QAAQ,CAAC,aAAa,CAAsB;IAEpD,6CAA6C;IAC7C,OAAO,CAAC,MAAM,CAAS;IAEvB,kFAAkF;IAClF,OAAO,CAAC,QAAQ,CAAS;IAEzB;;;;OAIG;IACH,OAAO,CAAC,cAAc,CAAqB;IAM3C;;OAEG;gBACS,MAAM,EAAE,oBAAoB;IA4BxC;;;;;;;;;;;OAWG;IACH,SAAS,CAAC,KAAK,EAAE,UAAU,GAAG,IAAI;IAgBlC;;;;;;;;OAQG;IACG,KAAK,IAAI,OAAO,CAAC,IAAI,CAAC;IAwB5B;;;;OAIG;IACH,KAAK,IAAI,IAAI;IASb;;;;;;;;;;;;;;OAcG;YACW,YAAY;IAgD1B;;;;;;;;;OASG;IACH,OAAO,CAAC,oBAAoB;CAgC7B"}