@framers/agentos-ext-streaming-stt-whisper 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +23 -0
- package/SKILL.md +62 -0
- package/dist/SlidingWindowBuffer.d.ts +93 -0
- package/dist/SlidingWindowBuffer.d.ts.map +1 -0
- package/dist/SlidingWindowBuffer.js +144 -0
- package/dist/SlidingWindowBuffer.js.map +1 -0
- package/dist/WhisperChunkSession.d.ts +123 -0
- package/dist/WhisperChunkSession.d.ts.map +1 -0
- package/dist/WhisperChunkSession.js +371 -0
- package/dist/WhisperChunkSession.js.map +1 -0
- package/dist/WhisperChunkedSTT.d.ts +72 -0
- package/dist/WhisperChunkedSTT.d.ts.map +1 -0
- package/dist/WhisperChunkedSTT.js +89 -0
- package/dist/WhisperChunkedSTT.js.map +1 -0
- package/dist/index.d.ts +70 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +78 -0
- package/dist/index.js.map +1 -0
- package/dist/types.d.ts +102 -0
- package/dist/types.d.ts.map +1 -0
- package/dist/types.js +11 -0
- package/dist/types.js.map +1 -0
- package/manifest.json +8 -0
- package/package.json +43 -0
- package/src/SlidingWindowBuffer.ts +176 -0
- package/src/WhisperChunkSession.ts +436 -0
- package/src/WhisperChunkedSTT.ts +111 -0
- package/src/index.ts +115 -0
- package/src/types.ts +122 -0
package/LICENSE
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2025 Framers
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
22
|
+
|
|
23
|
+
|
package/SKILL.md
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: streaming-stt-whisper
|
|
3
|
+
description: Chunked sliding-window streaming speech-to-text via OpenAI Whisper HTTP API
|
|
4
|
+
category: voice
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
# Whisper Chunked Streaming STT
|
|
8
|
+
|
|
9
|
+
Streaming speech-to-text using OpenAI's Whisper model via the `/v1/audio/transcriptions` HTTP API.
|
|
10
|
+
Audio is accumulated in a sliding-window ring buffer and sent as 1-second WAV chunks with 200 ms
|
|
11
|
+
overlap for continuity.
|
|
12
|
+
|
|
13
|
+
## Setup
|
|
14
|
+
|
|
15
|
+
Set `OPENAI_API_KEY` in your environment or agent secrets store.
|
|
16
|
+
|
|
17
|
+
## Features
|
|
18
|
+
|
|
19
|
+
- Works with any OpenAI-compatible Whisper endpoint (e.g. local Faster-Whisper, Groq, OpenRouter)
|
|
20
|
+
- Sliding-window ring buffer: 1 s chunks, 200 ms overlap to prevent word boundary clipping
|
|
21
|
+
- Inline WAV encoding — zero runtime dependencies (no `ws`, no native binaries)
|
|
22
|
+
- Previous chunk transcript forwarded as `prompt` for cross-chunk continuity
|
|
23
|
+
- Simple RMS energy detector emits `speech_start` / `speech_end` events
|
|
24
|
+
- On fetch failure, emits `error` and continues processing — no session crash
|
|
25
|
+
|
|
26
|
+
## Configuration
|
|
27
|
+
|
|
28
|
+
In `agent.config.json`:
|
|
29
|
+
|
|
30
|
+
```json
|
|
31
|
+
{
|
|
32
|
+
"voice": {
|
|
33
|
+
"stt": "whisper"
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
Provider-specific options via `providerOptions`:
|
|
39
|
+
|
|
40
|
+
```json
|
|
41
|
+
{
|
|
42
|
+
"voice": {
|
|
43
|
+
"stt": "whisper",
|
|
44
|
+
"providerOptions": {
|
|
45
|
+
"model": "whisper-1",
|
|
46
|
+
"language": "en",
|
|
47
|
+
"baseUrl": "https://api.openai.com"
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
## Events
|
|
54
|
+
|
|
55
|
+
| Event | Payload | Description |
|
|
56
|
+
|------------------------|-------------------|-------------------------------------------------|
|
|
57
|
+
| `interim_transcript` | `TranscriptEvent` | Emitted after each chunk is transcribed |
|
|
58
|
+
| `final_transcript` | `TranscriptEvent` | Emitted after flush() completes |
|
|
59
|
+
| `speech_start` | — | RMS energy crossed threshold (0.01) |
|
|
60
|
+
| `speech_end` | — | RMS energy dropped below threshold |
|
|
61
|
+
| `error` | `Error` | Fetch failure (session continues) |
|
|
62
|
+
| `close` | — | Session fully terminated |
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @file SlidingWindowBuffer.ts
|
|
3
|
+
* @description Ring buffer that accumulates Float32 audio frames into fixed-size chunks
|
|
4
|
+
* with configurable overlap between consecutive chunks.
|
|
5
|
+
*
|
|
6
|
+
* When {@link pushSamples} fills the internal buffer to {@link chunkSizeSamples}, it
|
|
7
|
+
* emits a `'chunk_ready'` event carrying the complete `Float32Array` chunk, copies the
|
|
8
|
+
* last {@link overlapSamples} samples to the head of the buffer as overlap context for
|
|
9
|
+
* the next chunk, and resets the write cursor accordingly.
|
|
10
|
+
*
|
|
11
|
+
* The overlap strategy prevents words straddling chunk boundaries from being silently
|
|
12
|
+
* dropped by the Whisper model.
|
|
13
|
+
*
|
|
14
|
+
* @module streaming-stt-whisper/SlidingWindowBuffer
|
|
15
|
+
*/
|
|
16
|
+
import { EventEmitter } from 'node:events';
|
|
17
|
+
/**
|
|
18
|
+
* Default chunk size in samples.
|
|
19
|
+
* At 16 kHz this corresponds to exactly 1 second of mono audio.
|
|
20
|
+
*/
|
|
21
|
+
export declare const DEFAULT_CHUNK_SIZE_SAMPLES = 16000;
|
|
22
|
+
/**
|
|
23
|
+
* Default overlap in samples carried forward to the next chunk.
|
|
24
|
+
* At 16 kHz this corresponds to 200 ms of audio context.
|
|
25
|
+
*/
|
|
26
|
+
export declare const DEFAULT_OVERLAP_SAMPLES = 3200;
|
|
27
|
+
/** Event map for {@link SlidingWindowBuffer}. */
|
|
28
|
+
export interface SlidingWindowBufferEvents {
|
|
29
|
+
/** Emitted when a complete chunk of {@link chunkSizeSamples} is ready. */
|
|
30
|
+
chunk_ready: [chunk: Float32Array];
|
|
31
|
+
}
|
|
32
|
+
/**
|
|
33
|
+
* Ring-buffer that accumulates raw PCM samples and emits fixed-size audio
|
|
34
|
+
* chunks with configurable overlap.
|
|
35
|
+
*
|
|
36
|
+
* @example
|
|
37
|
+
* ```ts
|
|
38
|
+
* const buf = new SlidingWindowBuffer(16_000, 3_200);
|
|
39
|
+
* buf.on('chunk_ready', (chunk) => sendToWhisper(chunk));
|
|
40
|
+
*
|
|
41
|
+
* microphone.on('frame', (f) => buf.pushSamples(f.samples));
|
|
42
|
+
* await buf.flush(); // emit any remaining samples
|
|
43
|
+
* ```
|
|
44
|
+
*/
|
|
45
|
+
export declare class SlidingWindowBuffer extends EventEmitter {
|
|
46
|
+
private readonly chunkSizeSamples;
|
|
47
|
+
private readonly overlapSamples;
|
|
48
|
+
/**
|
|
49
|
+
* Internal sample store. Sized to {@link chunkSizeSamples} so a single
|
|
50
|
+
* allocation is reused for the lifetime of the session.
|
|
51
|
+
*/
|
|
52
|
+
private buffer;
|
|
53
|
+
/**
|
|
54
|
+
* Current write position within {@link buffer}.
|
|
55
|
+
* Always in the range `[0, chunkSizeSamples)`.
|
|
56
|
+
*/
|
|
57
|
+
private writePos;
|
|
58
|
+
/**
|
|
59
|
+
* @param chunkSizeSamples - Number of samples per emitted chunk.
|
|
60
|
+
* Defaults to {@link DEFAULT_CHUNK_SIZE_SAMPLES} (1 s at 16 kHz).
|
|
61
|
+
* @param overlapSamples - Number of samples carried forward from each chunk
|
|
62
|
+
* to the start of the next. Must be less than `chunkSizeSamples`.
|
|
63
|
+
* Defaults to {@link DEFAULT_OVERLAP_SAMPLES} (200 ms at 16 kHz).
|
|
64
|
+
*/
|
|
65
|
+
constructor(chunkSizeSamples?: number, overlapSamples?: number);
|
|
66
|
+
/**
|
|
67
|
+
* Append audio samples to the internal buffer.
|
|
68
|
+
*
|
|
69
|
+
* If the incoming batch causes the buffer to reach or exceed
|
|
70
|
+
* {@link chunkSizeSamples}, one or more `'chunk_ready'` events are emitted
|
|
71
|
+
* before the remainder is retained for the next chunk. Each chunk includes
|
|
72
|
+
* an overlap region copied from the tail of the previous chunk.
|
|
73
|
+
*
|
|
74
|
+
* @param samples - Float32 PCM samples to append.
|
|
75
|
+
*/
|
|
76
|
+
pushSamples(samples: Float32Array): void;
|
|
77
|
+
/**
|
|
78
|
+
* Emit any samples currently held in the buffer as a final partial chunk.
|
|
79
|
+
*
|
|
80
|
+
* If the buffer contains no samples (`writePos === 0`), this is a no-op.
|
|
81
|
+
* After flushing, the buffer is reset to an empty state.
|
|
82
|
+
*/
|
|
83
|
+
flush(): void;
|
|
84
|
+
/**
|
|
85
|
+
* Clear all buffered samples and reset the write cursor to zero.
|
|
86
|
+
*
|
|
87
|
+
* Does NOT emit a `'chunk_ready'` event — use {@link flush} for that.
|
|
88
|
+
*/
|
|
89
|
+
reset(): void;
|
|
90
|
+
/** Number of samples currently held in the buffer. */
|
|
91
|
+
get bufferedSamples(): number;
|
|
92
|
+
}
|
|
93
|
+
//# sourceMappingURL=SlidingWindowBuffer.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"SlidingWindowBuffer.d.ts","sourceRoot":"","sources":["../src/SlidingWindowBuffer.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;GAcG;AAEH,OAAO,EAAE,YAAY,EAAE,MAAM,aAAa,CAAC;AAM3C;;;GAGG;AACH,eAAO,MAAM,0BAA0B,QAAS,CAAC;AAEjD;;;GAGG;AACH,eAAO,MAAM,uBAAuB,OAAQ,CAAC;AAM7C,iDAAiD;AACjD,MAAM,WAAW,yBAAyB;IACxC,0EAA0E;IAC1E,WAAW,EAAE,CAAC,KAAK,EAAE,YAAY,CAAC,CAAC;CACpC;AAMD;;;;;;;;;;;;GAYG;AACH,qBAAa,mBAAoB,SAAQ,YAAY;IA6BjD,OAAO,CAAC,QAAQ,CAAC,gBAAgB;IACjC,OAAO,CAAC,QAAQ,CAAC,cAAc;IAzBjC;;;OAGG;IACH,OAAO,CAAC,MAAM,CAAe;IAE7B;;;OAGG;IACH,OAAO,CAAC,QAAQ,CAAK;IAMrB;;;;;;OAMG;gBAEgB,gBAAgB,GAAE,MAAmC,EACrD,cAAc,GAAE,MAAgC;IAiBnE;;;;;;;;;OASG;IACH,WAAW,CAAC,OAAO,EAAE,YAAY,GAAG,IAAI;IAyBxC;;;;;OAKG;IACH,KAAK,IAAI,IAAI;IAQb;;;;OAIG;IACH,KAAK,IAAI,IAAI;IASb,sDAAsD;IACtD,IAAI,eAAe,IAAI,MAAM,CAE5B;CACF"}
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @file SlidingWindowBuffer.ts
|
|
3
|
+
* @description Ring buffer that accumulates Float32 audio frames into fixed-size chunks
|
|
4
|
+
* with configurable overlap between consecutive chunks.
|
|
5
|
+
*
|
|
6
|
+
* When {@link pushSamples} fills the internal buffer to {@link chunkSizeSamples}, it
|
|
7
|
+
* emits a `'chunk_ready'` event carrying the complete `Float32Array` chunk, copies the
|
|
8
|
+
* last {@link overlapSamples} samples to the head of the buffer as overlap context for
|
|
9
|
+
* the next chunk, and resets the write cursor accordingly.
|
|
10
|
+
*
|
|
11
|
+
* The overlap strategy prevents words straddling chunk boundaries from being silently
|
|
12
|
+
* dropped by the Whisper model.
|
|
13
|
+
*
|
|
14
|
+
* @module streaming-stt-whisper/SlidingWindowBuffer
|
|
15
|
+
*/
|
|
16
|
+
import { EventEmitter } from 'node:events';
|
|
17
|
+
// ---------------------------------------------------------------------------
|
|
18
|
+
// Constants
|
|
19
|
+
// ---------------------------------------------------------------------------
|
|
20
|
+
/**
|
|
21
|
+
* Default chunk size in samples.
|
|
22
|
+
* At 16 kHz this corresponds to exactly 1 second of mono audio.
|
|
23
|
+
*/
|
|
24
|
+
export const DEFAULT_CHUNK_SIZE_SAMPLES = 16_000;
|
|
25
|
+
/**
|
|
26
|
+
* Default overlap in samples carried forward to the next chunk.
|
|
27
|
+
* At 16 kHz this corresponds to 200 ms of audio context.
|
|
28
|
+
*/
|
|
29
|
+
export const DEFAULT_OVERLAP_SAMPLES = 3_200;
|
|
30
|
+
// ---------------------------------------------------------------------------
|
|
31
|
+
// Main class
|
|
32
|
+
// ---------------------------------------------------------------------------
|
|
33
|
+
/**
|
|
34
|
+
* Ring-buffer that accumulates raw PCM samples and emits fixed-size audio
|
|
35
|
+
* chunks with configurable overlap.
|
|
36
|
+
*
|
|
37
|
+
* @example
|
|
38
|
+
* ```ts
|
|
39
|
+
* const buf = new SlidingWindowBuffer(16_000, 3_200);
|
|
40
|
+
* buf.on('chunk_ready', (chunk) => sendToWhisper(chunk));
|
|
41
|
+
*
|
|
42
|
+
* microphone.on('frame', (f) => buf.pushSamples(f.samples));
|
|
43
|
+
* await buf.flush(); // emit any remaining samples
|
|
44
|
+
* ```
|
|
45
|
+
*/
|
|
46
|
+
export class SlidingWindowBuffer extends EventEmitter {
|
|
47
|
+
chunkSizeSamples;
|
|
48
|
+
overlapSamples;
|
|
49
|
+
// -------------------------------------------------------------------------
|
|
50
|
+
// Private state
|
|
51
|
+
// -------------------------------------------------------------------------
|
|
52
|
+
/**
|
|
53
|
+
* Internal sample store. Sized to {@link chunkSizeSamples} so a single
|
|
54
|
+
* allocation is reused for the lifetime of the session.
|
|
55
|
+
*/
|
|
56
|
+
buffer;
|
|
57
|
+
/**
|
|
58
|
+
* Current write position within {@link buffer}.
|
|
59
|
+
* Always in the range `[0, chunkSizeSamples)`.
|
|
60
|
+
*/
|
|
61
|
+
writePos = 0;
|
|
62
|
+
// -------------------------------------------------------------------------
|
|
63
|
+
// Constructor
|
|
64
|
+
// -------------------------------------------------------------------------
|
|
65
|
+
/**
|
|
66
|
+
* @param chunkSizeSamples - Number of samples per emitted chunk.
|
|
67
|
+
* Defaults to {@link DEFAULT_CHUNK_SIZE_SAMPLES} (1 s at 16 kHz).
|
|
68
|
+
* @param overlapSamples - Number of samples carried forward from each chunk
|
|
69
|
+
* to the start of the next. Must be less than `chunkSizeSamples`.
|
|
70
|
+
* Defaults to {@link DEFAULT_OVERLAP_SAMPLES} (200 ms at 16 kHz).
|
|
71
|
+
*/
|
|
72
|
+
constructor(chunkSizeSamples = DEFAULT_CHUNK_SIZE_SAMPLES, overlapSamples = DEFAULT_OVERLAP_SAMPLES) {
|
|
73
|
+
super();
|
|
74
|
+
this.chunkSizeSamples = chunkSizeSamples;
|
|
75
|
+
this.overlapSamples = overlapSamples;
|
|
76
|
+
if (overlapSamples >= chunkSizeSamples) {
|
|
77
|
+
throw new RangeError(`overlapSamples (${overlapSamples}) must be less than chunkSizeSamples (${chunkSizeSamples})`);
|
|
78
|
+
}
|
|
79
|
+
this.buffer = new Float32Array(chunkSizeSamples);
|
|
80
|
+
}
|
|
81
|
+
// -------------------------------------------------------------------------
|
|
82
|
+
// Public API
|
|
83
|
+
// -------------------------------------------------------------------------
|
|
84
|
+
/**
|
|
85
|
+
* Append audio samples to the internal buffer.
|
|
86
|
+
*
|
|
87
|
+
* If the incoming batch causes the buffer to reach or exceed
|
|
88
|
+
* {@link chunkSizeSamples}, one or more `'chunk_ready'` events are emitted
|
|
89
|
+
* before the remainder is retained for the next chunk. Each chunk includes
|
|
90
|
+
* an overlap region copied from the tail of the previous chunk.
|
|
91
|
+
*
|
|
92
|
+
* @param samples - Float32 PCM samples to append.
|
|
93
|
+
*/
|
|
94
|
+
pushSamples(samples) {
|
|
95
|
+
let srcOffset = 0;
|
|
96
|
+
while (srcOffset < samples.length) {
|
|
97
|
+
// How many samples can we copy into the current chunk before it is full?
|
|
98
|
+
const spaceLeft = this.chunkSizeSamples - this.writePos;
|
|
99
|
+
const copyCount = Math.min(spaceLeft, samples.length - srcOffset);
|
|
100
|
+
this.buffer.set(samples.subarray(srcOffset, srcOffset + copyCount), this.writePos);
|
|
101
|
+
this.writePos += copyCount;
|
|
102
|
+
srcOffset += copyCount;
|
|
103
|
+
if (this.writePos >= this.chunkSizeSamples) {
|
|
104
|
+
// Chunk is full — emit a copy (not a reference to the internal buffer).
|
|
105
|
+
this.emit('chunk_ready', this.buffer.slice());
|
|
106
|
+
// Copy the last `overlapSamples` to the beginning of the buffer so that
|
|
107
|
+
// the next chunk begins with audio context from the previous boundary.
|
|
108
|
+
const overlapStart = this.chunkSizeSamples - this.overlapSamples;
|
|
109
|
+
this.buffer.copyWithin(0, overlapStart, this.chunkSizeSamples);
|
|
110
|
+
this.writePos = this.overlapSamples;
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
}
|
|
114
|
+
/**
|
|
115
|
+
* Emit any samples currently held in the buffer as a final partial chunk.
|
|
116
|
+
*
|
|
117
|
+
* If the buffer contains no samples (`writePos === 0`), this is a no-op.
|
|
118
|
+
* After flushing, the buffer is reset to an empty state.
|
|
119
|
+
*/
|
|
120
|
+
flush() {
|
|
121
|
+
if (this.writePos === 0)
|
|
122
|
+
return;
|
|
123
|
+
// Emit only the samples that were actually written (not the whole buffer).
|
|
124
|
+
this.emit('chunk_ready', this.buffer.slice(0, this.writePos));
|
|
125
|
+
this.reset();
|
|
126
|
+
}
|
|
127
|
+
/**
|
|
128
|
+
* Clear all buffered samples and reset the write cursor to zero.
|
|
129
|
+
*
|
|
130
|
+
* Does NOT emit a `'chunk_ready'` event — use {@link flush} for that.
|
|
131
|
+
*/
|
|
132
|
+
reset() {
|
|
133
|
+
this.buffer = new Float32Array(this.chunkSizeSamples);
|
|
134
|
+
this.writePos = 0;
|
|
135
|
+
}
|
|
136
|
+
// -------------------------------------------------------------------------
|
|
137
|
+
// Accessors (useful for testing)
|
|
138
|
+
// -------------------------------------------------------------------------
|
|
139
|
+
/** Number of samples currently held in the buffer. */
|
|
140
|
+
get bufferedSamples() {
|
|
141
|
+
return this.writePos;
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
//# sourceMappingURL=SlidingWindowBuffer.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"SlidingWindowBuffer.js","sourceRoot":"","sources":["../src/SlidingWindowBuffer.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;GAcG;AAEH,OAAO,EAAE,YAAY,EAAE,MAAM,aAAa,CAAC;AAE3C,8EAA8E;AAC9E,YAAY;AACZ,8EAA8E;AAE9E;;;GAGG;AACH,MAAM,CAAC,MAAM,0BAA0B,GAAG,MAAM,CAAC;AAEjD;;;GAGG;AACH,MAAM,CAAC,MAAM,uBAAuB,GAAG,KAAK,CAAC;AAY7C,8EAA8E;AAC9E,aAAa;AACb,8EAA8E;AAE9E;;;;;;;;;;;;GAYG;AACH,MAAM,OAAO,mBAAoB,SAAQ,YAAY;IA6BhC;IACA;IA7BnB,4EAA4E;IAC5E,gBAAgB;IAChB,4EAA4E;IAE5E;;;OAGG;IACK,MAAM,CAAe;IAE7B;;;OAGG;IACK,QAAQ,GAAG,CAAC,CAAC;IAErB,4EAA4E;IAC5E,cAAc;IACd,4EAA4E;IAE5E;;;;;;OAMG;IACH,YACmB,mBAA2B,0BAA0B,EACrD,iBAAyB,uBAAuB;QAEjE,KAAK,EAAE,CAAC;QAHS,qBAAgB,GAAhB,gBAAgB,CAAqC;QACrD,mBAAc,GAAd,cAAc,CAAkC;QAIjE,IAAI,cAAc,IAAI,gBAAgB,EAAE,CAAC;YACvC,MAAM,IAAI,UAAU,CAClB,mBAAmB,cAAc,yCAAyC,gBAAgB,GAAG,CAC9F,CAAC;QACJ,CAAC;QAED,IAAI,CAAC,MAAM,GAAG,IAAI,YAAY,CAAC,gBAAgB,CAAC,CAAC;IACnD,CAAC;IAED,4EAA4E;IAC5E,aAAa;IACb,4EAA4E;IAE5E;;;;;;;;;OASG;IACH,WAAW,CAAC,OAAqB;QAC/B,IAAI,SAAS,GAAG,CAAC,CAAC;QAElB,OAAO,SAAS,GAAG,OAAO,CAAC,MAAM,EAAE,CAAC;YAClC,yEAAyE;YACzE,MAAM,SAAS,GAAG,IAAI,CAAC,gBAAgB,GAAG,IAAI,CAAC,QAAQ,CAAC;YACxD,MAAM,SAAS,GAAG,IAAI,CAAC,GAAG,CAAC,SAAS,EAAE,OAAO,CAAC,MAAM,GAAG,SAAS,CAAC,CAAC;YAElE,IAAI,CAAC,MAAM,CAAC,GAAG,CAAC,OAAO,CAAC,QAAQ,CAAC,SAAS,EAAE,SAAS,GAAG,SAAS,CAAC,EAAE,IAAI,CAAC,QAAQ,CAAC,CAAC;YACnF,IAAI,CAAC,QAAQ,IAAI,SAAS,CAAC;YAC3B,SAAS,IAAI,SAAS,CAAC;YAEvB,IAAI,IAAI,CAAC,QAAQ,IAAI,IAAI,CAAC,gBAAgB,EAAE,CAAC;gBAC3C,wEAAwE;gBACxE,IAAI,CAAC,IAAI,CAAC,aAAa,EAAE,IAAI,CAAC,MAAM,CAAC,KAAK,EAAE,CAAC,CAAC;gBAE9C,wEAAwE;gBACxE,uEAAuE;gBACvE,MAAM,YAAY,GAAG,IAAI,CAAC,gBAAgB,GAAG,IAAI,CAAC,cAAc,CAAC;gBACjE,IAAI,CAAC,MAAM,CAAC,UAAU,CAAC,CAAC,EAAE,YAAY,EAAE,IAAI,CAAC,gBAAgB,CAAC,CAAC;gBAC/D,IAAI,CAAC,QAAQ,GAAG,IAAI,CAAC,cAAc,CAAC;YACtC,CAAC;QACH,CAAC;IACH,CAAC;IAED;;;;;OAKG;IACH,KAAK;QACH,IAAI,IAAI,CAAC,QAAQ,KAAK,CAAC;YAAE,OAAO;QAEhC,2EAA2E;QAC3E,IAAI,CAAC,IAAI,CAAC,aAAa,EAAE,IAAI,CAAC,MAAM,CAAC,KAAK,CAAC,CAAC,EAAE,IAAI,CAAC,QAAQ,CAAC,CAAC,CAAC;QAC9D,IAAI,CAAC,KAAK,EAAE,CAAC;IACf,CAAC;IAED;;;;OAIG;IACH,KAAK;QACH,IAAI,CAAC,MAAM,GAAG,IAAI,YAAY,CAAC,IAAI,CAAC,gBAAgB,CAAC,CAAC;QACtD,IAAI,CAAC,QAAQ,GAAG,CAAC,CAAC;IACpB,CAAC;IAED,4EAA4E;IAC5E,iCAAiC;IACjC,4EAA4E;IAE5E,sDAAsD;IACtD,IAAI,eAAe;QACjB,OAAO,IAAI,CAAC,QAAQ,CAAC;IACvB,CAAC;CACF"}
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @file WhisperChunkSession.ts
|
|
3
|
+
* @description Active streaming STT session backed by the OpenAI Whisper HTTP API.
|
|
4
|
+
*
|
|
5
|
+
* {@link WhisperChunkSession} implements the `StreamingSTTSession` interface
|
|
6
|
+
* (EventEmitter-based) using a sliding-window ring buffer to accumulate audio
|
|
7
|
+
* into fixed-size chunks. Each chunk is encoded as a RIFF/WAV file and posted
|
|
8
|
+
* to the Whisper `/v1/audio/transcriptions` endpoint.
|
|
9
|
+
*
|
|
10
|
+
* ### Chunk lifecycle
|
|
11
|
+
* 1. {@link pushAudio} feeds PCM frames into the internal {@link SlidingWindowBuffer}.
|
|
12
|
+
* 2. When a full chunk is ready, {@link onChunkReady} is invoked.
|
|
13
|
+
* 3. The chunk is WAV-encoded and POST-ed to Whisper with multipart/form-data.
|
|
14
|
+
* 4. The parsed response is emitted as `'interim_transcript'`.
|
|
15
|
+
* 5. The response text becomes the `prompt` for the next API call (continuity).
|
|
16
|
+
*
|
|
17
|
+
* ### Speech detection
|
|
18
|
+
* A simple RMS energy threshold (`RMS_THRESHOLD = 0.01`) gates `speech_start`
|
|
19
|
+
* and `speech_end` events. This is not VAD — it is a lightweight proxy that
|
|
20
|
+
* avoids emitting events on pure-silence frames.
|
|
21
|
+
*
|
|
22
|
+
* ### Error resilience
|
|
23
|
+
* On fetch failure the error is emitted as an `'error'` event and the session
|
|
24
|
+
* continues processing subsequent chunks rather than terminating.
|
|
25
|
+
*
|
|
26
|
+
* @module streaming-stt-whisper/WhisperChunkSession
|
|
27
|
+
*/
|
|
28
|
+
import { EventEmitter } from 'node:events';
|
|
29
|
+
import type { WhisperChunkedConfig, AudioFrame } from './types.js';
|
|
30
|
+
/**
|
|
31
|
+
* Active chunked Whisper STT session.
|
|
32
|
+
*
|
|
33
|
+
* Construct via {@link WhisperChunkedSTT.startSession} rather than directly.
|
|
34
|
+
*
|
|
35
|
+
* @example
|
|
36
|
+
* ```ts
|
|
37
|
+
* const session = new WhisperChunkSession({ apiKey: process.env.OPENAI_API_KEY! });
|
|
38
|
+
*
|
|
39
|
+
* session.on('interim_transcript', (evt) => console.log('chunk:', evt.text));
|
|
40
|
+
* session.on('final_transcript', (evt) => console.log('done:', evt.text));
|
|
41
|
+
*
|
|
42
|
+
* microphone.on('frame', (f) => session.pushAudio(f));
|
|
43
|
+
* await session.flush();
|
|
44
|
+
* session.close();
|
|
45
|
+
* ```
|
|
46
|
+
*/
|
|
47
|
+
export declare class WhisperChunkSession extends EventEmitter {
|
|
48
|
+
/** Resolved Whisper API configuration. */
|
|
49
|
+
private readonly cfg;
|
|
50
|
+
/** Sliding-window ring buffer feeding audio chunks. */
|
|
51
|
+
private readonly slidingBuffer;
|
|
52
|
+
/** Whether {@link close} has been called. */
|
|
53
|
+
private closed;
|
|
54
|
+
/** Whether the session is currently in a speech segment (above RMS threshold). */
|
|
55
|
+
private inSpeech;
|
|
56
|
+
/**
|
|
57
|
+
* Transcript text from the most recently completed chunk.
|
|
58
|
+
* Forwarded as `prompt` to the next Whisper API call for cross-chunk
|
|
59
|
+
* lexical continuity.
|
|
60
|
+
*/
|
|
61
|
+
private previousPrompt;
|
|
62
|
+
/**
|
|
63
|
+
* @param config - Whisper session configuration. `apiKey` is required.
|
|
64
|
+
*/
|
|
65
|
+
constructor(config: WhisperChunkedConfig);
|
|
66
|
+
/**
|
|
67
|
+
* Feed a raw audio frame into the session.
|
|
68
|
+
*
|
|
69
|
+
* The frame's samples are appended to the sliding-window buffer. When the
|
|
70
|
+
* buffer accumulates a full chunk, {@link onChunkReady} is invoked.
|
|
71
|
+
*
|
|
72
|
+
* Speech detection is performed on every frame: if the RMS energy crosses
|
|
73
|
+
* {@link RMS_THRESHOLD}, `'speech_start'` is emitted on the first such frame
|
|
74
|
+
* and `'speech_end'` when energy falls back below the threshold.
|
|
75
|
+
*
|
|
76
|
+
* @param frame - Audio frame with normalised Float32 samples.
|
|
77
|
+
*/
|
|
78
|
+
pushAudio(frame: AudioFrame): void;
|
|
79
|
+
/**
|
|
80
|
+
* Flush any remaining buffered samples, transcribe the final partial chunk,
|
|
81
|
+
* and emit `'final_transcript'`.
|
|
82
|
+
*
|
|
83
|
+
* Must be called when the audio stream ends to ensure the tail of the
|
|
84
|
+
* recording is not silently discarded.
|
|
85
|
+
*
|
|
86
|
+
* @returns Promise that resolves when the final Whisper request completes.
|
|
87
|
+
*/
|
|
88
|
+
flush(): Promise<void>;
|
|
89
|
+
/**
|
|
90
|
+
* Immediately terminate the session.
|
|
91
|
+
*
|
|
92
|
+
* No further events are emitted after `close()`.
|
|
93
|
+
*/
|
|
94
|
+
close(): void;
|
|
95
|
+
/**
|
|
96
|
+
* Transcribe a ready audio chunk by POST-ing it to the Whisper API.
|
|
97
|
+
*
|
|
98
|
+
* Steps:
|
|
99
|
+
* 1. Encode the Float32 chunk as a RIFF/WAV `ArrayBuffer`.
|
|
100
|
+
* 2. Build a multipart/form-data body with the WAV blob.
|
|
101
|
+
* 3. POST to `${baseUrl}/v1/audio/transcriptions`.
|
|
102
|
+
* 4. Parse the `verbose_json` response into a {@link TranscriptEvent}.
|
|
103
|
+
* 5. Emit `'interim_transcript'` and save the text as `previousPrompt`.
|
|
104
|
+
*
|
|
105
|
+
* On any fetch error, `'error'` is emitted and the method returns normally
|
|
106
|
+
* so that subsequent chunks are still processed.
|
|
107
|
+
*
|
|
108
|
+
* @param chunk - Float32 PCM samples for one audio chunk.
|
|
109
|
+
*/
|
|
110
|
+
private onChunkReady;
|
|
111
|
+
/**
|
|
112
|
+
* Convert a Whisper `verbose_json` response into a {@link TranscriptEvent}.
|
|
113
|
+
*
|
|
114
|
+
* Word-level timestamps are sourced from the first segment's `words` array
|
|
115
|
+
* when available. Confidence is approximated from the segment `avg_logprob`
|
|
116
|
+
* (clamped to [0, 1]) when present; falls back to 1 otherwise.
|
|
117
|
+
*
|
|
118
|
+
* @param response - Parsed Whisper verbose_json response.
|
|
119
|
+
* @returns A `TranscriptEvent` suitable for emission.
|
|
120
|
+
*/
|
|
121
|
+
private parseWhisperResponse;
|
|
122
|
+
}
|
|
123
|
+
//# sourceMappingURL=WhisperChunkSession.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"WhisperChunkSession.d.ts","sourceRoot":"","sources":["../src/WhisperChunkSession.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;GA0BG;AAEH,OAAO,EAAE,YAAY,EAAE,MAAM,aAAa,CAAC;AAE3C,OAAO,KAAK,EACV,oBAAoB,EACpB,UAAU,EAKX,MAAM,YAAY,CAAC;AA6IpB;;;;;;;;;;;;;;;;GAgBG;AACH,qBAAa,mBAAoB,SAAQ,YAAY;IAKnD,0CAA0C;IAC1C,OAAO,CAAC,QAAQ,CAAC,GAAG,CACqB;IAEzC,uDAAuD;IACvD,OAAO,CAAC,QAAQ,CAAC,aAAa,CAAsB;IAEpD,6CAA6C;IAC7C,OAAO,CAAC,MAAM,CAAS;IAEvB,kFAAkF;IAClF,OAAO,CAAC,QAAQ,CAAS;IAEzB;;;;OAIG;IACH,OAAO,CAAC,cAAc,CAAqB;IAM3C;;OAEG;gBACS,MAAM,EAAE,oBAAoB;IA4BxC;;;;;;;;;;;OAWG;IACH,SAAS,CAAC,KAAK,EAAE,UAAU,GAAG,IAAI;IAgBlC;;;;;;;;OAQG;IACG,KAAK,IAAI,OAAO,CAAC,IAAI,CAAC;IAwB5B;;;;OAIG;IACH,KAAK,IAAI,IAAI;IASb;;;;;;;;;;;;;;OAcG;YACW,YAAY;IAgD1B;;;;;;;;;OASG;IACH,OAAO,CAAC,oBAAoB;CAgC7B"}
|