@framers/agentos-ext-streaming-stt-whisper 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +23 -0
- package/SKILL.md +62 -0
- package/dist/SlidingWindowBuffer.d.ts +93 -0
- package/dist/SlidingWindowBuffer.d.ts.map +1 -0
- package/dist/SlidingWindowBuffer.js +144 -0
- package/dist/SlidingWindowBuffer.js.map +1 -0
- package/dist/WhisperChunkSession.d.ts +123 -0
- package/dist/WhisperChunkSession.d.ts.map +1 -0
- package/dist/WhisperChunkSession.js +371 -0
- package/dist/WhisperChunkSession.js.map +1 -0
- package/dist/WhisperChunkedSTT.d.ts +72 -0
- package/dist/WhisperChunkedSTT.d.ts.map +1 -0
- package/dist/WhisperChunkedSTT.js +89 -0
- package/dist/WhisperChunkedSTT.js.map +1 -0
- package/dist/index.d.ts +70 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +78 -0
- package/dist/index.js.map +1 -0
- package/dist/types.d.ts +102 -0
- package/dist/types.d.ts.map +1 -0
- package/dist/types.js +11 -0
- package/dist/types.js.map +1 -0
- package/manifest.json +8 -0
- package/package.json +43 -0
- package/src/SlidingWindowBuffer.ts +176 -0
- package/src/WhisperChunkSession.ts +436 -0
- package/src/WhisperChunkedSTT.ts +111 -0
- package/src/index.ts +115 -0
- package/src/types.ts +122 -0
package/dist/index.js
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @file index.ts
|
|
3
|
+
* @description Pack factory for the Whisper Chunked Streaming STT extension pack.
|
|
4
|
+
*
|
|
5
|
+
* This module exports the main {@link createWhisperChunkedSTT} factory function
|
|
6
|
+
* and the {@link createExtensionPack} bridge function that conforms to the AgentOS
|
|
7
|
+
* manifest factory convention.
|
|
8
|
+
*
|
|
9
|
+
* ### Usage (direct)
|
|
10
|
+
* ```ts
|
|
11
|
+
* import { createWhisperChunkedSTT } from '@framers/agentos-ext-streaming-stt-whisper';
|
|
12
|
+
*
|
|
13
|
+
* const stt = createWhisperChunkedSTT(process.env.OPENAI_API_KEY!);
|
|
14
|
+
* const session = await stt.startSession({ language: 'en' });
|
|
15
|
+
* ```
|
|
16
|
+
*
|
|
17
|
+
* ### Usage (manifest-driven)
|
|
18
|
+
* ```json
|
|
19
|
+
* { "packs": [{ "module": "@framers/agentos-ext-streaming-stt-whisper" }] }
|
|
20
|
+
* ```
|
|
21
|
+
*
|
|
22
|
+
* @module streaming-stt-whisper
|
|
23
|
+
*/
|
|
24
|
+
import { WhisperChunkedSTT } from './WhisperChunkedSTT.js';
|
|
25
|
+
/** Kind constant matching packages/agentos/src/extensions/types.ts. */
|
|
26
|
+
const EXTENSION_KIND_STREAMING_STT = 'streaming-stt-provider';
|
|
27
|
+
// ---------------------------------------------------------------------------
|
|
28
|
+
// Factory
|
|
29
|
+
// ---------------------------------------------------------------------------
|
|
30
|
+
/**
|
|
31
|
+
* Create a standalone {@link WhisperChunkedSTT} instance.
|
|
32
|
+
*
|
|
33
|
+
* Use this when composing the provider programmatically outside of the
|
|
34
|
+
* AgentOS extension system.
|
|
35
|
+
*
|
|
36
|
+
* @param apiKey - OpenAI (or compatible) API key.
|
|
37
|
+
* @param baseUrl - Optional API base URL override.
|
|
38
|
+
* @returns Configured {@link WhisperChunkedSTT}.
|
|
39
|
+
*/
|
|
40
|
+
export function createWhisperChunkedSTT(apiKey, baseUrl) {
|
|
41
|
+
return new WhisperChunkedSTT(apiKey, baseUrl);
|
|
42
|
+
}
|
|
43
|
+
/**
|
|
44
|
+
* AgentOS manifest factory function.
|
|
45
|
+
*
|
|
46
|
+
* Reads the `OPENAI_API_KEY` secret from the context and returns an
|
|
47
|
+
* {@link ExtensionPack} containing a single `streaming-stt-provider` descriptor
|
|
48
|
+
* backed by {@link WhisperChunkedSTT}.
|
|
49
|
+
*
|
|
50
|
+
* @param context - Pack context supplied by the extension manager.
|
|
51
|
+
* @returns A fully configured {@link ExtensionPack}.
|
|
52
|
+
*/
|
|
53
|
+
export function createExtensionPack(context) {
|
|
54
|
+
const apiKey = context.getSecret?.('OPENAI_API_KEY') ?? '';
|
|
55
|
+
const baseUrl = context.options?.['baseUrl'];
|
|
56
|
+
const stt = new WhisperChunkedSTT(apiKey, baseUrl);
|
|
57
|
+
return {
|
|
58
|
+
id: 'streaming-stt-whisper',
|
|
59
|
+
descriptors: [
|
|
60
|
+
{
|
|
61
|
+
id: 'whisper-chunked-stt',
|
|
62
|
+
kind: EXTENSION_KIND_STREAMING_STT,
|
|
63
|
+
payload: stt,
|
|
64
|
+
enableByDefault: true,
|
|
65
|
+
metadata: { providerId: 'whisper-chunked' },
|
|
66
|
+
},
|
|
67
|
+
],
|
|
68
|
+
};
|
|
69
|
+
}
|
|
70
|
+
// ---------------------------------------------------------------------------
|
|
71
|
+
// Re-exports
|
|
72
|
+
// ---------------------------------------------------------------------------
|
|
73
|
+
export * from './types.js';
|
|
74
|
+
export { WhisperChunkedSTT } from './WhisperChunkedSTT.js';
|
|
75
|
+
export { WhisperChunkSession } from './WhisperChunkSession.js';
|
|
76
|
+
export { SlidingWindowBuffer } from './SlidingWindowBuffer.js';
|
|
77
|
+
export { DEFAULT_CHUNK_SIZE_SAMPLES, DEFAULT_OVERLAP_SAMPLES, } from './SlidingWindowBuffer.js';
|
|
78
|
+
//# sourceMappingURL=index.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"index.js","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;GAsBG;AAEH,OAAO,EAAE,iBAAiB,EAAE,MAAM,wBAAwB,CAAC;AA6B3D,uEAAuE;AACvE,MAAM,4BAA4B,GAAG,wBAAwB,CAAC;AAE9D,8EAA8E;AAC9E,UAAU;AACV,8EAA8E;AAE9E;;;;;;;;;GASG;AACH,MAAM,UAAU,uBAAuB,CAAC,MAAc,EAAE,OAAgB;IACtE,OAAO,IAAI,iBAAiB,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;AAChD,CAAC;AAED;;;;;;;;;GASG;AACH,MAAM,UAAU,mBAAmB,CAAC,OAA6B;IAC/D,MAAM,MAAM,GAAG,OAAO,CAAC,SAAS,EAAE,CAAC,gBAAgB,CAAC,IAAI,EAAE,CAAC;IAC3D,MAAM,OAAO,GAAG,OAAO,CAAC,OAAO,EAAE,CAAC,SAAS,CAAuB,CAAC;IACnE,MAAM,GAAG,GAAG,IAAI,iBAAiB,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;IAEnD,OAAO;QACL,EAAE,EAAE,uBAAuB;QAC3B,WAAW,EAAE;YACX;gBACE,EAAE,EAAE,qBAAqB;gBACzB,IAAI,EAAE,4BAA4B;gBAClC,OAAO,EAAE,GAAG;gBACZ,eAAe,EAAE,IAAI;gBACrB,QAAQ,EAAE,EAAE,UAAU,EAAE,iBAAiB,EAAE;aAC5C;SACF;KACF,CAAC;AACJ,CAAC;AAED,8EAA8E;AAC9E,aAAa;AACb,8EAA8E;AAE9E,cAAc,YAAY,CAAC;AAC3B,OAAO,EAAE,iBAAiB,EAAE,MAAM,wBAAwB,CAAC;AAC3D,OAAO,EAAE,mBAAmB,EAAE,MAAM,0BAA0B,CAAC;AAC/D,OAAO,EAAE,mBAAmB,EAAE,MAAM,0BAA0B,CAAC;AAC/D,OAAO,EACL,0BAA0B,EAC1B,uBAAuB,GACxB,MAAM,0BAA0B,CAAC"}
|
package/dist/types.d.ts
ADDED
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @file types.ts
|
|
3
|
+
* @description Whisper-specific configuration types for the chunked streaming STT extension pack.
|
|
4
|
+
*
|
|
5
|
+
* These types define the configuration for the sliding-window Whisper adapter that
|
|
6
|
+
* accumulates audio into 1-second chunks and sends them to the Whisper HTTP API.
|
|
7
|
+
*
|
|
8
|
+
* @module streaming-stt-whisper/types
|
|
9
|
+
*/
|
|
10
|
+
/**
|
|
11
|
+
* Configuration for the Whisper chunked streaming STT session.
|
|
12
|
+
*
|
|
13
|
+
* All fields except `apiKey` are optional — sensible defaults are applied.
|
|
14
|
+
*/
|
|
15
|
+
export interface WhisperChunkedConfig {
|
|
16
|
+
/**
|
|
17
|
+
* OpenAI API key (or compatible provider key).
|
|
18
|
+
* Read from `OPENAI_API_KEY` when constructed via {@link createExtensionPack}.
|
|
19
|
+
*/
|
|
20
|
+
apiKey: string;
|
|
21
|
+
/**
|
|
22
|
+
* Base URL for the Whisper API endpoint.
|
|
23
|
+
* Override to use a self-hosted or compatible provider (e.g. Groq, local Faster-Whisper).
|
|
24
|
+
* @defaultValue 'https://api.openai.com'
|
|
25
|
+
*/
|
|
26
|
+
baseUrl?: string;
|
|
27
|
+
/**
|
|
28
|
+
* Whisper model name.
|
|
29
|
+
* @defaultValue 'whisper-1'
|
|
30
|
+
* @see {@link https://platform.openai.com/docs/models/whisper}
|
|
31
|
+
*/
|
|
32
|
+
model?: string;
|
|
33
|
+
/**
|
|
34
|
+
* BCP-47 language hint (e.g. `'en'`, `'fr'`, `'de'`).
|
|
35
|
+
* When omitted Whisper auto-detects the language.
|
|
36
|
+
*/
|
|
37
|
+
language?: string;
|
|
38
|
+
/**
|
|
39
|
+
* Optional initial prompt to bias the first chunk's transcription.
|
|
40
|
+
* Subsequent chunks automatically receive the previous chunk's transcript as prompt.
|
|
41
|
+
* @see {@link https://platform.openai.com/docs/guides/speech-to-text/prompting}
|
|
42
|
+
*/
|
|
43
|
+
prompt?: string;
|
|
44
|
+
/**
|
|
45
|
+
* Size of each audio chunk in samples (at 16 kHz).
|
|
46
|
+
* @defaultValue 16000 (1 second at 16 kHz)
|
|
47
|
+
*/
|
|
48
|
+
chunkSizeSamples?: number;
|
|
49
|
+
/**
|
|
50
|
+
* Number of samples to carry forward from each chunk as overlap.
|
|
51
|
+
* Prevents words at chunk boundaries from being silently dropped.
|
|
52
|
+
* @defaultValue 3200 (200 ms at 16 kHz)
|
|
53
|
+
*/
|
|
54
|
+
overlapSamples?: number;
|
|
55
|
+
}
|
|
56
|
+
/** Minimal AudioFrame shape — mirrors packages/agentos/src/voice-pipeline/types.ts */
|
|
57
|
+
export interface AudioFrame {
|
|
58
|
+
samples: Float32Array;
|
|
59
|
+
sampleRate: number;
|
|
60
|
+
timestamp: number;
|
|
61
|
+
speakerHint?: string;
|
|
62
|
+
}
|
|
63
|
+
/** A single recognised word with timing metadata. */
|
|
64
|
+
export interface TranscriptWord {
|
|
65
|
+
word: string;
|
|
66
|
+
start: number;
|
|
67
|
+
end: number;
|
|
68
|
+
confidence: number;
|
|
69
|
+
speaker?: string;
|
|
70
|
+
}
|
|
71
|
+
/** A transcription result emitted by the session. */
|
|
72
|
+
export interface TranscriptEvent {
|
|
73
|
+
text: string;
|
|
74
|
+
confidence: number;
|
|
75
|
+
words: TranscriptWord[];
|
|
76
|
+
isFinal: boolean;
|
|
77
|
+
durationMs?: number;
|
|
78
|
+
}
|
|
79
|
+
/** A single segment from Whisper's verbose_json response. */
|
|
80
|
+
export interface WhisperSegment {
|
|
81
|
+
id: number;
|
|
82
|
+
start: number;
|
|
83
|
+
end: number;
|
|
84
|
+
text: string;
|
|
85
|
+
avg_logprob?: number;
|
|
86
|
+
words?: WhisperWord[];
|
|
87
|
+
}
|
|
88
|
+
/** A word-level entry from Whisper's verbose_json response (requires word timestamps). */
|
|
89
|
+
export interface WhisperWord {
|
|
90
|
+
word: string;
|
|
91
|
+
start: number;
|
|
92
|
+
end: number;
|
|
93
|
+
}
|
|
94
|
+
/** Top-level shape of the Whisper verbose_json transcription response. */
|
|
95
|
+
export interface WhisperTranscriptionResponse {
|
|
96
|
+
task?: string;
|
|
97
|
+
language?: string;
|
|
98
|
+
duration?: number;
|
|
99
|
+
text: string;
|
|
100
|
+
segments?: WhisperSegment[];
|
|
101
|
+
}
|
|
102
|
+
//# sourceMappingURL=types.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"types.d.ts","sourceRoot":"","sources":["../src/types.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH;;;;GAIG;AACH,MAAM,WAAW,oBAAoB;IACnC;;;OAGG;IACH,MAAM,EAAE,MAAM,CAAC;IAEf;;;;OAIG;IACH,OAAO,CAAC,EAAE,MAAM,CAAC;IAEjB;;;;OAIG;IACH,KAAK,CAAC,EAAE,MAAM,CAAC;IAEf;;;OAGG;IACH,QAAQ,CAAC,EAAE,MAAM,CAAC;IAElB;;;;OAIG;IACH,MAAM,CAAC,EAAE,MAAM,CAAC;IAEhB;;;OAGG;IACH,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAE1B;;;;OAIG;IACH,cAAc,CAAC,EAAE,MAAM,CAAC;CACzB;AAMD,sFAAsF;AACtF,MAAM,WAAW,UAAU;IACzB,OAAO,EAAE,YAAY,CAAC;IACtB,UAAU,EAAE,MAAM,CAAC;IACnB,SAAS,EAAE,MAAM,CAAC;IAClB,WAAW,CAAC,EAAE,MAAM,CAAC;CACtB;AAED,qDAAqD;AACrD,MAAM,WAAW,cAAc;IAC7B,IAAI,EAAE,MAAM,CAAC;IACb,KAAK,EAAE,MAAM,CAAC;IACd,GAAG,EAAE,MAAM,CAAC;IACZ,UAAU,EAAE,MAAM,CAAC;IACnB,OAAO,CAAC,EAAE,MAAM,CAAC;CAClB;AAED,qDAAqD;AACrD,MAAM,WAAW,eAAe;IAC9B,IAAI,EAAE,MAAM,CAAC;IACb,UAAU,EAAE,MAAM,CAAC;IACnB,KAAK,EAAE,cAAc,EAAE,CAAC;IACxB,OAAO,EAAE,OAAO,CAAC;IACjB,UAAU,CAAC,EAAE,MAAM,CAAC;CACrB;AAMD,6DAA6D;AAC7D,MAAM,WAAW,cAAc;IAC7B,EAAE,EAAE,MAAM,CAAC;IACX,KAAK,EAAE,MAAM,CAAC;IACd,GAAG,EAAE,MAAM,CAAC;IACZ,IAAI,EAAE,MAAM,CAAC;IACb,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,KAAK,CAAC,EAAE,WAAW,EAAE,CAAC;CACvB;AAED,0FAA0F;AAC1F,MAAM,WAAW,WAAW;IAC1B,IAAI,EAAE,MAAM,CAAC;IACb,KAAK,EAAE,MAAM,CAAC;IACd,GAAG,EAAE,MAAM,CAAC;CACb;AAED,0EAA0E;AAC1E,MAAM,WAAW,4BAA4B;IAC3C,IAAI,CAAC,EAAE,MAAM,CAAC;IACd,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,IAAI,EAAE,MAAM,CAAC;IACb,QAAQ,CAAC,EAAE,cAAc,EAAE,CAAC;CAC7B"}
|
package/dist/types.js
ADDED
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @file types.ts
|
|
3
|
+
* @description Whisper-specific configuration types for the chunked streaming STT extension pack.
|
|
4
|
+
*
|
|
5
|
+
* These types define the configuration for the sliding-window Whisper adapter that
|
|
6
|
+
* accumulates audio into 1-second chunks and sends them to the Whisper HTTP API.
|
|
7
|
+
*
|
|
8
|
+
* @module streaming-stt-whisper/types
|
|
9
|
+
*/
|
|
10
|
+
export {};
|
|
11
|
+
//# sourceMappingURL=types.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"types.js","sourceRoot":"","sources":["../src/types.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG"}
|
package/manifest.json
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "@framers/agentos-ext-streaming-stt-whisper",
|
|
3
|
+
"version": "0.1.0",
|
|
4
|
+
"description": "Chunked sliding-window streaming STT via OpenAI Whisper HTTP API",
|
|
5
|
+
"kind": "streaming-stt-provider",
|
|
6
|
+
"extensionId": "streaming-stt-whisper",
|
|
7
|
+
"entryPoint": "./dist/index.js"
|
|
8
|
+
}
|
package/package.json
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "@framers/agentos-ext-streaming-stt-whisper",
|
|
3
|
+
"version": "0.2.0",
|
|
4
|
+
"description": "Chunked sliding-window streaming STT via OpenAI Whisper HTTP API for AgentOS voice pipeline",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"main": "./dist/index.js",
|
|
7
|
+
"types": "./dist/index.d.ts",
|
|
8
|
+
"exports": {
|
|
9
|
+
".": {
|
|
10
|
+
"import": "./dist/index.js",
|
|
11
|
+
"types": "./dist/index.d.ts"
|
|
12
|
+
}
|
|
13
|
+
},
|
|
14
|
+
"files": [
|
|
15
|
+
"dist",
|
|
16
|
+
"src",
|
|
17
|
+
"SKILL.md",
|
|
18
|
+
"manifest.json"
|
|
19
|
+
],
|
|
20
|
+
"peerDependencies": {
|
|
21
|
+
"@framers/agentos": "^0.1.0"
|
|
22
|
+
},
|
|
23
|
+
"dependencies": {},
|
|
24
|
+
"devDependencies": {
|
|
25
|
+
"typescript": "^5.5.0",
|
|
26
|
+
"vitest": "^1.6.0",
|
|
27
|
+
"@framers/agentos": "0.1.94"
|
|
28
|
+
},
|
|
29
|
+
"license": "MIT",
|
|
30
|
+
"author": "Frame.dev",
|
|
31
|
+
"repository": {
|
|
32
|
+
"type": "git",
|
|
33
|
+
"url": "https://github.com/framersai/agentos-extensions.git",
|
|
34
|
+
"directory": "registry/curated/voice/streaming-stt-whisper"
|
|
35
|
+
},
|
|
36
|
+
"publishConfig": {
|
|
37
|
+
"access": "public"
|
|
38
|
+
},
|
|
39
|
+
"scripts": {
|
|
40
|
+
"build": "tsc -p tsconfig.json",
|
|
41
|
+
"test": "vitest run"
|
|
42
|
+
}
|
|
43
|
+
}
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @file SlidingWindowBuffer.ts
|
|
3
|
+
* @description Ring buffer that accumulates Float32 audio frames into fixed-size chunks
|
|
4
|
+
* with configurable overlap between consecutive chunks.
|
|
5
|
+
*
|
|
6
|
+
* When {@link pushSamples} fills the internal buffer to {@link chunkSizeSamples}, it
|
|
7
|
+
* emits a `'chunk_ready'` event carrying the complete `Float32Array` chunk, copies the
|
|
8
|
+
* last {@link overlapSamples} samples to the head of the buffer as overlap context for
|
|
9
|
+
* the next chunk, and resets the write cursor accordingly.
|
|
10
|
+
*
|
|
11
|
+
* The overlap strategy prevents words straddling chunk boundaries from being silently
|
|
12
|
+
* dropped by the Whisper model.
|
|
13
|
+
*
|
|
14
|
+
* @module streaming-stt-whisper/SlidingWindowBuffer
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
import { EventEmitter } from 'node:events';
|
|
18
|
+
|
|
19
|
+
// ---------------------------------------------------------------------------
|
|
20
|
+
// Constants
|
|
21
|
+
// ---------------------------------------------------------------------------
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* Default chunk size in samples.
|
|
25
|
+
* At 16 kHz this corresponds to exactly 1 second of mono audio.
|
|
26
|
+
*/
|
|
27
|
+
export const DEFAULT_CHUNK_SIZE_SAMPLES = 16_000;
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* Default overlap in samples carried forward to the next chunk.
|
|
31
|
+
* At 16 kHz this corresponds to 200 ms of audio context.
|
|
32
|
+
*/
|
|
33
|
+
export const DEFAULT_OVERLAP_SAMPLES = 3_200;
|
|
34
|
+
|
|
35
|
+
// ---------------------------------------------------------------------------
|
|
36
|
+
// Events interface (TypeScript augmentation for typed emit/on)
|
|
37
|
+
// ---------------------------------------------------------------------------
|
|
38
|
+
|
|
39
|
+
/** Event map for {@link SlidingWindowBuffer}. */
|
|
40
|
+
export interface SlidingWindowBufferEvents {
|
|
41
|
+
/** Emitted when a complete chunk of {@link chunkSizeSamples} is ready. */
|
|
42
|
+
chunk_ready: [chunk: Float32Array];
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
// ---------------------------------------------------------------------------
|
|
46
|
+
// Main class
|
|
47
|
+
// ---------------------------------------------------------------------------
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* Ring-buffer that accumulates raw PCM samples and emits fixed-size audio
|
|
51
|
+
* chunks with configurable overlap.
|
|
52
|
+
*
|
|
53
|
+
* @example
|
|
54
|
+
* ```ts
|
|
55
|
+
* const buf = new SlidingWindowBuffer(16_000, 3_200);
|
|
56
|
+
* buf.on('chunk_ready', (chunk) => sendToWhisper(chunk));
|
|
57
|
+
*
|
|
58
|
+
* microphone.on('frame', (f) => buf.pushSamples(f.samples));
|
|
59
|
+
* await buf.flush(); // emit any remaining samples
|
|
60
|
+
* ```
|
|
61
|
+
*/
|
|
62
|
+
export class SlidingWindowBuffer extends EventEmitter {
|
|
63
|
+
// -------------------------------------------------------------------------
|
|
64
|
+
// Private state
|
|
65
|
+
// -------------------------------------------------------------------------
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* Internal sample store. Sized to {@link chunkSizeSamples} so a single
|
|
69
|
+
* allocation is reused for the lifetime of the session.
|
|
70
|
+
*/
|
|
71
|
+
private buffer: Float32Array;
|
|
72
|
+
|
|
73
|
+
/**
|
|
74
|
+
* Current write position within {@link buffer}.
|
|
75
|
+
* Always in the range `[0, chunkSizeSamples)`.
|
|
76
|
+
*/
|
|
77
|
+
private writePos = 0;
|
|
78
|
+
|
|
79
|
+
// -------------------------------------------------------------------------
|
|
80
|
+
// Constructor
|
|
81
|
+
// -------------------------------------------------------------------------
|
|
82
|
+
|
|
83
|
+
/**
|
|
84
|
+
* @param chunkSizeSamples - Number of samples per emitted chunk.
|
|
85
|
+
* Defaults to {@link DEFAULT_CHUNK_SIZE_SAMPLES} (1 s at 16 kHz).
|
|
86
|
+
* @param overlapSamples - Number of samples carried forward from each chunk
|
|
87
|
+
* to the start of the next. Must be less than `chunkSizeSamples`.
|
|
88
|
+
* Defaults to {@link DEFAULT_OVERLAP_SAMPLES} (200 ms at 16 kHz).
|
|
89
|
+
*/
|
|
90
|
+
constructor(
|
|
91
|
+
private readonly chunkSizeSamples: number = DEFAULT_CHUNK_SIZE_SAMPLES,
|
|
92
|
+
private readonly overlapSamples: number = DEFAULT_OVERLAP_SAMPLES,
|
|
93
|
+
) {
|
|
94
|
+
super();
|
|
95
|
+
|
|
96
|
+
if (overlapSamples >= chunkSizeSamples) {
|
|
97
|
+
throw new RangeError(
|
|
98
|
+
`overlapSamples (${overlapSamples}) must be less than chunkSizeSamples (${chunkSizeSamples})`,
|
|
99
|
+
);
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
this.buffer = new Float32Array(chunkSizeSamples);
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
// -------------------------------------------------------------------------
|
|
106
|
+
// Public API
|
|
107
|
+
// -------------------------------------------------------------------------
|
|
108
|
+
|
|
109
|
+
/**
|
|
110
|
+
* Append audio samples to the internal buffer.
|
|
111
|
+
*
|
|
112
|
+
* If the incoming batch causes the buffer to reach or exceed
|
|
113
|
+
* {@link chunkSizeSamples}, one or more `'chunk_ready'` events are emitted
|
|
114
|
+
* before the remainder is retained for the next chunk. Each chunk includes
|
|
115
|
+
* an overlap region copied from the tail of the previous chunk.
|
|
116
|
+
*
|
|
117
|
+
* @param samples - Float32 PCM samples to append.
|
|
118
|
+
*/
|
|
119
|
+
pushSamples(samples: Float32Array): void {
|
|
120
|
+
let srcOffset = 0;
|
|
121
|
+
|
|
122
|
+
while (srcOffset < samples.length) {
|
|
123
|
+
// How many samples can we copy into the current chunk before it is full?
|
|
124
|
+
const spaceLeft = this.chunkSizeSamples - this.writePos;
|
|
125
|
+
const copyCount = Math.min(spaceLeft, samples.length - srcOffset);
|
|
126
|
+
|
|
127
|
+
this.buffer.set(samples.subarray(srcOffset, srcOffset + copyCount), this.writePos);
|
|
128
|
+
this.writePos += copyCount;
|
|
129
|
+
srcOffset += copyCount;
|
|
130
|
+
|
|
131
|
+
if (this.writePos >= this.chunkSizeSamples) {
|
|
132
|
+
// Chunk is full — emit a copy (not a reference to the internal buffer).
|
|
133
|
+
this.emit('chunk_ready', this.buffer.slice());
|
|
134
|
+
|
|
135
|
+
// Copy the last `overlapSamples` to the beginning of the buffer so that
|
|
136
|
+
// the next chunk begins with audio context from the previous boundary.
|
|
137
|
+
const overlapStart = this.chunkSizeSamples - this.overlapSamples;
|
|
138
|
+
this.buffer.copyWithin(0, overlapStart, this.chunkSizeSamples);
|
|
139
|
+
this.writePos = this.overlapSamples;
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
/**
|
|
145
|
+
* Emit any samples currently held in the buffer as a final partial chunk.
|
|
146
|
+
*
|
|
147
|
+
* If the buffer contains no samples (`writePos === 0`), this is a no-op.
|
|
148
|
+
* After flushing, the buffer is reset to an empty state.
|
|
149
|
+
*/
|
|
150
|
+
flush(): void {
|
|
151
|
+
if (this.writePos === 0) return;
|
|
152
|
+
|
|
153
|
+
// Emit only the samples that were actually written (not the whole buffer).
|
|
154
|
+
this.emit('chunk_ready', this.buffer.slice(0, this.writePos));
|
|
155
|
+
this.reset();
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
/**
|
|
159
|
+
* Clear all buffered samples and reset the write cursor to zero.
|
|
160
|
+
*
|
|
161
|
+
* Does NOT emit a `'chunk_ready'` event — use {@link flush} for that.
|
|
162
|
+
*/
|
|
163
|
+
reset(): void {
|
|
164
|
+
this.buffer = new Float32Array(this.chunkSizeSamples);
|
|
165
|
+
this.writePos = 0;
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
// -------------------------------------------------------------------------
|
|
169
|
+
// Accessors (useful for testing)
|
|
170
|
+
// -------------------------------------------------------------------------
|
|
171
|
+
|
|
172
|
+
/** Number of samples currently held in the buffer. */
|
|
173
|
+
get bufferedSamples(): number {
|
|
174
|
+
return this.writePos;
|
|
175
|
+
}
|
|
176
|
+
}
|