@spader/node-whisper-cpp 0.2.2 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/audio.d.ts +4 -0
- package/dist/audio.js +88 -0
- package/dist/errors.d.ts +9 -0
- package/dist/errors.js +18 -0
- package/dist/index.d.ts +12 -12
- package/dist/index.js +14 -22
- package/dist/loader.js +12 -9
- package/dist/model.d.ts +2 -0
- package/dist/model.js +37 -0
- package/dist/session.d.ts +2 -0
- package/dist/session.js +16 -0
- package/dist/transcribe.d.ts +3 -0
- package/dist/transcribe.js +118 -0
- package/dist/types.d.ts +152 -16
- package/package.json +19 -8
package/dist/audio.d.ts
ADDED
package/dist/audio.js
ADDED
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
import { readFile } from "node:fs/promises";
|
|
2
|
+
import { AudioDecodeError } from "./errors.js";
|
|
3
|
+
export const SAMPLE_RATE = 16000;
|
|
4
|
+
const tag = (bytes, offset) => String.fromCharCode(bytes[offset], bytes[offset + 1], bytes[offset + 2], bytes[offset + 3]);
|
|
5
|
+
const sampleReader = (view, base, format, bits) => {
|
|
6
|
+
if (format === 3 && bits === 32)
|
|
7
|
+
return (i) => view.getFloat32(base + i * 4, true);
|
|
8
|
+
if (format === 1 || format === 0xfffe) {
|
|
9
|
+
if (bits === 8)
|
|
10
|
+
return (i) => (view.getUint8(base + i) - 128) / 128;
|
|
11
|
+
if (bits === 16)
|
|
12
|
+
return (i) => view.getInt16(base + i * 2, true) / 32768;
|
|
13
|
+
if (bits === 24) {
|
|
14
|
+
return (i) => {
|
|
15
|
+
const o = base + i * 3;
|
|
16
|
+
const raw = view.getUint8(o) | (view.getUint8(o + 1) << 8) | (view.getUint8(o + 2) << 16);
|
|
17
|
+
return (raw & 0x800000 ? raw | ~0xffffff : raw) / 8388608;
|
|
18
|
+
};
|
|
19
|
+
}
|
|
20
|
+
if (bits === 32)
|
|
21
|
+
return (i) => view.getInt32(base + i * 4, true) / 2147483648;
|
|
22
|
+
}
|
|
23
|
+
throw new AudioDecodeError(`unsupported WAV encoding: format ${format}, ${bits}-bit`);
|
|
24
|
+
};
|
|
25
|
+
const decodeWav = (bytes) => {
|
|
26
|
+
const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength);
|
|
27
|
+
if (bytes.byteLength < 12 || tag(bytes, 0) !== "RIFF" || tag(bytes, 8) !== "WAVE") {
|
|
28
|
+
throw new AudioDecodeError("not a RIFF/WAVE file");
|
|
29
|
+
}
|
|
30
|
+
let format = 0;
|
|
31
|
+
let channels = 0;
|
|
32
|
+
let sampleRate = 0;
|
|
33
|
+
let bits = 0;
|
|
34
|
+
let dataOffset = -1;
|
|
35
|
+
let dataLength = 0;
|
|
36
|
+
let offset = 12;
|
|
37
|
+
while (offset + 8 <= bytes.byteLength) {
|
|
38
|
+
const id = tag(bytes, offset);
|
|
39
|
+
const size = view.getUint32(offset + 4, true);
|
|
40
|
+
const body = offset + 8;
|
|
41
|
+
if (id === "fmt " && size >= 16 && body + 16 <= bytes.byteLength) {
|
|
42
|
+
format = view.getUint16(body, true);
|
|
43
|
+
channels = view.getUint16(body + 2, true);
|
|
44
|
+
sampleRate = view.getUint32(body + 4, true);
|
|
45
|
+
bits = view.getUint16(body + 14, true);
|
|
46
|
+
}
|
|
47
|
+
else if (id === "data") {
|
|
48
|
+
dataOffset = body;
|
|
49
|
+
dataLength = Math.min(size, bytes.byteLength - body);
|
|
50
|
+
}
|
|
51
|
+
offset = body + size + (size & 1);
|
|
52
|
+
}
|
|
53
|
+
if (dataOffset < 0 || channels === 0 || sampleRate === 0) {
|
|
54
|
+
throw new AudioDecodeError("missing fmt or data chunk");
|
|
55
|
+
}
|
|
56
|
+
const read = sampleReader(view, dataOffset, format, bits);
|
|
57
|
+
const frames = Math.floor(dataLength / (channels * (bits / 8)));
|
|
58
|
+
const samples = new Float32Array(frames);
|
|
59
|
+
for (let i = 0; i < frames; i++) {
|
|
60
|
+
let sum = 0;
|
|
61
|
+
for (let c = 0; c < channels; c++)
|
|
62
|
+
sum += read(i * channels + c);
|
|
63
|
+
samples[i] = sum / channels;
|
|
64
|
+
}
|
|
65
|
+
return { samples, sampleRate, channels };
|
|
66
|
+
};
|
|
67
|
+
const resample = (input, from, to) => {
|
|
68
|
+
if (from === to)
|
|
69
|
+
return input;
|
|
70
|
+
const ratio = from / to;
|
|
71
|
+
const length = Math.floor(input.length / ratio);
|
|
72
|
+
const output = new Float32Array(length);
|
|
73
|
+
for (let i = 0; i < length; i++) {
|
|
74
|
+
const position = i * ratio;
|
|
75
|
+
const index = Math.floor(position);
|
|
76
|
+
const fraction = position - index;
|
|
77
|
+
const a = input[index] ?? 0;
|
|
78
|
+
const b = input[index + 1] ?? a;
|
|
79
|
+
output[i] = a + (b - a) * fraction;
|
|
80
|
+
}
|
|
81
|
+
return output;
|
|
82
|
+
};
|
|
83
|
+
export const fromWav = (data) => {
|
|
84
|
+
const bytes = data instanceof Uint8Array ? data : new Uint8Array(data);
|
|
85
|
+
const decoded = decodeWav(bytes);
|
|
86
|
+
return resample(decoded.samples, decoded.sampleRate, SAMPLE_RATE);
|
|
87
|
+
};
|
|
88
|
+
export const fromFile = async (path) => fromWav(await readFile(path));
|
package/dist/errors.d.ts
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
export declare class ModelLoadError extends Error {
|
|
2
|
+
constructor(model: string, cause?: unknown);
|
|
3
|
+
}
|
|
4
|
+
export declare class TranscribeError extends Error {
|
|
5
|
+
constructor(message: string, cause?: unknown);
|
|
6
|
+
}
|
|
7
|
+
export declare class AudioDecodeError extends Error {
|
|
8
|
+
constructor(message: string);
|
|
9
|
+
}
|
package/dist/errors.js
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
export class ModelLoadError extends Error {
|
|
2
|
+
constructor(model, cause) {
|
|
3
|
+
super(`failed to load whisper model: ${model}`, { cause });
|
|
4
|
+
this.name = "ModelLoadError";
|
|
5
|
+
}
|
|
6
|
+
}
|
|
7
|
+
export class TranscribeError extends Error {
|
|
8
|
+
constructor(message, cause) {
|
|
9
|
+
super(message, { cause });
|
|
10
|
+
this.name = "TranscribeError";
|
|
11
|
+
}
|
|
12
|
+
}
|
|
13
|
+
export class AudioDecodeError extends Error {
|
|
14
|
+
constructor(message) {
|
|
15
|
+
super(message);
|
|
16
|
+
this.name = "AudioDecodeError";
|
|
17
|
+
}
|
|
18
|
+
}
|
package/dist/index.d.ts
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
|
-
import
|
|
2
|
-
|
|
3
|
-
export
|
|
4
|
-
export
|
|
5
|
-
export declare
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
1
|
+
import * as audio from "./audio.js";
|
|
2
|
+
import * as platform from "./platform.js";
|
|
3
|
+
export type { Audio, Gpu, LoadOptions, Model, ModelInfo, Segment, Session, Token, TranscribeOptions, TranscriptionResult, TranscriptionStream, VadOptions, } from "./types.js";
|
|
4
|
+
export { AudioDecodeError, ModelLoadError, TranscribeError } from "./errors.js";
|
|
5
|
+
export declare const whisper: {
|
|
6
|
+
load: (options: string | import("./types.js").LoadOptions) => Promise<import("./types.js").Model>;
|
|
7
|
+
audio: typeof audio;
|
|
8
|
+
platform: typeof platform;
|
|
9
|
+
readonly version: string;
|
|
10
|
+
systemInfo: () => string;
|
|
11
|
+
};
|
|
12
|
+
export default whisper;
|
package/dist/index.js
CHANGED
|
@@ -1,23 +1,15 @@
|
|
|
1
|
+
import * as audio from "./audio.js";
|
|
1
2
|
import { loadAddon } from "./loader.js";
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
}
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
this.native = new (loadAddon().WhisperContext)(options);
|
|
16
|
-
}
|
|
17
|
-
transcribe(options) {
|
|
18
|
-
return this.native.transcribe(options);
|
|
19
|
-
}
|
|
20
|
-
free() {
|
|
21
|
-
this.native.free();
|
|
22
|
-
}
|
|
23
|
-
}
|
|
3
|
+
import { load } from "./model.js";
|
|
4
|
+
import * as platform from "./platform.js";
|
|
5
|
+
export { AudioDecodeError, ModelLoadError, TranscribeError } from "./errors.js";
|
|
6
|
+
export const whisper = {
|
|
7
|
+
load,
|
|
8
|
+
audio,
|
|
9
|
+
platform,
|
|
10
|
+
get version() {
|
|
11
|
+
return loadAddon().version();
|
|
12
|
+
},
|
|
13
|
+
systemInfo: () => loadAddon().systemInfo(),
|
|
14
|
+
};
|
|
15
|
+
export default whisper;
|
package/dist/loader.js
CHANGED
|
@@ -14,16 +14,19 @@ function resolvePlatformPackageDir(name) {
|
|
|
14
14
|
return join(currentDir, "..", "packages", "platform", name.slice(prefix.length));
|
|
15
15
|
}
|
|
16
16
|
}
|
|
17
|
-
|
|
18
|
-
addon: null
|
|
19
|
-
};
|
|
20
|
-
export function loadAddon() {
|
|
21
|
-
if (store.addon != null)
|
|
22
|
-
return store.addon;
|
|
17
|
+
function defaultAddonPath() {
|
|
23
18
|
const triple = detect();
|
|
24
19
|
const name = `@spader/node-whisper-cpp-${triple}`;
|
|
25
20
|
const dir = resolvePlatformPackageDir(name);
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
21
|
+
return join(dir, "bins", "whisper-addon.node");
|
|
22
|
+
}
|
|
23
|
+
const cache = new Map();
|
|
24
|
+
export function loadAddon() {
|
|
25
|
+
const addonPath = process.env.NODE_WHISPER_CPP_ADDON || defaultAddonPath();
|
|
26
|
+
const cached = cache.get(addonPath);
|
|
27
|
+
if (cached)
|
|
28
|
+
return cached;
|
|
29
|
+
const addon = require(addonPath);
|
|
30
|
+
cache.set(addonPath, addon);
|
|
31
|
+
return addon;
|
|
29
32
|
}
|
package/dist/model.d.ts
ADDED
package/dist/model.js
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
import { ModelLoadError } from "./errors.js";
|
|
2
|
+
import { loadAddon } from "./loader.js";
|
|
3
|
+
import { createSession } from "./session.js";
|
|
4
|
+
import { runStream, runTranscribe } from "./transcribe.js";
|
|
5
|
+
const open = (options) => {
|
|
6
|
+
const gpu = options.gpu ?? true;
|
|
7
|
+
try {
|
|
8
|
+
return new (loadAddon().WhisperContext)({
|
|
9
|
+
model: options.model,
|
|
10
|
+
useGpu: gpu !== false,
|
|
11
|
+
gpuDevice: typeof gpu === "object" ? gpu.device ?? 0 : 0,
|
|
12
|
+
flashAttn: options.flashAttention ?? false,
|
|
13
|
+
});
|
|
14
|
+
}
|
|
15
|
+
catch (cause) {
|
|
16
|
+
throw new ModelLoadError(options.model, cause);
|
|
17
|
+
}
|
|
18
|
+
};
|
|
19
|
+
export const load = async (options) => {
|
|
20
|
+
const native = open(typeof options === "string" ? { model: options } : options);
|
|
21
|
+
const info = native.modelInfo();
|
|
22
|
+
let disposed = false;
|
|
23
|
+
const dispose = () => {
|
|
24
|
+
if (disposed)
|
|
25
|
+
return;
|
|
26
|
+
disposed = true;
|
|
27
|
+
native.free();
|
|
28
|
+
};
|
|
29
|
+
return {
|
|
30
|
+
info,
|
|
31
|
+
transcribe: (audio, options) => runTranscribe(native, audio, options),
|
|
32
|
+
transcribeStream: (audio, options) => runStream(native, audio, options),
|
|
33
|
+
createSession: () => createSession(native.createState()),
|
|
34
|
+
dispose,
|
|
35
|
+
[Symbol.dispose]: dispose,
|
|
36
|
+
};
|
|
37
|
+
};
|
package/dist/session.js
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { runStream, runTranscribe } from "./transcribe.js";
|
|
2
|
+
export const createSession = (native) => {
|
|
3
|
+
let disposed = false;
|
|
4
|
+
const dispose = () => {
|
|
5
|
+
if (disposed)
|
|
6
|
+
return;
|
|
7
|
+
disposed = true;
|
|
8
|
+
native.free();
|
|
9
|
+
};
|
|
10
|
+
return {
|
|
11
|
+
transcribe: (audio, options) => runTranscribe(native, audio, options),
|
|
12
|
+
transcribeStream: (audio, options) => runStream(native, audio, options),
|
|
13
|
+
dispose,
|
|
14
|
+
[Symbol.dispose]: dispose,
|
|
15
|
+
};
|
|
16
|
+
};
|
|
@@ -0,0 +1,3 @@
|
|
|
1
|
+
import type { Audio, NativeState, TranscribeOptions, TranscriptionResult, TranscriptionStream } from "./types.js";
|
|
2
|
+
export declare const runTranscribe: (native: NativeState, audio: Audio, options?: TranscribeOptions) => Promise<TranscriptionResult>;
|
|
3
|
+
export declare const runStream: (native: NativeState, audio: Audio, options?: TranscribeOptions) => TranscriptionStream;
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
import { TranscribeError } from "./errors.js";
|
|
2
|
+
const locks = new WeakMap();
|
|
3
|
+
const withLock = (native, fn) => {
|
|
4
|
+
const previous = locks.get(native) ?? Promise.resolve();
|
|
5
|
+
const run = previous.then(fn, fn);
|
|
6
|
+
locks.set(native, run.then(() => { }, () => { }));
|
|
7
|
+
return run;
|
|
8
|
+
};
|
|
9
|
+
const makeAbort = (signal) => {
|
|
10
|
+
if (!signal)
|
|
11
|
+
return undefined;
|
|
12
|
+
const flag = new Int32Array(new SharedArrayBuffer(4));
|
|
13
|
+
if (signal.aborted)
|
|
14
|
+
Atomics.store(flag, 0, 1);
|
|
15
|
+
else
|
|
16
|
+
signal.addEventListener("abort", () => Atomics.store(flag, 0, 1), { once: true });
|
|
17
|
+
return flag;
|
|
18
|
+
};
|
|
19
|
+
const toNative = (audio, options, onSegment) => {
|
|
20
|
+
const vad = options.vad;
|
|
21
|
+
const vadOptions = typeof vad === "object" ? vad : undefined;
|
|
22
|
+
return {
|
|
23
|
+
pcm: audio,
|
|
24
|
+
language: options.language ?? "auto",
|
|
25
|
+
translate: options.translate,
|
|
26
|
+
threads: options.threads,
|
|
27
|
+
prompt: options.prompt,
|
|
28
|
+
offsetMs: options.offsetMs,
|
|
29
|
+
durationMs: options.durationMs,
|
|
30
|
+
maxLen: options.maxLen,
|
|
31
|
+
splitOnWord: options.splitOnWord,
|
|
32
|
+
maxTokens: options.maxTokens,
|
|
33
|
+
tokenTimestamps: options.tokenTimestamps,
|
|
34
|
+
diarize: options.diarize,
|
|
35
|
+
temperature: options.temperature,
|
|
36
|
+
temperatureInc: options.temperatureInc,
|
|
37
|
+
beamSize: options.beamSize,
|
|
38
|
+
bestOf: options.bestOf,
|
|
39
|
+
noSpeechThreshold: options.noSpeechThreshold,
|
|
40
|
+
entropyThreshold: options.entropyThreshold,
|
|
41
|
+
logProbThreshold: options.logProbThreshold,
|
|
42
|
+
suppressBlank: options.suppressBlank,
|
|
43
|
+
suppressNonSpeech: options.suppressNonSpeech,
|
|
44
|
+
vad: vad === undefined ? undefined : Boolean(vad),
|
|
45
|
+
vadModel: vadOptions?.model,
|
|
46
|
+
vadThreshold: vadOptions?.threshold,
|
|
47
|
+
vadMinSpeechMs: vadOptions?.minSpeechMs,
|
|
48
|
+
vadMinSilenceMs: vadOptions?.minSilenceMs,
|
|
49
|
+
vadMaxSpeechSeconds: vadOptions?.maxSpeechSeconds,
|
|
50
|
+
vadSpeechPadMs: vadOptions?.speechPadMs,
|
|
51
|
+
vadSamplesOverlap: vadOptions?.samplesOverlap,
|
|
52
|
+
abort: makeAbort(options.signal),
|
|
53
|
+
onProgress: options.onProgress,
|
|
54
|
+
onSegment,
|
|
55
|
+
};
|
|
56
|
+
};
|
|
57
|
+
const seconds = (centiseconds) => centiseconds / 100;
|
|
58
|
+
const mapToken = (token) => ({
|
|
59
|
+
text: token.text,
|
|
60
|
+
start: seconds(token.t0),
|
|
61
|
+
end: seconds(token.t1),
|
|
62
|
+
probability: token.p,
|
|
63
|
+
});
|
|
64
|
+
const mapSegment = (segment) => ({
|
|
65
|
+
text: segment.text.trim(),
|
|
66
|
+
start: seconds(segment.t0),
|
|
67
|
+
end: seconds(segment.t1),
|
|
68
|
+
noSpeechProb: segment.noSpeechProb,
|
|
69
|
+
...(segment.speakerTurn !== undefined && { speakerTurn: segment.speakerTurn }),
|
|
70
|
+
...(segment.tokens && { tokens: segment.tokens.map(mapToken) }),
|
|
71
|
+
});
|
|
72
|
+
const mapResult = (result) => {
|
|
73
|
+
const segments = result.segments.map(mapSegment);
|
|
74
|
+
return {
|
|
75
|
+
text: segments
|
|
76
|
+
.map((segment) => segment.text)
|
|
77
|
+
.join(" ")
|
|
78
|
+
.trim(),
|
|
79
|
+
language: result.language,
|
|
80
|
+
segments,
|
|
81
|
+
};
|
|
82
|
+
};
|
|
83
|
+
const run = (native, audio, options, onSegment) => withLock(native, () => native.transcribe(toNative(audio, options, onSegment)).then(mapResult, (cause) => {
|
|
84
|
+
if (options.signal?.aborted)
|
|
85
|
+
throw new DOMException("transcription aborted", "AbortError");
|
|
86
|
+
throw new TranscribeError(cause instanceof Error ? cause.message : "transcription failed", cause);
|
|
87
|
+
}));
|
|
88
|
+
export const runTranscribe = (native, audio, options = {}) => run(native, audio, options);
|
|
89
|
+
export const runStream = (native, audio, options = {}) => {
|
|
90
|
+
const queue = [];
|
|
91
|
+
let wake = null;
|
|
92
|
+
let finished = false;
|
|
93
|
+
const signal = () => {
|
|
94
|
+
wake?.();
|
|
95
|
+
wake = null;
|
|
96
|
+
};
|
|
97
|
+
const result = run(native, audio, options, (segment) => {
|
|
98
|
+
queue.push(mapSegment(segment));
|
|
99
|
+
signal();
|
|
100
|
+
});
|
|
101
|
+
const done = result.finally(() => {
|
|
102
|
+
finished = true;
|
|
103
|
+
signal();
|
|
104
|
+
});
|
|
105
|
+
done.catch(() => { });
|
|
106
|
+
const iterator = async function* () {
|
|
107
|
+
while (true) {
|
|
108
|
+
while (queue.length)
|
|
109
|
+
yield queue.shift();
|
|
110
|
+
if (finished) {
|
|
111
|
+
await done;
|
|
112
|
+
return;
|
|
113
|
+
}
|
|
114
|
+
await new Promise((resolve) => (wake = resolve));
|
|
115
|
+
}
|
|
116
|
+
};
|
|
117
|
+
return Object.assign(done, { [Symbol.asyncIterator]: iterator });
|
|
118
|
+
};
|
package/dist/types.d.ts
CHANGED
|
@@ -1,26 +1,162 @@
|
|
|
1
|
-
export
|
|
1
|
+
export type Audio = Float32Array;
|
|
2
|
+
export type Gpu = boolean | {
|
|
3
|
+
device?: number;
|
|
4
|
+
};
|
|
5
|
+
export type LoadOptions = {
|
|
2
6
|
model: string;
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
7
|
+
gpu?: Gpu;
|
|
8
|
+
flashAttention?: boolean;
|
|
9
|
+
};
|
|
10
|
+
export type ModelInfo = {
|
|
11
|
+
type: string;
|
|
12
|
+
multilingual: boolean;
|
|
13
|
+
vocabSize: number;
|
|
14
|
+
audioContextSize: number;
|
|
15
|
+
textContextSize: number;
|
|
16
|
+
};
|
|
17
|
+
export type VadOptions = {
|
|
18
|
+
model: string;
|
|
19
|
+
threshold?: number;
|
|
20
|
+
minSpeechMs?: number;
|
|
21
|
+
minSilenceMs?: number;
|
|
22
|
+
maxSpeechSeconds?: number;
|
|
23
|
+
speechPadMs?: number;
|
|
24
|
+
samplesOverlap?: number;
|
|
25
|
+
};
|
|
26
|
+
export type TranscribeOptions = {
|
|
27
|
+
language?: string;
|
|
28
|
+
translate?: boolean;
|
|
29
|
+
threads?: number;
|
|
30
|
+
prompt?: string;
|
|
31
|
+
offsetMs?: number;
|
|
32
|
+
durationMs?: number;
|
|
33
|
+
maxLen?: number;
|
|
34
|
+
splitOnWord?: boolean;
|
|
35
|
+
maxTokens?: number;
|
|
36
|
+
tokenTimestamps?: boolean;
|
|
37
|
+
diarize?: boolean;
|
|
38
|
+
temperature?: number;
|
|
39
|
+
temperatureInc?: number;
|
|
40
|
+
beamSize?: number;
|
|
41
|
+
bestOf?: number;
|
|
42
|
+
noSpeechThreshold?: number;
|
|
43
|
+
entropyThreshold?: number;
|
|
44
|
+
logProbThreshold?: number;
|
|
45
|
+
suppressBlank?: boolean;
|
|
46
|
+
suppressNonSpeech?: boolean;
|
|
47
|
+
vad?: boolean | VadOptions;
|
|
48
|
+
signal?: AbortSignal;
|
|
49
|
+
onProgress?: (progress: number) => void;
|
|
50
|
+
};
|
|
51
|
+
export type Token = {
|
|
52
|
+
text: string;
|
|
53
|
+
start: number;
|
|
54
|
+
end: number;
|
|
55
|
+
probability: number;
|
|
56
|
+
};
|
|
57
|
+
export type Segment = {
|
|
58
|
+
text: string;
|
|
59
|
+
start: number;
|
|
60
|
+
end: number;
|
|
61
|
+
noSpeechProb: number;
|
|
62
|
+
speakerTurn?: boolean;
|
|
63
|
+
tokens?: Token[];
|
|
64
|
+
};
|
|
65
|
+
export type TranscriptionResult = {
|
|
66
|
+
text: string;
|
|
67
|
+
language: string;
|
|
68
|
+
segments: Segment[];
|
|
69
|
+
};
|
|
70
|
+
export type TranscriptionStream = AsyncIterable<Segment> & Promise<TranscriptionResult>;
|
|
71
|
+
export type Session = {
|
|
72
|
+
transcribe(audio: Audio, options?: TranscribeOptions): Promise<TranscriptionResult>;
|
|
73
|
+
transcribeStream(audio: Audio, options?: TranscribeOptions): TranscriptionStream;
|
|
74
|
+
dispose(): void;
|
|
75
|
+
[Symbol.dispose](): void;
|
|
76
|
+
};
|
|
77
|
+
export type Model = {
|
|
78
|
+
readonly info: ModelInfo;
|
|
79
|
+
transcribe(audio: Audio, options?: TranscribeOptions): Promise<TranscriptionResult>;
|
|
80
|
+
transcribeStream(audio: Audio, options?: TranscribeOptions): TranscriptionStream;
|
|
81
|
+
createSession(): Session;
|
|
82
|
+
dispose(): void;
|
|
83
|
+
[Symbol.dispose](): void;
|
|
84
|
+
};
|
|
85
|
+
export type NativeToken = {
|
|
86
|
+
text: string;
|
|
8
87
|
t0: number;
|
|
9
88
|
t1: number;
|
|
89
|
+
p: number;
|
|
90
|
+
};
|
|
91
|
+
export type NativeSegment = {
|
|
10
92
|
text: string;
|
|
11
|
-
|
|
12
|
-
|
|
93
|
+
t0: number;
|
|
94
|
+
t1: number;
|
|
95
|
+
noSpeechProb: number;
|
|
96
|
+
speakerTurn?: boolean;
|
|
97
|
+
tokens?: NativeToken[];
|
|
98
|
+
};
|
|
99
|
+
export type NativeResult = {
|
|
100
|
+
language: string;
|
|
101
|
+
segments: NativeSegment[];
|
|
102
|
+
};
|
|
103
|
+
export type NativeTranscribeOptions = {
|
|
13
104
|
pcm: Float32Array;
|
|
14
105
|
language?: string;
|
|
106
|
+
translate?: boolean;
|
|
15
107
|
threads?: number;
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
108
|
+
prompt?: string;
|
|
109
|
+
offsetMs?: number;
|
|
110
|
+
durationMs?: number;
|
|
111
|
+
maxLen?: number;
|
|
112
|
+
splitOnWord?: boolean;
|
|
113
|
+
maxTokens?: number;
|
|
114
|
+
tokenTimestamps?: boolean;
|
|
115
|
+
diarize?: boolean;
|
|
116
|
+
temperature?: number;
|
|
117
|
+
temperatureInc?: number;
|
|
118
|
+
beamSize?: number;
|
|
119
|
+
bestOf?: number;
|
|
120
|
+
noSpeechThreshold?: number;
|
|
121
|
+
entropyThreshold?: number;
|
|
122
|
+
logProbThreshold?: number;
|
|
123
|
+
suppressBlank?: boolean;
|
|
124
|
+
suppressNonSpeech?: boolean;
|
|
125
|
+
vad?: boolean;
|
|
126
|
+
vadModel?: string;
|
|
127
|
+
vadThreshold?: number;
|
|
128
|
+
vadMinSpeechMs?: number;
|
|
129
|
+
vadMinSilenceMs?: number;
|
|
130
|
+
vadMaxSpeechSeconds?: number;
|
|
131
|
+
vadSpeechPadMs?: number;
|
|
132
|
+
vadSamplesOverlap?: number;
|
|
133
|
+
abort?: Int32Array;
|
|
134
|
+
onSegment?: (segment: NativeSegment) => void;
|
|
135
|
+
onProgress?: (progress: number) => void;
|
|
136
|
+
};
|
|
137
|
+
export type NativeModelInfo = {
|
|
138
|
+
type: string;
|
|
139
|
+
multilingual: boolean;
|
|
140
|
+
vocabSize: number;
|
|
141
|
+
audioContextSize: number;
|
|
142
|
+
textContextSize: number;
|
|
143
|
+
};
|
|
144
|
+
export type NativeContextOptions = {
|
|
145
|
+
model: string;
|
|
146
|
+
useGpu: boolean;
|
|
147
|
+
gpuDevice: number;
|
|
148
|
+
flashAttn: boolean;
|
|
149
|
+
};
|
|
150
|
+
export type NativeState = {
|
|
151
|
+
transcribe(options: NativeTranscribeOptions): Promise<NativeResult>;
|
|
20
152
|
free(): void;
|
|
21
|
-
}
|
|
22
|
-
export
|
|
23
|
-
|
|
153
|
+
};
|
|
154
|
+
export type NativeContext = NativeState & {
|
|
155
|
+
createState(): NativeState;
|
|
156
|
+
modelInfo(): NativeModelInfo;
|
|
157
|
+
};
|
|
158
|
+
export type NativeAddon = {
|
|
159
|
+
WhisperContext: new (options: NativeContextOptions) => NativeContext;
|
|
24
160
|
version(): string;
|
|
25
161
|
systemInfo(): string;
|
|
26
|
-
}
|
|
162
|
+
};
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@spader/node-whisper-cpp",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.5.0",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"packageManager": "bun@1.3.9",
|
|
6
6
|
"main": "./dist/index.js",
|
|
@@ -25,16 +25,25 @@
|
|
|
25
25
|
],
|
|
26
26
|
"scripts": {
|
|
27
27
|
"build": "bun run ./tools/build.ts all",
|
|
28
|
-
"
|
|
29
|
-
"
|
|
30
|
-
"
|
|
28
|
+
"act:build": "bun run ./tools/act.ts build",
|
|
29
|
+
"model": "bun run ./tools/model.ts",
|
|
30
|
+
"fixtures:audio": "bun run ./test/utils/audio-fixtures.ts",
|
|
31
|
+
"smoke": "NODE_WHISPER_CPP_BACKEND=cpu bun run ./tools/build.ts all --ci && bun run test:integration && bun run test:smoke",
|
|
32
|
+
"test": "bun test ./test/unit",
|
|
33
|
+
"test:unit": "bun test ./test/unit",
|
|
34
|
+
"test:integration": "bun test ./test/integration",
|
|
35
|
+
"test:smoke": "NODE_WHISPER_CPP_BACKEND=cpu bun test ./test/smoke",
|
|
36
|
+
"example:basic": "bun run ./example/basic.ts",
|
|
37
|
+
"example:streaming": "bun run ./example/streaming.ts",
|
|
38
|
+
"example:concurrent": "bun run ./example/concurrent.ts",
|
|
39
|
+
"example:kitchen-sink": "bun run ./example/kitchen-sink.ts",
|
|
31
40
|
"clean": "bun run ./tools/clean.ts"
|
|
32
41
|
},
|
|
33
42
|
"optionalDependencies": {
|
|
34
|
-
"@spader/node-whisper-cpp-arm64-darwin-metal": "0.
|
|
35
|
-
"@spader/node-whisper-cpp-arm64-darwin-cpu": "0.
|
|
36
|
-
"@spader/node-whisper-cpp-x64-linux-cpu-gnu": "0.
|
|
37
|
-
"@spader/node-whisper-cpp-x64-linux-cuda-gnu": "0.
|
|
43
|
+
"@spader/node-whisper-cpp-arm64-darwin-metal": "0.5.0",
|
|
44
|
+
"@spader/node-whisper-cpp-arm64-darwin-cpu": "0.5.0",
|
|
45
|
+
"@spader/node-whisper-cpp-x64-linux-cpu-gnu": "0.5.0",
|
|
46
|
+
"@spader/node-whisper-cpp-x64-linux-cuda-gnu": "0.5.0"
|
|
38
47
|
},
|
|
39
48
|
"repository": {
|
|
40
49
|
"type": "git",
|
|
@@ -47,6 +56,8 @@
|
|
|
47
56
|
"node-addon-api": "^8.3.1"
|
|
48
57
|
},
|
|
49
58
|
"devDependencies": {
|
|
59
|
+
"@actions/core": "^1.11.1",
|
|
60
|
+
"@actions/exec": "^1.1.1",
|
|
50
61
|
"@clack/prompts": "^1.0.0",
|
|
51
62
|
"@types/bun": "^1.2.21",
|
|
52
63
|
"@types/node": "^24.3.0",
|