nixamp 0.25.1 → 0.26.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +108 -7
- package/dist/captions.d.ts +7 -0
- package/dist/captions.js +92 -28
- package/dist/live-voice.d.ts +70 -0
- package/dist/live-voice.js +299 -0
- package/dist/server.d.ts +2 -0
- package/dist/server.js +154 -5
- package/dist/speaker-turns.d.ts +32 -0
- package/dist/speaker-turns.js +30 -0
- package/dist/speech.d.ts +47 -0
- package/dist/speech.js +49 -4
- package/dist/transcript-client.d.ts +2 -2
- package/dist/transcript-client.js +5 -2
- package/dist/transcripts.d.ts +5 -0
- package/dist/transcripts.js +8 -3
- package/dist/translate-jobs.js +13 -7
- package/dist/translate.d.ts +2 -0
- package/dist/translate.js +9 -0
- package/dist/voice-profile.d.ts +7 -0
- package/dist/voice-profile.js +53 -0
- package/dist/warm.js +3 -3
- package/package.json +1 -1
- package/src/captions.ts +86 -25
- package/src/live-voice.ts +255 -0
- package/src/server.ts +104 -5
- package/src/speaker-turns.ts +34 -0
- package/src/speech.ts +46 -7
- package/src/transcript-client.ts +4 -0
- package/src/transcripts.ts +13 -3
- package/src/translate-jobs.ts +13 -7
- package/src/translate.ts +7 -1
- package/src/voice-profile.ts +46 -0
- package/src/warm.ts +3 -3
- package/web/dist/assets/{hls-3VKVEQE3-GlzSYi0P.js → hls-3VKVEQE3-vgax_tk1.js} +1 -1
- package/web/dist/assets/index-BzjrTOLf.js +1 -0
- package/web/dist/assets/index-D2Iy07pG.css +1 -0
- package/web/dist/assets/{mpegts-KIbyW_RL.js → mpegts-DmcUOiHq.js} +1 -1
- package/web/dist/assets/{mpegts-LO6RVLD6-Fs_vbGcB.js → mpegts-LO6RVLD6-CH3EQi6L.js} +1 -1
- package/web/dist/index.html +42 -22
- package/web/dist/sw.js +6 -6
- package/web/dist/assets/index-DJHHB_63.js +0 -1
- package/web/dist/assets/index-oyp61Kly.css +0 -1
package/dist/speech.d.ts
CHANGED
|
@@ -16,6 +16,8 @@ export declare const SECONDS_PER_MINUTE = 300;
|
|
|
16
16
|
/** How many may wait for the one CPU. Past this, the honest answer is "later". */
|
|
17
17
|
export declare const QUEUE_LIMIT = 8;
|
|
18
18
|
export declare const DEFAULT_MODEL = "onnx-community/whisper-base";
|
|
19
|
+
/** Stored live lines from older recognition settings must be heard again. */
|
|
20
|
+
export declare const NATIVE_REVISION = "native-v2";
|
|
19
21
|
export declare class SpeechError extends Error {
|
|
20
22
|
readonly status: number;
|
|
21
23
|
constructor(message: string, status: number);
|
|
@@ -79,10 +81,53 @@ export declare function resample(samples: Float32Array, from: number, to: number
|
|
|
79
81
|
export declare function encodeWav(samples: Float32Array, rate?: number): Uint8Array;
|
|
80
82
|
/** Whisper's own spacing and blank-audio tokens, tidied into a line. */
|
|
81
83
|
export declare function tidy(text: string): string;
|
|
84
|
+
/** Reject decoding loops, without removing normal emphasis like “no, no, no”. */
|
|
85
|
+
export declare function reliableText(text: string, seconds: number): string;
|
|
86
|
+
export declare function quietSamples(pcm: Float32Array): boolean;
|
|
82
87
|
/** A two-letter language code, or nothing: Whisper guesses when not told. */
|
|
83
88
|
export declare function languageOf(value: unknown): string | undefined;
|
|
89
|
+
/** What the pipeline answers: the words, and the pieces with their timing when asked. */
|
|
90
|
+
interface AsrOutput {
|
|
91
|
+
text: string;
|
|
92
|
+
chunks?: {
|
|
93
|
+
timestamp: [number, number | null];
|
|
94
|
+
text: string;
|
|
95
|
+
}[];
|
|
96
|
+
}
|
|
97
|
+
/** The pipeline, and the parts under it that language detection needs. */
|
|
98
|
+
export interface AsrPipeline {
|
|
99
|
+
(audio: Float32Array, options: Record<string, unknown>): Promise<AsrOutput | AsrOutput[]>;
|
|
100
|
+
model: {
|
|
101
|
+
(inputs: Record<string, unknown>): Promise<{
|
|
102
|
+
logits: {
|
|
103
|
+
data: Float32Array | number[];
|
|
104
|
+
};
|
|
105
|
+
}>;
|
|
106
|
+
generation_config: {
|
|
107
|
+
decoder_start_token_id: number;
|
|
108
|
+
is_multilingual?: boolean;
|
|
109
|
+
lang_to_id?: Record<string, number>;
|
|
110
|
+
};
|
|
111
|
+
};
|
|
112
|
+
processor: (audio: Float32Array) => Promise<{
|
|
113
|
+
input_features: unknown;
|
|
114
|
+
}>;
|
|
115
|
+
}
|
|
116
|
+
/** The module, as much of it as this file touches. Typed here so the import can be by name. */
|
|
117
|
+
export interface Transformers {
|
|
118
|
+
env: {
|
|
119
|
+
cacheDir?: string;
|
|
120
|
+
allowLocalModels?: boolean;
|
|
121
|
+
};
|
|
122
|
+
Tensor: new (type: string, data: BigInt64Array, dims: number[]) => unknown;
|
|
123
|
+
pipeline(task: "automatic-speech-recognition", model: string, options: {
|
|
124
|
+
dtype: string;
|
|
125
|
+
}): Promise<AsrPipeline>;
|
|
126
|
+
}
|
|
84
127
|
/** Load the library, or say why not with a status. */
|
|
85
128
|
export declare function loadTransformers<T>(): Promise<T>;
|
|
129
|
+
/** Detection is from each recording, never from the caption translation selection. */
|
|
130
|
+
export declare function whisperRecognizer(transformers: Transformers, recognize: AsrPipeline): Recognizer;
|
|
86
131
|
export declare class Speech {
|
|
87
132
|
readonly model: string;
|
|
88
133
|
private readonly cacheDir;
|
|
@@ -112,5 +157,7 @@ export declare class Speech {
|
|
|
112
157
|
language?: string;
|
|
113
158
|
timestamps?: boolean;
|
|
114
159
|
by?: string;
|
|
160
|
+
deadline?: number;
|
|
115
161
|
}): Promise<Heard>;
|
|
116
162
|
}
|
|
163
|
+
export {};
|
package/dist/speech.js
CHANGED
|
@@ -47,6 +47,8 @@ export const SECONDS_PER_MINUTE = 300;
|
|
|
47
47
|
/** How many may wait for the one CPU. Past this, the honest answer is "later". */
|
|
48
48
|
export const QUEUE_LIMIT = 8;
|
|
49
49
|
export const DEFAULT_MODEL = "onnx-community/whisper-base";
|
|
50
|
+
/** Stored live lines from older recognition settings must be heard again. */
|
|
51
|
+
export const NATIVE_REVISION = "native-v2";
|
|
50
52
|
export class SpeechError extends Error {
|
|
51
53
|
status;
|
|
52
54
|
constructor(message, status) {
|
|
@@ -189,6 +191,33 @@ export function encodeWav(samples, rate = RATE) {
|
|
|
189
191
|
export function tidy(text) {
|
|
190
192
|
return text.replace(/\[[A-Z_ ]+\]|\([A-Za-z ]+\)/g, " ").replace(/\s+/g, " ").trim();
|
|
191
193
|
}
|
|
194
|
+
/** Reject decoding loops, without removing normal emphasis like “no, no, no”. */
|
|
195
|
+
export function reliableText(text, seconds) {
|
|
196
|
+
const clean = text.replace(/\s+/g, " ").trim();
|
|
197
|
+
if (clean.length > Math.max(120, seconds * 55))
|
|
198
|
+
return "";
|
|
199
|
+
const words = clean.toLowerCase().match(/[\p{L}\p{N}']+/gu) ?? [];
|
|
200
|
+
for (let size = 1; size <= Math.min(20, Math.floor(words.length / 4)); size++) {
|
|
201
|
+
for (let start = 0; start + size * 4 <= words.length; start++) {
|
|
202
|
+
let repeated = true;
|
|
203
|
+
for (let i = size; i < size * 4; i++) {
|
|
204
|
+
if (words[start + i] !== words[start + i % size]) {
|
|
205
|
+
repeated = false;
|
|
206
|
+
break;
|
|
207
|
+
}
|
|
208
|
+
}
|
|
209
|
+
if (repeated)
|
|
210
|
+
return "";
|
|
211
|
+
}
|
|
212
|
+
}
|
|
213
|
+
return clean;
|
|
214
|
+
}
|
|
215
|
+
export function quietSamples(pcm) {
|
|
216
|
+
let energy = 0;
|
|
217
|
+
for (const sample of pcm)
|
|
218
|
+
energy += sample * sample;
|
|
219
|
+
return pcm.length === 0 || Math.sqrt(energy / pcm.length) < 0.004;
|
|
220
|
+
}
|
|
192
221
|
/** A two-letter language code, or nothing: Whisper guesses when not told. */
|
|
193
222
|
export function languageOf(value) {
|
|
194
223
|
if (typeof value !== "string")
|
|
@@ -247,13 +276,25 @@ async function loadWhisper(model, cacheDir) {
|
|
|
247
276
|
const transformers = await loadTransformers();
|
|
248
277
|
transformers.env.cacheDir = cacheDir;
|
|
249
278
|
const recognize = await transformers.pipeline("automatic-speech-recognition", model, { dtype: "q8" });
|
|
279
|
+
return whisperRecognizer(transformers, recognize);
|
|
280
|
+
}
|
|
281
|
+
/** Detection is from each recording, never from the caption translation selection. */
|
|
282
|
+
export function whisperRecognizer(transformers, recognize) {
|
|
250
283
|
return async (pcm, { language, timestamps }) => {
|
|
251
|
-
const
|
|
284
|
+
const multilingual = recognize.model.generation_config.is_multilingual !== false;
|
|
285
|
+
const spoken = language ?? (multilingual ? await detectLanguage(transformers, recognize, pcm) : "en");
|
|
286
|
+
if (multilingual && !spoken)
|
|
287
|
+
throw new SpeechError("could not detect the audio language; try another speech segment", 422);
|
|
252
288
|
const heard = await recognize(pcm, {
|
|
253
289
|
// Whisper hears thirty seconds at a time; longer is heard in overlapping pieces.
|
|
254
290
|
chunk_length_s: 30,
|
|
255
291
|
stride_length_s: 5,
|
|
256
|
-
...(
|
|
292
|
+
...(multilingual ? { language: spoken, task: "transcribe" } : {}),
|
|
293
|
+
// A five-second clip used to be allowed hundreds of tokens of hallucinated loops.
|
|
294
|
+
max_new_tokens: Math.min(384, Math.max(32, Math.ceil(Math.min(30, pcm.length / RATE) * 10) + 16)),
|
|
295
|
+
do_sample: false,
|
|
296
|
+
num_beams: 1,
|
|
297
|
+
no_repeat_ngram_size: 6,
|
|
257
298
|
...(timestamps ? { return_timestamps: true } : {}),
|
|
258
299
|
});
|
|
259
300
|
const pieces = Array.isArray(heard) ? heard : [heard];
|
|
@@ -344,10 +385,14 @@ export class Speech {
|
|
|
344
385
|
if (wav.samples.length < wav.rate / 10)
|
|
345
386
|
return { text: "", seconds };
|
|
346
387
|
const pcm = resample(wav.samples, wav.rate, RATE);
|
|
388
|
+
if (quietSamples(pcm))
|
|
389
|
+
return { text: "", seconds };
|
|
347
390
|
if (this.waiting >= QUEUE_LIMIT)
|
|
348
391
|
throw new SpeechError("too many people are talking at once; try again in a moment", 503);
|
|
349
392
|
this.waiting += 1;
|
|
350
393
|
const turn = this.tail.then(async () => {
|
|
394
|
+
if (options.deadline && this.now() > options.deadline)
|
|
395
|
+
throw new SpeechError("this live audio is too old; waiting for the next segment", 408);
|
|
351
396
|
const recognize = await this.ear();
|
|
352
397
|
return recognize(pcm, {
|
|
353
398
|
...(options.language ? { language: options.language } : {}),
|
|
@@ -359,10 +404,10 @@ export class Speech {
|
|
|
359
404
|
try {
|
|
360
405
|
const heard = await turn;
|
|
361
406
|
const segments = heard.segments
|
|
362
|
-
?.map((segment) => ({ start: segment.start, end: segment.end, text: tidy(segment.text) }))
|
|
407
|
+
?.map((segment) => ({ start: segment.start, end: segment.end, text: reliableText(tidy(segment.text), segment.end - segment.start) }))
|
|
363
408
|
.filter((segment) => segment.text !== "");
|
|
364
409
|
return {
|
|
365
|
-
text: tidy(heard.text),
|
|
410
|
+
text: reliableText(tidy(heard.text), seconds),
|
|
366
411
|
seconds,
|
|
367
412
|
...(heard.language ? { language: heard.language } : {}),
|
|
368
413
|
...(segments ? { segments } : {}),
|
|
@@ -45,7 +45,7 @@ export type Got<T> = {
|
|
|
45
45
|
error: string;
|
|
46
46
|
};
|
|
47
47
|
/** The transcript of some media in a language ("" for the original), if the store has one. */
|
|
48
|
-
export declare function fetchTranscript(signed: Signed, id: string, language?: string, fetcher?: typeof fetch): Promise<Got<StoredTranscript>>;
|
|
48
|
+
export declare function fetchTranscript(signed: Signed, id: string, language?: string, fetcher?: typeof fetch, cachedOnly?: boolean): Promise<Got<StoredTranscript>>;
|
|
49
49
|
export interface LinesToKeep {
|
|
50
50
|
media: string;
|
|
51
51
|
language: string;
|
|
@@ -68,7 +68,7 @@ export interface Translated {
|
|
|
68
68
|
model: string;
|
|
69
69
|
}
|
|
70
70
|
/** Texts in another language. `from` may be "" when nixamp.com should tell. */
|
|
71
|
-
export declare function translateTexts(signed: Signed, texts: string[], from: string, to: string, fetcher?: typeof fetch): Promise<Got<Translated>>;
|
|
71
|
+
export declare function translateTexts(signed: Signed, texts: string[], from: string, to: string, fetcher?: typeof fetch, signal?: AbortSignal): Promise<Got<Translated>>;
|
|
72
72
|
/** What a machine tells nixamp.com about a file it has, for the record at /hash/<id>. */
|
|
73
73
|
export interface MediaToKeep {
|
|
74
74
|
name?: string;
|
|
@@ -11,10 +11,12 @@ function unreachable(site, error) {
|
|
|
11
11
|
return { ok: false, status: 0, error: `could not reach ${site}: ${error.message}` };
|
|
12
12
|
}
|
|
13
13
|
/** The transcript of some media in a language ("" for the original), if the store has one. */
|
|
14
|
-
export async function fetchTranscript(signed, id, language = "", fetcher = fetch) {
|
|
14
|
+
export async function fetchTranscript(signed, id, language = "", fetcher = fetch, cachedOnly = false) {
|
|
15
15
|
const url = new URL(`${base(signed.site)}/api/v1/transcripts/${encodeURIComponent(id)}`);
|
|
16
16
|
if (language)
|
|
17
17
|
url.searchParams.set("language", language);
|
|
18
|
+
if (cachedOnly)
|
|
19
|
+
url.searchParams.set("cached", "1");
|
|
18
20
|
try {
|
|
19
21
|
const response = await fetcher(url.toString(), { headers: { authorization: `Bearer ${signed.token}` } });
|
|
20
22
|
return await asJson(response);
|
|
@@ -38,12 +40,13 @@ export async function keepLines(signed, id, ask, fetcher = fetch) {
|
|
|
38
40
|
}
|
|
39
41
|
}
|
|
40
42
|
/** Texts in another language. `from` may be "" when nixamp.com should tell. */
|
|
41
|
-
export async function translateTexts(signed, texts, from, to, fetcher = fetch) {
|
|
43
|
+
export async function translateTexts(signed, texts, from, to, fetcher = fetch, signal) {
|
|
42
44
|
try {
|
|
43
45
|
const response = await fetcher(`${base(signed.site)}/api/v1/translate`, {
|
|
44
46
|
method: "POST",
|
|
45
47
|
headers: { authorization: `Bearer ${signed.token}`, "content-type": "application/json" },
|
|
46
48
|
body: JSON.stringify({ texts, from, to }),
|
|
49
|
+
signal,
|
|
47
50
|
});
|
|
48
51
|
return await asJson(response);
|
|
49
52
|
}
|
package/dist/transcripts.d.ts
CHANGED
|
@@ -4,6 +4,11 @@ export interface TranscriptLine {
|
|
|
4
4
|
start: number;
|
|
5
5
|
end: number;
|
|
6
6
|
text: string;
|
|
7
|
+
/** Live transcripts may change language within one programme. */
|
|
8
|
+
language?: string;
|
|
9
|
+
revision?: string;
|
|
10
|
+
original?: string;
|
|
11
|
+
voiceProfile?: "lower" | "higher" | "unknown";
|
|
7
12
|
}
|
|
8
13
|
export type MediaKind = "file" | "url" | "live";
|
|
9
14
|
export interface Transcript {
|
package/dist/transcripts.js
CHANGED
|
@@ -120,14 +120,19 @@ export function linesFrom(value) {
|
|
|
120
120
|
for (const one of parsed) {
|
|
121
121
|
if (!one || typeof one !== "object")
|
|
122
122
|
continue;
|
|
123
|
-
const { start, end, text } = one;
|
|
123
|
+
const { start, end, text, language, revision, original, voiceProfile } = one;
|
|
124
124
|
if (typeof text !== "string" || typeof start !== "number" || !Number.isFinite(start) || start < 0)
|
|
125
125
|
continue;
|
|
126
126
|
const words = text.replace(/\s+/g, " ").trim().slice(0, MAX_LINE_CHARS);
|
|
127
127
|
if (words === "")
|
|
128
128
|
continue;
|
|
129
129
|
const until = typeof end === "number" && Number.isFinite(end) && end > start ? end : start;
|
|
130
|
-
lines.push({ start: round(start), end: round(until), text: words
|
|
130
|
+
lines.push({ start: round(start), end: round(until), text: words,
|
|
131
|
+
...(typeof language === "string" && /^[a-z]{2,3}$/.test(language) ? { language } : {}),
|
|
132
|
+
...(typeof revision === "string" ? { revision: revision.slice(0, 32) } : {}),
|
|
133
|
+
...(typeof original === "string" ? { original: original.slice(0, MAX_LINE_CHARS) } : {}),
|
|
134
|
+
...(voiceProfile === "lower" || voiceProfile === "higher" || voiceProfile === "unknown" ? { voiceProfile } : {}),
|
|
135
|
+
});
|
|
131
136
|
}
|
|
132
137
|
return lines;
|
|
133
138
|
}
|
|
@@ -306,7 +311,7 @@ export class Transcripts {
|
|
|
306
311
|
await this.ensure();
|
|
307
312
|
const { rows } = language === ""
|
|
308
313
|
? await this.db.query(`SELECT * FROM transcripts WHERE id = $1 AND translated_from IS NULL
|
|
309
|
-
ORDER BY complete DESC, updated_at DESC LIMIT 1`, [id])
|
|
314
|
+
ORDER BY (language = '') DESC, complete DESC, updated_at DESC LIMIT 1`, [id])
|
|
310
315
|
: await this.db.query("SELECT * FROM transcripts WHERE id = $1 AND language = $2 LIMIT 1", [id, language]);
|
|
311
316
|
const row = rows[0];
|
|
312
317
|
return row ? rowToTranscript(row) : null;
|
package/dist/translate-jobs.js
CHANGED
|
@@ -46,11 +46,12 @@ export class StoredTranslations {
|
|
|
46
46
|
const missing = StoredTranslations.missing(original, translation);
|
|
47
47
|
if (translation && missing.length === 0)
|
|
48
48
|
return { status: 200, transcript: translation };
|
|
49
|
-
|
|
49
|
+
const sources = new Set(missing.map((line) => line.language || original.language));
|
|
50
|
+
if (sources.has(""))
|
|
50
51
|
return { status: 409, error: "the language this was heard in is not known, so it cannot be translated" };
|
|
51
52
|
if (!this.translator)
|
|
52
53
|
return { status: 503, error: "this nixamp cannot translate: no model here. nixamp.com can." };
|
|
53
|
-
if (!this.translator.can(
|
|
54
|
+
if ([...sources].some((source) => !this.translator.can(source, language))) {
|
|
54
55
|
return { status: 409, error: `there is no model here from ${original.language} to ${language}` };
|
|
55
56
|
}
|
|
56
57
|
const key = `${id}|${language}`;
|
|
@@ -95,19 +96,24 @@ export class StoredTranslations {
|
|
|
95
96
|
let model = "";
|
|
96
97
|
for (let at = 0; at < missing.length; at += BATCH) {
|
|
97
98
|
const batch = missing.slice(at, at + BATCH);
|
|
98
|
-
const
|
|
99
|
-
|
|
100
|
-
|
|
99
|
+
const lines = [];
|
|
100
|
+
for (const source of new Set(batch.map((line) => line.language || original.language))) {
|
|
101
|
+
const group = batch.filter((line) => (line.language || original.language) === source);
|
|
102
|
+
const done = await translator.translate(group.map((line) => line.text), source, language);
|
|
103
|
+
model = done.model;
|
|
104
|
+
lines.push(...group.map((line, i) => ({ ...line, language, original: line.text, text: done.texts[i] ?? "" })).filter((line) => line.text !== ""));
|
|
105
|
+
}
|
|
106
|
+
lines.sort((a, b) => a.start - b.start);
|
|
101
107
|
made.push(...lines);
|
|
102
108
|
await this.store.save({
|
|
103
|
-
media: original.media, language, translatedFrom: original.language, model, title: original.title, by, lines,
|
|
109
|
+
media: original.media, language, translatedFrom: original.language || "mul", model, title: original.title, by, lines,
|
|
104
110
|
});
|
|
105
111
|
job.progress.done += batch.length;
|
|
106
112
|
}
|
|
107
113
|
if (original.complete) {
|
|
108
114
|
// Whole, like the original: the pieces are replaced by the lot, and nothing partial touches it again.
|
|
109
115
|
await this.store.save({
|
|
110
|
-
media: original.media, language, translatedFrom: original.language, model, title: original.title, by, lines: made, complete: true,
|
|
116
|
+
media: original.media, language, translatedFrom: original.language || "mul", model, title: original.title, by, lines: made, complete: true,
|
|
111
117
|
});
|
|
112
118
|
}
|
|
113
119
|
if (missing.length > 0)
|
package/dist/translate.d.ts
CHANGED
|
@@ -59,6 +59,7 @@ export declare class Translator {
|
|
|
59
59
|
private readonly keep;
|
|
60
60
|
private readonly pairs;
|
|
61
61
|
private readonly loaded;
|
|
62
|
+
private readonly downloadCooldown;
|
|
62
63
|
private tail;
|
|
63
64
|
private waiting;
|
|
64
65
|
private readonly asked;
|
|
@@ -82,5 +83,6 @@ export declare class Translator {
|
|
|
82
83
|
*/
|
|
83
84
|
translate(texts: string[], from: string, to: string, options?: {
|
|
84
85
|
by?: string;
|
|
86
|
+
deadline?: number;
|
|
85
87
|
}): Promise<Translated>;
|
|
86
88
|
}
|
package/dist/translate.js
CHANGED
|
@@ -126,6 +126,7 @@ export class Translator {
|
|
|
126
126
|
keep;
|
|
127
127
|
pairs;
|
|
128
128
|
loaded = new Map();
|
|
129
|
+
downloadCooldown = new Map();
|
|
129
130
|
tail = Promise.resolve();
|
|
130
131
|
waiting = 0;
|
|
131
132
|
asked = new Map();
|
|
@@ -147,6 +148,8 @@ export class Translator {
|
|
|
147
148
|
}
|
|
148
149
|
pair(from, to) {
|
|
149
150
|
const model = modelFor(from, to);
|
|
151
|
+
if ((this.downloadCooldown.get(model) ?? 0) > this.now())
|
|
152
|
+
return Promise.reject(new SpeechError("the translation model download was rate limited; retry in a minute", 503));
|
|
150
153
|
const held = this.loaded.get(model);
|
|
151
154
|
if (held) {
|
|
152
155
|
held.usedAt = this.now();
|
|
@@ -155,6 +158,8 @@ export class Translator {
|
|
|
155
158
|
const pair = this.load(model, this.cacheDir).catch((error) => {
|
|
156
159
|
// A failed load is tried again next time, not remembered forever.
|
|
157
160
|
this.loaded.delete(model);
|
|
161
|
+
if (/\b429\b|rate.?limit/i.test(String(error)))
|
|
162
|
+
this.downloadCooldown.set(model, this.now() + 60_000);
|
|
158
163
|
throw error;
|
|
159
164
|
});
|
|
160
165
|
this.loaded.set(model, { pair, usedAt: this.now() });
|
|
@@ -231,12 +236,16 @@ export class Translator {
|
|
|
231
236
|
throw new SpeechError("too much is being translated at once; try again in a moment", 503);
|
|
232
237
|
this.waiting += 1;
|
|
233
238
|
const turn = this.tail.then(async () => {
|
|
239
|
+
const fresh = () => { if (options.deadline && this.now() > options.deadline)
|
|
240
|
+
throw new SpeechError("live translation expired while waiting; try the next line", 503); };
|
|
241
|
+
fresh();
|
|
234
242
|
// Only the lines with something in them go to the model; the rest keep their place.
|
|
235
243
|
const spoken = texts.map((text) => text.trim());
|
|
236
244
|
const which = spoken.map((text, i) => (text === "" ? -1 : i)).filter((i) => i >= 0);
|
|
237
245
|
let current = which.map((i) => spoken[i]);
|
|
238
246
|
for (const [a, b] of hops) {
|
|
239
247
|
const pair = await this.pair(a, b);
|
|
248
|
+
fresh();
|
|
240
249
|
current = await pair.translate(current);
|
|
241
250
|
}
|
|
242
251
|
const out = [...spoken];
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
/** Acoustic matching only: pitch does not establish a speaker's gender or identity. */
|
|
2
|
+
export type VoiceProfile = "lower" | "higher" | "unknown";
|
|
3
|
+
/** Median fundamental frequency from periodic voiced frames, sampled at 8 kHz.
|
|
4
|
+
* Ambiguous pitch, overlapping voices and unvoiced sound use the selected default.
|
|
5
|
+
* This is deliberately inexpensive and does not load another model beside ASR.
|
|
6
|
+
*/
|
|
7
|
+
export declare function voiceProfile(pcm: Buffer): VoiceProfile;
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
/** Median fundamental frequency from periodic voiced frames, sampled at 8 kHz.
|
|
2
|
+
* Ambiguous pitch, overlapping voices and unvoiced sound use the selected default.
|
|
3
|
+
* This is deliberately inexpensive and does not load another model beside ASR.
|
|
4
|
+
*/
|
|
5
|
+
export function voiceProfile(pcm) {
|
|
6
|
+
const pitches = [];
|
|
7
|
+
const frame = 320;
|
|
8
|
+
const samples = Math.floor(pcm.length / 4);
|
|
9
|
+
for (let start = 0; start + frame < samples; start += 1600) {
|
|
10
|
+
const values = new Float32Array(frame);
|
|
11
|
+
let mean = 0;
|
|
12
|
+
for (let i = 0; i < frame; i++)
|
|
13
|
+
mean += values[i] = pcm.readInt16LE((start + i) * 4) / 32768;
|
|
14
|
+
mean /= frame;
|
|
15
|
+
let energy = 0;
|
|
16
|
+
for (let i = 0; i < frame; i++) {
|
|
17
|
+
values[i] = values[i] - mean;
|
|
18
|
+
energy += values[i] ** 2;
|
|
19
|
+
}
|
|
20
|
+
if (Math.sqrt(energy / frame) < 0.01)
|
|
21
|
+
continue;
|
|
22
|
+
let best = 0;
|
|
23
|
+
let period = 0;
|
|
24
|
+
// Prefer the first strong peak; a later multiple of the period has the same correlation.
|
|
25
|
+
const correlations = [];
|
|
26
|
+
for (let lag = 20; lag <= 114; lag++) {
|
|
27
|
+
let cross = 0, left = 0, right = 0;
|
|
28
|
+
for (let i = 0; i < frame - lag; i++) {
|
|
29
|
+
const a = values[i], b = values[i + lag];
|
|
30
|
+
cross += a * b;
|
|
31
|
+
left += a * a;
|
|
32
|
+
right += b * b;
|
|
33
|
+
}
|
|
34
|
+
correlations[lag] = cross / Math.sqrt(left * right || 1);
|
|
35
|
+
}
|
|
36
|
+
for (let lag = 21; lag < 114; lag++) {
|
|
37
|
+
const score = correlations[lag];
|
|
38
|
+
if (score > 0.8 && score > correlations[lag - 1] && score >= correlations[lag + 1] && score > best + 0.03) {
|
|
39
|
+
best = score;
|
|
40
|
+
period = lag;
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
if (period)
|
|
44
|
+
pitches.push(8000 / period);
|
|
45
|
+
}
|
|
46
|
+
if (pitches.length < 3)
|
|
47
|
+
return "unknown";
|
|
48
|
+
pitches.sort((a, b) => a - b);
|
|
49
|
+
const median = pitches[Math.floor(pitches.length / 2)];
|
|
50
|
+
const lower = pitches.filter((pitch) => pitch < 155).length / pitches.length;
|
|
51
|
+
const higher = pitches.filter((pitch) => pitch > 185).length / pitches.length;
|
|
52
|
+
return median < 155 && lower >= 0.75 ? "lower" : median > 185 && higher >= 0.75 ? "higher" : "unknown";
|
|
53
|
+
}
|
package/dist/warm.js
CHANGED
|
@@ -7,14 +7,14 @@
|
|
|
7
7
|
* ask after each one, and the first person to speak waited for it. The
|
|
8
8
|
* models land in NIXAMP_STT_CACHE and the image keeps them.
|
|
9
9
|
*
|
|
10
|
-
* NIXAMP_MT_WARM names the pairs
|
|
11
|
-
*
|
|
10
|
+
* NIXAMP_MT_WARM names the pairs: German and Swedish both ways with
|
|
11
|
+
* English, and Spanish both ways with English and German by default.
|
|
12
12
|
* Exits non-zero when anything could not be fetched, so a build does not
|
|
13
13
|
* quietly ship without its ear.
|
|
14
14
|
*/
|
|
15
15
|
import { Speech } from "./speech.js";
|
|
16
16
|
import { Translator } from "./translate.js";
|
|
17
|
-
const DEFAULT_PAIRS = "en-de,en-sv,de-en,sv-en";
|
|
17
|
+
const DEFAULT_PAIRS = "en-de,en-sv,de-en,sv-en,es-en,en-es,es-de,de-es";
|
|
18
18
|
async function main() {
|
|
19
19
|
const speech = new Speech();
|
|
20
20
|
console.error(`warming ${speech.model}...`);
|