@chatpanel/gateway 0.6.58 → 0.6.60

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@chatpanel/gateway",
3
- "version": "0.6.58",
3
+ "version": "0.6.60",
4
4
  "description": "Local privacy gateway — redacts PII out of OpenAI/Anthropic API traffic before it reaches a model, then restores it in the reply. Point opencode, codex, aider, Claude Code, etc. at it.",
5
5
  "type": "module",
6
6
  "bin": {
package/src/server.js CHANGED
@@ -44,6 +44,7 @@ import { TTS_MODEL_CATALOG, TTS_VOICES, isKnownTtsModel, isValidCustomTtsId, isK
44
44
  import { ttsDestination, synthesizeRemote, isValidRemoteVoice } from './tts-remote.js';
45
45
  import { rawOrtAvailable } from './ort.js';
46
46
  import * as ttsVoices from './tts-voices.js';
47
+ import { resolveTtsVoice } from './tts-voice-resolve.js';
47
48
  import { resolvePro, checkQuota, consume, usage } from './freegate.js';
48
49
  import { publicConfig, applyConfigPatch, applyNerModelSelection, persistConfig, configPath } from './configstore.js';
49
50
  import { resolveDestination, aggregateModelsAsync, listDestinations } from './router.js';
@@ -54,7 +55,7 @@ import * as openai from './openai.js';
54
55
  import * as responses from './responses.js';
55
56
  import * as anthropic from './anthropic.js';
56
57
 
57
- export const VERSION = '0.6.58';
58
+ export const VERSION = '0.6.60';
58
59
 
59
60
  // WARM search tier — SQLite + FTS5 record store (falls back to an encrypted-JSON
60
61
  // store if SQLite can't load), fed by the extension's ingest sync + backup-ingest.
@@ -1078,8 +1079,15 @@ export function createGateway(cfg = loadConfig()) {
1078
1079
  const curId = ttsVoices.parseCustomVoice(cfg.tts.voice || '');
1079
1080
  const isCustom = !!(curId && ttsVoices.getVoice(curId));
1080
1081
  if (wantsCustom && !isCustom) {
1081
- const saved = ttsVoices.listVoices();
1082
- cfg.tts.voice = saved.length ? `custom:${saved[0].id}` : '';
1082
+ // Switching to a cloning model does not mean "start speaking as the user".
1083
+ // A model with built-in speakers gets its default speaker; only SpeechT5,
1084
+ // which has no built-ins, falls back to a saved voice.
1085
+ if (ttsModelEngineOf(id) === 'pocket-tts') {
1086
+ cfg.tts.voice = DEFAULT_POCKET_VOICE;
1087
+ } else {
1088
+ const saved = ttsVoices.listVoices();
1089
+ cfg.tts.voice = saved.length ? `custom:${saved[0].id}` : '';
1090
+ }
1083
1091
  } else if (!wantsCustom && isCustom) {
1084
1092
  cfg.tts.voice = DEFAULT_TTS_VOICE;
1085
1093
  }
@@ -1280,76 +1288,13 @@ export function createGateway(cfg = loadConfig()) {
1280
1288
  // SpeechT5 takes a recorded embedding, VITS takes neither. Resolving first
1281
1289
  // meant a `custom:` voice could reach a freshly-loaded Kokoro and fail deep
1282
1290
  // in the engine with "invalid voice id".
1283
- let useVoice = voice;
1284
- let speakerEmbedding = null;
1285
- let customId = ttsVoices.parseCustomVoice(voice);
1286
-
1287
- // A Pocket built-in speaker is a NAME, not an embedding, so it has to be
1288
- // recognised before the custom-voice path which would otherwise demand a
1289
- // recording for a model that ships eight voices of its own.
1290
- if (ttsEngine.isPocket() && isPocketVoice(rawVoice || useVoice)) {
1291
- useVoice = rawVoice || useVoice;
1292
- customId = null;
1293
- } else if (ttsEngine.supportsCustomVoices()) {
1294
- // This model speaks ONLY in a recorded voice. If the configured one names
1295
- // a built-in (switching model does not rewrite `voice`) or points at a
1296
- // voice since deleted, fall back to the most recent saved one — the
1297
- // caller asked to be spoken to, not for that exact voice. A voice named
1298
- // EXPLICITLY in the request still fails loudly.
1299
- let rec = customId ? ttsVoices.getVoice(customId) : null;
1300
- if (!rec && !rawVoice) {
1301
- const saved = ttsVoices.listVoices();
1302
- if (saved.length) { customId = saved[0].id; rec = ttsVoices.getVoice(customId); }
1303
- // Pocket can fall back to a built-in speaker; SpeechT5 has none, so for
1304
- // that one "no saved voice" really is the end of the road.
1305
- else if (ttsEngine.isPocket()) { useVoice = DEFAULT_POCKET_VOICE; customId = null; }
1306
- }
1307
- if (!rec && !customId && ttsEngine.isPocket()) {
1308
- // resolved to a built-in above — nothing more to look up
1309
- } else if (!rec) {
1310
- return sendJson(res, customId ? 404 : 400, {
1311
- error: {
1312
- message: customId ? 'no such saved voice' : 'this model speaks in a voice you record — add one in Settings → Text-to-speech',
1313
- type: 'bad_voice',
1314
- },
1315
- });
1316
- }
1317
- // Which print to hand over depends on the engine: Pocket TTS takes its
1318
- // Mimi conditioning, SpeechT5 the 512-d x-vector. A voice saved before
1319
- // the pocket bundle existed has only the latter.
1320
- if (ttsEngine.isPocket()) {
1321
- const pk = ttsVoices.getPocketVoice(customId);
1322
- if (!pk) {
1323
- return sendJson(res, 409, {
1324
- error: {
1325
- message: 'this voice was saved without a Pocket TTS conditioning — record it again with Pocket TTS selected',
1326
- type: 'voice_kind_missing',
1327
- },
1328
- });
1329
- }
1330
- speakerEmbedding = pk;
1331
- } else {
1332
- speakerEmbedding = rec.vec;
1333
- }
1334
- useVoice = `custom:${customId}`;
1335
- } else if (customId) {
1336
- // Explicitly asked for a recorded voice this model cannot use — say so.
1337
- // Inherited from config, though, it is just a stale setting, and refusing
1338
- // to speak at all is a worse answer than speaking in the default voice.
1339
- if (rawVoice) {
1340
- return sendJson(res, 409, {
1341
- error: { message: `the active model (${ttsEngine.arch()}) cannot use a recorded voice — switch to SpeechT5`, type: 'voice_unsupported' },
1342
- });
1343
- }
1344
- customId = null;
1345
- useVoice = DEFAULT_TTS_VOICE;
1346
- } else if (ttsEngine.isPocket() && !isPocketVoice(useVoice)) {
1347
- // Pocket has its own speaker namespace; a Kokoro voice name here means the
1348
- // config was carried over from another model, so fall back to its default.
1349
- useVoice = DEFAULT_POCKET_VOICE;
1350
- } else if (ttsEngine.supportsVoices() && !ttsEngine.isPocket() && !(isKnownVoice(useVoice) && isValidVoiceId(useVoice))) {
1351
- return sendJson(res, 400, { error: { message: 'unknown or invalid voice', type: 'bad_voice' } });
1352
- }
1291
+ const picked = resolveTtsVoice({
1292
+ requested: rawVoice, configured: voice, engine: ttsEngine, voices: ttsVoices,
1293
+ isPocketVoice, isKnownVoice, isValidVoiceId,
1294
+ defaultVoice: DEFAULT_TTS_VOICE, defaultPocketVoice: DEFAULT_POCKET_VOICE,
1295
+ });
1296
+ if (!picked.ok) return sendJson(res, picked.status, { error: { message: picked.message, type: picked.type } });
1297
+ const { voice: useVoice, customId, speakerEmbedding } = picked;
1353
1298
 
1354
1299
  const pcm = await ttsEngine.synth(text, { voice: useVoice, speed, speakerEmbedding });
1355
1300
  // The ACTIVE model's rate, not the constant: a VITS/MMS model emits 16 kHz
@@ -1361,7 +1306,9 @@ export function createGateway(cfg = loadConfig()) {
1361
1306
  'Content-Length': String(out.length),
1362
1307
  'Cache-Control': 'no-store',
1363
1308
  'X-Tts-Sample-Rate': String(rate),
1364
- ...(customId ? { 'X-Tts-Voice': `custom:${customId}` } : {}),
1309
+ // Say which voice actually spoke, so a caller (or a person debugging one)
1310
+ // can check the setting took. Single-speaker models have nothing to say.
1311
+ ...(customId || ttsEngine.isPocket() || ttsEngine.supportsVoices() ? { 'X-Tts-Voice': useVoice } : {}),
1365
1312
  });
1366
1313
  return res.end(out);
1367
1314
  } catch (e) {
@@ -1389,7 +1336,9 @@ export function createGateway(cfg = loadConfig()) {
1389
1336
  // same composable model as everything else: any stage, with or without).
1390
1337
  // `diarize: true` (+ optional `speakerLabel` to pin the mic channel to a
1391
1338
  // name) attaches a speaker to each final.
1392
- const { id } = sttEngine.createSession({ lang: body?.lang, redact: body?.redact === true, diarize: wantDiarize, speakerLabel: body?.speakerLabel });
1339
+ // `endSilenceMs` (additive) lets a voice conversation wait longer for a
1340
+ // sentence to finish than dictation into a text box needs to.
1341
+ const { id } = sttEngine.createSession({ lang: body?.lang, redact: body?.redact === true, diarize: wantDiarize, speakerLabel: body?.speakerLabel, endSilenceMs: body?.endSilenceMs });
1393
1342
  return sendJson(res, 201, { id, state: sttEngine.state() });
1394
1343
  } catch (e) {
1395
1344
  return sendJson(res, e.code === 'too_many_sessions' ? 429 : 500, { error: { message: e.message, type: e.code || 'stt_error' } });
package/src/stt-engine.js CHANGED
@@ -238,13 +238,26 @@ let _decodeChain = Promise.resolve(); // whisper is effectively single-threaded
238
238
  export function sessionCount() { return _sessions.size; }
239
239
 
240
240
  /** @param {{ lang?: string, redact?: boolean, diarize?: boolean, speakerLabel?: string }} [opts] */
241
- export function createSession({ lang, redact = false, diarize: diarizeOpt = false, speakerLabel = null } = {}) {
241
+ // How long a pause commits a segment. The default suits dictation into a text box,
242
+ // where a final just appends and a short pause costs nothing. In a VOICE
243
+ // conversation every final is SENT as a question, so a 700ms pause mid-thought
244
+ // ("I want to… um… go to Google") sends half a sentence. Callers may ask for more.
245
+ const END_SILENCE_MIN_MS = 300;
246
+ const END_SILENCE_MAX_MS = 3000;
247
+ export function clampEndSilence(ms) {
248
+ const n = Number(ms);
249
+ if (!Number.isFinite(n)) return SILENCE_FINAL_MS;
250
+ return Math.min(END_SILENCE_MAX_MS, Math.max(END_SILENCE_MIN_MS, Math.round(n)));
251
+ }
252
+
253
+ export function createSession({ lang, redact = false, diarize: diarizeOpt = false, speakerLabel = null, endSilenceMs = SILENCE_FINAL_MS } = {}) {
242
254
  if (_sessions.size >= MAX_SESSIONS) {
243
255
  const e = /** @type {Error & { code?: string }} */ (new Error('too many concurrent dictation sessions'));
244
256
  e.code = 'too_many_sessions'; throw e;
245
257
  }
246
258
  const s = {
247
259
  id: randomUUID(),
260
+ endSilenceMs: clampEndSilence(endSilenceMs),
248
261
  lang: typeof lang === 'string' && lang ? lang.slice(0, 12) : null,
249
262
  langTried: false, // language auto-detect runs once per session (multilingual models)
250
263
  // Opaque to this engine: the server applies the redaction hop to finals when
@@ -374,7 +387,7 @@ async function decodeSession(s, { flush = false } = {}) {
374
387
  s.lastDecodeAt = Date.now();
375
388
 
376
389
  const audio = concatBuffer(s);
377
- const tail = Math.round((SILENCE_FINAL_MS / 1000) * SAMPLE_RATE);
390
+ const tail = Math.round(((s.endSilenceMs || SILENCE_FINAL_MS) / 1000) * SAMPLE_RATE);
378
391
  const trailingQuiet = audio.length > tail && rms(audio, audio.length - tail) < SILENCE_RMS;
379
392
 
380
393
  // Nothing but room tone? Don't decode (whisper hallucinates on silence) and
@@ -0,0 +1,78 @@
1
+ // Which voice a /tts request actually speaks in — decided in ONE place, with every
2
+ // dependency injected so the table below can be tested without loading a model.
3
+ //
4
+ // The rule that shapes this: a saved clone is a CHOICE, never a default. Someone's
5
+ // own voice print must not start speaking because the config was empty, named a
6
+ // voice from another model, or pointed at a voice since deleted. A model with
7
+ // built-in speakers (Pocket) falls back to one of those; only SpeechT5, which has
8
+ // no built-ins, has nothing to fall back to but a recording.
9
+ //
10
+ // A voice named EXPLICITLY in the request never falls back — it succeeds or fails
11
+ // loudly, so a caller that asked for a specific voice can trust what it got.
12
+ //
13
+ // Returns either { ok: true, voice, customId, speakerEmbedding }
14
+ // or { ok: false, status, type, message }.
15
+
16
+ export function resolveTtsVoice({
17
+ requested = null, // body.voice, or null when the caller left it to the config
18
+ configured, // the voice the request would use absent an override
19
+ engine, // { isPocket(), supportsCustomVoices(), supportsVoices(), arch() }
20
+ voices, // { parseCustomVoice, getVoice, listVoices, getPocketVoice }
21
+ isPocketVoice,
22
+ isKnownVoice,
23
+ isValidVoiceId,
24
+ defaultVoice, // Kokoro's default
25
+ defaultPocketVoice, // Pocket's built-in default
26
+ }) {
27
+ const voice = requested || configured;
28
+ const pocket = engine.isPocket();
29
+ let customId = voices.parseCustomVoice(voice);
30
+
31
+ // A Pocket built-in speaker is a NAME, not an embedding, so it is recognised
32
+ // before the custom-voice path — which would otherwise demand a recording for a
33
+ // model that ships eight voices of its own.
34
+ if (pocket && isPocketVoice(voice)) return { ok: true, voice, customId: null, speakerEmbedding: null };
35
+
36
+ if (engine.supportsCustomVoices()) {
37
+ let rec = customId ? voices.getVoice(customId) : null;
38
+ if (!rec && !requested) {
39
+ // Nothing valid was configured and nothing was asked for.
40
+ if (pocket) return { ok: true, voice: defaultPocketVoice, customId: null, speakerEmbedding: null };
41
+ const saved = voices.listVoices();
42
+ if (saved.length) { customId = saved[0].id; rec = voices.getVoice(customId); }
43
+ }
44
+ if (!rec) {
45
+ return customId
46
+ ? { ok: false, status: 404, type: 'bad_voice', message: 'no such saved voice' }
47
+ : { ok: false, status: 400, type: 'bad_voice', message: 'this model speaks in a voice you record — add one in Settings → Text-to-speech' };
48
+ }
49
+ // Which print to hand over depends on the engine: Pocket TTS takes its Mimi
50
+ // conditioning, SpeechT5 the 512-d x-vector. A voice saved before the pocket
51
+ // bundle existed has only the latter.
52
+ let speakerEmbedding;
53
+ if (pocket) {
54
+ speakerEmbedding = voices.getPocketVoice(customId);
55
+ if (!speakerEmbedding) {
56
+ return { ok: false, status: 409, type: 'voice_kind_missing', message: 'this voice was saved without a Pocket TTS conditioning — record it again with Pocket TTS selected' };
57
+ }
58
+ } else {
59
+ speakerEmbedding = rec.vec;
60
+ }
61
+ return { ok: true, voice: `custom:${customId}`, customId, speakerEmbedding };
62
+ }
63
+
64
+ if (customId) {
65
+ // Explicitly asked for a recorded voice this model cannot use — say so.
66
+ // Inherited from config, though, it is just a stale setting, and refusing to
67
+ // speak at all is a worse answer than speaking in the default voice.
68
+ if (requested) {
69
+ return { ok: false, status: 409, type: 'voice_unsupported', message: `the active model (${engine.arch()}) cannot use a recorded voice — switch to SpeechT5` };
70
+ }
71
+ return { ok: true, voice: defaultVoice, customId: null, speakerEmbedding: null };
72
+ }
73
+ if (pocket && !isPocketVoice(voice)) return { ok: true, voice: defaultPocketVoice, customId: null, speakerEmbedding: null };
74
+ if (engine.supportsVoices() && !pocket && !(isKnownVoice(voice) && isValidVoiceId(voice))) {
75
+ return { ok: false, status: 400, type: 'bad_voice', message: 'unknown or invalid voice' };
76
+ }
77
+ return { ok: true, voice, customId: null, speakerEmbedding: null };
78
+ }