@crossworks/voice-client 0.230.43
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE.md +135 -0
- package/package.json +20 -0
- package/src/adapters/registry.ts +296 -0
- package/src/adapters/retry.ts +193 -0
- package/src/adapters/types.ts +866 -0
- package/src/audio-tags.test.ts +221 -0
- package/src/audio-tags.ts +191 -0
- package/src/catalog.test.ts +144 -0
- package/src/catalog.ts +237 -0
- package/src/catalogs/anthropic.ts +135 -0
- package/src/catalogs/assemblyai.ts +54 -0
- package/src/catalogs/copilot.ts +63 -0
- package/src/catalogs/deepgram.ts +61 -0
- package/src/catalogs/deepseek.ts +85 -0
- package/src/catalogs/elevenlabs.ts +244 -0
- package/src/catalogs/google.ts +332 -0
- package/src/catalogs/huggingface.ts +180 -0
- package/src/catalogs/openai-image.ts +62 -0
- package/src/catalogs/openai-vision.ts +53 -0
- package/src/catalogs/openrouter.ts +221 -0
- package/src/catalogs/xai.ts +330 -0
- package/src/index.ts +48 -0
- package/src/providers.test.ts +172 -0
- package/src/providers.ts +262 -0
- package/src/types.ts +137 -0
- package/tsconfig.json +4 -0
- package/tsconfig.tsbuildinfo +1 -0
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Tests for the audio-tag composition + stripping helpers.
|
|
3
|
+
*
|
|
4
|
+
* Why these exist:
|
|
5
|
+
* 1. `composeAudioTagInstructions` is what makes Saskia use the
|
|
6
|
+
* right tags — if the paragraph it produces is malformed, the
|
|
7
|
+
* LLM either ignores it or hallucinates fake tags. Locking down
|
|
8
|
+
* the output shape catches future regressions.
|
|
9
|
+
*
|
|
10
|
+
* 2. `stripAudioTags` is the safety net that keeps bracketed tags
|
|
11
|
+
* out of text-mode replies. It must:
|
|
12
|
+
* - Remove `[laughs]` / `[whispers]` / `[strong british accent]`
|
|
13
|
+
* - NEVER touch markdown links `[label](url)` (these are not
|
|
14
|
+
* audio tags; stripping them would break formatted replies)
|
|
15
|
+
* - NEVER touch citation markers `[1]`, `[2,3]` (digits/commas
|
|
16
|
+
* excluded by the pattern)
|
|
17
|
+
* - Clean up whitespace so the strip doesn't leave double
|
|
18
|
+
* spaces or stranded line breaks
|
|
19
|
+
*/
|
|
20
|
+
|
|
21
|
+
import { describe, expect, it } from 'vitest';
|
|
22
|
+
import { composeAudioTagInstructions, stripAudioTags } from './audio-tags';
|
|
23
|
+
import { ELEVENLABS_V3_AUDIO_TAGS } from './catalogs/elevenlabs';
|
|
24
|
+
import { XAI_WRAPPING_TAGS } from './catalogs/xai';
|
|
25
|
+
|
|
26
|
+
describe('composeAudioTagInstructions', () => {
|
|
27
|
+
it('returns empty string for an empty tag list', () => {
|
|
28
|
+
// No-op when the active TTS has no tags — caller concatenates
|
|
29
|
+
// unconditionally, so this MUST be exactly '' (not whitespace).
|
|
30
|
+
expect(composeAudioTagInstructions([])).toBe('');
|
|
31
|
+
});
|
|
32
|
+
|
|
33
|
+
it('includes every tag passed in', () => {
|
|
34
|
+
const out = composeAudioTagInstructions(ELEVENLABS_V3_AUDIO_TAGS);
|
|
35
|
+
for (const t of ELEVENLABS_V3_AUDIO_TAGS) {
|
|
36
|
+
expect(out).toContain(t.tag);
|
|
37
|
+
expect(out).toContain(t.description);
|
|
38
|
+
}
|
|
39
|
+
});
|
|
40
|
+
|
|
41
|
+
it('groups by category — emotion / reaction / delivery etc.', () => {
|
|
42
|
+
// The prompt's readability comes from grouping. Lock in that
|
|
43
|
+
// each category we use shows up as a section header.
|
|
44
|
+
const out = composeAudioTagInstructions(ELEVENLABS_V3_AUDIO_TAGS);
|
|
45
|
+
expect(out).toContain('reaction:');
|
|
46
|
+
expect(out).toContain('emotion:');
|
|
47
|
+
expect(out).toContain('delivery:');
|
|
48
|
+
expect(out).toContain('cognitive:');
|
|
49
|
+
expect(out).toContain('tone:');
|
|
50
|
+
});
|
|
51
|
+
|
|
52
|
+
it('includes the "use sparingly" guidance', () => {
|
|
53
|
+
// Saskia tends to over-use new affordances. The prompt explicitly
|
|
54
|
+
// tells her one or two tags per voice reply is plenty. Without
|
|
55
|
+
// this guidance she peppers every line.
|
|
56
|
+
const out = composeAudioTagInstructions(ELEVENLABS_V3_AUDIO_TAGS);
|
|
57
|
+
expect(out).toMatch(/sparingly|one or two/i);
|
|
58
|
+
});
|
|
59
|
+
|
|
60
|
+
it('mentions the auto-strip behaviour for text replies', () => {
|
|
61
|
+
// Tells the LLM it's safe to use tags even if she ends up routed
|
|
62
|
+
// text-out. Otherwise she'd hedge and skip them.
|
|
63
|
+
const out = composeAudioTagInstructions(ELEVENLABS_V3_AUDIO_TAGS);
|
|
64
|
+
expect(out).toMatch(/strip|text/i);
|
|
65
|
+
});
|
|
66
|
+
|
|
67
|
+
it('returns empty string when both inline and wrapping are empty', () => {
|
|
68
|
+
expect(composeAudioTagInstructions([], [])).toBe('');
|
|
69
|
+
});
|
|
70
|
+
|
|
71
|
+
it('renders a wrapping-tag section with <name>…</name> forms', () => {
|
|
72
|
+
const out = composeAudioTagInstructions([], XAI_WRAPPING_TAGS);
|
|
73
|
+
// The open/close form is derived from the bare name.
|
|
74
|
+
expect(out).toContain('<whisper>…</whisper>');
|
|
75
|
+
expect(out).toContain('<soft>…</soft>');
|
|
76
|
+
// Every wrapping tag's description lands in the prompt.
|
|
77
|
+
for (const t of XAI_WRAPPING_TAGS) {
|
|
78
|
+
expect(out).toContain(t.description);
|
|
79
|
+
}
|
|
80
|
+
});
|
|
81
|
+
|
|
82
|
+
it('groups wrapping tags by their categories (volume/pitch/pacing/style)', () => {
|
|
83
|
+
const out = composeAudioTagInstructions([], XAI_WRAPPING_TAGS);
|
|
84
|
+
expect(out).toContain('volume:');
|
|
85
|
+
expect(out).toContain('pitch:');
|
|
86
|
+
expect(out).toContain('pacing:');
|
|
87
|
+
expect(out).toContain('style:');
|
|
88
|
+
});
|
|
89
|
+
|
|
90
|
+
it('renders both vocabularies together when both are passed', () => {
|
|
91
|
+
const out = composeAudioTagInstructions(ELEVENLABS_V3_AUDIO_TAGS, XAI_WRAPPING_TAGS);
|
|
92
|
+
expect(out).toContain('['); // an inline tag
|
|
93
|
+
expect(out).toContain('<whisper>…</whisper>'); // a wrapping tag
|
|
94
|
+
// One combined paragraph — single header, not two.
|
|
95
|
+
expect(out.match(/## Voice expression/g)?.length).toBe(1);
|
|
96
|
+
});
|
|
97
|
+
});
|
|
98
|
+
|
|
99
|
+
describe('stripAudioTags', () => {
|
|
100
|
+
it('strips a single-word tag', () => {
|
|
101
|
+
const { text, stripped } = stripAudioTags('Hey [laughs] that was funny.');
|
|
102
|
+
expect(text).toBe('Hey that was funny.');
|
|
103
|
+
expect(stripped).toBe(1);
|
|
104
|
+
});
|
|
105
|
+
|
|
106
|
+
it('strips multi-word tags', () => {
|
|
107
|
+
// ElevenLabs has tags like [strong British accent] and
|
|
108
|
+
// [resigned tone] — multi-word; the pattern allows spaces.
|
|
109
|
+
const { text } = stripAudioTags('[resigned tone] Fine.');
|
|
110
|
+
expect(text).toBe('Fine.');
|
|
111
|
+
});
|
|
112
|
+
|
|
113
|
+
it('strips multiple tags in one string', () => {
|
|
114
|
+
const { text, stripped } = stripAudioTags(
|
|
115
|
+
'[whispers] keep it quiet — [laughs] not THAT quiet.',
|
|
116
|
+
);
|
|
117
|
+
expect(text).toBe('keep it quiet — not THAT quiet.');
|
|
118
|
+
expect(stripped).toBe(2);
|
|
119
|
+
});
|
|
120
|
+
|
|
121
|
+
it('preserves markdown links — [label](url) is NOT a tag', () => {
|
|
122
|
+
// The negative lookahead on `(` is the critical mechanism. If
|
|
123
|
+
// this breaks, replies with markdown links lose their visible
|
|
124
|
+
// text.
|
|
125
|
+
const { text, stripped } = stripAudioTags('See the [docs](https://example.com) for details.');
|
|
126
|
+
expect(text).toBe('See the [docs](https://example.com) for details.');
|
|
127
|
+
expect(stripped).toBe(0);
|
|
128
|
+
});
|
|
129
|
+
|
|
130
|
+
it('preserves citation markers — [1] [2,3] are NOT tags', () => {
|
|
131
|
+
// Citation markers contain digits/commas; our pattern requires
|
|
132
|
+
// letters only. Lock down the behaviour.
|
|
133
|
+
const { text, stripped } = stripAudioTags('Per [1] and [2,3], the result holds.');
|
|
134
|
+
expect(text).toBe('Per [1] and [2,3], the result holds.');
|
|
135
|
+
expect(stripped).toBe(0);
|
|
136
|
+
});
|
|
137
|
+
|
|
138
|
+
it('handles an empty / null input gracefully', () => {
|
|
139
|
+
expect(stripAudioTags('').text).toBe('');
|
|
140
|
+
expect(stripAudioTags('').stripped).toBe(0);
|
|
141
|
+
});
|
|
142
|
+
|
|
143
|
+
it('counts tags removed in the return value', () => {
|
|
144
|
+
// The agent runtime puts this on the trace step meta so we can
|
|
145
|
+
// tell, post-hoc, whether the LLM was using tags this turn.
|
|
146
|
+
const { stripped } = stripAudioTags('[laughs] [sighs] [whispers] hi');
|
|
147
|
+
expect(stripped).toBe(3);
|
|
148
|
+
});
|
|
149
|
+
|
|
150
|
+
it('does not introduce double spaces after stripping', () => {
|
|
151
|
+
const { text } = stripAudioTags('Hey [laughs] there.');
|
|
152
|
+
// Should be one space between 'Hey' and 'there.', not two.
|
|
153
|
+
expect(text).not.toMatch(/ {2}/);
|
|
154
|
+
});
|
|
155
|
+
|
|
156
|
+
it('handles tags at the start of a line cleanly', () => {
|
|
157
|
+
const { text } = stripAudioTags('[whispers] secret message.');
|
|
158
|
+
expect(text).toBe('secret message.');
|
|
159
|
+
});
|
|
160
|
+
|
|
161
|
+
it('does not match unbracketed text that looks tag-shaped', () => {
|
|
162
|
+
// `laughs` without brackets is just a word.
|
|
163
|
+
const { text } = stripAudioTags('She laughs a lot.');
|
|
164
|
+
expect(text).toBe('She laughs a lot.');
|
|
165
|
+
});
|
|
166
|
+
|
|
167
|
+
it('caps the matchable token length (defensive)', () => {
|
|
168
|
+
// A very long bracketed string (e.g. 200 chars) shouldn't be
|
|
169
|
+
// treated as a tag — the pattern caps token length at 40 chars
|
|
170
|
+
// to avoid eating long-bracketed editorial inserts. Test this
|
|
171
|
+
// by passing a long bracketed phrase.
|
|
172
|
+
const long = '[' + 'word '.repeat(20).trim() + ']'; // ~100 chars
|
|
173
|
+
const { text } = stripAudioTags(`prefix ${long} suffix`);
|
|
174
|
+
// The long bracketed string should be preserved (NOT stripped).
|
|
175
|
+
expect(text).toContain(long);
|
|
176
|
+
});
|
|
177
|
+
|
|
178
|
+
// ─── Wrapping tags (<whisper>…</whisper>) ─────────────────────────
|
|
179
|
+
|
|
180
|
+
it('strips wrapping tag markers but keeps the inner text', () => {
|
|
181
|
+
const { text } = stripAudioTags("I'll tell you. <whisper>it's a secret</whisper>");
|
|
182
|
+
expect(text).toBe("I'll tell you. it's a secret");
|
|
183
|
+
});
|
|
184
|
+
|
|
185
|
+
it('counts each wrapping marker removed', () => {
|
|
186
|
+
// open + close = 2 markers for one pair.
|
|
187
|
+
const { stripped } = stripAudioTags('<soft>quietly</soft>');
|
|
188
|
+
expect(stripped).toBe(2);
|
|
189
|
+
});
|
|
190
|
+
|
|
191
|
+
it('strips every documented xAI wrapping name', () => {
|
|
192
|
+
for (const t of XAI_WRAPPING_TAGS) {
|
|
193
|
+
const { text } = stripAudioTags(`<${t.name}>hello</${t.name}>`);
|
|
194
|
+
expect(text).toBe('hello');
|
|
195
|
+
}
|
|
196
|
+
});
|
|
197
|
+
|
|
198
|
+
it('is case-insensitive on wrapping tag names', () => {
|
|
199
|
+
const { text } = stripAudioTags('<Whisper>psst</WHISPER>');
|
|
200
|
+
expect(text).toBe('psst');
|
|
201
|
+
});
|
|
202
|
+
|
|
203
|
+
it('handles inline and wrapping tags mixed in one reply', () => {
|
|
204
|
+
const { text } = stripAudioTags('[sigh] fine. <whisper>but keep it quiet</whisper> [laughs]');
|
|
205
|
+
expect(text).toBe('fine. but keep it quiet');
|
|
206
|
+
});
|
|
207
|
+
|
|
208
|
+
it('does NOT touch autolinks or generic HTML — only known speech names', () => {
|
|
209
|
+
// `<https://…>` autolinks, emails, and arbitrary HTML must survive;
|
|
210
|
+
// the stripper matches an explicit name allowlist, not any `<word>`.
|
|
211
|
+
const samples = [
|
|
212
|
+
'See <https://example.com> for more.',
|
|
213
|
+
'Email <jason@schoeman.me> if stuck.',
|
|
214
|
+
'A <div> and a <span> walk in.',
|
|
215
|
+
'Markdown *em* and _underscore_ stay.',
|
|
216
|
+
];
|
|
217
|
+
for (const s of samples) {
|
|
218
|
+
expect(stripAudioTags(s).text).toBe(s);
|
|
219
|
+
}
|
|
220
|
+
});
|
|
221
|
+
});
|
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Helpers for audio-tag composition and sanitisation.
|
|
3
|
+
*
|
|
4
|
+
* Two jobs:
|
|
5
|
+
* 1. Build the system-prompt paragraph that tells the chat agent
|
|
6
|
+
* which tags the active TTS will honour. Saskia uses this to
|
|
7
|
+
* emit `[laughs]` / `[whispers]` / `[sighs]` etc. in voice
|
|
8
|
+
* replies — but only when the configured TTS actually renders
|
|
9
|
+
* them (ElevenLabs v3 does; OpenAI tts-1 doesn't; the tag
|
|
10
|
+
* adapter is the source of truth).
|
|
11
|
+
*
|
|
12
|
+
* 2. Strip audio tags from text-mode replies before they go out
|
|
13
|
+
* on the chat surface. The LLM doesn't always know in advance
|
|
14
|
+
* whether the reply will go as text or voice (`[VOICE]` opt-in
|
|
15
|
+
* is decided AFTER the LLM finishes), so we let her use tags
|
|
16
|
+
* freely and pull them out if she ends up routed to sendMessage.
|
|
17
|
+
*
|
|
18
|
+
* Both helpers are pure (no I/O, no DB, no provider calls), so they
|
|
19
|
+
* test cleanly and can be called from either runtime or the web's
|
|
20
|
+
* server-action layer.
|
|
21
|
+
*/
|
|
22
|
+
|
|
23
|
+
import type { AudioTag, WrappingTag } from './adapters/types';
|
|
24
|
+
|
|
25
|
+
/** Group a tag list by its `category`, preserving first-seen order. */
|
|
26
|
+
function groupByCategory<T extends { category?: string }>(tags: readonly T[]): Map<string, T[]> {
|
|
27
|
+
const byCategory = new Map<string, T[]>();
|
|
28
|
+
for (const t of tags) {
|
|
29
|
+
const cat = t.category ?? 'other';
|
|
30
|
+
if (!byCategory.has(cat)) byCategory.set(cat, []);
|
|
31
|
+
byCategory.get(cat)!.push(t);
|
|
32
|
+
}
|
|
33
|
+
return byCategory;
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* Render a system-prompt paragraph listing the supported voice tags.
|
|
38
|
+
* The paragraph is appended to the agent's base system_prompt before
|
|
39
|
+
* the chat call. Returns an empty string when both lists are empty, so
|
|
40
|
+
* the caller can unconditionally concatenate the result without
|
|
41
|
+
* churning the prompt for tag-less TTS providers.
|
|
42
|
+
*
|
|
43
|
+
* Two vocabularies, one paragraph:
|
|
44
|
+
* - `inline` — point-in-time `[bracket]` cues (`[laughs]`, `[pause]`).
|
|
45
|
+
* - `wrapping` — angle-bracket pairs that style a span
|
|
46
|
+
* (`<whisper>…</whisper>`, `<soft>…</soft>`). xAI Grok voice today.
|
|
47
|
+
*
|
|
48
|
+
* `wrapping` defaults to `[]` so existing callers that only pass inline
|
|
49
|
+
* tags keep their exact output.
|
|
50
|
+
*/
|
|
51
|
+
export function composeAudioTagInstructions(
|
|
52
|
+
inline: readonly AudioTag[],
|
|
53
|
+
wrapping: readonly WrappingTag[] = [],
|
|
54
|
+
): string {
|
|
55
|
+
if (inline.length === 0 && wrapping.length === 0) return '';
|
|
56
|
+
|
|
57
|
+
// Group by category so the prompt is readable. The model handles
|
|
58
|
+
// long flat lists fine but grouped lists land more reliably in our
|
|
59
|
+
// testing — and humans editing the prompt later find it easier.
|
|
60
|
+
const inlineSections: string[] = [];
|
|
61
|
+
for (const [cat, list] of groupByCategory(inline)) {
|
|
62
|
+
const items = list.map((t) => ` ${t.tag} — ${t.description}`).join('\n');
|
|
63
|
+
inlineSections.push(`${cat}:\n${items}`);
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
const wrappingSections: string[] = [];
|
|
67
|
+
for (const [cat, list] of groupByCategory(wrapping)) {
|
|
68
|
+
const items = list.map((t) => ` <${t.name}>…</${t.name}> — ${t.description}`).join('\n');
|
|
69
|
+
wrappingSections.push(`${cat}:\n${items}`);
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
const lines: string[] = [
|
|
73
|
+
'',
|
|
74
|
+
'## Voice expression — speech tags',
|
|
75
|
+
'',
|
|
76
|
+
'When your reply will be spoken aloud (voice-in or [VOICE] opt-in),',
|
|
77
|
+
'you can use speech tags to add warmth, beats, and emotion. These',
|
|
78
|
+
'tags ONLY work with the currently-configured voice model; using',
|
|
79
|
+
'them sparingly is more effective than using them often.',
|
|
80
|
+
'',
|
|
81
|
+
];
|
|
82
|
+
|
|
83
|
+
if (inlineSections.length > 0) {
|
|
84
|
+
lines.push(
|
|
85
|
+
'Inline tags — written verbatim with square brackets, they fire at',
|
|
86
|
+
'the point they sit (e.g. [laughs] becomes a chuckle right there):',
|
|
87
|
+
'',
|
|
88
|
+
...inlineSections,
|
|
89
|
+
'',
|
|
90
|
+
);
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
if (wrappingSections.length > 0) {
|
|
94
|
+
lines.push(
|
|
95
|
+
'Wrapping tags — angle-bracket pairs that style the whole phrase',
|
|
96
|
+
'they surround (e.g. <whisper>keep this quiet</whisper>). Always',
|
|
97
|
+
'close the tag you open:',
|
|
98
|
+
'',
|
|
99
|
+
...wrappingSections,
|
|
100
|
+
'',
|
|
101
|
+
);
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
lines.push(
|
|
105
|
+
'Rules of thumb:',
|
|
106
|
+
'- One or two tags per voice reply is usually plenty.',
|
|
107
|
+
'- Place an inline tag right before the line it should affect;',
|
|
108
|
+
' wrap only the exact words a wrapping tag should style.',
|
|
109
|
+
'- If the reply ends up text rather than voice, the tags are',
|
|
110
|
+
' stripped automatically — feel free to use them and not worry.',
|
|
111
|
+
'',
|
|
112
|
+
);
|
|
113
|
+
|
|
114
|
+
return lines.join('\n');
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
/**
|
|
118
|
+
* The wrapping-tag names the stripper recognises. This is a curated
|
|
119
|
+
* superset of every provider's wrapping vocabulary (xAI Grok today)
|
|
120
|
+
* plus a few defensive synonyms — used ONLY by {@link stripAudioTags}
|
|
121
|
+
* as the safety net for text-out replies. We match by explicit name
|
|
122
|
+
* rather than "any `<word>`" so we never touch autolinks
|
|
123
|
+
* (`<https://…>`), email brackets, or real HTML/markdown the model may
|
|
124
|
+
* legitimately emit. Keep it conservative: only add names that are
|
|
125
|
+
* unambiguously speech-delivery styles.
|
|
126
|
+
*/
|
|
127
|
+
const WRAPPING_TAG_NAMES = [
|
|
128
|
+
'whisper',
|
|
129
|
+
'soft',
|
|
130
|
+
'loud',
|
|
131
|
+
'quiet',
|
|
132
|
+
'slow',
|
|
133
|
+
'fast',
|
|
134
|
+
'high',
|
|
135
|
+
'low',
|
|
136
|
+
'emphasis',
|
|
137
|
+
'singing',
|
|
138
|
+
'sing-song',
|
|
139
|
+
'shout',
|
|
140
|
+
] as const;
|
|
141
|
+
|
|
142
|
+
/**
|
|
143
|
+
* Strip voice tags from a reply that's going out as plain text. Two
|
|
144
|
+
* vocabularies are removed:
|
|
145
|
+
*
|
|
146
|
+
* - Inline `[word]` / `[word phrase]` cues (`[laughs]`, `[pause]`).
|
|
147
|
+
* Permissive pattern so a not-yet-catalogued bracket tag still gets
|
|
148
|
+
* stripped. Intentionally does NOT strip:
|
|
149
|
+
* · Markdown link text `[label](url)` — the `(` after `]` excludes it.
|
|
150
|
+
* · Citation markers `[1]` / `[2,3]` — digits/commas excluded.
|
|
151
|
+
* · Code blocks with brackets — this runs on chat content, not
|
|
152
|
+
* tool output. Add a fence-aware variant if that changes.
|
|
153
|
+
*
|
|
154
|
+
* - Wrapping `<name>…</name>` markers (`<whisper>`, `<soft>`, …). The
|
|
155
|
+
* INNER TEXT is kept; only the angle-bracket markers are removed,
|
|
156
|
+
* so `<whisper>it's a secret</whisper>` → `it's a secret`. Matched
|
|
157
|
+
* against {@link WRAPPING_TAG_NAMES} so generic angle-bracket
|
|
158
|
+
* content (autolinks, HTML) is left alone.
|
|
159
|
+
*
|
|
160
|
+
* Returns the cleaned text plus a count of markers removed (inline tags
|
|
161
|
+
* + wrapping markers) so callers can log or surface the strip.
|
|
162
|
+
*/
|
|
163
|
+
export function stripAudioTags(text: string): { text: string; stripped: number } {
|
|
164
|
+
if (!text) return { text: '', stripped: 0 };
|
|
165
|
+
let stripped = 0;
|
|
166
|
+
|
|
167
|
+
// Inline `[token]` — letters/spaces only (so `[laughs softly]` works),
|
|
168
|
+
// no digits/commas, not followed by `(` (markdown link). The negative
|
|
169
|
+
// lookahead on `(` is the distinguishing trick.
|
|
170
|
+
const inlinePattern = /\[([a-zA-Z][a-zA-Z\s]{0,40})\](?!\()/g;
|
|
171
|
+
let cleaned = text.replace(inlinePattern, () => {
|
|
172
|
+
stripped++;
|
|
173
|
+
return '';
|
|
174
|
+
});
|
|
175
|
+
|
|
176
|
+
// Wrapping `<name>` / `</name>` for the known speech-style names.
|
|
177
|
+
// Case-insensitive; removes the markers, keeps the inner text.
|
|
178
|
+
const wrappingPattern = new RegExp(`</?(?:${WRAPPING_TAG_NAMES.join('|')})\\s*>`, 'gi');
|
|
179
|
+
cleaned = cleaned.replace(wrappingPattern, () => {
|
|
180
|
+
stripped++;
|
|
181
|
+
return '';
|
|
182
|
+
});
|
|
183
|
+
|
|
184
|
+
// Collapse runs of whitespace introduced by the strip, then trim
|
|
185
|
+
// leading/trailing space without losing inline structure.
|
|
186
|
+
const tidied = cleaned
|
|
187
|
+
.replace(/[ \t]+/g, ' ')
|
|
188
|
+
.replace(/ ?\n ?/g, '\n')
|
|
189
|
+
.trim();
|
|
190
|
+
return { text: tidied, stripped };
|
|
191
|
+
}
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Tests for the static TTS/STT catalogue.
|
|
3
|
+
*
|
|
4
|
+
* Why these tests exist: the catalogue is hand-maintained (OpenAI has
|
|
5
|
+
* no `/v1/audio/voices` endpoint) and feeds the agent-settings UI. If
|
|
6
|
+
* a model→voices mapping is wrong, the dropdown shows voices that
|
|
7
|
+
* the API will refuse at runtime. These tests lock down the
|
|
8
|
+
* documented model/voice combinations as of May 2026 so a typo or
|
|
9
|
+
* accidental drop is caught at PR time, not at the next voice
|
|
10
|
+
* message.
|
|
11
|
+
*
|
|
12
|
+
* The DISCOVERY layer (live `/v1/models` filtering) is integration-
|
|
13
|
+
* shaped — it needs a real OpenAI key — so it's not tested here.
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
import { describe, expect, it } from 'vitest';
|
|
17
|
+
import {
|
|
18
|
+
ALL_OPENAI_VOICES,
|
|
19
|
+
OPENAI_STT_MODELS,
|
|
20
|
+
OPENAI_TTS_MODELS,
|
|
21
|
+
VOICE_DESCRIPTIONS,
|
|
22
|
+
getSttModel,
|
|
23
|
+
getTtsModel,
|
|
24
|
+
isOpenAiVoice,
|
|
25
|
+
voicesForModel,
|
|
26
|
+
} from './catalog';
|
|
27
|
+
|
|
28
|
+
describe('OPENAI_TTS_MODELS', () => {
|
|
29
|
+
it('contains all currently-published TTS models', () => {
|
|
30
|
+
const ids = OPENAI_TTS_MODELS.map((m) => m.id);
|
|
31
|
+
// Lock down the exact set so a model rename or drop is loud.
|
|
32
|
+
expect(new Set(ids)).toEqual(new Set(['gpt-4o-mini-tts', 'tts-1', 'tts-1-hd']));
|
|
33
|
+
});
|
|
34
|
+
|
|
35
|
+
it('lists gpt-4o-mini-tts first (recommended)', () => {
|
|
36
|
+
// The UI relies on catalog order — gpt-4o-mini-tts is the
|
|
37
|
+
// recommended default, so it must lead the dropdown.
|
|
38
|
+
expect(OPENAI_TTS_MODELS[0]?.id).toBe('gpt-4o-mini-tts');
|
|
39
|
+
});
|
|
40
|
+
|
|
41
|
+
it('marks gpt-4o-mini-tts as the only model supporting instructions', () => {
|
|
42
|
+
// `instructions` is the "speak warmly" style-steering parameter.
|
|
43
|
+
// It's silently ignored by tts-1 / tts-1-hd; the UI greys out the
|
|
44
|
+
// input when those models are selected, so this flag must be
|
|
45
|
+
// accurate.
|
|
46
|
+
const flags = Object.fromEntries(OPENAI_TTS_MODELS.map((m) => [m.id, m.supportsInstructions]));
|
|
47
|
+
expect(flags['gpt-4o-mini-tts']).toBe(true);
|
|
48
|
+
expect(flags['tts-1']).toBe(false);
|
|
49
|
+
expect(flags['tts-1-hd']).toBe(false);
|
|
50
|
+
});
|
|
51
|
+
|
|
52
|
+
it('gpt-4o-mini-tts ships the full 13 voices', () => {
|
|
53
|
+
const m = getTtsModel('gpt-4o-mini-tts');
|
|
54
|
+
expect(m).not.toBeNull();
|
|
55
|
+
expect(m!.voices.length).toBe(13);
|
|
56
|
+
// The signature additions over tts-1 must be present.
|
|
57
|
+
expect(m!.voices).toContain('ballad');
|
|
58
|
+
expect(m!.voices).toContain('verse');
|
|
59
|
+
expect(m!.voices).toContain('marin');
|
|
60
|
+
expect(m!.voices).toContain('cedar');
|
|
61
|
+
});
|
|
62
|
+
|
|
63
|
+
it('tts-1 and tts-1-hd ship the same 9 voices (no instructions support)', () => {
|
|
64
|
+
const a = getTtsModel('tts-1');
|
|
65
|
+
const b = getTtsModel('tts-1-hd');
|
|
66
|
+
expect(a).not.toBeNull();
|
|
67
|
+
expect(b).not.toBeNull();
|
|
68
|
+
expect(a!.voices.length).toBe(9);
|
|
69
|
+
expect(b!.voices.length).toBe(9);
|
|
70
|
+
expect(new Set(a!.voices)).toEqual(new Set(b!.voices));
|
|
71
|
+
// The expansion-only voices must NOT be in the older models.
|
|
72
|
+
for (const exp of ['ballad', 'verse', 'marin', 'cedar']) {
|
|
73
|
+
expect(a!.voices).not.toContain(exp);
|
|
74
|
+
expect(b!.voices).not.toContain(exp);
|
|
75
|
+
}
|
|
76
|
+
});
|
|
77
|
+
});
|
|
78
|
+
|
|
79
|
+
describe('OPENAI_STT_MODELS', () => {
|
|
80
|
+
it('contains whisper-1 + the gpt-4o transcription variants', () => {
|
|
81
|
+
const ids = OPENAI_STT_MODELS.map((m) => m.id);
|
|
82
|
+
expect(new Set(ids)).toEqual(
|
|
83
|
+
new Set(['whisper-1', 'gpt-4o-mini-transcribe', 'gpt-4o-transcribe']),
|
|
84
|
+
);
|
|
85
|
+
});
|
|
86
|
+
|
|
87
|
+
it('lists whisper-1 first (stable, cheapest)', () => {
|
|
88
|
+
// We default to whisper-1 for new STT workers because it's the
|
|
89
|
+
// longest-stable variant. Catalog order drives that default.
|
|
90
|
+
expect(OPENAI_STT_MODELS[0]?.id).toBe('whisper-1');
|
|
91
|
+
});
|
|
92
|
+
});
|
|
93
|
+
|
|
94
|
+
describe('voicesForModel', () => {
|
|
95
|
+
it('returns the model-specific voice list with descriptions', () => {
|
|
96
|
+
const voices = voicesForModel('gpt-4o-mini-tts');
|
|
97
|
+
expect(voices.length).toBe(13);
|
|
98
|
+
const nova = voices.find((v) => v.id === 'nova');
|
|
99
|
+
expect(nova?.description).toContain('Saskia');
|
|
100
|
+
});
|
|
101
|
+
|
|
102
|
+
it('returns 9 voices for tts-1', () => {
|
|
103
|
+
expect(voicesForModel('tts-1').length).toBe(9);
|
|
104
|
+
});
|
|
105
|
+
|
|
106
|
+
it('returns an empty array for an unknown model (caller decides fallback)', () => {
|
|
107
|
+
// If the operator types a custom model id (e.g. a fine-tuned one),
|
|
108
|
+
// we don't have a voice list. Return [] and let the caller decide
|
|
109
|
+
// whether to fall back to the full catalogue or refuse.
|
|
110
|
+
expect(voicesForModel('made-up-model')).toEqual([]);
|
|
111
|
+
});
|
|
112
|
+
});
|
|
113
|
+
|
|
114
|
+
describe('isOpenAiVoice', () => {
|
|
115
|
+
it('accepts every voice in ALL_OPENAI_VOICES', () => {
|
|
116
|
+
for (const v of ALL_OPENAI_VOICES) {
|
|
117
|
+
expect(isOpenAiVoice(v)).toBe(true);
|
|
118
|
+
}
|
|
119
|
+
});
|
|
120
|
+
|
|
121
|
+
it('rejects typos and unknown names', () => {
|
|
122
|
+
expect(isOpenAiVoice('novah')).toBe(false);
|
|
123
|
+
expect(isOpenAiVoice('Nova')).toBe(false);
|
|
124
|
+
expect(isOpenAiVoice('')).toBe(false);
|
|
125
|
+
});
|
|
126
|
+
});
|
|
127
|
+
|
|
128
|
+
describe('VOICE_DESCRIPTIONS', () => {
|
|
129
|
+
it('has a description for every voice in ALL_OPENAI_VOICES (no gaps)', () => {
|
|
130
|
+
// The UI dropdown reads from this; a missing entry would render
|
|
131
|
+
// as "voice — undefined" and look broken.
|
|
132
|
+
for (const v of ALL_OPENAI_VOICES) {
|
|
133
|
+
expect(VOICE_DESCRIPTIONS[v]).toBeTruthy();
|
|
134
|
+
expect(VOICE_DESCRIPTIONS[v]?.length ?? 0).toBeGreaterThan(0);
|
|
135
|
+
}
|
|
136
|
+
});
|
|
137
|
+
});
|
|
138
|
+
|
|
139
|
+
describe('getSttModel / getTtsModel', () => {
|
|
140
|
+
it('returns null for unknown ids (no nil-throwing)', () => {
|
|
141
|
+
expect(getTtsModel('tts-99')).toBeNull();
|
|
142
|
+
expect(getSttModel('whisper-99')).toBeNull();
|
|
143
|
+
});
|
|
144
|
+
});
|