@athanlab/mcp 0.0.0-stage → 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +222 -2
- package/dist/api/client.js +282 -0
- package/dist/api/client.js.map +1 -0
- package/dist/api/errors.js +284 -0
- package/dist/api/errors.js.map +1 -0
- package/dist/api/features.js +123 -0
- package/dist/api/features.js.map +1 -0
- package/dist/api/types.js +15 -0
- package/dist/api/types.js.map +1 -0
- package/dist/batch.js +394 -0
- package/dist/batch.js.map +1 -0
- package/dist/clock.js +31 -0
- package/dist/clock.js.map +1 -0
- package/dist/config.js +101 -0
- package/dist/config.js.map +1 -0
- package/dist/files.js +167 -0
- package/dist/files.js.map +1 -0
- package/dist/guide.js +96 -0
- package/dist/guide.js.map +1 -0
- package/dist/index.js +66 -0
- package/dist/index.js.map +1 -0
- package/dist/lock.js +158 -0
- package/dist/lock.js.map +1 -0
- package/dist/log.js +50 -0
- package/dist/log.js.map +1 -0
- package/dist/script.js +869 -0
- package/dist/script.js.map +1 -0
- package/dist/server.js +76 -0
- package/dist/server.js.map +1 -0
- package/dist/speak.js +95 -0
- package/dist/speak.js.map +1 -0
- package/dist/speech.js +230 -0
- package/dist/speech.js.map +1 -0
- package/dist/timeline.js +37 -0
- package/dist/timeline.js.map +1 -0
- package/dist/tools.js +617 -0
- package/dist/tools.js.map +1 -0
- package/dist/version.js +6 -0
- package/dist/version.js.map +1 -0
- package/dist/wav.js +169 -0
- package/dist/wav.js.map +1 -0
- package/package.json +51 -4
- package/skill/athanlab-voice/SKILL.md +166 -0
package/dist/tools.js
ADDED
|
@@ -0,0 +1,617 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
2
|
+
import { formatError, toAthanLabError, AthanLabError } from './api/errors.js';
|
|
3
|
+
import { JOB_STATUSES, NUMBER_MODES, OUTPUT_FORMATS } from './api/types.js';
|
|
4
|
+
import { ensureDir, sanitizeFilename, uniquePath, writeResponseToFile } from './files.js';
|
|
5
|
+
import { redact } from './log.js';
|
|
6
|
+
import { CONFIRM_CHARACTERS_THRESHOLD, CONFIRM_LINES_THRESHOLD, runScript } from './script.js';
|
|
7
|
+
import { runSpeak, DEFAULT_SPEAK_TIMEOUT_SECONDS } from './speak.js';
|
|
8
|
+
import { checkJobId, downloadJobAudio, getJob, jobOutcomeError, listJobs, quoteSpeech, stamp, summarizeJob, } from './speech.js';
|
|
9
|
+
const MAX_PREVIEW_BYTES = 25 * 1024 * 1024;
|
|
10
|
+
// ── Shared schemas ───────────────────────────────
|
|
11
|
+
const voiceId = z.string().min(1).max(128)
|
|
12
|
+
.describe('A voice id from athanlab_list_voices. Leave it out to use the API default voice.');
|
|
13
|
+
const jobId = z.string().regex(/^[0-9a-f]{32}$/, 'a job id is 32 lowercase hex characters')
|
|
14
|
+
.describe('The job id (32 lowercase hex characters) from athanlab_speak, athanlab_list_jobs or a timeout result.');
|
|
15
|
+
const speechText = z.string().min(1).max(5000)
|
|
16
|
+
.describe('Burmese text to speak, 1–5,000 characters (UTF-16 units: an emoji counts 2). Write Burmese script, end sentences with ။ ! or ?. See the athanlab-writing-guide.');
|
|
17
|
+
const numberMode = z.enum(NUMBER_MODES)
|
|
18
|
+
.describe('How digits are read: smart (default: Burmese digits as words, Latin digits as written), place (all digits as Burmese place words), digits (Burmese, one digit at a time — phone numbers), english (English number words), english_digits (English, one digit at a time).');
|
|
19
|
+
const outputFormat = z.enum(OUTPUT_FORMATS);
|
|
20
|
+
const filename = z.string().min(1).max(200)
|
|
21
|
+
.describe('Optional file name (no folders: only the last path segment is used, the extension is forced to the format). Written inside the server\'s output directory; never overwrites — an existing name gets a suffix.');
|
|
22
|
+
const segmentOut = z.object({
|
|
23
|
+
index: z.number(),
|
|
24
|
+
start_seconds: z.number(),
|
|
25
|
+
end_seconds: z.number(),
|
|
26
|
+
characters: z.number(),
|
|
27
|
+
});
|
|
28
|
+
const timingOut = {
|
|
29
|
+
speech_start_seconds: z.number().nullable().describe('Where speech starts in the file (seconds); null until the API reports timing.'),
|
|
30
|
+
speech_end_seconds: z.number().nullable().describe('Where speech ends in the file (seconds); null until the API reports timing.'),
|
|
31
|
+
segments: z.array(segmentOut).nullable().describe('Timed spans of the audio; null until the API reports timing.'),
|
|
32
|
+
};
|
|
33
|
+
// Values the API owns (statuses, formats, sources) are plain strings in the
|
|
34
|
+
// output schemas: the API may add values within v1, and a client validating
|
|
35
|
+
// structuredContent must not reject a job because of one.
|
|
36
|
+
const jobSummaryShape = {
|
|
37
|
+
job_id: z.string(),
|
|
38
|
+
status: z.string().describe('processing, succeeded, failed or cancelled.'),
|
|
39
|
+
progress: z.number().nullable(),
|
|
40
|
+
created_at: z.string(),
|
|
41
|
+
completed_at: z.string().nullable(),
|
|
42
|
+
voice_id: z.string().nullable(),
|
|
43
|
+
output_format: z.string(),
|
|
44
|
+
number_mode: z.string(),
|
|
45
|
+
characters: z.number(),
|
|
46
|
+
characters_charged: z.number(),
|
|
47
|
+
characters_refunded: z.number(),
|
|
48
|
+
audio: z.object({ format: z.string(), duration_seconds: z.number(), expires_at: z.string() }).nullable(),
|
|
49
|
+
...timingOut,
|
|
50
|
+
error: z.object({ code: z.string(), message: z.string(), retryable: z.boolean() }).nullable(),
|
|
51
|
+
metadata: z.record(z.string(), z.string()),
|
|
52
|
+
};
|
|
53
|
+
const jobSummaryOut = z.object(jobSummaryShape);
|
|
54
|
+
function speedShape(features) {
|
|
55
|
+
if (!features.speed)
|
|
56
|
+
return {};
|
|
57
|
+
const { min, max } = features.speed;
|
|
58
|
+
return {
|
|
59
|
+
speed: z.number().min(min).max(max).optional()
|
|
60
|
+
.describe(`Speaking pace, ${min}–${max} in steps of 0.01 (1 = normal). Pitch is preserved; the price does not change.`),
|
|
61
|
+
};
|
|
62
|
+
}
|
|
63
|
+
function replacementsShape(features) {
|
|
64
|
+
if (!features.replacements)
|
|
65
|
+
return {};
|
|
66
|
+
const limits = features.replacements;
|
|
67
|
+
return {
|
|
68
|
+
replacements: z.array(z.object({
|
|
69
|
+
from: z.string().min(1).max(limits.fromMaxLength).describe('Literal text as written in `text` (no regex).'),
|
|
70
|
+
to: z.string().max(limits.toMaxLength).describe('What to say instead (Burmese script). "" deletes the match.'),
|
|
71
|
+
case_sensitive: z.boolean().optional().describe('Default true; false folds ASCII A–Z only.'),
|
|
72
|
+
})).max(limits.maxRules).optional()
|
|
73
|
+
.describe(`Pronunciation rules applied only to what is spoken, e.g. [{"from": "AthanLab", "to": "အသံလက်ဘ်"}]. Up to ${limits.maxRules}. Billing stays on the length of text.`),
|
|
74
|
+
};
|
|
75
|
+
}
|
|
76
|
+
function scrub(value, secrets) {
|
|
77
|
+
return JSON.parse(redact(JSON.stringify(value), secrets));
|
|
78
|
+
}
|
|
79
|
+
/**
|
|
80
|
+
* notifications/progress for a client that sent a progressToken.
|
|
81
|
+
*
|
|
82
|
+
* MCP requires progress to increase with every notification, but a job can
|
|
83
|
+
* sit at the same progress for minutes (a single-piece job reports 0 until it
|
|
84
|
+
* finishes; a cold start takes over a minute) — and a client that keeps its
|
|
85
|
+
* request alive on progress must still hear from us on every poll. So a
|
|
86
|
+
* report that does not move progress goes out anyway as a heartbeat whose
|
|
87
|
+
* value creeps up by a shrinking step (n + k/(k+1)): always increasing, never
|
|
88
|
+
* reaching the next whole value. Callers report whole numbers.
|
|
89
|
+
*/
|
|
90
|
+
export function progressReporter(extra) {
|
|
91
|
+
const token = extra?._meta?.progressToken;
|
|
92
|
+
if (token === undefined || !extra)
|
|
93
|
+
return undefined;
|
|
94
|
+
let last = Number.NEGATIVE_INFINITY;
|
|
95
|
+
let base = 0;
|
|
96
|
+
let beats = 0;
|
|
97
|
+
return (progress, total, message) => {
|
|
98
|
+
let value;
|
|
99
|
+
if (progress > last) {
|
|
100
|
+
value = progress;
|
|
101
|
+
base = progress;
|
|
102
|
+
beats = 0;
|
|
103
|
+
}
|
|
104
|
+
else {
|
|
105
|
+
beats++;
|
|
106
|
+
value = base + beats / (beats + 1);
|
|
107
|
+
if (!(value > last))
|
|
108
|
+
return;
|
|
109
|
+
}
|
|
110
|
+
last = value;
|
|
111
|
+
void extra.sendNotification({
|
|
112
|
+
method: 'notifications/progress',
|
|
113
|
+
params: { progressToken: token, progress: value, ...(total !== undefined ? { total } : {}), ...(message ? { message } : {}) },
|
|
114
|
+
}).catch(() => { });
|
|
115
|
+
};
|
|
116
|
+
}
|
|
117
|
+
function makeRunner(deps) {
|
|
118
|
+
return async function run(tool, fn) {
|
|
119
|
+
try {
|
|
120
|
+
const { text, structured, isError } = await fn();
|
|
121
|
+
return {
|
|
122
|
+
content: [{ type: 'text', text: redact(text, deps.secrets) }],
|
|
123
|
+
structuredContent: scrub(structured, deps.secrets),
|
|
124
|
+
...(isError ? { isError: true } : {}),
|
|
125
|
+
};
|
|
126
|
+
}
|
|
127
|
+
catch (err) {
|
|
128
|
+
const e = toAthanLabError(err);
|
|
129
|
+
deps.ctx.logger.warn('tool call failed', { tool, code: e.code, status: e.status, request_id: e.requestId });
|
|
130
|
+
// No structuredContent on errors: clients validate any structuredContent
|
|
131
|
+
// against the tool's outputSchema, error or not. The second block is the
|
|
132
|
+
// same error as JSON, for agents that branch on fields.
|
|
133
|
+
return {
|
|
134
|
+
isError: true,
|
|
135
|
+
content: [
|
|
136
|
+
{ type: 'text', text: redact(formatError(e), deps.secrets) },
|
|
137
|
+
{ type: 'text', text: redact(JSON.stringify({ error: e.toJSON() }), deps.secrets) },
|
|
138
|
+
],
|
|
139
|
+
};
|
|
140
|
+
}
|
|
141
|
+
};
|
|
142
|
+
}
|
|
143
|
+
function speakInput(args, features, fallbackFormat) {
|
|
144
|
+
return {
|
|
145
|
+
text: args.text,
|
|
146
|
+
voice_id: args.voice_id,
|
|
147
|
+
output_format: args.output_format ?? fallbackFormat,
|
|
148
|
+
number_mode: args.number_mode,
|
|
149
|
+
...(features.speed && args.speed !== undefined ? { speed: args.speed } : {}),
|
|
150
|
+
...(features.replacements && args.replacements ? { replacements: args.replacements } : {}),
|
|
151
|
+
metadata: { client: 'athanlab-mcp' },
|
|
152
|
+
};
|
|
153
|
+
}
|
|
154
|
+
function normalizeSpeak(result) {
|
|
155
|
+
if (result.status === 'succeeded') {
|
|
156
|
+
return {
|
|
157
|
+
status: 'succeeded',
|
|
158
|
+
path: result.path,
|
|
159
|
+
job_id: result.job_id,
|
|
160
|
+
output_format: result.output_format,
|
|
161
|
+
duration_seconds: result.duration_seconds,
|
|
162
|
+
speech_start_seconds: result.speech_start_seconds,
|
|
163
|
+
speech_end_seconds: result.speech_end_seconds,
|
|
164
|
+
segments: result.segments,
|
|
165
|
+
characters_charged: result.characters_charged,
|
|
166
|
+
voice_id: result.voice_id,
|
|
167
|
+
bytes: result.bytes,
|
|
168
|
+
replayed: result.replayed,
|
|
169
|
+
progress: 1,
|
|
170
|
+
message: null,
|
|
171
|
+
};
|
|
172
|
+
}
|
|
173
|
+
return {
|
|
174
|
+
status: 'processing',
|
|
175
|
+
path: null,
|
|
176
|
+
job_id: result.job_id,
|
|
177
|
+
output_format: result.output_format,
|
|
178
|
+
duration_seconds: null,
|
|
179
|
+
speech_start_seconds: null,
|
|
180
|
+
speech_end_seconds: null,
|
|
181
|
+
segments: null,
|
|
182
|
+
characters_charged: result.characters_charged,
|
|
183
|
+
voice_id: null,
|
|
184
|
+
bytes: null,
|
|
185
|
+
replayed: null,
|
|
186
|
+
progress: result.progress,
|
|
187
|
+
message: result.message,
|
|
188
|
+
};
|
|
189
|
+
}
|
|
190
|
+
function jobLine(s) {
|
|
191
|
+
const audio = s.audio ? `, ${s.audio.duration_seconds.toFixed(1)} s ${s.audio.format}` : '';
|
|
192
|
+
const error = s.error ? `, ${s.error.code}` : '';
|
|
193
|
+
return `${s.job_id} ${s.status}${s.status === 'processing' && s.progress !== null ? ` ${Math.round(s.progress * 100)}%` : ''} ${s.characters} chars${audio}${error} (${s.created_at})`;
|
|
194
|
+
}
|
|
195
|
+
// ── Registration ─────────────────────────────────
|
|
196
|
+
export function registerTools(server, deps) {
|
|
197
|
+
const { ctx } = deps;
|
|
198
|
+
const { features, config } = ctx;
|
|
199
|
+
const run = makeRunner(deps);
|
|
200
|
+
const outputDirNote = `Files are written to ${config.outputDir}.`;
|
|
201
|
+
server.registerTool('athanlab_list_voices', {
|
|
202
|
+
title: 'List AthanLab voices',
|
|
203
|
+
description: 'Lists the voices this API key can speak with: AthanLab\'s official Burmese voices (source "athanlab") and the voices in the account\'s own library (source "user"), plus default_voice_id — the voice used when a speech call names none. Call this first to pick a voice_id for athanlab_speak / athanlab_speak_script. Free; read-only.',
|
|
204
|
+
inputSchema: {},
|
|
205
|
+
outputSchema: {
|
|
206
|
+
voices: z.array(z.object({
|
|
207
|
+
id: z.string(),
|
|
208
|
+
name: z.string(),
|
|
209
|
+
category: z.string(),
|
|
210
|
+
source: z.string().describe('athanlab (official) or user (your library).'),
|
|
211
|
+
is_default: z.boolean(),
|
|
212
|
+
})),
|
|
213
|
+
default_voice_id: z.string().nullable(),
|
|
214
|
+
},
|
|
215
|
+
annotations: { title: 'List voices', readOnlyHint: true, idempotentHint: true, openWorldHint: true },
|
|
216
|
+
}, async () => run('athanlab_list_voices', async () => {
|
|
217
|
+
const { data } = await ctx.client.json('/voices');
|
|
218
|
+
const voices = data.data.map(v => ({ id: v.id, name: v.name, category: v.category, source: v.source, is_default: v.is_default }));
|
|
219
|
+
const text = voices.length
|
|
220
|
+
? `${voices.length} voice(s):\n${voices.map(v => `- ${v.id} — ${v.name} (${v.category}, ${v.source})${v.is_default ? ' [default]' : ''}`).join('\n')}${data.default_voice_id ? '' : '\nThe API default voice is unavailable right now: pass voice_id explicitly.'}`
|
|
221
|
+
: 'No voices are available to this key.';
|
|
222
|
+
return { text, structured: { voices, default_voice_id: data.default_voice_id } };
|
|
223
|
+
}));
|
|
224
|
+
server.registerTool('athanlab_preview_voice', {
|
|
225
|
+
title: 'Preview an AthanLab voice',
|
|
226
|
+
description: `Returns a signed URL of a short sample of one voice (valid about an hour; no API key needed to play it), so the user can hear a voice before choosing it. With save: true the sample is also downloaded into the output directory. Free. ${outputDirNote}`,
|
|
227
|
+
inputSchema: {
|
|
228
|
+
voice_id: z.string().min(1).max(128).describe('A voice id from athanlab_list_voices.'),
|
|
229
|
+
save: z.boolean().default(false).describe('Also download the sample into the output directory.'),
|
|
230
|
+
},
|
|
231
|
+
outputSchema: {
|
|
232
|
+
voice_id: z.string(),
|
|
233
|
+
name: z.string(),
|
|
234
|
+
url: z.string(),
|
|
235
|
+
expires_in: z.number(),
|
|
236
|
+
path: z.string().nullable(),
|
|
237
|
+
bytes: z.number().nullable(),
|
|
238
|
+
},
|
|
239
|
+
annotations: { title: 'Preview voice', readOnlyHint: false, destructiveHint: false, idempotentHint: false, openWorldHint: true },
|
|
240
|
+
}, async (args) => run('athanlab_preview_voice', async () => {
|
|
241
|
+
const { data } = await ctx.client.json(`/voices/${encodeURIComponent(args.voice_id)}/preview`);
|
|
242
|
+
let savedPath = null;
|
|
243
|
+
let bytes = null;
|
|
244
|
+
if (args.save) {
|
|
245
|
+
const url = new URL(data.url);
|
|
246
|
+
if (url.protocol !== 'https:' && !(url.protocol === 'http:' && /^(localhost|127\.0\.0\.1)$/.test(url.hostname))) {
|
|
247
|
+
throw AthanLabError.client('invalid_response', 'The sample URL is not https; not downloading it.');
|
|
248
|
+
}
|
|
249
|
+
// A signed storage URL: fetched WITHOUT the API key, which only ever goes to the API.
|
|
250
|
+
const fetchImpl = ctx.fetchImpl ?? fetch;
|
|
251
|
+
const res = await fetchImpl(url.toString(), { signal: AbortSignal.timeout(120_000) });
|
|
252
|
+
if (!res.ok) {
|
|
253
|
+
throw new AthanLabError({ code: 'download_failed', message: `The sample download answered HTTP ${res.status}.`, source: 'network', retryable: res.status >= 500 });
|
|
254
|
+
}
|
|
255
|
+
const type = res.headers.get('content-type') ?? '';
|
|
256
|
+
const fromPath = url.pathname.toLowerCase().match(/\.(wav|mp3)$/)?.[1];
|
|
257
|
+
const ext = /mpeg|mp3/.test(type) ? 'mp3' : /wav/.test(type) ? 'wav' : fromPath ?? 'wav';
|
|
258
|
+
await ensureDir(config.outputDir);
|
|
259
|
+
const name = sanitizeFilename(`voice-preview-${args.voice_id}`, ext, 'voice-preview');
|
|
260
|
+
savedPath = await uniquePath(config.outputDir, name, stamp(ctx.clock.now()));
|
|
261
|
+
bytes = await writeResponseToFile(res, savedPath, MAX_PREVIEW_BYTES);
|
|
262
|
+
}
|
|
263
|
+
const text = `Sample of ${data.name} (${data.id}): ${data.url}\nThe link expires in ${Math.round(data.expires_in / 60)} minutes.${savedPath ? `\nSaved to ${savedPath}` : ''}`;
|
|
264
|
+
return { text, structured: { voice_id: data.id, name: data.name, url: data.url, expires_in: data.expires_in, path: savedPath, bytes } };
|
|
265
|
+
}));
|
|
266
|
+
server.registerTool('athanlab_quote', {
|
|
267
|
+
title: 'Quote speech (free dry run)',
|
|
268
|
+
description: 'Validates a text exactly as athanlab_speak would and returns what it would cost — without creating a job or charging anything. Returns characters (the exact charge: UTF-16 length), dispatches (how many pieces the text is split into), spendable (the account\'s balance now) and sufficient. Use it before long texts, or to check that a text will be accepted (it surfaces unspeakable_text, text_too_long and voice_not_found errors for free).',
|
|
269
|
+
inputSchema: {
|
|
270
|
+
text: speechText,
|
|
271
|
+
voice_id: voiceId.optional(),
|
|
272
|
+
number_mode: numberMode.optional(),
|
|
273
|
+
...speedShape(features),
|
|
274
|
+
...replacementsShape(features),
|
|
275
|
+
},
|
|
276
|
+
outputSchema: {
|
|
277
|
+
characters: z.number(),
|
|
278
|
+
dispatches: z.number(),
|
|
279
|
+
spendable: z.number(),
|
|
280
|
+
sufficient: z.boolean(),
|
|
281
|
+
max_chars_per_call: z.number().nullable(),
|
|
282
|
+
within_max_chars_per_call: z.boolean(),
|
|
283
|
+
replacements: z.unknown().optional(),
|
|
284
|
+
},
|
|
285
|
+
annotations: { title: 'Quote', readOnlyHint: true, idempotentHint: true, openWorldHint: true },
|
|
286
|
+
}, async (args) => run('athanlab_quote', async () => {
|
|
287
|
+
ctx.client.assertKey();
|
|
288
|
+
const input = speakInput(args, features, 'mp3');
|
|
289
|
+
const quote = await quoteSpeech(ctx, { ...input, metadata: undefined });
|
|
290
|
+
const cap = config.maxCharsPerCall;
|
|
291
|
+
const within = cap === null || quote.characters <= cap;
|
|
292
|
+
const structured = {
|
|
293
|
+
characters: quote.characters,
|
|
294
|
+
dispatches: quote.dispatches,
|
|
295
|
+
spendable: quote.spendable,
|
|
296
|
+
sufficient: quote.sufficient,
|
|
297
|
+
max_chars_per_call: cap,
|
|
298
|
+
within_max_chars_per_call: within,
|
|
299
|
+
};
|
|
300
|
+
if (quote.replacements !== undefined)
|
|
301
|
+
structured.replacements = quote.replacements;
|
|
302
|
+
const text = `Would cost ${quote.characters} characters (${quote.dispatches} piece${quote.dispatches === 1 ? '' : 's'}). The account can spend ${quote.spendable}: ${quote.sufficient ? 'enough' : 'NOT enough'}.${within ? '' : ` Above this server's ATHANLAB_MAX_CHARS_PER_CALL (${cap}): athanlab_speak would refuse it.`} Nothing was charged.`;
|
|
303
|
+
return { text, structured };
|
|
304
|
+
}));
|
|
305
|
+
server.registerTool('athanlab_speak', {
|
|
306
|
+
title: 'Speak Burmese text',
|
|
307
|
+
description: `Turns one Burmese text (up to 5,000 characters) into an audio file and returns its path. COSTS the text's length in characters from the account balance (failed jobs are refunded). Creates an asynchronous job, waits for it (usually seconds to a few minutes) and downloads the audio. If it does not finish within timeout_seconds the job keeps running: the result has status "processing" and the job_id — then use athanlab_get_job / athanlab_download_audio and do NOT call athanlab_speak again for the same text (that would pay twice). For dialogue or many lines, use athanlab_speak_script. Read the athanlab-writing-guide first. ${outputDirNote}`,
|
|
308
|
+
inputSchema: {
|
|
309
|
+
text: speechText,
|
|
310
|
+
voice_id: voiceId.optional(),
|
|
311
|
+
output_format: outputFormat.default('wav').describe('wav (default; exact, for editing) or mp3 (smaller).'),
|
|
312
|
+
number_mode: numberMode.optional(),
|
|
313
|
+
...speedShape(features),
|
|
314
|
+
...replacementsShape(features),
|
|
315
|
+
filename: filename.optional(),
|
|
316
|
+
timeout_seconds: z.number().int().min(30).max(3600).default(DEFAULT_SPEAK_TIMEOUT_SECONDS)
|
|
317
|
+
.describe('How long to wait for the job before returning its job_id instead (default 900).'),
|
|
318
|
+
idempotency_key: z.string().regex(/^[\x20-\x7E]{1,255}$/).optional()
|
|
319
|
+
.describe('Only to retry a call that failed with an unknown outcome (network error): pass the idempotency_key from that error so the same text is never charged twice. Leave it out otherwise.'),
|
|
320
|
+
},
|
|
321
|
+
outputSchema: {
|
|
322
|
+
status: z.enum(['succeeded', 'processing']),
|
|
323
|
+
path: z.string().nullable().describe('Absolute path of the audio file; null while processing.'),
|
|
324
|
+
job_id: z.string(),
|
|
325
|
+
output_format: outputFormat,
|
|
326
|
+
duration_seconds: z.number().nullable(),
|
|
327
|
+
...timingOut,
|
|
328
|
+
characters_charged: z.number().nullable(),
|
|
329
|
+
voice_id: z.string().nullable(),
|
|
330
|
+
bytes: z.number().nullable(),
|
|
331
|
+
replayed: z.boolean().nullable(),
|
|
332
|
+
progress: z.number().nullable(),
|
|
333
|
+
message: z.string().nullable(),
|
|
334
|
+
},
|
|
335
|
+
annotations: { title: 'Speak (costs characters)', readOnlyHint: false, destructiveHint: false, idempotentHint: false, openWorldHint: true },
|
|
336
|
+
}, async (args, extra) => run('athanlab_speak', async () => {
|
|
337
|
+
const progress = progressReporter(extra);
|
|
338
|
+
const started = ctx.clock.now();
|
|
339
|
+
// One notification per poll (a heartbeat when progress did not move), with the time waited so far.
|
|
340
|
+
const onProgress = progress
|
|
341
|
+
? (job) => {
|
|
342
|
+
const percent = Math.round((job.progress ?? 0) * 100);
|
|
343
|
+
const seconds = Math.round((ctx.clock.now() - started) / 1000);
|
|
344
|
+
progress(percent, 100, `Job ${job.id}: ${job.status}${job.status === 'processing' && percent > 0 ? ` ${percent}%` : ''}, ${seconds} s`);
|
|
345
|
+
}
|
|
346
|
+
: undefined;
|
|
347
|
+
const result = await runSpeak(ctx, speakInput(args, features, 'wav'), {
|
|
348
|
+
filename: args.filename,
|
|
349
|
+
timeoutSeconds: args.timeout_seconds,
|
|
350
|
+
idempotencyKey: args.idempotency_key,
|
|
351
|
+
signal: extra?.signal,
|
|
352
|
+
...(onProgress ? { onProgress } : {}),
|
|
353
|
+
});
|
|
354
|
+
const structured = normalizeSpeak(result);
|
|
355
|
+
const text = result.status === 'succeeded'
|
|
356
|
+
? `Saved ${result.path} (${result.duration_seconds.toFixed(2)} s ${result.output_format}, ${result.characters_charged} characters charged${result.replayed ? ', replayed — not charged again' : ''}). Job ${result.job_id}.`
|
|
357
|
+
: result.message;
|
|
358
|
+
return { text, structured };
|
|
359
|
+
}));
|
|
360
|
+
server.registerTool('athanlab_speak_script', {
|
|
361
|
+
title: 'Speak a multi-line script',
|
|
362
|
+
description: `Generates a dialogue or script line by line: one audio file per line (named from the line's id or speaker and its position) plus manifest.json with every line on one timeline — consecutive lines of the same speaker are separated by gaps.sentence seconds, a speaker change by gaps.turn — and optionally combined.wav (combine: true, WAV only). Each line's voice is its voice_id, else voices[speaker], else the default voice. COSTS the characters of every new line. It quotes all lines first (free); above ${CONFIRM_CHARACTERS_THRESHOLD.toLocaleString('en-US')} characters or ${CONFIRM_LINES_THRESHOLD} lines it returns the quote WITHOUT creating anything unless confirm is true — show the quote to the user, then call again with confirm: true. RESUMABLE: re-running the same call (same lines and out_subdir) skips finished lines and follows running jobs without paying again; changing one line's text regenerates only that line; a line whose earlier audio is gone (deleted after 30 days) is quoted and charged again like a new line. Run ONE call per out_subdir at a time: a second call on the same folder while one is running returns script_running and does nothing. Stops creating jobs on plan_required, insufficient_characters or key_budget_exceeded and reports which lines are done. ${outputDirNote}`,
|
|
363
|
+
inputSchema: {
|
|
364
|
+
lines: z.array(z.object({
|
|
365
|
+
id: z.string().min(1).max(64).optional().describe('Your label for the line (used in its file name).'),
|
|
366
|
+
speaker: z.string().min(1).max(100).optional().describe('Who speaks it; maps to a voice through `voices`.'),
|
|
367
|
+
text: z.string().min(1).max(5000).describe('One sentence-sized line of Burmese text.'),
|
|
368
|
+
voice_id: voiceId.optional(),
|
|
369
|
+
})).min(1).max(200).describe('The lines, in order (1–200).'),
|
|
370
|
+
voices: z.record(z.string(), z.string().min(1).max(128)).optional()
|
|
371
|
+
.describe('speaker → voice_id, e.g. {"narrator": "athanlab-default-female-v1"}.'),
|
|
372
|
+
gaps: z.object({
|
|
373
|
+
sentence: z.number().min(0).max(10).default(0.4).describe('Seconds between consecutive lines of the same speaker (default 0.4).'),
|
|
374
|
+
turn: z.number().min(0).max(10).default(0.55).describe('Seconds at a speaker change (default 0.55).'),
|
|
375
|
+
}).optional(),
|
|
376
|
+
output_format: outputFormat.default('wav').describe('wav (default) or mp3. combine needs wav.'),
|
|
377
|
+
number_mode: numberMode.optional(),
|
|
378
|
+
combine: z.boolean().default(false).describe('Also write combined.wav: every line joined on the timeline with silent gaps (WAV only).'),
|
|
379
|
+
confirm: z.boolean().default(false).describe(`Required above ${CONFIRM_CHARACTERS_THRESHOLD.toLocaleString('en-US')} characters or ${CONFIRM_LINES_THRESHOLD} lines; otherwise only the quote is returned.`),
|
|
380
|
+
out_subdir: z.string().min(1).max(100).optional()
|
|
381
|
+
.describe('Folder (inside the output directory) for this script. Reuse it to resume or to regenerate changed lines; defaults to a name derived from the script.'),
|
|
382
|
+
timeout_seconds: z.number().int().min(60).max(3600).default(900)
|
|
383
|
+
.describe('Stop waiting after this long (default 900); running jobs continue and a re-run picks them up.'),
|
|
384
|
+
},
|
|
385
|
+
outputSchema: {
|
|
386
|
+
status: z.enum(['completed', 'partial', 'stopped', 'timed_out', 'confirmation_required']),
|
|
387
|
+
directory: z.string(),
|
|
388
|
+
manifest_path: z.string().nullable(),
|
|
389
|
+
combined_path: z.string().nullable(),
|
|
390
|
+
quote: z.object({
|
|
391
|
+
characters: z.number(),
|
|
392
|
+
lines: z.number(),
|
|
393
|
+
dispatches: z.number(),
|
|
394
|
+
spendable: z.number().nullable(),
|
|
395
|
+
sufficient: z.boolean().nullable(),
|
|
396
|
+
lines_skipped: z.number(),
|
|
397
|
+
lines_resumed: z.number(),
|
|
398
|
+
}).nullable(),
|
|
399
|
+
lines_total: z.number(),
|
|
400
|
+
lines_succeeded: z.number(),
|
|
401
|
+
lines_skipped: z.number(),
|
|
402
|
+
lines_failed: z.number(),
|
|
403
|
+
lines_pending: z.number(),
|
|
404
|
+
characters_charged: z.number(),
|
|
405
|
+
total_duration_seconds: z.number().nullable(),
|
|
406
|
+
timeline_complete: z.boolean(),
|
|
407
|
+
stopped: z.record(z.string(), z.unknown()).nullable(),
|
|
408
|
+
combine_error: z.record(z.string(), z.unknown()).nullable(),
|
|
409
|
+
used_batch: z.boolean(),
|
|
410
|
+
message: z.string(),
|
|
411
|
+
lines: z.array(z.object({
|
|
412
|
+
index: z.number(),
|
|
413
|
+
id: z.string().nullable(),
|
|
414
|
+
speaker: z.string().nullable(),
|
|
415
|
+
status: z.string(),
|
|
416
|
+
file: z.string().nullable(),
|
|
417
|
+
job_id: z.string().nullable(),
|
|
418
|
+
duration_seconds: z.number().nullable(),
|
|
419
|
+
start_seconds: z.number().nullable(),
|
|
420
|
+
end_seconds: z.number().nullable(),
|
|
421
|
+
error: z.object({ code: z.string(), message: z.string() }).nullable(),
|
|
422
|
+
})),
|
|
423
|
+
},
|
|
424
|
+
annotations: { title: 'Speak script (costs characters)', readOnlyHint: false, destructiveHint: true, idempotentHint: false, openWorldHint: true },
|
|
425
|
+
}, async (args, extra) => run('athanlab_speak_script', async () => {
|
|
426
|
+
const progress = progressReporter(extra);
|
|
427
|
+
const input = {
|
|
428
|
+
lines: args.lines,
|
|
429
|
+
voices: args.voices,
|
|
430
|
+
gaps: args.gaps,
|
|
431
|
+
output_format: args.output_format,
|
|
432
|
+
number_mode: args.number_mode,
|
|
433
|
+
combine: args.combine,
|
|
434
|
+
confirm: args.confirm,
|
|
435
|
+
out_subdir: args.out_subdir,
|
|
436
|
+
timeout_seconds: args.timeout_seconds,
|
|
437
|
+
};
|
|
438
|
+
const result = await runScript(ctx, input, {
|
|
439
|
+
signal: extra?.signal,
|
|
440
|
+
...(progress ? { onProgress: (done, total, message) => progress(done, total, message) } : {}),
|
|
441
|
+
});
|
|
442
|
+
return { text: result.message, structured: result };
|
|
443
|
+
}));
|
|
444
|
+
server.registerTool('athanlab_get_job', {
|
|
445
|
+
title: 'Get a speech job',
|
|
446
|
+
description: 'Returns one speech job: status (processing, succeeded, failed, cancelled), progress, characters charged and refunded, audio duration and expiry, and the failure reason. Use it to check on a job returned by a timed-out athanlab_speak. Free; read-only.',
|
|
447
|
+
inputSchema: { job_id: jobId },
|
|
448
|
+
outputSchema: jobSummaryShape,
|
|
449
|
+
annotations: { title: 'Get job', readOnlyHint: true, idempotentHint: true, openWorldHint: true },
|
|
450
|
+
}, async (args) => run('athanlab_get_job', async () => {
|
|
451
|
+
ctx.client.assertKey();
|
|
452
|
+
const summary = summarizeJob(await getJob(ctx, args.job_id));
|
|
453
|
+
const next = summary.status === 'succeeded' && summary.audio
|
|
454
|
+
? ' Save it with athanlab_download_audio.'
|
|
455
|
+
: summary.status === 'processing' ? ' Check again in a few seconds.' : '';
|
|
456
|
+
return { text: `${jobLine(summary)}.${next}`, structured: summary };
|
|
457
|
+
}));
|
|
458
|
+
server.registerTool('athanlab_list_jobs', {
|
|
459
|
+
title: 'List speech jobs',
|
|
460
|
+
description: 'Lists the account\'s speech jobs, newest first (all keys of the account share one history). Filter by status; page with cursor (the previous page\'s next_cursor). Use status "processing" to find jobs still running. Free; read-only.',
|
|
461
|
+
inputSchema: {
|
|
462
|
+
status: z.enum(JOB_STATUSES).optional().describe('Only jobs in this status.'),
|
|
463
|
+
limit: z.number().int().min(1).max(100).optional().describe('Jobs per page, 1–100 (default 20).'),
|
|
464
|
+
cursor: z.string().min(1).max(512).optional().describe('next_cursor from the previous page, unchanged.'),
|
|
465
|
+
},
|
|
466
|
+
outputSchema: {
|
|
467
|
+
jobs: z.array(jobSummaryOut),
|
|
468
|
+
has_more: z.boolean(),
|
|
469
|
+
next_cursor: z.string().nullable(),
|
|
470
|
+
},
|
|
471
|
+
annotations: { title: 'List jobs', readOnlyHint: true, idempotentHint: true, openWorldHint: true },
|
|
472
|
+
}, async (args) => run('athanlab_list_jobs', async () => {
|
|
473
|
+
ctx.client.assertKey();
|
|
474
|
+
const list = await listJobs(ctx, { status: args.status, limit: args.limit, cursor: args.cursor });
|
|
475
|
+
const jobs = list.data.map(summarizeJob);
|
|
476
|
+
const text = jobs.length
|
|
477
|
+
? `${jobs.length} job(s)${list.has_more ? ' (more: pass next_cursor)' : ''}:\n${jobs.map(jobLine).join('\n')}`
|
|
478
|
+
: 'No jobs.';
|
|
479
|
+
return { text, structured: { jobs, has_more: list.has_more, next_cursor: list.next_cursor } };
|
|
480
|
+
}));
|
|
481
|
+
server.registerTool('athanlab_cancel_job', {
|
|
482
|
+
title: 'Cancel a speech job',
|
|
483
|
+
description: 'Stops a job that is still processing. The part already generated stays charged; the rest is refunded (characters_refunded). Cancelling an already cancelled job returns it again; a finished job cannot be cancelled (job_finished). Works even when the key is read-only.',
|
|
484
|
+
inputSchema: { job_id: jobId },
|
|
485
|
+
outputSchema: jobSummaryShape,
|
|
486
|
+
annotations: { title: 'Cancel job', readOnlyHint: false, destructiveHint: true, idempotentHint: true, openWorldHint: true },
|
|
487
|
+
}, async (args) => run('athanlab_cancel_job', async () => {
|
|
488
|
+
checkJobId(args.job_id);
|
|
489
|
+
const res = await ctx.client.json(`/speech/${args.job_id}/cancel`, { method: 'POST' });
|
|
490
|
+
const summary = summarizeJob(res.data);
|
|
491
|
+
return {
|
|
492
|
+
text: `Job ${summary.job_id} ${summary.status}: ${summary.characters_refunded} of ${summary.characters_charged} characters refunded.`,
|
|
493
|
+
structured: summary,
|
|
494
|
+
};
|
|
495
|
+
}));
|
|
496
|
+
server.registerTool('athanlab_download_audio', {
|
|
497
|
+
title: 'Download a job\'s audio',
|
|
498
|
+
description: `Saves the audio of a succeeded job to a file and returns its path — for jobs from a timed-out athanlab_speak, from athanlab_list_jobs, or to get the other format (a job made as wav can be downloaded as mp3 and vice versa). Free (the job was paid when created). Audio is kept 30 days after the job was created. ${outputDirNote}`,
|
|
499
|
+
inputSchema: {
|
|
500
|
+
job_id: jobId,
|
|
501
|
+
format: outputFormat.optional().describe('wav or mp3; defaults to the job\'s own format.'),
|
|
502
|
+
filename: filename.optional(),
|
|
503
|
+
},
|
|
504
|
+
outputSchema: {
|
|
505
|
+
path: z.string(),
|
|
506
|
+
job_id: z.string(),
|
|
507
|
+
format: z.string(),
|
|
508
|
+
bytes: z.number(),
|
|
509
|
+
duration_seconds: z.number().nullable(),
|
|
510
|
+
...timingOut,
|
|
511
|
+
},
|
|
512
|
+
annotations: { title: 'Download audio', readOnlyHint: false, destructiveHint: false, idempotentHint: false, openWorldHint: true },
|
|
513
|
+
}, async (args, extra) => run('athanlab_download_audio', async () => {
|
|
514
|
+
ctx.client.assertKey();
|
|
515
|
+
const job = await getJob(ctx, args.job_id);
|
|
516
|
+
if (job.status === 'processing') {
|
|
517
|
+
throw new AthanLabError({
|
|
518
|
+
code: 'job_not_finished',
|
|
519
|
+
message: `Job ${job.id} is still processing (${Math.round((job.progress ?? 0) * 100)}%).`,
|
|
520
|
+
source: 'client',
|
|
521
|
+
retryable: true,
|
|
522
|
+
jobId: job.id,
|
|
523
|
+
});
|
|
524
|
+
}
|
|
525
|
+
if (job.status !== 'succeeded')
|
|
526
|
+
throw jobOutcomeError(job);
|
|
527
|
+
if (!job.audio) {
|
|
528
|
+
throw new AthanLabError({ code: 'audio_expired', message: `The audio of job ${job.id} is no longer available (kept 30 days).`, source: 'client', jobId: job.id });
|
|
529
|
+
}
|
|
530
|
+
const format = args.format ?? job.input.output_format;
|
|
531
|
+
await ensureDir(config.outputDir);
|
|
532
|
+
const name = sanitizeFilename(args.filename, format, `athanlab-${job.id}`);
|
|
533
|
+
const target = await uniquePath(config.outputDir, name, stamp(ctx.clock.now()));
|
|
534
|
+
let bytes;
|
|
535
|
+
try {
|
|
536
|
+
bytes = await downloadJobAudio(ctx, job.id, format, target, extra?.signal);
|
|
537
|
+
}
|
|
538
|
+
catch (err) {
|
|
539
|
+
const e = toAthanLabError(err);
|
|
540
|
+
e.jobId = job.id;
|
|
541
|
+
throw e;
|
|
542
|
+
}
|
|
543
|
+
const summary = summarizeJob(job);
|
|
544
|
+
return {
|
|
545
|
+
text: `Saved ${target} (${format}, ${bytes} bytes${job.audio ? `, ${job.audio.duration_seconds.toFixed(2)} s` : ''}).`,
|
|
546
|
+
structured: {
|
|
547
|
+
path: target,
|
|
548
|
+
job_id: job.id,
|
|
549
|
+
format,
|
|
550
|
+
bytes,
|
|
551
|
+
duration_seconds: job.audio?.duration_seconds ?? null,
|
|
552
|
+
speech_start_seconds: summary.speech_start_seconds,
|
|
553
|
+
speech_end_seconds: summary.speech_end_seconds,
|
|
554
|
+
segments: summary.segments,
|
|
555
|
+
},
|
|
556
|
+
};
|
|
557
|
+
}));
|
|
558
|
+
server.registerTool('athanlab_usage', {
|
|
559
|
+
title: 'AthanLab balance and limits',
|
|
560
|
+
description: 'Returns the account\'s character balance and API limits: plan, whether this key may create jobs (entitled — false means read-only: Max plan or founder access needed), the monthly allowance and when it resets, Athan Tokens, spendable (monthly remaining + tokens), jobs running now against max_concurrent_jobs, and this key\'s monthly budget. Free; read-only.',
|
|
561
|
+
inputSchema: {},
|
|
562
|
+
outputSchema: {
|
|
563
|
+
plan: z.string(),
|
|
564
|
+
entitled: z.boolean(),
|
|
565
|
+
upgrade_url: z.string().nullable(),
|
|
566
|
+
monthly: z.object({ limit: z.number(), used: z.number(), remaining: z.number(), resets_at: z.string().nullable() }),
|
|
567
|
+
tokens: z.object({ balance: z.number() }),
|
|
568
|
+
spendable: z.number(),
|
|
569
|
+
active_jobs: z.number(),
|
|
570
|
+
max_concurrent_jobs: z.number(),
|
|
571
|
+
job_slots_available: z.number(),
|
|
572
|
+
key: z.object({
|
|
573
|
+
id: z.string(),
|
|
574
|
+
monthly_char_budget: z.number().nullable(),
|
|
575
|
+
used_this_month: z.number().nullable(),
|
|
576
|
+
remaining: z.number().nullable(),
|
|
577
|
+
resets_at: z.string().nullable(),
|
|
578
|
+
}),
|
|
579
|
+
},
|
|
580
|
+
annotations: { title: 'Usage', readOnlyHint: true, idempotentHint: true, openWorldHint: true },
|
|
581
|
+
}, async () => run('athanlab_usage', async () => {
|
|
582
|
+
const { data } = await ctx.client.json('/usage');
|
|
583
|
+
const slots = Math.max(0, data.max_concurrent_jobs - data.active_jobs);
|
|
584
|
+
const budget = data.key.monthly_char_budget !== null
|
|
585
|
+
? ` This key's budget: ${data.key.remaining} of ${data.key.monthly_char_budget} left this month.`
|
|
586
|
+
: '';
|
|
587
|
+
const text = `Plan ${data.plan}${data.entitled ? '' : ' — READ-ONLY: creating jobs needs Max (or founder access)'}. Spendable: ${data.spendable} characters (monthly ${data.monthly.remaining} of ${data.monthly.limit}${data.monthly.resets_at ? `, resets ${data.monthly.resets_at}` : ''}; tokens ${data.tokens.balance}). Jobs running: ${data.active_jobs}/${data.max_concurrent_jobs}.${budget}`;
|
|
588
|
+
return {
|
|
589
|
+
text,
|
|
590
|
+
structured: {
|
|
591
|
+
// Field by field: the output schema is strict, and the API may add fields.
|
|
592
|
+
plan: data.plan,
|
|
593
|
+
entitled: data.entitled,
|
|
594
|
+
upgrade_url: data.upgrade_url ?? null,
|
|
595
|
+
monthly: {
|
|
596
|
+
limit: data.monthly.limit,
|
|
597
|
+
used: data.monthly.used,
|
|
598
|
+
remaining: data.monthly.remaining,
|
|
599
|
+
resets_at: data.monthly.resets_at ?? null,
|
|
600
|
+
},
|
|
601
|
+
tokens: { balance: data.tokens.balance },
|
|
602
|
+
spendable: data.spendable,
|
|
603
|
+
active_jobs: data.active_jobs,
|
|
604
|
+
max_concurrent_jobs: data.max_concurrent_jobs,
|
|
605
|
+
job_slots_available: slots,
|
|
606
|
+
key: {
|
|
607
|
+
id: data.key.id,
|
|
608
|
+
monthly_char_budget: data.key.monthly_char_budget ?? null,
|
|
609
|
+
used_this_month: data.key.used_this_month ?? null,
|
|
610
|
+
remaining: data.key.remaining ?? null,
|
|
611
|
+
resets_at: data.key.resets_at ?? null,
|
|
612
|
+
},
|
|
613
|
+
},
|
|
614
|
+
};
|
|
615
|
+
}));
|
|
616
|
+
}
|
|
617
|
+
//# sourceMappingURL=tools.js.map
|