@crossworks/voice-client 0.230.43

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,866 @@
1
+ /**
2
+ * Adapter interfaces — the boundary between the rest of Mantle and
3
+ * provider-specific HTTP calls.
4
+ *
5
+ * One interface per capability. Each provider that wants to be "wired"
6
+ * for a capability implements the matching interface. The runtime
7
+ * looks up the right adapter by `providerId` (which comes from the
8
+ * worker's `provider` column) and calls through it. The dispatch
9
+ * surface is identical regardless of provider — the adapter handles
10
+ * all the per-provider quirks (auth header shape, response decoding,
11
+ * voice/model naming).
12
+ *
13
+ * Why we own these instead of using LiteLLM:
14
+ * - End-to-end type safety. The adapter's input/output shapes are
15
+ * fixed at compile time.
16
+ * - No proxy service to operate on the VPS.
17
+ * - Adapters are ~50-150 LOC each. Total maintenance cost is small
18
+ * because we only adapter the providers we actually use.
19
+ * - LiteLLM's open-source code remains a reference we can lift from
20
+ * when a provider's API shape is surprising.
21
+ *
22
+ * Each adapter exposes its `providerId` (must match the catalogue in
23
+ * `providers.ts`) and an `adapterName` used purely for logs/traces so
24
+ * we can tell at a glance which adapter ran a given call.
25
+ */
26
+
27
+ import type { ProviderId } from '../providers';
28
+ import type {
29
+ SynthesizeOptions,
30
+ SynthesizeResult,
31
+ SttParam,
32
+ TranscribeOptions,
33
+ TranscribeResult,
34
+ TtsParam,
35
+ } from '../types';
36
+ import type { TtsModelInfo, SttModelInfo } from '../catalog';
37
+ import type { DiscoveryResult } from '../catalog';
38
+
39
+ /** Common shape every adapter exposes — used by the registry to log
40
+ * and surface the right one in errors. */
41
+ export interface AdapterMeta {
42
+ /** Matches one of the ids in SUPPORTED_PROVIDERS. */
43
+ readonly providerId: ProviderId;
44
+ /** Human-readable for logs. Convention: '<provider>-<capability>'
45
+ * (e.g. 'openai-tts', 'elevenlabs-tts'). */
46
+ readonly adapterName: string;
47
+ }
48
+
49
+ /** An inline audio tag the model interprets as a performance cue.
50
+ * ElevenLabs v3 calls these "audio tags"; xAI's voice models use a
51
+ * similar `[giggle]`/`[laugh]` convention. The shape is uniform: an
52
+ * exact bracket-wrapped token + a short description for the LLM's
53
+ * benefit so it knows when to use which.
54
+ *
55
+ * Inline tags are *point-in-time* — they fire at the spot they sit.
56
+ * For tags that style a whole *span* of text (e.g. xAI's
57
+ * `<whisper>…</whisper>`), see {@link WrappingTag}. */
58
+ export type AudioTag = {
59
+ /** Exact form including brackets, e.g. `[laughs]`, `[whispers]`. */
60
+ tag: string;
61
+ /** One-line hint for the LLM. The composer joins these into the
62
+ * system-prompt paragraph so Saskia knows what each tag does. */
63
+ description: string;
64
+ /** Coarse grouping so the UI hint can render them in sections. */
65
+ category?: 'emotion' | 'reaction' | 'delivery' | 'cognitive' | 'tone' | 'accent';
66
+ };
67
+
68
+ /** A wrapping speech tag — an angle-bracket pair that styles the whole
69
+ * phrase it surrounds, e.g. xAI Grok's `<whisper>secret</whisper>`,
70
+ * `<soft>…</soft>`, `<slow>…</slow>`. Distinct from {@link AudioTag}:
71
+ * inline tags are point-in-time cues, wrapping tags apply a delivery
72
+ * style across a span.
73
+ *
74
+ * The framework treats these the same way it treats inline tags:
75
+ * adapters advertise the set their model honours (`supportedWrappingTags`),
76
+ * the prompt composer tells Saskia she may use them, and the text-out
77
+ * path strips them (keeping the inner text) so the markers never leak
78
+ * into a plain-text reply. */
79
+ export type WrappingTag = {
80
+ /** Short lower-case name without brackets, e.g. `whisper`, `soft`.
81
+ * The open/close forms are derived as `<name>` / `</name>`. */
82
+ name: string;
83
+ /** One-line hint for the LLM — what the style does and when to reach
84
+ * for it. Joined into the system-prompt paragraph. */
85
+ description: string;
86
+ /** Coarse grouping for the UI hint (volume / pitch / pacing / style). */
87
+ category?: 'volume' | 'pitch' | 'pacing' | 'style';
88
+ };
89
+
90
+ export interface TtsDispatcher extends AdapterMeta {
91
+ /** The subset of {@link SynthesizeOptions} this adapter actually puts on the
92
+ * wire, excluding `voice`/`model`/`text` which every provider takes.
93
+ * REQUIRED and it must be honest: the caller reports every requested option
94
+ * outside this set rather than letting it look like it applied. Where the
95
+ * gate is per-MODEL instead of per-provider, the adapter additionally
96
+ * returns `warnings` from the call. Same contract as
97
+ * {@link ImageGenDispatcher.supports}. */
98
+ supports: readonly TtsParam[];
99
+
100
+ /** Synthesise speech and return audio bytes. The runtime then hands
101
+ * these to Telegram's sendVoice (when format='opus') or to an
102
+ * `<audio>` element on the web. */
103
+ synthesize(opts: SynthesizeOptions): Promise<SynthesizeResult>;
104
+
105
+ /** Optional. Live-discover which TTS models the API key can use.
106
+ * For OpenAI we hit /v1/models and intersect with the catalogue;
107
+ * for ElevenLabs we'd hit /v1/models too but the catalogue is
108
+ * different. If absent, the UI falls back to whatever static
109
+ * catalogue the adapter consults. */
110
+ discoverModels?(apiKey: string): Promise<DiscoveryResult<TtsModelInfo>>;
111
+
112
+ /** Optional. Returns the voices a given model supports. For OpenAI
113
+ * this is a static per-model list (we hardcode 13 voices for
114
+ * gpt-4o-mini-tts, 9 for tts-1). For ElevenLabs this would be a
115
+ * live `/v1/voices` query that includes the user's cloned voices.
116
+ * When absent, callers fall through to whatever default they
117
+ * rendered the form with. */
118
+ voicesForModel?(
119
+ modelId: string,
120
+ apiKey?: string,
121
+ ): Promise<Array<{ id: string; description: string }>>;
122
+
123
+ /** Optional. Inline audio tags the model interprets — `[laughs]`,
124
+ * `[whispers]`, `[sighs]`, etc. ElevenLabs v3 has the richest set;
125
+ * xAI's voice models support a smaller subset; OpenAI's
126
+ * gpt-4o-mini-tts uses the `instructions` parameter instead and
127
+ * returns an empty list here.
128
+ *
129
+ * The runtime queries this when building the chat agent's prompt
130
+ * so the LLM only emits tags the active TTS will render. Adapters
131
+ * whose providers ignore unknown tags can safely return [] —
132
+ * Saskia won't try to use them. */
133
+ supportedAudioTags?(modelId: string): readonly AudioTag[];
134
+
135
+ /** Optional. Wrapping speech tags the model honours —
136
+ * `<whisper>…</whisper>`, `<soft>…</soft>`, `<slow>…</slow>`, etc.
137
+ * (xAI Grok voice today). Same contract as {@link supportedAudioTags}
138
+ * but for span-styling rather than point-in-time cues. Adapters
139
+ * whose providers have no angle-bracket wrapping vocabulary return
140
+ * [] (or omit this), and the runtime simply won't advertise any. */
141
+ supportedWrappingTags?(modelId: string): readonly WrappingTag[];
142
+ }
143
+
144
+ /** A chat model entry. The fields are intentionally generic — provider
145
+ * catalogs (xAI, HF, OpenAI…) all describe their chat models in
146
+ * roughly this shape. Pricing is included so the UI can show a
147
+ * rough cost-per-1M-tokens estimate when picking a model. */
148
+ export interface ChatModelInfo {
149
+ id: string;
150
+ label: string;
151
+ description: string;
152
+ /** Context window in tokens. */
153
+ contextTokens?: number;
154
+ /** Capabilities beyond plain chat. */
155
+ capabilities?: readonly ('vision' | 'reasoning' | 'function_calling' | 'json_mode')[];
156
+ /** USD per 1M input tokens. Approximate; for UI hints only. */
157
+ inputPricePer1M?: number;
158
+ /** USD per 1M output tokens. */
159
+ outputPricePer1M?: number;
160
+ }
161
+
162
+ /** Result of a chat completion. Mirrors the OpenAI shape since every
163
+ * adapter we're likely to write speaks that dialect.
164
+ *
165
+ * Cache + cost fields are optional because not every provider reports
166
+ * them. The runtime treats `undefined` as "this provider doesn't tell
167
+ * us" and falls back to the static price table for cost; cache fields
168
+ * stay zero in the trace. Adapters MUST populate `tokensIn`/`tokensOut`
169
+ * when the provider returns them (every chat API in the catalogue does)
170
+ * so the trace's token + fallback-cost numbers stay accurate. */
171
+ export interface ChatResult {
172
+ text: string;
173
+ model: string;
174
+ /** Tool calls the model wants to make. When non-empty, the
175
+ * tool-loop dispatches each one and feeds results back. Adapters
176
+ * normalise from their provider's native shape (Anthropic
177
+ * `tool_use` blocks, Google `functionCall` parts, OpenAI
178
+ * `tool_calls`) to this single grammar.
179
+ *
180
+ * Important: when toolCalls is non-empty, `text` may be empty —
181
+ * the model emitted only tool calls and no narrative this turn. */
182
+ toolCalls?: ChatToolCall[];
183
+ tokensIn?: number;
184
+ tokensOut?: number;
185
+ /** Tokens served from the provider's prompt cache, billed at the
186
+ * reduced cache-read rate. Anthropic returns this as
187
+ * `cache_read_input_tokens`; OpenAI as `prompt_tokens_details.cached_tokens`;
188
+ * Google as `usageMetadata.cachedContentTokenCount`; OpenRouter
189
+ * aliases all of the above as `cache_read_input_tokens` /
190
+ * `cached_tokens`. The trace records this separately from
191
+ * `tokensIn` so the cost dashboard can show the actual cache-read
192
+ * savings on the responder path. */
193
+ cacheReadTokens?: number;
194
+ /** Tokens *written* into the provider's prompt cache, billed at the
195
+ * cache-write rate (Anthropic charges ~1.25× input for these). Only
196
+ * Anthropic surfaces this distinctly (`cache_creation_input_tokens`);
197
+ * every other provider folds it into `tokensIn`. */
198
+ cacheWriteTokens?: number;
199
+ /** Provider-reported USD cost for the call, in dollars (not micro-USD).
200
+ * Currently only OpenRouter populates this via `usage.cost` — and it
201
+ * matters because OR rolls in vendor surcharges that the static price
202
+ * table doesn't know about (e.g. Perplexity's per-search fee). Direct
203
+ * providers return `undefined` here; the trace falls back to
204
+ * `fallbackCostMicroUsd(model, ...)`. */
205
+ reportedCostUsd?: number;
206
+ /** Provider reasoning blocks from this turn (OpenRouter `reasoning_details`).
207
+ * When the model thought before answering, the tool-loop stores these on the
208
+ * assistant message it appends so the NEXT request can echo them back —
209
+ * required for thinking continuity across a tool round-trip (see
210
+ * {@link ChatAssistantMessage.reasoningDetails}). Undefined when the turn
211
+ * produced no reasoning. */
212
+ reasoningDetails?: ReasoningDetail[];
213
+ /** Why the model stopped, normalised across providers. Undefined when the
214
+ * adapter doesn't report one (most don't yet) — treat that as "unknown",
215
+ * NOT as `'stop'`.
216
+ *
217
+ * This exists because a truncated answer and a policy-blocked answer are
218
+ * otherwise indistinguishable from a genuinely short one: every provider
219
+ * returns HTTP 200 with little or no text in all three cases. Without this
220
+ * field a caller cannot tell "the model finished" from "the model was cut
221
+ * off at maxTokens" or "the provider refused". Adapters map their native
222
+ * value onto this small grammar:
223
+ *
224
+ * - 'stop' — finished normally
225
+ * - 'length' — hit the output-token ceiling, reply is TRUNCATED
226
+ * - 'tool_calls' — stopped to call tools (expect `toolCalls`)
227
+ * - 'content_filter' — provider blocked it (safety/recitation/blocklist)
228
+ * - 'error' — provider signalled a malformed generation
229
+ * - 'other' — reported, but not one of the above */
230
+ finishReason?: ChatFinishReason;
231
+ }
232
+
233
+ /** Normalised stop reason. Deliberately a small closed set: adapters collapse
234
+ * their provider's longer enum onto these, since callers only ever need to
235
+ * branch on "was this reply complete, truncated, or refused". */
236
+ export type ChatFinishReason =
237
+ 'stop' | 'length' | 'tool_calls' | 'content_filter' | 'error' | 'other';
238
+
239
+ /** A single tool the model can call. Mirrors the OpenAI function-tool
240
+ * shape since every adapter we're likely to talk to either accepts
241
+ * this natively (OpenRouter / xAI / HF) or translates from it
242
+ * (Anthropic `tool_use`, Google `functionDeclarations`). */
243
+ export interface ChatToolDefinition {
244
+ type: 'function';
245
+ function: {
246
+ name: string;
247
+ description: string;
248
+ /** JSON Schema describing the tool's arguments. Pass the same
249
+ * shape your `tools` table's `input_schema` carries. */
250
+ parameters: Record<string, unknown>;
251
+ };
252
+ }
253
+
254
+ /** One tool call returned by the model. Adapters normalise from
255
+ * provider-specific shapes (Anthropic `tool_use` blocks, Google
256
+ * `functionCall` parts) to this single shape so the tool-loop only
257
+ * iterates one grammar. */
258
+ export interface ChatToolCall {
259
+ /** Provider-assigned id, used to pair the tool result back in the
260
+ * next request. For Anthropic this is the `tool_use.id` (toolu_*);
261
+ * for OpenAI-shape providers it's `tool_calls[].id` (call_*); for
262
+ * Google we synthesise an id since Gemini's functionCall has no
263
+ * natural id field. */
264
+ id: string;
265
+ type: 'function';
266
+ function: {
267
+ name: string;
268
+ /** Stringified JSON args. Anthropic returns a parsed object as
269
+ * `input`; the adapter stringifies it here so the loop sees a
270
+ * single shape. Callers parse via @mantle/agent-runtime/tool-args
271
+ * defensively because not every model emits valid JSON. */
272
+ arguments: string;
273
+ };
274
+ }
275
+
276
+ /** A tool message in the conversation — i.e. the user-facing surface
277
+ * of "here's what the tool returned." Adapters translate to the
278
+ * provider's native tool-result shape (Anthropic emits this as a
279
+ * `user` message with a `tool_result` block; Google as a `function`
280
+ * message with `functionResponse`; OpenAI/OR/HF as a `tool` message). */
281
+ export type ChatToolMessage = {
282
+ role: 'tool';
283
+ toolCallId: string;
284
+ /** The tool result, already serialised to a string. */
285
+ content: string;
286
+ /** The tool threw / returned an error. Cache-aware adapters (Anthropic) set
287
+ * the provider's `is_error` flag so the model treats it as a failure rather
288
+ * than inferring from the serialised body. Providers without the concept
289
+ * ignore it. */
290
+ isError?: boolean;
291
+ };
292
+
293
+ /** Assistant turn — can carry text content, tool calls, or both. The
294
+ * tool-loop pushes this back into the conversation after each LLM
295
+ * round so the next request sees the model's prior tool_use + the
296
+ * matching tool_result pairs. */
297
+ export type ChatAssistantMessage = {
298
+ role: 'assistant';
299
+ /** Null when the model only emitted tool calls (no text turn). */
300
+ content: string | null;
301
+ toolCalls?: ChatToolCall[];
302
+ /** Provider reasoning blocks carried over from this assistant turn, replayed
303
+ * on the NEXT request so thinking continuity survives a tool round-trip.
304
+ * Anthropic (via OpenRouter) signs each thinking block and REJECTS a turn
305
+ * that contains tool_use but omits the preceding signed block — so when the
306
+ * responder thinks AND calls a tool, these must echo back unchanged. Opaque
307
+ * to the runtime; only the originating adapter reads them. */
308
+ reasoningDetails?: ReasoningDetail[];
309
+ };
310
+
311
+ /** One provider reasoning block (OpenRouter `reasoning_details` element:
312
+ * `reasoning.text` | `reasoning.encrypted` | `reasoning.summary`). Kept as a
313
+ * loose record so the runtime can carry it verbatim without depending on the
314
+ * provider SDK's union; the adapter that produced it is the only thing that
315
+ * interprets the shape. The `signature`/`data` fields are integrity-critical —
316
+ * echo them back exactly as received. */
317
+ export type ReasoningDetail = {
318
+ type: string;
319
+ index?: number;
320
+ text?: string | null;
321
+ data?: string | null;
322
+ summary?: string | null;
323
+ signature?: string | null;
324
+ format?: string | null;
325
+ id?: string | null;
326
+ };
327
+
328
+ /** A multi-modal user content part. Used when the responder receives
329
+ * an image attachment — the runtime builds `[{type:'text', text}, {type:
330
+ * 'image_url', imageUrl: {url, detail}}]` so vision-capable models
331
+ * see both the text and the image. Adapters that target a
332
+ * vision-capable provider translate; adapters that don't either
333
+ * extract just the text or warn at the call boundary. */
334
+ export type ChatUserContentPart =
335
+ | { type: 'text'; text: string }
336
+ | {
337
+ type: 'image_url';
338
+ imageUrl: { url: string; detail?: 'auto' | 'low' | 'high' };
339
+ };
340
+
341
+ /** A system-message content block. Used when the system prompt is
342
+ * composed of multiple cacheable segments (the responder splits its
343
+ * system prompt into the persona block + the conversation digest
344
+ * block, each with its own cache_control marker so prefix matches
345
+ * hit on shorter shared subsequences). */
346
+ export type ChatSystemContentPart = {
347
+ type: 'text';
348
+ text: string;
349
+ cacheControl?: { type: 'ephemeral' };
350
+ };
351
+
352
+ /** The tool-loop-shaped message union. Wider than the simpler
353
+ * string-content shape because tool-loop calls grow assistant turns
354
+ * with toolCalls, tool result messages, multi-modal user turns
355
+ * (image + text), and multi-segment cacheable system blocks (persona
356
+ * + digest each marked separately on Anthropic-shape providers).
357
+ * The 3a chat-shaped workers structurally satisfy this union with a
358
+ * plain `{role, content: string}` array. */
359
+ export type ChatToolLoopMessage =
360
+ | { role: 'system'; content: string | ChatSystemContentPart[] }
361
+ | { role: 'user'; content: string | ChatUserContentPart[] }
362
+ | ChatAssistantMessage
363
+ | ChatToolMessage;
364
+
365
+ /** Prompt-cache hints for the adapter. Anthropic and (less aggressively)
366
+ * OpenAI bill cached input at a fraction of the fresh rate when the
367
+ * caller marks cache breakpoints. The runtime tells the adapter where
368
+ * the breakpoints should land via this struct; adapters that talk to
369
+ * providers without prompt caching ignore the field entirely.
370
+ *
371
+ * Two breakpoint kinds matter today:
372
+ * - **systemPrompt** — mark the system block as cacheable. This is the
373
+ * dominant cost-saving on the responder path: the persona + skills
374
+ * block doesn't change turn-to-turn, so caching it pays back from
375
+ * the second call onward.
376
+ * - **lastUserMessage** — mark the most recent user message as a
377
+ * cache write point. Useful for the tool-loop pattern where the
378
+ * re-sent conversation history grows monotonically; marking the
379
+ * prior turn's last user msg makes the next call read it back as
380
+ * a cache hit. */
381
+ export interface ChatCacheControl {
382
+ systemPrompt?: boolean;
383
+ lastUserMessage?: boolean;
384
+ }
385
+
386
+ /** Reasoning-depth tiers, ascending. Provider-neutral: OpenRouter takes these
387
+ * verbatim as `reasoning.effort`, Anthropic as `output_config.effort`, Copilot
388
+ * as `reasoning_effort`.
389
+ *
390
+ * Restated here rather than imported from `@mantle/content` on purpose —
391
+ * `@mantle/voice` is the low-level adapter layer and must not pull in content's
392
+ * db/files/search dependency tree for five string literals. The two lists are
393
+ * pinned together by a compile-time assertion in `@mantle/assistant-runtime`,
394
+ * which already depends on both, so they cannot drift silently.
395
+ *
396
+ * No `none`: "off" is expressed by omitting the field, because models flagged
397
+ * `reasoning.mandatory` in GET /models reject an explicit none. */
398
+ export type ThinkingEffort = 'low' | 'medium' | 'high' | 'xhigh' | 'max';
399
+
400
+ export interface ChatOptions {
401
+ apiKey: string;
402
+ model: string;
403
+ /** Standard chat-completion messages. The adapter is free to
404
+ * transform these into the provider's native shape (e.g. Anthropic's
405
+ * separate `system` field), but we present a uniform interface.
406
+ *
407
+ * The grammar is `ChatToolLoopMessage[]` — wide enough to carry the
408
+ * tool-loop path (assistant turns with `toolCalls`, `tool` result
409
+ * messages) AND structurally compatible with the simpler shape every
410
+ * chat-shaped worker (extractor / summarizer / reflector) emits: a
411
+ * plain `{role, content: string}[]` array is a valid
412
+ * ChatToolLoopMessage[] for `system`/`user` roles. */
413
+ messages: ChatToolLoopMessage[];
414
+ /** Tools the model is allowed to call. Adapters translate this to
415
+ * the provider's native function-tool shape. When omitted or empty,
416
+ * the adapter MUST NOT send the field — some providers reject empty
417
+ * tool arrays. */
418
+ tools?: ChatToolDefinition[];
419
+ /** Steering hint for tool selection. The runtime only uses 'auto'
420
+ * (default) and 'none' today; we expose the OpenAI-style union so
421
+ * forcing a specific tool is a non-breaking addition later. */
422
+ toolChoice?: 'auto' | 'none';
423
+ temperature?: number;
424
+ maxTokens?: number;
425
+ topP?: number;
426
+ /** Thinking enable + budget hint. When > 0, reasoning-capable adapters ask the
427
+ * provider to think before answering and stream the reasoning back as
428
+ * `reasoning` deltas. The number is a token-budget HINT honoured only by
429
+ * providers that still take one (OpenRouter `reasoning.max_tokens`); on
430
+ * current Claude models it acts purely as on/off — the native Anthropic
431
+ * adapter sends `thinking:{type:'adaptive', display:'summarized'}` (the old
432
+ * `budget_tokens` form 400s on Opus 4.7/4.8 + Fable) and drops sampling
433
+ * params (also rejected in that mode). Omitted/0 ⇒ no requested thinking (a
434
+ * model may still emit incidental reasoning). Intended for the responder turn
435
+ * only — background workers leave it unset. Adapters without a reasoning mode
436
+ * ignore it.
437
+ *
438
+ * ⚠️ Enabling this on a multi-round tool loop additionally requires echoing
439
+ * prior `thinking` blocks back to the provider each iteration (Anthropic 400s
440
+ * otherwise) — that capture/replay is NOT yet implemented, so the runner must
441
+ * not set this by default until it is. */
442
+ thinkingBudget?: number;
443
+ /** Thinking EFFORT tier — the control providers actually honour now. Where
444
+ * {@link thinkingBudget} asks for N tokens of reasoning, this asks for a
445
+ * depth (`low`…`max`) and lets the provider size it.
446
+ *
447
+ * Prefer this. Budget-based thinking has been removed upstream on current
448
+ * models (Sonnet 5, Claude 4.7): `reasoning.max_tokens` is accepted-but-
449
+ * ignored, and effort maps to Anthropic's `output_config.effort`. On the
450
+ * OpenRouter path the budget was never even transmitted — its `reasoning`
451
+ * shape has no max_tokens field, so it serialised to `{}`.
452
+ *
453
+ * Adapters that have no effort concept ignore it and may still read
454
+ * `thinkingBudget` as an on/off signal. Both fields are set together by the
455
+ * runtime, so an adapter can use whichever its provider understands. */
456
+ thinkingEffort?: ThinkingEffort;
457
+ /** Retries AFTER the first attempt on transient errors (429/5xx/network/
458
+ * timeout), with exponential backoff + jitter. Undefined ⇒
459
+ * DEFAULT_MAX_RETRIES (2); 0 disables. Honored by withChatRetry, which the
460
+ * registry applies to the direct-provider adapters (OpenRouter relies on
461
+ * its SDK's own retries). */
462
+ maxRetries?: number;
463
+ /** Provider-neutral prompt-cache hints. See {@link ChatCacheControl}.
464
+ * Adapters that don't talk to a cache-aware provider ignore this. */
465
+ cacheControl?: ChatCacheControl;
466
+ /** Optional provider-specific overrides — adapter chooses what to honour.
467
+ * Used for things like xAI's `reasoning_effort` or HF's `:fastest`
468
+ * routing suffix. */
469
+ extra?: Record<string, unknown>;
470
+ /** Per-route base URL override for self-hosted / OpenAI-compatible chat
471
+ * servers. Lets a `local` chat route target a specific host (a LAN/tailnet
472
+ * box). The `local-chat` adapter honours it; fixed-endpoint cloud adapters
473
+ * ignore it. Mirrors `EmbedRequest.baseUrl`. */
474
+ baseUrl?: string;
475
+ /** When true, the request is dispatched through the Tailscale forward-proxy
476
+ * ({@link tailnetFetch}) so a `baseUrl` pointing at a tailnet host reaches a
477
+ * box behind NAT. Honoured by the `local-chat` adapter; inert when no proxy
478
+ * is configured. */
479
+ viaTailnet?: boolean;
480
+ /** Cancellation signal for the request — wired by the tool loop so a user can
481
+ * STOP an in-flight streamed turn. A streaming adapter passes it to the
482
+ * underlying fetch and, on abort, stops reading and returns the partial reply
483
+ * assembled so far (rather than throwing). Adapters that don't honour it just
484
+ * run to completion. */
485
+ signal?: AbortSignal;
486
+ }
487
+
488
+ /** A user-visible delta surfaced by `chatStream` as the model produces output.
489
+ * Deliberately minimal: only the visible reply text and the model's reasoning
490
+ * stream. Tool-call argument fragments are accumulated INTERNALLY by the adapter
491
+ * (so the resolved `ChatResult.toolCalls` is fully assembled, identical to
492
+ * `chat()`) — they're machinery, not something the user reads, so they're not
493
+ * surfaced here. */
494
+ export type ChatStreamDelta = { type: 'text'; text: string } | { type: 'reasoning'; text: string };
495
+
496
+ /** Sink the caller passes to `chatStream` to receive deltas as they arrive. The
497
+ * adapter calls it synchronously per chunk and never awaits it — the caller
498
+ * fans it onto the ephemeral live bus and must not let it throw back into the
499
+ * stream loop. */
500
+ export type ChatStreamSink = (delta: ChatStreamDelta) => void;
501
+
502
+ export interface ChatDispatcher extends AdapterMeta {
503
+ /** One-shot chat completion. */
504
+ chat(opts: ChatOptions): Promise<ChatResult>;
505
+ /** Streaming variant of `chat`: emits text/reasoning deltas to `onDelta` as
506
+ * they arrive AND resolves to the same fully-assembled `ChatResult` `chat()`
507
+ * would return (text, toolCalls, usage) — so it's a drop-in. The ChatResult is
508
+ * the durable answer; the deltas are ephemeral decoration. Optional + additive:
509
+ * a provider that can't stream omits it and callers fall back to `chat()`. */
510
+ chatStream?(opts: ChatOptions, onDelta: ChatStreamSink): Promise<ChatResult>;
511
+ /** Live-discover available chat models. Adapter does the cross-
512
+ * reference between provider's /v1/models response and its own
513
+ * static catalog. */
514
+ discoverModels?(apiKey: string): Promise<DiscoveryResult<ChatModelInfo>>;
515
+ /** Adapters expose their static catalog for the UI to render before
516
+ * discovery completes (or when discovery isn't supported). */
517
+ staticCatalog?(): readonly ChatModelInfo[];
518
+ }
519
+
520
+ export interface SttDispatcher extends AdapterMeta {
521
+ /** The request options this adapter forwards, completing the convention
522
+ * shared with {@link TtsDispatcher.supports} and
523
+ * {@link ImageGenDispatcher.supports}.
524
+ *
525
+ * Thin by nature: transcription's normalized surface is a language hint and
526
+ * nothing else, so every wired adapter currently declares `['language']`.
527
+ * It is still worth declaring rather than assuming — a 2026-08 docs sweep
528
+ * found xAI's `format` flag silently depending on `language`, and the value
529
+ * of these lists is that they are checked claims rather than beliefs. What
530
+ * each provider REPORTS BACK varies far more than what it accepts; see
531
+ * `TranscribeResult`, where several adapters can only return nulls. */
532
+ supports: readonly SttParam[];
533
+
534
+ /** Transcribe an audio buffer. Buffer must be in one of the formats
535
+ * the provider accepts; the adapter handles the multipart encoding
536
+ * and any provider-specific MIME wrangling. */
537
+ transcribe(audio: Buffer, opts: TranscribeOptions): Promise<TranscribeResult>;
538
+
539
+ /** Optional. Live-discover available transcription models. */
540
+ discoverModels?(apiKey: string): Promise<DiscoveryResult<SttModelInfo>>;
541
+ }
542
+
543
+ // ─── Vision ─────────────────────────────────────────────────────────
544
+
545
+ /** A vision-capable model entry. Same generic shape as ChatModelInfo;
546
+ * the form's vision-worker dropdown renders these so operators pick
547
+ * from a list rather than typing a model id by hand. */
548
+ export interface VisionModelInfo {
549
+ id: string;
550
+ label: string;
551
+ description: string;
552
+ /** Context window in tokens. */
553
+ contextTokens?: number;
554
+ /** USD per 1M input tokens. Approximate; for UI hints only. */
555
+ inputPricePer1M?: number;
556
+ /** USD per 1M output tokens. */
557
+ outputPricePer1M?: number;
558
+ /** Roughly which tier the model fits. 'fast' = use for high volume;
559
+ * 'balanced' = default; 'quality' = harder images, lower throughput. */
560
+ tier?: 'fast' | 'balanced' | 'quality';
561
+ }
562
+
563
+ export interface VisionExtractOptions {
564
+ apiKey: string;
565
+ /** MIME of the image bytes (image/jpeg, image/png, image/webp, image/gif).
566
+ * Each provider accepts a slightly different list — adapters error
567
+ * on unsupported MIMEs rather than silently sending bytes the API
568
+ * will reject downstream. */
569
+ mimeType: string;
570
+ /** User-side prompt — "transcribe this page verbatim", "summarize
571
+ * the diagram", "extract action items as a list". Adapters pass it
572
+ * through alongside the image, framed as the user turn. */
573
+ prompt: string;
574
+ /** Optional system-level steering (used the same way as a chat
575
+ * worker's system prompt). Operators set this on the worker row;
576
+ * callers forward it here. */
577
+ systemPrompt?: string;
578
+ /** Model id. Falls back to the adapter's documented default if
579
+ * omitted. */
580
+ model?: string;
581
+ /** Max output tokens. Vision-LLMs can run long without one. */
582
+ maxTokens?: number;
583
+ }
584
+
585
+ export interface VisionExtractResult {
586
+ /** Extracted text. Trimmed; empty string on no-output, not null. */
587
+ text: string;
588
+ /** Model that did the work. May be a more specific id than the
589
+ * caller passed in (e.g. dated Claude variants). */
590
+ model: string;
591
+ /** Token usage when the provider returns it. */
592
+ tokensIn?: number;
593
+ tokensOut?: number;
594
+ }
595
+
596
+ export interface VisionDispatcher extends AdapterMeta {
597
+ /** Extract text/structure from an image. The adapter handles the
598
+ * per-provider message shape, MIME validation, and image encoding
599
+ * (base64 vs URL). Caller hands raw bytes + a prompt; result is
600
+ * trimmed text. */
601
+ extract(image: Buffer, opts: VisionExtractOptions): Promise<VisionExtractResult>;
602
+
603
+ /** Optional. Extract text from a document (PDF) sent NATIVELY to the model —
604
+ * no rasterization. Providers whose API accepts a document content block
605
+ * (Anthropic, Google) implement this; the runtime PREFERS it over
606
+ * rasterize→per-page image OCR for PDFs (whole-document context, real layout
607
+ * and tables, one call, no PNG-conversion fidelity loss). Adapters that
608
+ * can't take a document natively (OpenAI, xAI) omit it, and the caller falls
609
+ * back to rasterizing the pages through `extract`. `opts.mimeType` is the
610
+ * document MIME (e.g. 'application/pdf'). */
611
+ extractDocument?(document: Buffer, opts: VisionExtractOptions): Promise<VisionExtractResult>;
612
+
613
+ /** Live-discover which vision-capable models the api key can use.
614
+ * Implementation parity with ChatDispatcher.discoverModels — when
615
+ * absent, the form falls back to the adapter's static catalog. */
616
+ discoverModels?(apiKey: string): Promise<import('../catalog').DiscoveryResult<VisionModelInfo>>;
617
+
618
+ /** Static catalog the UI renders before live discovery returns or
619
+ * when the adapter doesn't support discovery. */
620
+ staticCatalog?(): readonly VisionModelInfo[];
621
+ }
622
+
623
+ // ─── Image generation ───────────────────────────────────────────────
624
+
625
+ /** A generatable image model entry. Same shape conventions as
626
+ * ChatModelInfo/VisionModelInfo: id + label + description so the UI
627
+ * has rich dropdown options without each form re-parsing provider
628
+ * docs. */
629
+ export interface ImageGenModelInfo {
630
+ id: string;
631
+ label: string;
632
+ description: string;
633
+ /** Native resolutions the model accepts. Adapters reject sizes
634
+ * outside this list with a clear error. Free-form when undefined
635
+ * (HF models, where the underlying model decides). */
636
+ supportedSizes?: readonly string[];
637
+ /** Steerable styles, when the model supports them (DALL-E 3:
638
+ * 'vivid' | 'natural'). Undefined = no style steering. */
639
+ supportedStyles?: readonly string[];
640
+ /** Quality tiers the model accepts. Genuinely provider-specific:
641
+ * DALL-E 3 takes 'standard' | 'hd', gpt-image-1 takes
642
+ * 'low' | 'medium' | 'high' | 'auto', OpenRouter normalizes to the
643
+ * latter. Undefined = no quality control. */
644
+ supportedQualities?: readonly string[];
645
+ /** Aspect ratios the model accepts, for providers that steer by ratio
646
+ * rather than by pixel size. Undefined = not ratio-steerable. */
647
+ supportedAspectRatios?: readonly string[];
648
+ /** USD per image at default size. UI hint only. */
649
+ pricePerImage?: number;
650
+ /** Latency tier — useful when picking between same-provider models. */
651
+ tier?: 'fast' | 'balanced' | 'quality';
652
+ }
653
+
654
+ /** A request option an image-gen adapter may or may not forward.
655
+ *
656
+ * Declared per adapter so an option that will not survive the trip is
657
+ * REPORTED rather than silently discarded. Silent drop is the failure this
658
+ * interface exists to kill: an operator set size/style/quality on an
659
+ * OpenRouter worker, the UI showed them saved, and the adapter sent
660
+ * neither — with nothing in the trace to say so. See
661
+ * {@link ImageGenDispatcher.supports}. */
662
+ export type ImageGenParam =
663
+ 'size' | 'aspectRatio' | 'style' | 'quality' | 'negativePrompt' | 'seed' | 'inputImages';
664
+
665
+ /** One option the adapter chose not to send, and why.
666
+ *
667
+ * `supports` is declared per PROVIDER, but the real gate is often per MODEL:
668
+ * the OpenAI images endpoint takes `style` on dall-e-3 and not on
669
+ * gpt-image-1, `quality` on those two and not on dall-e-2. A provider-level
670
+ * list cannot express that, so the adapter — the only layer that knows which
671
+ * model it is actually talking to — reports the difference here rather than
672
+ * dropping it. (Same role as `warnings` on the AI SDK's image result.) */
673
+ export type ImageGenWarning = {
674
+ param: ImageGenParam;
675
+ /** Model-specific reason, phrased for a reader deciding what to do next. */
676
+ reason: string;
677
+ };
678
+
679
+ /** A picture handed IN to a generation: the reference for an edit or a
680
+ * variation. Bytes rather than a URL, because the source is a file node in
681
+ * the owner's own store and providers take base64 or multipart, never a link
682
+ * only we can reach. */
683
+ export interface ImageGenInput {
684
+ bytes: Buffer;
685
+ mimeType: string;
686
+ /** Original filename, used where the provider takes multipart. */
687
+ filename?: string;
688
+ }
689
+
690
+ export interface GenerateImageOptions {
691
+ apiKey: string;
692
+ /** Free-form prompt. No length cap at the interface layer; per-
693
+ * provider limits get enforced inside the adapter. */
694
+ prompt: string;
695
+ /** Reference images. Present ⇒ this is an EDIT, not a fresh generation:
696
+ * "make the sky orange in this one". An adapter that does not declare
697
+ * `inputImages` in `supports` must never be handed these — generating from
698
+ * scratch when an edit was asked for produces a confidently wrong picture
699
+ * and bills for it, so the caller refuses BEFORE spending. */
700
+ inputImages?: readonly ImageGenInput[];
701
+ /** Model id. Defaults to the adapter's documented default. */
702
+ model?: string;
703
+ /** Native resolution, e.g. '1024x1024' or '1792x1024'. Adapters
704
+ * validate against ImageGenModelInfo.supportedSizes. OpenRouter also
705
+ * accepts a tier ('1K' / '2K' / '4K'). */
706
+ size?: string;
707
+ /** Aspect ratio, e.g. '16:9'. The natural control for providers that
708
+ * steer by ratio (Imagen) or normalize across them (OpenRouter);
709
+ * ignored by the fixed-size APIs. */
710
+ aspectRatio?: string;
711
+ /** Style hint (provider-specific: DALL-E 3 = 'vivid'/'natural';
712
+ * others may ignore). */
713
+ style?: string;
714
+ /** Quality tier — DALL-E 3 uses 'standard'/'hd'; HF models may use
715
+ * 'fast'/'balanced'/'quality'. Adapters ignore unknown values. */
716
+ quality?: string;
717
+ /** Negative prompt — what the image should NOT contain. Honoured
718
+ * by HF + Imagen; OpenAI doesn't accept it (silently ignored). */
719
+ negativePrompt?: string;
720
+ /** Random seed for reproducibility, when the provider exposes one. */
721
+ seed?: number;
722
+ }
723
+
724
+ export interface GenerateImageResult {
725
+ /** Generated image bytes. Adapters always materialize the bytes
726
+ * even when the provider returns a URL — callers shouldn't have
727
+ * to deal with URL-vs-bytes asymmetry across providers. */
728
+ bytes: Buffer;
729
+ /** MIME of the returned image — typically 'image/png' but Imagen
730
+ * may return jpeg, HF may return jpeg/png depending on model. */
731
+ mimeType: string;
732
+ /** Model id that did the work. Echoed back for traces. */
733
+ model: string;
734
+ /** Options the adapter deliberately did NOT send, because this MODEL has
735
+ * no such control. Empty/absent means everything the caller passed went on
736
+ * the wire. Callers must surface these rather than let a requested option
737
+ * look like it applied. */
738
+ warnings?: readonly ImageGenWarning[];
739
+ /** Provider's revised prompt, when it returns one (DALL-E 3 rewrites
740
+ * the user prompt for safety+quality and surfaces the revision).
741
+ * Caller can pass this back as `revised_prompt` metadata on the
742
+ * saved file node so the operator sees what the model actually
743
+ * rendered against. */
744
+ revisedPrompt?: string;
745
+ }
746
+
747
+ export interface ImageGenDispatcher extends AdapterMeta {
748
+ /** The subset of {@link GenerateImageOptions} this adapter actually puts
749
+ * on the wire. REQUIRED, and it must be honest: the caller reports every
750
+ * requested option outside this set back to the model and into the trace
751
+ * rather than pretending it applied. An adapter that starts forwarding a
752
+ * new option must add it here in the same commit. */
753
+ supports: readonly ImageGenParam[];
754
+
755
+ /** Generate an image from a prompt. Throws on auth/network/quota
756
+ * errors with the provider's verbatim message slice — same
757
+ * convention as the other dispatchers. */
758
+ generate(opts: GenerateImageOptions): Promise<GenerateImageResult>;
759
+
760
+ /** Static catalog the UI renders for the model dropdown. Most
761
+ * image-gen providers don't expose a programmatic models list
762
+ * (or they do but it returns chat models too) — so we ship a
763
+ * curated static list and don't implement discoverModels. */
764
+ staticCatalog(): readonly ImageGenModelInfo[];
765
+ }
766
+
767
+ // ─── Embedding ──────────────────────────────────────────────────────
768
+
769
+ /** A text→vector model entry. Embedding pricing is input-only
770
+ * (the response is the vector, not a token stream — no output cost
771
+ * to bill). Native `dimensions` is exposed so the form can verify
772
+ * compatibility with the brain's `vector(768)` column before save. */
773
+ export interface EmbeddingModelInfo {
774
+ id: string;
775
+ label: string;
776
+ description: string;
777
+ /** Maximum input tokens accepted in a single call. */
778
+ contextTokens?: number;
779
+ /** USD per 1M input tokens. */
780
+ inputPricePer1M?: number;
781
+ /** Output vector dimension as the provider documents it. The form
782
+ * uses this to drive the dim-mismatch save block before the
783
+ * operator gets a 'won't insert' surprise at runtime. */
784
+ dimensions?: number;
785
+ /** Accepts non-text inputs (image / audio / file). Only OpenRouter's
786
+ * multimodal route and Google's gemini-embedding-2-preview today. */
787
+ multimodal?: boolean;
788
+ }
789
+
790
+ /** Inputs the embedding dispatchers accept. Same shape as the
791
+ * pre-adapter @mantle/embeddings package — keeps the public API
792
+ * stable while the dispatch path swaps underneath. */
793
+ export type EmbedInput =
794
+ | string
795
+ | { type: 'text'; text: string }
796
+ | { type: 'image'; url: string }
797
+ | { type: 'audio'; url: string }
798
+ | { type: 'file'; url: string; mimeType?: string };
799
+
800
+ export interface EmbedRequest {
801
+ apiKey: string;
802
+ model: string;
803
+ /** Single-element array for single-text calls; batch is the common path
804
+ * (extractor + recall batch many at once for cache + API efficiency). */
805
+ input: EmbedInput[];
806
+ /** Truncate to this dim where supported. OpenAI's text-embedding-3-*
807
+ * family honours it (truncation by MRL); Google's gemini-embedding-2
808
+ * honours it as `output_dimensionality`. Everything else ignores. */
809
+ dimensions?: number;
810
+ /** Per-call base URL override for self-hosted / OpenAI-compatible routes.
811
+ * Lets the embedding config point primary and backup at different hosts
812
+ * serving the SAME model (failover). The `local` adapter honours it;
813
+ * fixed-endpoint cloud adapters ignore it. */
814
+ baseUrl?: string;
815
+ /** When true, the request is dispatched through the Tailscale forward-proxy
816
+ * ({@link tailnetFetch}) so a `baseUrl` pointing at a tailnet host reaches a
817
+ * box behind NAT. Honoured by the `local-embedding` adapter; inert when no
818
+ * proxy is configured. */
819
+ viaTailnet?: boolean;
820
+ /** Texts per HTTP request for the `local-embedding` adapter. Lets the embedding
821
+ * config tune throughput per-owner (small on a CPU box so a request fits the
822
+ * timeout, large on a GPU). Null/undefined → the adapter's own
823
+ * `MANTLE_LOCAL_EMBED_BATCH` env → 16. Ignored by cloud adapters. */
824
+ localEmbedBatchSize?: number;
825
+ /** Per-request timeout (ms) for the `local-embedding` adapter. Null/undefined
826
+ * → `MANTLE_LOCAL_EMBED_TIMEOUT_MS` env → 120000. Ignored by cloud adapters. */
827
+ localEmbedTimeoutMs?: number;
828
+ }
829
+
830
+ export interface EmbedResult {
831
+ vectors: number[][];
832
+ /** Server-reported model id (so callers can verify their slug landed
833
+ * where they expected — direct providers sometimes alias). */
834
+ model: string;
835
+ /** Total input tokens billed. Undefined when the provider doesn't
836
+ * report (Cohere v2 omits, some HF routes too). */
837
+ tokensIn?: number;
838
+ }
839
+
840
+ export interface EmbeddingDispatcher extends AdapterMeta {
841
+ /** Embed a batch. Adapters that don't support multimodal input throw
842
+ * a clear error on non-text items rather than silently truncating —
843
+ * surfaces "you picked a text-only model but sent an image" at the
844
+ * point of failure rather than as a confused empty result. */
845
+ embed(req: EmbedRequest): Promise<EmbedResult>;
846
+
847
+ /** Optional: whether this adapter will accept the given input. Text-only
848
+ * adapters return false for non-text items so the caller can route
849
+ * multimodal inputs to OpenRouter (the only multimodal-capable path)
850
+ * instead of failing at request time. Default = true. */
851
+ acceptsInput?(input: EmbedInput): boolean;
852
+
853
+ /** Live-discover available embedding models. Each adapter does the
854
+ * cross-reference between its provider's list endpoint and its own
855
+ * static catalog. OpenRouter publishes `/v1/embeddings/models`
856
+ * separately from `/v1/models`; OpenAI returns embeddings inline
857
+ * in `/v1/models` (filtered by id pattern); Google requires a
858
+ * capability filter (`supportedGenerationMethods` includes
859
+ * `embedContent`). */
860
+ discoverModels?(apiKey: string): Promise<DiscoveryResult<EmbeddingModelInfo>>;
861
+
862
+ /** Curated fallback list. Used by the workers form when no API key
863
+ * is configured yet (so the dropdown isn't empty in create mode)
864
+ * AND as a last-resort if discovery errors. */
865
+ staticCatalog?(): readonly EmbeddingModelInfo[];
866
+ }