@crossworks/voice-client 0.230.43
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE.md +135 -0
- package/package.json +20 -0
- package/src/adapters/registry.ts +296 -0
- package/src/adapters/retry.ts +193 -0
- package/src/adapters/types.ts +866 -0
- package/src/audio-tags.test.ts +221 -0
- package/src/audio-tags.ts +191 -0
- package/src/catalog.test.ts +144 -0
- package/src/catalog.ts +237 -0
- package/src/catalogs/anthropic.ts +135 -0
- package/src/catalogs/assemblyai.ts +54 -0
- package/src/catalogs/copilot.ts +63 -0
- package/src/catalogs/deepgram.ts +61 -0
- package/src/catalogs/deepseek.ts +85 -0
- package/src/catalogs/elevenlabs.ts +244 -0
- package/src/catalogs/google.ts +332 -0
- package/src/catalogs/huggingface.ts +180 -0
- package/src/catalogs/openai-image.ts +62 -0
- package/src/catalogs/openai-vision.ts +53 -0
- package/src/catalogs/openrouter.ts +221 -0
- package/src/catalogs/xai.ts +330 -0
- package/src/index.ts +48 -0
- package/src/providers.test.ts +172 -0
- package/src/providers.ts +262 -0
- package/src/types.ts +137 -0
- package/tsconfig.json +4 -0
- package/tsconfig.tsbuildinfo +1 -0
|
@@ -0,0 +1,866 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Adapter interfaces — the boundary between the rest of Mantle and
|
|
3
|
+
* provider-specific HTTP calls.
|
|
4
|
+
*
|
|
5
|
+
* One interface per capability. Each provider that wants to be "wired"
|
|
6
|
+
* for a capability implements the matching interface. The runtime
|
|
7
|
+
* looks up the right adapter by `providerId` (which comes from the
|
|
8
|
+
* worker's `provider` column) and calls through it. The dispatch
|
|
9
|
+
* surface is identical regardless of provider — the adapter handles
|
|
10
|
+
* all the per-provider quirks (auth header shape, response decoding,
|
|
11
|
+
* voice/model naming).
|
|
12
|
+
*
|
|
13
|
+
* Why we own these instead of using LiteLLM:
|
|
14
|
+
* - End-to-end type safety. The adapter's input/output shapes are
|
|
15
|
+
* fixed at compile time.
|
|
16
|
+
* - No proxy service to operate on the VPS.
|
|
17
|
+
* - Adapters are ~50-150 LOC each. Total maintenance cost is small
|
|
18
|
+
* because we only adapter the providers we actually use.
|
|
19
|
+
* - LiteLLM's open-source code remains a reference we can lift from
|
|
20
|
+
* when a provider's API shape is surprising.
|
|
21
|
+
*
|
|
22
|
+
* Each adapter exposes its `providerId` (must match the catalogue in
|
|
23
|
+
* `providers.ts`) and an `adapterName` used purely for logs/traces so
|
|
24
|
+
* we can tell at a glance which adapter ran a given call.
|
|
25
|
+
*/
|
|
26
|
+
|
|
27
|
+
import type { ProviderId } from '../providers';
|
|
28
|
+
import type {
|
|
29
|
+
SynthesizeOptions,
|
|
30
|
+
SynthesizeResult,
|
|
31
|
+
SttParam,
|
|
32
|
+
TranscribeOptions,
|
|
33
|
+
TranscribeResult,
|
|
34
|
+
TtsParam,
|
|
35
|
+
} from '../types';
|
|
36
|
+
import type { TtsModelInfo, SttModelInfo } from '../catalog';
|
|
37
|
+
import type { DiscoveryResult } from '../catalog';
|
|
38
|
+
|
|
39
|
+
/** Common shape every adapter exposes — used by the registry to log
|
|
40
|
+
* and surface the right one in errors. */
|
|
41
|
+
export interface AdapterMeta {
|
|
42
|
+
/** Matches one of the ids in SUPPORTED_PROVIDERS. */
|
|
43
|
+
readonly providerId: ProviderId;
|
|
44
|
+
/** Human-readable for logs. Convention: '<provider>-<capability>'
|
|
45
|
+
* (e.g. 'openai-tts', 'elevenlabs-tts'). */
|
|
46
|
+
readonly adapterName: string;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
/** An inline audio tag the model interprets as a performance cue.
|
|
50
|
+
* ElevenLabs v3 calls these "audio tags"; xAI's voice models use a
|
|
51
|
+
* similar `[giggle]`/`[laugh]` convention. The shape is uniform: an
|
|
52
|
+
* exact bracket-wrapped token + a short description for the LLM's
|
|
53
|
+
* benefit so it knows when to use which.
|
|
54
|
+
*
|
|
55
|
+
* Inline tags are *point-in-time* — they fire at the spot they sit.
|
|
56
|
+
* For tags that style a whole *span* of text (e.g. xAI's
|
|
57
|
+
* `<whisper>…</whisper>`), see {@link WrappingTag}. */
|
|
58
|
+
export type AudioTag = {
|
|
59
|
+
/** Exact form including brackets, e.g. `[laughs]`, `[whispers]`. */
|
|
60
|
+
tag: string;
|
|
61
|
+
/** One-line hint for the LLM. The composer joins these into the
|
|
62
|
+
* system-prompt paragraph so Saskia knows what each tag does. */
|
|
63
|
+
description: string;
|
|
64
|
+
/** Coarse grouping so the UI hint can render them in sections. */
|
|
65
|
+
category?: 'emotion' | 'reaction' | 'delivery' | 'cognitive' | 'tone' | 'accent';
|
|
66
|
+
};
|
|
67
|
+
|
|
68
|
+
/** A wrapping speech tag — an angle-bracket pair that styles the whole
|
|
69
|
+
* phrase it surrounds, e.g. xAI Grok's `<whisper>secret</whisper>`,
|
|
70
|
+
* `<soft>…</soft>`, `<slow>…</slow>`. Distinct from {@link AudioTag}:
|
|
71
|
+
* inline tags are point-in-time cues, wrapping tags apply a delivery
|
|
72
|
+
* style across a span.
|
|
73
|
+
*
|
|
74
|
+
* The framework treats these the same way it treats inline tags:
|
|
75
|
+
* adapters advertise the set their model honours (`supportedWrappingTags`),
|
|
76
|
+
* the prompt composer tells Saskia she may use them, and the text-out
|
|
77
|
+
* path strips them (keeping the inner text) so the markers never leak
|
|
78
|
+
* into a plain-text reply. */
|
|
79
|
+
export type WrappingTag = {
|
|
80
|
+
/** Short lower-case name without brackets, e.g. `whisper`, `soft`.
|
|
81
|
+
* The open/close forms are derived as `<name>` / `</name>`. */
|
|
82
|
+
name: string;
|
|
83
|
+
/** One-line hint for the LLM — what the style does and when to reach
|
|
84
|
+
* for it. Joined into the system-prompt paragraph. */
|
|
85
|
+
description: string;
|
|
86
|
+
/** Coarse grouping for the UI hint (volume / pitch / pacing / style). */
|
|
87
|
+
category?: 'volume' | 'pitch' | 'pacing' | 'style';
|
|
88
|
+
};
|
|
89
|
+
|
|
90
|
+
export interface TtsDispatcher extends AdapterMeta {
|
|
91
|
+
/** The subset of {@link SynthesizeOptions} this adapter actually puts on the
|
|
92
|
+
* wire, excluding `voice`/`model`/`text` which every provider takes.
|
|
93
|
+
* REQUIRED and it must be honest: the caller reports every requested option
|
|
94
|
+
* outside this set rather than letting it look like it applied. Where the
|
|
95
|
+
* gate is per-MODEL instead of per-provider, the adapter additionally
|
|
96
|
+
* returns `warnings` from the call. Same contract as
|
|
97
|
+
* {@link ImageGenDispatcher.supports}. */
|
|
98
|
+
supports: readonly TtsParam[];
|
|
99
|
+
|
|
100
|
+
/** Synthesise speech and return audio bytes. The runtime then hands
|
|
101
|
+
* these to Telegram's sendVoice (when format='opus') or to an
|
|
102
|
+
* `<audio>` element on the web. */
|
|
103
|
+
synthesize(opts: SynthesizeOptions): Promise<SynthesizeResult>;
|
|
104
|
+
|
|
105
|
+
/** Optional. Live-discover which TTS models the API key can use.
|
|
106
|
+
* For OpenAI we hit /v1/models and intersect with the catalogue;
|
|
107
|
+
* for ElevenLabs we'd hit /v1/models too but the catalogue is
|
|
108
|
+
* different. If absent, the UI falls back to whatever static
|
|
109
|
+
* catalogue the adapter consults. */
|
|
110
|
+
discoverModels?(apiKey: string): Promise<DiscoveryResult<TtsModelInfo>>;
|
|
111
|
+
|
|
112
|
+
/** Optional. Returns the voices a given model supports. For OpenAI
|
|
113
|
+
* this is a static per-model list (we hardcode 13 voices for
|
|
114
|
+
* gpt-4o-mini-tts, 9 for tts-1). For ElevenLabs this would be a
|
|
115
|
+
* live `/v1/voices` query that includes the user's cloned voices.
|
|
116
|
+
* When absent, callers fall through to whatever default they
|
|
117
|
+
* rendered the form with. */
|
|
118
|
+
voicesForModel?(
|
|
119
|
+
modelId: string,
|
|
120
|
+
apiKey?: string,
|
|
121
|
+
): Promise<Array<{ id: string; description: string }>>;
|
|
122
|
+
|
|
123
|
+
/** Optional. Inline audio tags the model interprets — `[laughs]`,
|
|
124
|
+
* `[whispers]`, `[sighs]`, etc. ElevenLabs v3 has the richest set;
|
|
125
|
+
* xAI's voice models support a smaller subset; OpenAI's
|
|
126
|
+
* gpt-4o-mini-tts uses the `instructions` parameter instead and
|
|
127
|
+
* returns an empty list here.
|
|
128
|
+
*
|
|
129
|
+
* The runtime queries this when building the chat agent's prompt
|
|
130
|
+
* so the LLM only emits tags the active TTS will render. Adapters
|
|
131
|
+
* whose providers ignore unknown tags can safely return [] —
|
|
132
|
+
* Saskia won't try to use them. */
|
|
133
|
+
supportedAudioTags?(modelId: string): readonly AudioTag[];
|
|
134
|
+
|
|
135
|
+
/** Optional. Wrapping speech tags the model honours —
|
|
136
|
+
* `<whisper>…</whisper>`, `<soft>…</soft>`, `<slow>…</slow>`, etc.
|
|
137
|
+
* (xAI Grok voice today). Same contract as {@link supportedAudioTags}
|
|
138
|
+
* but for span-styling rather than point-in-time cues. Adapters
|
|
139
|
+
* whose providers have no angle-bracket wrapping vocabulary return
|
|
140
|
+
* [] (or omit this), and the runtime simply won't advertise any. */
|
|
141
|
+
supportedWrappingTags?(modelId: string): readonly WrappingTag[];
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
/** A chat model entry. The fields are intentionally generic — provider
|
|
145
|
+
* catalogs (xAI, HF, OpenAI…) all describe their chat models in
|
|
146
|
+
* roughly this shape. Pricing is included so the UI can show a
|
|
147
|
+
* rough cost-per-1M-tokens estimate when picking a model. */
|
|
148
|
+
export interface ChatModelInfo {
|
|
149
|
+
id: string;
|
|
150
|
+
label: string;
|
|
151
|
+
description: string;
|
|
152
|
+
/** Context window in tokens. */
|
|
153
|
+
contextTokens?: number;
|
|
154
|
+
/** Capabilities beyond plain chat. */
|
|
155
|
+
capabilities?: readonly ('vision' | 'reasoning' | 'function_calling' | 'json_mode')[];
|
|
156
|
+
/** USD per 1M input tokens. Approximate; for UI hints only. */
|
|
157
|
+
inputPricePer1M?: number;
|
|
158
|
+
/** USD per 1M output tokens. */
|
|
159
|
+
outputPricePer1M?: number;
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
/** Result of a chat completion. Mirrors the OpenAI shape since every
|
|
163
|
+
* adapter we're likely to write speaks that dialect.
|
|
164
|
+
*
|
|
165
|
+
* Cache + cost fields are optional because not every provider reports
|
|
166
|
+
* them. The runtime treats `undefined` as "this provider doesn't tell
|
|
167
|
+
* us" and falls back to the static price table for cost; cache fields
|
|
168
|
+
* stay zero in the trace. Adapters MUST populate `tokensIn`/`tokensOut`
|
|
169
|
+
* when the provider returns them (every chat API in the catalogue does)
|
|
170
|
+
* so the trace's token + fallback-cost numbers stay accurate. */
|
|
171
|
+
export interface ChatResult {
|
|
172
|
+
text: string;
|
|
173
|
+
model: string;
|
|
174
|
+
/** Tool calls the model wants to make. When non-empty, the
|
|
175
|
+
* tool-loop dispatches each one and feeds results back. Adapters
|
|
176
|
+
* normalise from their provider's native shape (Anthropic
|
|
177
|
+
* `tool_use` blocks, Google `functionCall` parts, OpenAI
|
|
178
|
+
* `tool_calls`) to this single grammar.
|
|
179
|
+
*
|
|
180
|
+
* Important: when toolCalls is non-empty, `text` may be empty —
|
|
181
|
+
* the model emitted only tool calls and no narrative this turn. */
|
|
182
|
+
toolCalls?: ChatToolCall[];
|
|
183
|
+
tokensIn?: number;
|
|
184
|
+
tokensOut?: number;
|
|
185
|
+
/** Tokens served from the provider's prompt cache, billed at the
|
|
186
|
+
* reduced cache-read rate. Anthropic returns this as
|
|
187
|
+
* `cache_read_input_tokens`; OpenAI as `prompt_tokens_details.cached_tokens`;
|
|
188
|
+
* Google as `usageMetadata.cachedContentTokenCount`; OpenRouter
|
|
189
|
+
* aliases all of the above as `cache_read_input_tokens` /
|
|
190
|
+
* `cached_tokens`. The trace records this separately from
|
|
191
|
+
* `tokensIn` so the cost dashboard can show the actual cache-read
|
|
192
|
+
* savings on the responder path. */
|
|
193
|
+
cacheReadTokens?: number;
|
|
194
|
+
/** Tokens *written* into the provider's prompt cache, billed at the
|
|
195
|
+
* cache-write rate (Anthropic charges ~1.25× input for these). Only
|
|
196
|
+
* Anthropic surfaces this distinctly (`cache_creation_input_tokens`);
|
|
197
|
+
* every other provider folds it into `tokensIn`. */
|
|
198
|
+
cacheWriteTokens?: number;
|
|
199
|
+
/** Provider-reported USD cost for the call, in dollars (not micro-USD).
|
|
200
|
+
* Currently only OpenRouter populates this via `usage.cost` — and it
|
|
201
|
+
* matters because OR rolls in vendor surcharges that the static price
|
|
202
|
+
* table doesn't know about (e.g. Perplexity's per-search fee). Direct
|
|
203
|
+
* providers return `undefined` here; the trace falls back to
|
|
204
|
+
* `fallbackCostMicroUsd(model, ...)`. */
|
|
205
|
+
reportedCostUsd?: number;
|
|
206
|
+
/** Provider reasoning blocks from this turn (OpenRouter `reasoning_details`).
|
|
207
|
+
* When the model thought before answering, the tool-loop stores these on the
|
|
208
|
+
* assistant message it appends so the NEXT request can echo them back —
|
|
209
|
+
* required for thinking continuity across a tool round-trip (see
|
|
210
|
+
* {@link ChatAssistantMessage.reasoningDetails}). Undefined when the turn
|
|
211
|
+
* produced no reasoning. */
|
|
212
|
+
reasoningDetails?: ReasoningDetail[];
|
|
213
|
+
/** Why the model stopped, normalised across providers. Undefined when the
|
|
214
|
+
* adapter doesn't report one (most don't yet) — treat that as "unknown",
|
|
215
|
+
* NOT as `'stop'`.
|
|
216
|
+
*
|
|
217
|
+
* This exists because a truncated answer and a policy-blocked answer are
|
|
218
|
+
* otherwise indistinguishable from a genuinely short one: every provider
|
|
219
|
+
* returns HTTP 200 with little or no text in all three cases. Without this
|
|
220
|
+
* field a caller cannot tell "the model finished" from "the model was cut
|
|
221
|
+
* off at maxTokens" or "the provider refused". Adapters map their native
|
|
222
|
+
* value onto this small grammar:
|
|
223
|
+
*
|
|
224
|
+
* - 'stop' — finished normally
|
|
225
|
+
* - 'length' — hit the output-token ceiling, reply is TRUNCATED
|
|
226
|
+
* - 'tool_calls' — stopped to call tools (expect `toolCalls`)
|
|
227
|
+
* - 'content_filter' — provider blocked it (safety/recitation/blocklist)
|
|
228
|
+
* - 'error' — provider signalled a malformed generation
|
|
229
|
+
* - 'other' — reported, but not one of the above */
|
|
230
|
+
finishReason?: ChatFinishReason;
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
/** Normalised stop reason. Deliberately a small closed set: adapters collapse
|
|
234
|
+
* their provider's longer enum onto these, since callers only ever need to
|
|
235
|
+
* branch on "was this reply complete, truncated, or refused". */
|
|
236
|
+
export type ChatFinishReason =
|
|
237
|
+
'stop' | 'length' | 'tool_calls' | 'content_filter' | 'error' | 'other';
|
|
238
|
+
|
|
239
|
+
/** A single tool the model can call. Mirrors the OpenAI function-tool
|
|
240
|
+
* shape since every adapter we're likely to talk to either accepts
|
|
241
|
+
* this natively (OpenRouter / xAI / HF) or translates from it
|
|
242
|
+
* (Anthropic `tool_use`, Google `functionDeclarations`). */
|
|
243
|
+
export interface ChatToolDefinition {
|
|
244
|
+
type: 'function';
|
|
245
|
+
function: {
|
|
246
|
+
name: string;
|
|
247
|
+
description: string;
|
|
248
|
+
/** JSON Schema describing the tool's arguments. Pass the same
|
|
249
|
+
* shape your `tools` table's `input_schema` carries. */
|
|
250
|
+
parameters: Record<string, unknown>;
|
|
251
|
+
};
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
/** One tool call returned by the model. Adapters normalise from
|
|
255
|
+
* provider-specific shapes (Anthropic `tool_use` blocks, Google
|
|
256
|
+
* `functionCall` parts) to this single shape so the tool-loop only
|
|
257
|
+
* iterates one grammar. */
|
|
258
|
+
export interface ChatToolCall {
|
|
259
|
+
/** Provider-assigned id, used to pair the tool result back in the
|
|
260
|
+
* next request. For Anthropic this is the `tool_use.id` (toolu_*);
|
|
261
|
+
* for OpenAI-shape providers it's `tool_calls[].id` (call_*); for
|
|
262
|
+
* Google we synthesise an id since Gemini's functionCall has no
|
|
263
|
+
* natural id field. */
|
|
264
|
+
id: string;
|
|
265
|
+
type: 'function';
|
|
266
|
+
function: {
|
|
267
|
+
name: string;
|
|
268
|
+
/** Stringified JSON args. Anthropic returns a parsed object as
|
|
269
|
+
* `input`; the adapter stringifies it here so the loop sees a
|
|
270
|
+
* single shape. Callers parse via @mantle/agent-runtime/tool-args
|
|
271
|
+
* defensively because not every model emits valid JSON. */
|
|
272
|
+
arguments: string;
|
|
273
|
+
};
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
/** A tool message in the conversation — i.e. the user-facing surface
|
|
277
|
+
* of "here's what the tool returned." Adapters translate to the
|
|
278
|
+
* provider's native tool-result shape (Anthropic emits this as a
|
|
279
|
+
* `user` message with a `tool_result` block; Google as a `function`
|
|
280
|
+
* message with `functionResponse`; OpenAI/OR/HF as a `tool` message). */
|
|
281
|
+
export type ChatToolMessage = {
|
|
282
|
+
role: 'tool';
|
|
283
|
+
toolCallId: string;
|
|
284
|
+
/** The tool result, already serialised to a string. */
|
|
285
|
+
content: string;
|
|
286
|
+
/** The tool threw / returned an error. Cache-aware adapters (Anthropic) set
|
|
287
|
+
* the provider's `is_error` flag so the model treats it as a failure rather
|
|
288
|
+
* than inferring from the serialised body. Providers without the concept
|
|
289
|
+
* ignore it. */
|
|
290
|
+
isError?: boolean;
|
|
291
|
+
};
|
|
292
|
+
|
|
293
|
+
/** Assistant turn — can carry text content, tool calls, or both. The
|
|
294
|
+
* tool-loop pushes this back into the conversation after each LLM
|
|
295
|
+
* round so the next request sees the model's prior tool_use + the
|
|
296
|
+
* matching tool_result pairs. */
|
|
297
|
+
export type ChatAssistantMessage = {
|
|
298
|
+
role: 'assistant';
|
|
299
|
+
/** Null when the model only emitted tool calls (no text turn). */
|
|
300
|
+
content: string | null;
|
|
301
|
+
toolCalls?: ChatToolCall[];
|
|
302
|
+
/** Provider reasoning blocks carried over from this assistant turn, replayed
|
|
303
|
+
* on the NEXT request so thinking continuity survives a tool round-trip.
|
|
304
|
+
* Anthropic (via OpenRouter) signs each thinking block and REJECTS a turn
|
|
305
|
+
* that contains tool_use but omits the preceding signed block — so when the
|
|
306
|
+
* responder thinks AND calls a tool, these must echo back unchanged. Opaque
|
|
307
|
+
* to the runtime; only the originating adapter reads them. */
|
|
308
|
+
reasoningDetails?: ReasoningDetail[];
|
|
309
|
+
};
|
|
310
|
+
|
|
311
|
+
/** One provider reasoning block (OpenRouter `reasoning_details` element:
|
|
312
|
+
* `reasoning.text` | `reasoning.encrypted` | `reasoning.summary`). Kept as a
|
|
313
|
+
* loose record so the runtime can carry it verbatim without depending on the
|
|
314
|
+
* provider SDK's union; the adapter that produced it is the only thing that
|
|
315
|
+
* interprets the shape. The `signature`/`data` fields are integrity-critical —
|
|
316
|
+
* echo them back exactly as received. */
|
|
317
|
+
export type ReasoningDetail = {
|
|
318
|
+
type: string;
|
|
319
|
+
index?: number;
|
|
320
|
+
text?: string | null;
|
|
321
|
+
data?: string | null;
|
|
322
|
+
summary?: string | null;
|
|
323
|
+
signature?: string | null;
|
|
324
|
+
format?: string | null;
|
|
325
|
+
id?: string | null;
|
|
326
|
+
};
|
|
327
|
+
|
|
328
|
+
/** A multi-modal user content part. Used when the responder receives
|
|
329
|
+
* an image attachment — the runtime builds `[{type:'text', text}, {type:
|
|
330
|
+
* 'image_url', imageUrl: {url, detail}}]` so vision-capable models
|
|
331
|
+
* see both the text and the image. Adapters that target a
|
|
332
|
+
* vision-capable provider translate; adapters that don't either
|
|
333
|
+
* extract just the text or warn at the call boundary. */
|
|
334
|
+
export type ChatUserContentPart =
|
|
335
|
+
| { type: 'text'; text: string }
|
|
336
|
+
| {
|
|
337
|
+
type: 'image_url';
|
|
338
|
+
imageUrl: { url: string; detail?: 'auto' | 'low' | 'high' };
|
|
339
|
+
};
|
|
340
|
+
|
|
341
|
+
/** A system-message content block. Used when the system prompt is
|
|
342
|
+
* composed of multiple cacheable segments (the responder splits its
|
|
343
|
+
* system prompt into the persona block + the conversation digest
|
|
344
|
+
* block, each with its own cache_control marker so prefix matches
|
|
345
|
+
* hit on shorter shared subsequences). */
|
|
346
|
+
export type ChatSystemContentPart = {
|
|
347
|
+
type: 'text';
|
|
348
|
+
text: string;
|
|
349
|
+
cacheControl?: { type: 'ephemeral' };
|
|
350
|
+
};
|
|
351
|
+
|
|
352
|
+
/** The tool-loop-shaped message union. Wider than the simpler
|
|
353
|
+
* string-content shape because tool-loop calls grow assistant turns
|
|
354
|
+
* with toolCalls, tool result messages, multi-modal user turns
|
|
355
|
+
* (image + text), and multi-segment cacheable system blocks (persona
|
|
356
|
+
* + digest each marked separately on Anthropic-shape providers).
|
|
357
|
+
* The 3a chat-shaped workers structurally satisfy this union with a
|
|
358
|
+
* plain `{role, content: string}` array. */
|
|
359
|
+
export type ChatToolLoopMessage =
|
|
360
|
+
| { role: 'system'; content: string | ChatSystemContentPart[] }
|
|
361
|
+
| { role: 'user'; content: string | ChatUserContentPart[] }
|
|
362
|
+
| ChatAssistantMessage
|
|
363
|
+
| ChatToolMessage;
|
|
364
|
+
|
|
365
|
+
/** Prompt-cache hints for the adapter. Anthropic and (less aggressively)
|
|
366
|
+
* OpenAI bill cached input at a fraction of the fresh rate when the
|
|
367
|
+
* caller marks cache breakpoints. The runtime tells the adapter where
|
|
368
|
+
* the breakpoints should land via this struct; adapters that talk to
|
|
369
|
+
* providers without prompt caching ignore the field entirely.
|
|
370
|
+
*
|
|
371
|
+
* Two breakpoint kinds matter today:
|
|
372
|
+
* - **systemPrompt** — mark the system block as cacheable. This is the
|
|
373
|
+
* dominant cost-saving on the responder path: the persona + skills
|
|
374
|
+
* block doesn't change turn-to-turn, so caching it pays back from
|
|
375
|
+
* the second call onward.
|
|
376
|
+
* - **lastUserMessage** — mark the most recent user message as a
|
|
377
|
+
* cache write point. Useful for the tool-loop pattern where the
|
|
378
|
+
* re-sent conversation history grows monotonically; marking the
|
|
379
|
+
* prior turn's last user msg makes the next call read it back as
|
|
380
|
+
* a cache hit. */
|
|
381
|
+
export interface ChatCacheControl {
|
|
382
|
+
systemPrompt?: boolean;
|
|
383
|
+
lastUserMessage?: boolean;
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
/** Reasoning-depth tiers, ascending. Provider-neutral: OpenRouter takes these
|
|
387
|
+
* verbatim as `reasoning.effort`, Anthropic as `output_config.effort`, Copilot
|
|
388
|
+
* as `reasoning_effort`.
|
|
389
|
+
*
|
|
390
|
+
* Restated here rather than imported from `@mantle/content` on purpose —
|
|
391
|
+
* `@mantle/voice` is the low-level adapter layer and must not pull in content's
|
|
392
|
+
* db/files/search dependency tree for five string literals. The two lists are
|
|
393
|
+
* pinned together by a compile-time assertion in `@mantle/assistant-runtime`,
|
|
394
|
+
* which already depends on both, so they cannot drift silently.
|
|
395
|
+
*
|
|
396
|
+
* No `none`: "off" is expressed by omitting the field, because models flagged
|
|
397
|
+
* `reasoning.mandatory` in GET /models reject an explicit none. */
|
|
398
|
+
export type ThinkingEffort = 'low' | 'medium' | 'high' | 'xhigh' | 'max';
|
|
399
|
+
|
|
400
|
+
export interface ChatOptions {
|
|
401
|
+
apiKey: string;
|
|
402
|
+
model: string;
|
|
403
|
+
/** Standard chat-completion messages. The adapter is free to
|
|
404
|
+
* transform these into the provider's native shape (e.g. Anthropic's
|
|
405
|
+
* separate `system` field), but we present a uniform interface.
|
|
406
|
+
*
|
|
407
|
+
* The grammar is `ChatToolLoopMessage[]` — wide enough to carry the
|
|
408
|
+
* tool-loop path (assistant turns with `toolCalls`, `tool` result
|
|
409
|
+
* messages) AND structurally compatible with the simpler shape every
|
|
410
|
+
* chat-shaped worker (extractor / summarizer / reflector) emits: a
|
|
411
|
+
* plain `{role, content: string}[]` array is a valid
|
|
412
|
+
* ChatToolLoopMessage[] for `system`/`user` roles. */
|
|
413
|
+
messages: ChatToolLoopMessage[];
|
|
414
|
+
/** Tools the model is allowed to call. Adapters translate this to
|
|
415
|
+
* the provider's native function-tool shape. When omitted or empty,
|
|
416
|
+
* the adapter MUST NOT send the field — some providers reject empty
|
|
417
|
+
* tool arrays. */
|
|
418
|
+
tools?: ChatToolDefinition[];
|
|
419
|
+
/** Steering hint for tool selection. The runtime only uses 'auto'
|
|
420
|
+
* (default) and 'none' today; we expose the OpenAI-style union so
|
|
421
|
+
* forcing a specific tool is a non-breaking addition later. */
|
|
422
|
+
toolChoice?: 'auto' | 'none';
|
|
423
|
+
temperature?: number;
|
|
424
|
+
maxTokens?: number;
|
|
425
|
+
topP?: number;
|
|
426
|
+
/** Thinking enable + budget hint. When > 0, reasoning-capable adapters ask the
|
|
427
|
+
* provider to think before answering and stream the reasoning back as
|
|
428
|
+
* `reasoning` deltas. The number is a token-budget HINT honoured only by
|
|
429
|
+
* providers that still take one (OpenRouter `reasoning.max_tokens`); on
|
|
430
|
+
* current Claude models it acts purely as on/off — the native Anthropic
|
|
431
|
+
* adapter sends `thinking:{type:'adaptive', display:'summarized'}` (the old
|
|
432
|
+
* `budget_tokens` form 400s on Opus 4.7/4.8 + Fable) and drops sampling
|
|
433
|
+
* params (also rejected in that mode). Omitted/0 ⇒ no requested thinking (a
|
|
434
|
+
* model may still emit incidental reasoning). Intended for the responder turn
|
|
435
|
+
* only — background workers leave it unset. Adapters without a reasoning mode
|
|
436
|
+
* ignore it.
|
|
437
|
+
*
|
|
438
|
+
* ⚠️ Enabling this on a multi-round tool loop additionally requires echoing
|
|
439
|
+
* prior `thinking` blocks back to the provider each iteration (Anthropic 400s
|
|
440
|
+
* otherwise) — that capture/replay is NOT yet implemented, so the runner must
|
|
441
|
+
* not set this by default until it is. */
|
|
442
|
+
thinkingBudget?: number;
|
|
443
|
+
/** Thinking EFFORT tier — the control providers actually honour now. Where
|
|
444
|
+
* {@link thinkingBudget} asks for N tokens of reasoning, this asks for a
|
|
445
|
+
* depth (`low`…`max`) and lets the provider size it.
|
|
446
|
+
*
|
|
447
|
+
* Prefer this. Budget-based thinking has been removed upstream on current
|
|
448
|
+
* models (Sonnet 5, Claude 4.7): `reasoning.max_tokens` is accepted-but-
|
|
449
|
+
* ignored, and effort maps to Anthropic's `output_config.effort`. On the
|
|
450
|
+
* OpenRouter path the budget was never even transmitted — its `reasoning`
|
|
451
|
+
* shape has no max_tokens field, so it serialised to `{}`.
|
|
452
|
+
*
|
|
453
|
+
* Adapters that have no effort concept ignore it and may still read
|
|
454
|
+
* `thinkingBudget` as an on/off signal. Both fields are set together by the
|
|
455
|
+
* runtime, so an adapter can use whichever its provider understands. */
|
|
456
|
+
thinkingEffort?: ThinkingEffort;
|
|
457
|
+
/** Retries AFTER the first attempt on transient errors (429/5xx/network/
|
|
458
|
+
* timeout), with exponential backoff + jitter. Undefined ⇒
|
|
459
|
+
* DEFAULT_MAX_RETRIES (2); 0 disables. Honored by withChatRetry, which the
|
|
460
|
+
* registry applies to the direct-provider adapters (OpenRouter relies on
|
|
461
|
+
* its SDK's own retries). */
|
|
462
|
+
maxRetries?: number;
|
|
463
|
+
/** Provider-neutral prompt-cache hints. See {@link ChatCacheControl}.
|
|
464
|
+
* Adapters that don't talk to a cache-aware provider ignore this. */
|
|
465
|
+
cacheControl?: ChatCacheControl;
|
|
466
|
+
/** Optional provider-specific overrides — adapter chooses what to honour.
|
|
467
|
+
* Used for things like xAI's `reasoning_effort` or HF's `:fastest`
|
|
468
|
+
* routing suffix. */
|
|
469
|
+
extra?: Record<string, unknown>;
|
|
470
|
+
/** Per-route base URL override for self-hosted / OpenAI-compatible chat
|
|
471
|
+
* servers. Lets a `local` chat route target a specific host (a LAN/tailnet
|
|
472
|
+
* box). The `local-chat` adapter honours it; fixed-endpoint cloud adapters
|
|
473
|
+
* ignore it. Mirrors `EmbedRequest.baseUrl`. */
|
|
474
|
+
baseUrl?: string;
|
|
475
|
+
/** When true, the request is dispatched through the Tailscale forward-proxy
|
|
476
|
+
* ({@link tailnetFetch}) so a `baseUrl` pointing at a tailnet host reaches a
|
|
477
|
+
* box behind NAT. Honoured by the `local-chat` adapter; inert when no proxy
|
|
478
|
+
* is configured. */
|
|
479
|
+
viaTailnet?: boolean;
|
|
480
|
+
/** Cancellation signal for the request — wired by the tool loop so a user can
|
|
481
|
+
* STOP an in-flight streamed turn. A streaming adapter passes it to the
|
|
482
|
+
* underlying fetch and, on abort, stops reading and returns the partial reply
|
|
483
|
+
* assembled so far (rather than throwing). Adapters that don't honour it just
|
|
484
|
+
* run to completion. */
|
|
485
|
+
signal?: AbortSignal;
|
|
486
|
+
}
|
|
487
|
+
|
|
488
|
+
/** A user-visible delta surfaced by `chatStream` as the model produces output.
|
|
489
|
+
* Deliberately minimal: only the visible reply text and the model's reasoning
|
|
490
|
+
* stream. Tool-call argument fragments are accumulated INTERNALLY by the adapter
|
|
491
|
+
* (so the resolved `ChatResult.toolCalls` is fully assembled, identical to
|
|
492
|
+
* `chat()`) — they're machinery, not something the user reads, so they're not
|
|
493
|
+
* surfaced here. */
|
|
494
|
+
export type ChatStreamDelta = { type: 'text'; text: string } | { type: 'reasoning'; text: string };
|
|
495
|
+
|
|
496
|
+
/** Sink the caller passes to `chatStream` to receive deltas as they arrive. The
|
|
497
|
+
* adapter calls it synchronously per chunk and never awaits it — the caller
|
|
498
|
+
* fans it onto the ephemeral live bus and must not let it throw back into the
|
|
499
|
+
* stream loop. */
|
|
500
|
+
export type ChatStreamSink = (delta: ChatStreamDelta) => void;
|
|
501
|
+
|
|
502
|
+
export interface ChatDispatcher extends AdapterMeta {
|
|
503
|
+
/** One-shot chat completion. */
|
|
504
|
+
chat(opts: ChatOptions): Promise<ChatResult>;
|
|
505
|
+
/** Streaming variant of `chat`: emits text/reasoning deltas to `onDelta` as
|
|
506
|
+
* they arrive AND resolves to the same fully-assembled `ChatResult` `chat()`
|
|
507
|
+
* would return (text, toolCalls, usage) — so it's a drop-in. The ChatResult is
|
|
508
|
+
* the durable answer; the deltas are ephemeral decoration. Optional + additive:
|
|
509
|
+
* a provider that can't stream omits it and callers fall back to `chat()`. */
|
|
510
|
+
chatStream?(opts: ChatOptions, onDelta: ChatStreamSink): Promise<ChatResult>;
|
|
511
|
+
/** Live-discover available chat models. Adapter does the cross-
|
|
512
|
+
* reference between provider's /v1/models response and its own
|
|
513
|
+
* static catalog. */
|
|
514
|
+
discoverModels?(apiKey: string): Promise<DiscoveryResult<ChatModelInfo>>;
|
|
515
|
+
/** Adapters expose their static catalog for the UI to render before
|
|
516
|
+
* discovery completes (or when discovery isn't supported). */
|
|
517
|
+
staticCatalog?(): readonly ChatModelInfo[];
|
|
518
|
+
}
|
|
519
|
+
|
|
520
|
+
export interface SttDispatcher extends AdapterMeta {
|
|
521
|
+
/** The request options this adapter forwards, completing the convention
|
|
522
|
+
* shared with {@link TtsDispatcher.supports} and
|
|
523
|
+
* {@link ImageGenDispatcher.supports}.
|
|
524
|
+
*
|
|
525
|
+
* Thin by nature: transcription's normalized surface is a language hint and
|
|
526
|
+
* nothing else, so every wired adapter currently declares `['language']`.
|
|
527
|
+
* It is still worth declaring rather than assuming — a 2026-08 docs sweep
|
|
528
|
+
* found xAI's `format` flag silently depending on `language`, and the value
|
|
529
|
+
* of these lists is that they are checked claims rather than beliefs. What
|
|
530
|
+
* each provider REPORTS BACK varies far more than what it accepts; see
|
|
531
|
+
* `TranscribeResult`, where several adapters can only return nulls. */
|
|
532
|
+
supports: readonly SttParam[];
|
|
533
|
+
|
|
534
|
+
/** Transcribe an audio buffer. Buffer must be in one of the formats
|
|
535
|
+
* the provider accepts; the adapter handles the multipart encoding
|
|
536
|
+
* and any provider-specific MIME wrangling. */
|
|
537
|
+
transcribe(audio: Buffer, opts: TranscribeOptions): Promise<TranscribeResult>;
|
|
538
|
+
|
|
539
|
+
/** Optional. Live-discover available transcription models. */
|
|
540
|
+
discoverModels?(apiKey: string): Promise<DiscoveryResult<SttModelInfo>>;
|
|
541
|
+
}
|
|
542
|
+
|
|
543
|
+
// ─── Vision ─────────────────────────────────────────────────────────
|
|
544
|
+
|
|
545
|
+
/** A vision-capable model entry. Same generic shape as ChatModelInfo;
|
|
546
|
+
* the form's vision-worker dropdown renders these so operators pick
|
|
547
|
+
* from a list rather than typing a model id by hand. */
|
|
548
|
+
export interface VisionModelInfo {
|
|
549
|
+
id: string;
|
|
550
|
+
label: string;
|
|
551
|
+
description: string;
|
|
552
|
+
/** Context window in tokens. */
|
|
553
|
+
contextTokens?: number;
|
|
554
|
+
/** USD per 1M input tokens. Approximate; for UI hints only. */
|
|
555
|
+
inputPricePer1M?: number;
|
|
556
|
+
/** USD per 1M output tokens. */
|
|
557
|
+
outputPricePer1M?: number;
|
|
558
|
+
/** Roughly which tier the model fits. 'fast' = use for high volume;
|
|
559
|
+
* 'balanced' = default; 'quality' = harder images, lower throughput. */
|
|
560
|
+
tier?: 'fast' | 'balanced' | 'quality';
|
|
561
|
+
}
|
|
562
|
+
|
|
563
|
+
export interface VisionExtractOptions {
|
|
564
|
+
apiKey: string;
|
|
565
|
+
/** MIME of the image bytes (image/jpeg, image/png, image/webp, image/gif).
|
|
566
|
+
* Each provider accepts a slightly different list — adapters error
|
|
567
|
+
* on unsupported MIMEs rather than silently sending bytes the API
|
|
568
|
+
* will reject downstream. */
|
|
569
|
+
mimeType: string;
|
|
570
|
+
/** User-side prompt — "transcribe this page verbatim", "summarize
|
|
571
|
+
* the diagram", "extract action items as a list". Adapters pass it
|
|
572
|
+
* through alongside the image, framed as the user turn. */
|
|
573
|
+
prompt: string;
|
|
574
|
+
/** Optional system-level steering (used the same way as a chat
|
|
575
|
+
* worker's system prompt). Operators set this on the worker row;
|
|
576
|
+
* callers forward it here. */
|
|
577
|
+
systemPrompt?: string;
|
|
578
|
+
/** Model id. Falls back to the adapter's documented default if
|
|
579
|
+
* omitted. */
|
|
580
|
+
model?: string;
|
|
581
|
+
/** Max output tokens. Vision-LLMs can run long without one. */
|
|
582
|
+
maxTokens?: number;
|
|
583
|
+
}
|
|
584
|
+
|
|
585
|
+
export interface VisionExtractResult {
|
|
586
|
+
/** Extracted text. Trimmed; empty string on no-output, not null. */
|
|
587
|
+
text: string;
|
|
588
|
+
/** Model that did the work. May be a more specific id than the
|
|
589
|
+
* caller passed in (e.g. dated Claude variants). */
|
|
590
|
+
model: string;
|
|
591
|
+
/** Token usage when the provider returns it. */
|
|
592
|
+
tokensIn?: number;
|
|
593
|
+
tokensOut?: number;
|
|
594
|
+
}
|
|
595
|
+
|
|
596
|
+
export interface VisionDispatcher extends AdapterMeta {
|
|
597
|
+
/** Extract text/structure from an image. The adapter handles the
|
|
598
|
+
* per-provider message shape, MIME validation, and image encoding
|
|
599
|
+
* (base64 vs URL). Caller hands raw bytes + a prompt; result is
|
|
600
|
+
* trimmed text. */
|
|
601
|
+
extract(image: Buffer, opts: VisionExtractOptions): Promise<VisionExtractResult>;
|
|
602
|
+
|
|
603
|
+
/** Optional. Extract text from a document (PDF) sent NATIVELY to the model —
|
|
604
|
+
* no rasterization. Providers whose API accepts a document content block
|
|
605
|
+
* (Anthropic, Google) implement this; the runtime PREFERS it over
|
|
606
|
+
* rasterize→per-page image OCR for PDFs (whole-document context, real layout
|
|
607
|
+
* and tables, one call, no PNG-conversion fidelity loss). Adapters that
|
|
608
|
+
* can't take a document natively (OpenAI, xAI) omit it, and the caller falls
|
|
609
|
+
* back to rasterizing the pages through `extract`. `opts.mimeType` is the
|
|
610
|
+
* document MIME (e.g. 'application/pdf'). */
|
|
611
|
+
extractDocument?(document: Buffer, opts: VisionExtractOptions): Promise<VisionExtractResult>;
|
|
612
|
+
|
|
613
|
+
/** Live-discover which vision-capable models the api key can use.
|
|
614
|
+
* Implementation parity with ChatDispatcher.discoverModels — when
|
|
615
|
+
* absent, the form falls back to the adapter's static catalog. */
|
|
616
|
+
discoverModels?(apiKey: string): Promise<import('../catalog').DiscoveryResult<VisionModelInfo>>;
|
|
617
|
+
|
|
618
|
+
/** Static catalog the UI renders before live discovery returns or
|
|
619
|
+
* when the adapter doesn't support discovery. */
|
|
620
|
+
staticCatalog?(): readonly VisionModelInfo[];
|
|
621
|
+
}
|
|
622
|
+
|
|
623
|
+
// ─── Image generation ───────────────────────────────────────────────
|
|
624
|
+
|
|
625
|
+
/** A generatable image model entry. Same shape conventions as
|
|
626
|
+
* ChatModelInfo/VisionModelInfo: id + label + description so the UI
|
|
627
|
+
* has rich dropdown options without each form re-parsing provider
|
|
628
|
+
* docs. */
|
|
629
|
+
export interface ImageGenModelInfo {
|
|
630
|
+
id: string;
|
|
631
|
+
label: string;
|
|
632
|
+
description: string;
|
|
633
|
+
/** Native resolutions the model accepts. Adapters reject sizes
|
|
634
|
+
* outside this list with a clear error. Free-form when undefined
|
|
635
|
+
* (HF models, where the underlying model decides). */
|
|
636
|
+
supportedSizes?: readonly string[];
|
|
637
|
+
/** Steerable styles, when the model supports them (DALL-E 3:
|
|
638
|
+
* 'vivid' | 'natural'). Undefined = no style steering. */
|
|
639
|
+
supportedStyles?: readonly string[];
|
|
640
|
+
/** Quality tiers the model accepts. Genuinely provider-specific:
|
|
641
|
+
* DALL-E 3 takes 'standard' | 'hd', gpt-image-1 takes
|
|
642
|
+
* 'low' | 'medium' | 'high' | 'auto', OpenRouter normalizes to the
|
|
643
|
+
* latter. Undefined = no quality control. */
|
|
644
|
+
supportedQualities?: readonly string[];
|
|
645
|
+
/** Aspect ratios the model accepts, for providers that steer by ratio
|
|
646
|
+
* rather than by pixel size. Undefined = not ratio-steerable. */
|
|
647
|
+
supportedAspectRatios?: readonly string[];
|
|
648
|
+
/** USD per image at default size. UI hint only. */
|
|
649
|
+
pricePerImage?: number;
|
|
650
|
+
/** Latency tier — useful when picking between same-provider models. */
|
|
651
|
+
tier?: 'fast' | 'balanced' | 'quality';
|
|
652
|
+
}
|
|
653
|
+
|
|
654
|
+
/** A request option an image-gen adapter may or may not forward.
|
|
655
|
+
*
|
|
656
|
+
* Declared per adapter so an option that will not survive the trip is
|
|
657
|
+
* REPORTED rather than silently discarded. Silent drop is the failure this
|
|
658
|
+
* interface exists to kill: an operator set size/style/quality on an
|
|
659
|
+
* OpenRouter worker, the UI showed them saved, and the adapter sent
|
|
660
|
+
* neither — with nothing in the trace to say so. See
|
|
661
|
+
* {@link ImageGenDispatcher.supports}. */
|
|
662
|
+
export type ImageGenParam =
|
|
663
|
+
'size' | 'aspectRatio' | 'style' | 'quality' | 'negativePrompt' | 'seed' | 'inputImages';
|
|
664
|
+
|
|
665
|
+
/** One option the adapter chose not to send, and why.
|
|
666
|
+
*
|
|
667
|
+
* `supports` is declared per PROVIDER, but the real gate is often per MODEL:
|
|
668
|
+
* the OpenAI images endpoint takes `style` on dall-e-3 and not on
|
|
669
|
+
* gpt-image-1, `quality` on those two and not on dall-e-2. A provider-level
|
|
670
|
+
* list cannot express that, so the adapter — the only layer that knows which
|
|
671
|
+
* model it is actually talking to — reports the difference here rather than
|
|
672
|
+
* dropping it. (Same role as `warnings` on the AI SDK's image result.) */
|
|
673
|
+
export type ImageGenWarning = {
|
|
674
|
+
param: ImageGenParam;
|
|
675
|
+
/** Model-specific reason, phrased for a reader deciding what to do next. */
|
|
676
|
+
reason: string;
|
|
677
|
+
};
|
|
678
|
+
|
|
679
|
+
/** A picture handed IN to a generation: the reference for an edit or a
|
|
680
|
+
* variation. Bytes rather than a URL, because the source is a file node in
|
|
681
|
+
* the owner's own store and providers take base64 or multipart, never a link
|
|
682
|
+
* only we can reach. */
|
|
683
|
+
export interface ImageGenInput {
|
|
684
|
+
bytes: Buffer;
|
|
685
|
+
mimeType: string;
|
|
686
|
+
/** Original filename, used where the provider takes multipart. */
|
|
687
|
+
filename?: string;
|
|
688
|
+
}
|
|
689
|
+
|
|
690
|
+
export interface GenerateImageOptions {
|
|
691
|
+
apiKey: string;
|
|
692
|
+
/** Free-form prompt. No length cap at the interface layer; per-
|
|
693
|
+
* provider limits get enforced inside the adapter. */
|
|
694
|
+
prompt: string;
|
|
695
|
+
/** Reference images. Present ⇒ this is an EDIT, not a fresh generation:
|
|
696
|
+
* "make the sky orange in this one". An adapter that does not declare
|
|
697
|
+
* `inputImages` in `supports` must never be handed these — generating from
|
|
698
|
+
* scratch when an edit was asked for produces a confidently wrong picture
|
|
699
|
+
* and bills for it, so the caller refuses BEFORE spending. */
|
|
700
|
+
inputImages?: readonly ImageGenInput[];
|
|
701
|
+
/** Model id. Defaults to the adapter's documented default. */
|
|
702
|
+
model?: string;
|
|
703
|
+
/** Native resolution, e.g. '1024x1024' or '1792x1024'. Adapters
|
|
704
|
+
* validate against ImageGenModelInfo.supportedSizes. OpenRouter also
|
|
705
|
+
* accepts a tier ('1K' / '2K' / '4K'). */
|
|
706
|
+
size?: string;
|
|
707
|
+
/** Aspect ratio, e.g. '16:9'. The natural control for providers that
|
|
708
|
+
* steer by ratio (Imagen) or normalize across them (OpenRouter);
|
|
709
|
+
* ignored by the fixed-size APIs. */
|
|
710
|
+
aspectRatio?: string;
|
|
711
|
+
/** Style hint (provider-specific: DALL-E 3 = 'vivid'/'natural';
|
|
712
|
+
* others may ignore). */
|
|
713
|
+
style?: string;
|
|
714
|
+
/** Quality tier — DALL-E 3 uses 'standard'/'hd'; HF models may use
|
|
715
|
+
* 'fast'/'balanced'/'quality'. Adapters ignore unknown values. */
|
|
716
|
+
quality?: string;
|
|
717
|
+
/** Negative prompt — what the image should NOT contain. Honoured
|
|
718
|
+
* by HF + Imagen; OpenAI doesn't accept it (silently ignored). */
|
|
719
|
+
negativePrompt?: string;
|
|
720
|
+
/** Random seed for reproducibility, when the provider exposes one. */
|
|
721
|
+
seed?: number;
|
|
722
|
+
}
|
|
723
|
+
|
|
724
|
+
export interface GenerateImageResult {
|
|
725
|
+
/** Generated image bytes. Adapters always materialize the bytes
|
|
726
|
+
* even when the provider returns a URL — callers shouldn't have
|
|
727
|
+
* to deal with URL-vs-bytes asymmetry across providers. */
|
|
728
|
+
bytes: Buffer;
|
|
729
|
+
/** MIME of the returned image — typically 'image/png' but Imagen
|
|
730
|
+
* may return jpeg, HF may return jpeg/png depending on model. */
|
|
731
|
+
mimeType: string;
|
|
732
|
+
/** Model id that did the work. Echoed back for traces. */
|
|
733
|
+
model: string;
|
|
734
|
+
/** Options the adapter deliberately did NOT send, because this MODEL has
|
|
735
|
+
* no such control. Empty/absent means everything the caller passed went on
|
|
736
|
+
* the wire. Callers must surface these rather than let a requested option
|
|
737
|
+
* look like it applied. */
|
|
738
|
+
warnings?: readonly ImageGenWarning[];
|
|
739
|
+
/** Provider's revised prompt, when it returns one (DALL-E 3 rewrites
|
|
740
|
+
* the user prompt for safety+quality and surfaces the revision).
|
|
741
|
+
* Caller can pass this back as `revised_prompt` metadata on the
|
|
742
|
+
* saved file node so the operator sees what the model actually
|
|
743
|
+
* rendered against. */
|
|
744
|
+
revisedPrompt?: string;
|
|
745
|
+
}
|
|
746
|
+
|
|
747
|
+
export interface ImageGenDispatcher extends AdapterMeta {
|
|
748
|
+
/** The subset of {@link GenerateImageOptions} this adapter actually puts
|
|
749
|
+
* on the wire. REQUIRED, and it must be honest: the caller reports every
|
|
750
|
+
* requested option outside this set back to the model and into the trace
|
|
751
|
+
* rather than pretending it applied. An adapter that starts forwarding a
|
|
752
|
+
* new option must add it here in the same commit. */
|
|
753
|
+
supports: readonly ImageGenParam[];
|
|
754
|
+
|
|
755
|
+
/** Generate an image from a prompt. Throws on auth/network/quota
|
|
756
|
+
* errors with the provider's verbatim message slice — same
|
|
757
|
+
* convention as the other dispatchers. */
|
|
758
|
+
generate(opts: GenerateImageOptions): Promise<GenerateImageResult>;
|
|
759
|
+
|
|
760
|
+
/** Static catalog the UI renders for the model dropdown. Most
|
|
761
|
+
* image-gen providers don't expose a programmatic models list
|
|
762
|
+
* (or they do but it returns chat models too) — so we ship a
|
|
763
|
+
* curated static list and don't implement discoverModels. */
|
|
764
|
+
staticCatalog(): readonly ImageGenModelInfo[];
|
|
765
|
+
}
|
|
766
|
+
|
|
767
|
+
// ─── Embedding ──────────────────────────────────────────────────────
|
|
768
|
+
|
|
769
|
+
/** A text→vector model entry. Embedding pricing is input-only
|
|
770
|
+
* (the response is the vector, not a token stream — no output cost
|
|
771
|
+
* to bill). Native `dimensions` is exposed so the form can verify
|
|
772
|
+
* compatibility with the brain's `vector(768)` column before save. */
|
|
773
|
+
export interface EmbeddingModelInfo {
|
|
774
|
+
id: string;
|
|
775
|
+
label: string;
|
|
776
|
+
description: string;
|
|
777
|
+
/** Maximum input tokens accepted in a single call. */
|
|
778
|
+
contextTokens?: number;
|
|
779
|
+
/** USD per 1M input tokens. */
|
|
780
|
+
inputPricePer1M?: number;
|
|
781
|
+
/** Output vector dimension as the provider documents it. The form
|
|
782
|
+
* uses this to drive the dim-mismatch save block before the
|
|
783
|
+
* operator gets a 'won't insert' surprise at runtime. */
|
|
784
|
+
dimensions?: number;
|
|
785
|
+
/** Accepts non-text inputs (image / audio / file). Only OpenRouter's
|
|
786
|
+
* multimodal route and Google's gemini-embedding-2-preview today. */
|
|
787
|
+
multimodal?: boolean;
|
|
788
|
+
}
|
|
789
|
+
|
|
790
|
+
/** Inputs the embedding dispatchers accept. Same shape as the
|
|
791
|
+
* pre-adapter @mantle/embeddings package — keeps the public API
|
|
792
|
+
* stable while the dispatch path swaps underneath. */
|
|
793
|
+
export type EmbedInput =
|
|
794
|
+
| string
|
|
795
|
+
| { type: 'text'; text: string }
|
|
796
|
+
| { type: 'image'; url: string }
|
|
797
|
+
| { type: 'audio'; url: string }
|
|
798
|
+
| { type: 'file'; url: string; mimeType?: string };
|
|
799
|
+
|
|
800
|
+
export interface EmbedRequest {
|
|
801
|
+
apiKey: string;
|
|
802
|
+
model: string;
|
|
803
|
+
/** Single-element array for single-text calls; batch is the common path
|
|
804
|
+
* (extractor + recall batch many at once for cache + API efficiency). */
|
|
805
|
+
input: EmbedInput[];
|
|
806
|
+
/** Truncate to this dim where supported. OpenAI's text-embedding-3-*
|
|
807
|
+
* family honours it (truncation by MRL); Google's gemini-embedding-2
|
|
808
|
+
* honours it as `output_dimensionality`. Everything else ignores. */
|
|
809
|
+
dimensions?: number;
|
|
810
|
+
/** Per-call base URL override for self-hosted / OpenAI-compatible routes.
|
|
811
|
+
* Lets the embedding config point primary and backup at different hosts
|
|
812
|
+
* serving the SAME model (failover). The `local` adapter honours it;
|
|
813
|
+
* fixed-endpoint cloud adapters ignore it. */
|
|
814
|
+
baseUrl?: string;
|
|
815
|
+
/** When true, the request is dispatched through the Tailscale forward-proxy
|
|
816
|
+
* ({@link tailnetFetch}) so a `baseUrl` pointing at a tailnet host reaches a
|
|
817
|
+
* box behind NAT. Honoured by the `local-embedding` adapter; inert when no
|
|
818
|
+
* proxy is configured. */
|
|
819
|
+
viaTailnet?: boolean;
|
|
820
|
+
/** Texts per HTTP request for the `local-embedding` adapter. Lets the embedding
|
|
821
|
+
* config tune throughput per-owner (small on a CPU box so a request fits the
|
|
822
|
+
* timeout, large on a GPU). Null/undefined → the adapter's own
|
|
823
|
+
* `MANTLE_LOCAL_EMBED_BATCH` env → 16. Ignored by cloud adapters. */
|
|
824
|
+
localEmbedBatchSize?: number;
|
|
825
|
+
/** Per-request timeout (ms) for the `local-embedding` adapter. Null/undefined
|
|
826
|
+
* → `MANTLE_LOCAL_EMBED_TIMEOUT_MS` env → 120000. Ignored by cloud adapters. */
|
|
827
|
+
localEmbedTimeoutMs?: number;
|
|
828
|
+
}
|
|
829
|
+
|
|
830
|
+
export interface EmbedResult {
|
|
831
|
+
vectors: number[][];
|
|
832
|
+
/** Server-reported model id (so callers can verify their slug landed
|
|
833
|
+
* where they expected — direct providers sometimes alias). */
|
|
834
|
+
model: string;
|
|
835
|
+
/** Total input tokens billed. Undefined when the provider doesn't
|
|
836
|
+
* report (Cohere v2 omits, some HF routes too). */
|
|
837
|
+
tokensIn?: number;
|
|
838
|
+
}
|
|
839
|
+
|
|
840
|
+
export interface EmbeddingDispatcher extends AdapterMeta {
|
|
841
|
+
/** Embed a batch. Adapters that don't support multimodal input throw
|
|
842
|
+
* a clear error on non-text items rather than silently truncating —
|
|
843
|
+
* surfaces "you picked a text-only model but sent an image" at the
|
|
844
|
+
* point of failure rather than as a confused empty result. */
|
|
845
|
+
embed(req: EmbedRequest): Promise<EmbedResult>;
|
|
846
|
+
|
|
847
|
+
/** Optional: whether this adapter will accept the given input. Text-only
|
|
848
|
+
* adapters return false for non-text items so the caller can route
|
|
849
|
+
* multimodal inputs to OpenRouter (the only multimodal-capable path)
|
|
850
|
+
* instead of failing at request time. Default = true. */
|
|
851
|
+
acceptsInput?(input: EmbedInput): boolean;
|
|
852
|
+
|
|
853
|
+
/** Live-discover available embedding models. Each adapter does the
|
|
854
|
+
* cross-reference between its provider's list endpoint and its own
|
|
855
|
+
* static catalog. OpenRouter publishes `/v1/embeddings/models`
|
|
856
|
+
* separately from `/v1/models`; OpenAI returns embeddings inline
|
|
857
|
+
* in `/v1/models` (filtered by id pattern); Google requires a
|
|
858
|
+
* capability filter (`supportedGenerationMethods` includes
|
|
859
|
+
* `embedContent`). */
|
|
860
|
+
discoverModels?(apiKey: string): Promise<DiscoveryResult<EmbeddingModelInfo>>;
|
|
861
|
+
|
|
862
|
+
/** Curated fallback list. Used by the workers form when no API key
|
|
863
|
+
* is configured yet (so the dropdown isn't empty in create mode)
|
|
864
|
+
* AND as a last-resort if discovery errors. */
|
|
865
|
+
staticCatalog?(): readonly EmbeddingModelInfo[];
|
|
866
|
+
}
|