@xberg-io/liter-llm 1.11.3 → 1.17.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -3
- package/index.d.ts +385 -128
- package/index.js +624 -69
- package/liter-llm-node.darwin-arm64.node +0 -0
- package/liter-llm-node.darwin-x64.node +0 -0
- package/liter-llm-node.linux-arm64-gnu.node +0 -0
- package/liter-llm-node.linux-x64-gnu.node +0 -0
- package/liter-llm-node.win32-arm64-msvc.node +0 -0
- package/liter-llm-node.win32-x64-msvc.node +0 -0
- package/package.json +13 -7
package/index.d.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
// This file is auto-generated by alef — DO NOT EDIT.
|
|
2
|
-
// alef:hash:
|
|
2
|
+
// alef:hash:6bae00d6b31bd5ec9e80699322018c3a78359cee1f06b9e2b7ffbdc6f1d4bc21
|
|
3
3
|
// To regenerate: alef generate
|
|
4
|
-
// To verify freshness: alef verify
|
|
4
|
+
// To verify freshness: alef verify
|
|
5
5
|
/* eslint-disable */
|
|
6
6
|
|
|
7
7
|
export type JsonValue = string | number | boolean | null | JsonValue[] | { [key: string]: JsonValue };
|
|
@@ -10,8 +10,8 @@ export type JsonValue = string | number | boolean | null | JsonValue[] | { [key:
|
|
|
10
10
|
* Return all provider configs from the registry.
|
|
11
11
|
*
|
|
12
12
|
* Useful for tooling, documentation generation, or runtime enumeration.
|
|
13
|
-
* Returns the public
|
|
14
|
-
* To query capability flags for a specific provider use
|
|
13
|
+
* Returns the public `ProviderConfig` slice (without capability flags).
|
|
14
|
+
* To query capability flags for a specific provider use `capabilities`.
|
|
15
15
|
*/
|
|
16
16
|
export declare function allProviders(): Array<ProviderConfig>;
|
|
17
17
|
|
|
@@ -31,8 +31,8 @@ export declare function capabilities(providerName: string): ProviderCapabilities
|
|
|
31
31
|
* Assert that `current_len + incoming` does not exceed `limit`.
|
|
32
32
|
*
|
|
33
33
|
* Call this before appending `incoming` bytes to any buffer that must
|
|
34
|
-
* stay below `limit`. Returns `Err(LiterLlmError
|
|
35
|
-
* and emits a `tracing
|
|
34
|
+
* stay below `limit`. Returns `Err(LiterLlmError.Streaming)` on overflow
|
|
35
|
+
* and emits a `tracing.warn!` with context.
|
|
36
36
|
*/
|
|
37
37
|
export declare function checkBound(context: string, currentLen: number, incoming: number, limit: number): void;
|
|
38
38
|
|
|
@@ -40,6 +40,9 @@ export declare function checkBound(context: string, currentLen: number, incoming
|
|
|
40
40
|
* Remove all guardrails from the global registry.
|
|
41
41
|
*
|
|
42
42
|
* Primarily useful in tests to reset state between test cases.
|
|
43
|
+
*
|
|
44
|
+
* If the lock was poisoned by a panicking guardrail on a previous access,
|
|
45
|
+
* the poisoned state is recovered rather than propagating the panic.
|
|
43
46
|
*/
|
|
44
47
|
export declare function clear(): void;
|
|
45
48
|
|
|
@@ -48,7 +51,7 @@ export declare function clear(): void;
|
|
|
48
51
|
* `completion_cost_with_cache`, and `model_info` to the
|
|
49
52
|
* embedded catalog.
|
|
50
53
|
*
|
|
51
|
-
* Primarily a test seam (see
|
|
54
|
+
* Primarily a test seam (see `install_catalog_overlay_from_str`); also
|
|
52
55
|
* usable by long-running processes that want to abandon a runtime refresh.
|
|
53
56
|
*/
|
|
54
57
|
export declare function clearCatalogOverlay(): void;
|
|
@@ -78,9 +81,9 @@ export declare function completionCost(model: string, promptTokens: number, comp
|
|
|
78
81
|
* input rate.
|
|
79
82
|
*
|
|
80
83
|
* Returns `None` if the model is not present in the embedded pricing
|
|
81
|
-
* registry, mirroring
|
|
84
|
+
* registry, mirroring `completion_cost`.
|
|
82
85
|
*
|
|
83
|
-
* When the model has
|
|
86
|
+
* When the model has `ModelPricing.tiers`, the tier whose
|
|
84
87
|
* `min_context_tokens` is the highest value `<= prompt_tokens` supplies the
|
|
85
88
|
* input/output/cache rates for the whole call; models without tiers (or
|
|
86
89
|
* when `prompt_tokens` is below every tier threshold) use the base rates
|
|
@@ -99,13 +102,13 @@ export declare function completionCostWithCache(model: string, promptTokens: num
|
|
|
99
102
|
export declare function complexProviderNames(): Array<string>;
|
|
100
103
|
|
|
101
104
|
/**
|
|
102
|
-
* Count tokens for a full
|
|
105
|
+
* Count tokens for a full `ChatCompletionRequest`.
|
|
103
106
|
*
|
|
104
107
|
* Sums tokens across all message text contents plus a per-message overhead
|
|
105
108
|
* of ~4 tokens (for role, separators, and formatting metadata). Tool
|
|
106
109
|
* definitions and multimodal content parts (images, audio, documents) are
|
|
107
110
|
* not counted — only textual content contributes to the token total.
|
|
108
|
-
* @throws Returns
|
|
111
|
+
* @throws Returns `LiterLlmError.BadRequest` if the tokenizer cannot be loaded or
|
|
109
112
|
* if tokenization fails for any message.
|
|
110
113
|
*/
|
|
111
114
|
export declare function countRequestTokens(model: string, req?: ChatCompletionRequest | undefined | null): number;
|
|
@@ -116,7 +119,7 @@ export declare function countRequestTokens(model: string, req?: ChatCompletionRe
|
|
|
116
119
|
* The tokenizer is resolved from the model name prefix (e.g. `"gpt-4o"` maps
|
|
117
120
|
* to the `Xenova/gpt-4o` HuggingFace tokenizer). Tokenizers are cached after
|
|
118
121
|
* first load.
|
|
119
|
-
* @throws Returns
|
|
122
|
+
* @throws Returns `LiterLlmError.BadRequest` if the tokenizer cannot be loaded
|
|
120
123
|
* (e.g. network failure on first use) or if tokenization itself fails.
|
|
121
124
|
*/
|
|
122
125
|
export declare function countTokens(model: string, text: string): number;
|
|
@@ -126,8 +129,8 @@ export declare function countTokens(model: string, text: string): number;
|
|
|
126
129
|
*
|
|
127
130
|
* This is the primary binding entry-point. All parameters except `api_key`
|
|
128
131
|
* are optional — omitting them uses the same defaults as
|
|
129
|
-
*
|
|
130
|
-
* @throws Returns
|
|
132
|
+
* `ClientConfigBuilder`.
|
|
133
|
+
* @throws Returns `LiterLlmError` if the underlying HTTP client cannot be
|
|
131
134
|
* constructed, or if the resolved provider configuration is invalid.
|
|
132
135
|
*/
|
|
133
136
|
export declare function createClient(apiKey: string, baseUrl?: string | undefined | null, timeoutSecs?: number | undefined | null, maxRetries?: number | undefined | null, modelHint?: string | undefined | null): DefaultClient;
|
|
@@ -136,13 +139,13 @@ export declare function createClient(apiKey: string, baseUrl?: string | undefine
|
|
|
136
139
|
* Create a new LLM client from a JSON string.
|
|
137
140
|
*
|
|
138
141
|
* The JSON object accepts the same fields as `liter-llm.toml` (snake_case).
|
|
139
|
-
* @throws Returns
|
|
142
|
+
* @throws Returns `LiterLlmError.BadRequest` if `json` is not valid JSON or
|
|
140
143
|
* contains unknown fields.
|
|
141
144
|
*/
|
|
142
145
|
export declare function createClientFromJson(json: string): DefaultClient;
|
|
143
146
|
|
|
144
147
|
/**
|
|
145
|
-
* Decode a base64 data URL into
|
|
148
|
+
* Decode a base64 data URL into `DecodedDataUrl`.
|
|
146
149
|
*
|
|
147
150
|
* Returns `None` for:
|
|
148
151
|
* - Non-data URLs (strings that do not start with `"data:"`).
|
|
@@ -157,7 +160,7 @@ export declare function decodeDataUrl(url: string): DecodedDataUrl | null;
|
|
|
157
160
|
/**
|
|
158
161
|
* Encode bytes as a base64 data URL: `data:<mime>;base64,<b64>`.
|
|
159
162
|
*
|
|
160
|
-
* `mime` defaults to
|
|
163
|
+
* `mime` defaults to `IMAGE_PNG` when `None`.
|
|
161
164
|
*/
|
|
162
165
|
export declare function encodeDataUrl(bytes: Uint8Array, mime?: string | undefined | null): string;
|
|
163
166
|
|
|
@@ -169,7 +172,7 @@ export declare function encodeDataUrl(bytes: Uint8Array, mime?: string | undefin
|
|
|
169
172
|
* another rustls crypto provider has already been installed is safe: the
|
|
170
173
|
* `Err` from `install_default()` is silently ignored.
|
|
171
174
|
*
|
|
172
|
-
* Called automatically by every internal `reqwest
|
|
175
|
+
* Called automatically by every internal `reqwest.Client` constructor
|
|
173
176
|
* (auth providers, default HTTP client). Bindings and downstream consumers
|
|
174
177
|
* reach those constructors transitively, so no manual init is required.
|
|
175
178
|
*
|
|
@@ -186,9 +189,9 @@ export declare function ensureCryptoProvider(): void;
|
|
|
186
189
|
* the network and disk cache entirely.
|
|
187
190
|
*
|
|
188
191
|
* Parses and flattens `catalog_json` with the same
|
|
189
|
-
*
|
|
192
|
+
* `registry_from_catalog_str` logic used for the embedded catalog and the
|
|
190
193
|
* network refresh path, then atomically swaps it in as the active overlay.
|
|
191
|
-
* A parse failure returns
|
|
194
|
+
* A parse failure returns `CatalogRefreshError.Parse` and leaves any
|
|
192
195
|
* existing overlay untouched.
|
|
193
196
|
*
|
|
194
197
|
* This is primarily a testable seam: it lets tests exercise overlay
|
|
@@ -202,16 +205,13 @@ export declare function installCatalogOverlayFromStr(catalogJson: string): void;
|
|
|
202
205
|
* Content shape for assistant messages.
|
|
203
206
|
*
|
|
204
207
|
* `#[serde(untagged)]` means providers returning a plain scalar string for the
|
|
205
|
-
* `content` field still deserialise correctly into `AssistantContent
|
|
208
|
+
* `content` field still deserialise correctly into `AssistantContent.Text(_)`.
|
|
206
209
|
* Providers returning an array of typed parts (e.g. after an image-generation
|
|
207
|
-
* or audio-synthesis request) deserialise into `AssistantContent
|
|
210
|
+
* or audio-synthesis request) deserialise into `AssistantContent.Parts(_)`.
|
|
208
211
|
*/
|
|
209
|
-
export
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
/** Structured parts — text, refusals, output images, output audio. */
|
|
213
|
-
Parts = "Parts",
|
|
214
|
-
}
|
|
212
|
+
export type AssistantContent =
|
|
213
|
+
| string
|
|
214
|
+
| Array<AssistantPart>
|
|
215
215
|
|
|
216
216
|
/** Assistant's response to a user message. */
|
|
217
217
|
export interface AssistantMessage {
|
|
@@ -225,7 +225,12 @@ export interface AssistantMessage {
|
|
|
225
225
|
readonly name?: string
|
|
226
226
|
/** Tool calls the model wants to execute, if any. */
|
|
227
227
|
readonly toolCalls?: Array<ToolCall>
|
|
228
|
-
/**
|
|
228
|
+
/**
|
|
229
|
+
* Refusal reason, if the model declined to respond per safety policies.
|
|
230
|
+
*
|
|
231
|
+
* OpenAI's response schema requires this key to be present even when null,
|
|
232
|
+
* so it is deliberately not `skip_serializing_if`.
|
|
233
|
+
*/
|
|
229
234
|
readonly refusal?: string
|
|
230
235
|
/** Deprecated legacy function_call field; retained for API compatibility. */
|
|
231
236
|
readonly functionCall?: FunctionCall
|
|
@@ -248,7 +253,12 @@ export type AssistantPart =
|
|
|
248
253
|
| { type: 'output_image'; imageUrl: ImageUrl }
|
|
249
254
|
| { type: 'output_audio'; audio: AudioContent }
|
|
250
255
|
|
|
251
|
-
/**
|
|
256
|
+
/**
|
|
257
|
+
* Audio content part for speech-capable models.
|
|
258
|
+
*
|
|
259
|
+
* No `deny_unknown_fields`: shared with the response side (see
|
|
260
|
+
* `AssistantPart.OutputAudio`), same rationale as `ImageUrl` (#51).
|
|
261
|
+
*/
|
|
252
262
|
export interface AudioContent {
|
|
253
263
|
/** Base64-encoded audio data. */
|
|
254
264
|
readonly data?: string
|
|
@@ -373,6 +383,30 @@ export declare enum BatchStatus {
|
|
|
373
383
|
Cancelled = "cancelled",
|
|
374
384
|
}
|
|
375
385
|
|
|
386
|
+
/**
|
|
387
|
+
* AWS Bedrock configuration.
|
|
388
|
+
*
|
|
389
|
+
* All fields are optional; anything left unset falls back to the standard
|
|
390
|
+
* AWS environment variables (`AWS_DEFAULT_REGION` / `AWS_REGION`,
|
|
391
|
+
* `AWS_ACCESS_KEY_ID`, `AWS_SECRET_ACCESS_KEY`, `AWS_SESSION_TOKEN`,
|
|
392
|
+
* `BEDROCK_CROSS_REGION`).
|
|
393
|
+
*
|
|
394
|
+
* Implements `Debug` manually (see below) so the AWS credential fields are
|
|
395
|
+
* redacted rather than printed in full.
|
|
396
|
+
*/
|
|
397
|
+
export interface BedrockConfig {
|
|
398
|
+
/** AWS region (e.g. `"us-east-1"`). */
|
|
399
|
+
readonly region?: string
|
|
400
|
+
/** Cross-region inference profile prefix (e.g. `"us"`). */
|
|
401
|
+
readonly crossRegionPrefix?: string
|
|
402
|
+
/** Explicit AWS access key ID. */
|
|
403
|
+
readonly accessKeyId?: string
|
|
404
|
+
/** Explicit AWS secret access key. */
|
|
405
|
+
readonly secretAccessKey?: string
|
|
406
|
+
/** Explicit AWS session token (temporary credentials). */
|
|
407
|
+
readonly sessionToken?: string
|
|
408
|
+
}
|
|
409
|
+
|
|
376
410
|
/** Configuration for budget enforcement. */
|
|
377
411
|
export interface BudgetConfig {
|
|
378
412
|
/** Maximum total spend across all models, in USD. `None` means unlimited. */
|
|
@@ -402,7 +436,7 @@ export interface CacheConfig {
|
|
|
402
436
|
}
|
|
403
437
|
|
|
404
438
|
/**
|
|
405
|
-
* Plain-data configuration for
|
|
439
|
+
* Plain-data configuration for `refresh_catalog`.
|
|
406
440
|
*
|
|
407
441
|
* Deliberately FFI/binding-friendly: no `Duration` or `PathBuf`, just
|
|
408
442
|
* primitives that translate directly across language boundaries.
|
|
@@ -410,14 +444,14 @@ export interface CacheConfig {
|
|
|
410
444
|
export interface CatalogRefreshConfig {
|
|
411
445
|
/**
|
|
412
446
|
* Runtime catalog refresh is entirely opt-in: when `false`,
|
|
413
|
-
*
|
|
414
|
-
* `Ok(
|
|
447
|
+
* `refresh_catalog` is a no-op that returns
|
|
448
|
+
* `Ok(``RefreshOutcome.Disabled``)` without touching the network,
|
|
415
449
|
* the filesystem, or the overlay registry.
|
|
416
450
|
*/
|
|
417
451
|
readonly enabled?: boolean
|
|
418
452
|
/**
|
|
419
453
|
* Source URL to fetch `catalog.json` from. Must be `https`. Defaults to
|
|
420
|
-
*
|
|
454
|
+
* `DEFAULT_CATALOG_URL`; configurable so self-hosted mirrors work.
|
|
421
455
|
*/
|
|
422
456
|
readonly sourceUrl?: string
|
|
423
457
|
/**
|
|
@@ -427,7 +461,7 @@ export interface CatalogRefreshConfig {
|
|
|
427
461
|
readonly ttlSeconds?: number
|
|
428
462
|
/**
|
|
429
463
|
* Filesystem path for the on-disk cache. `None` uses a default path
|
|
430
|
-
* under `std
|
|
464
|
+
* under `std.env.temp_dir()`.
|
|
431
465
|
*/
|
|
432
466
|
readonly cachePath?: string
|
|
433
467
|
}
|
|
@@ -461,9 +495,26 @@ export interface ChatCompletionRequest {
|
|
|
461
495
|
readonly model?: string
|
|
462
496
|
/** Conversation history from oldest to newest. */
|
|
463
497
|
readonly messages?: Array<Message>
|
|
464
|
-
/**
|
|
498
|
+
/**
|
|
499
|
+
* Sampling temperature. Higher increases randomness, lower is more deterministic.
|
|
500
|
+
* Defaults to 1.0.
|
|
501
|
+
*
|
|
502
|
+
* The accepted range depends on the provider the request is routed to. OpenAI-compatible
|
|
503
|
+
* providers accept `[0.0, 2.0]`; Anthropic and Amazon Bedrock both cap it at `1.0`, and
|
|
504
|
+
* for those two a value above the cap is rejected with a `BadRequest` error before the
|
|
505
|
+
* request is sent, rather than being silently clamped or left for the provider to reject.
|
|
506
|
+
*
|
|
507
|
+
* No range is enforced for providers whose own documentation does not state one — the
|
|
508
|
+
* value is forwarded and the provider decides. Consult the target provider's reference
|
|
509
|
+
* rather than assuming `[0.0, 2.0]` is portable.
|
|
510
|
+
*/
|
|
465
511
|
readonly temperature?: number
|
|
466
|
-
/**
|
|
512
|
+
/**
|
|
513
|
+
* Nucleus sampling parameter. Lower is more focused.
|
|
514
|
+
*
|
|
515
|
+
* Accepted ranges vary by provider (most document `[0.0, 1.0]`, but this is not
|
|
516
|
+
* universal — check the target provider's own documentation for its exact bounds).
|
|
517
|
+
*/
|
|
467
518
|
readonly topP?: number
|
|
468
519
|
/** Number of chat completions to generate. Defaults to 1. */
|
|
469
520
|
readonly n?: number
|
|
@@ -500,7 +551,7 @@ export interface ChatCompletionRequest {
|
|
|
500
551
|
readonly streamOptions?: StreamOptions
|
|
501
552
|
/** Random seed for reproducible outputs. Provider support varies. */
|
|
502
553
|
readonly seed?: number
|
|
503
|
-
/** Reasoning effort level (low, medium, high) for extended-thinking models. */
|
|
554
|
+
/** Reasoning effort level (minimal, low, medium, high, max) for extended-thinking models. */
|
|
504
555
|
readonly reasoningEffort?: ReasoningEffort
|
|
505
556
|
/**
|
|
506
557
|
* Output modalities to request from the model.
|
|
@@ -509,6 +560,49 @@ export interface ChatCompletionRequest {
|
|
|
509
560
|
* translates these to `generationConfig.responseModalities` (uppercase).
|
|
510
561
|
*/
|
|
511
562
|
readonly modalities?: Array<Modality>
|
|
563
|
+
/** Whether to return log probabilities of the output tokens. */
|
|
564
|
+
readonly logprobs?: boolean
|
|
565
|
+
/**
|
|
566
|
+
* Number of most-likely tokens to return log probabilities for, `0..=20`.
|
|
567
|
+
* Requires `logprobs` to be `true`.
|
|
568
|
+
*/
|
|
569
|
+
readonly topLogprobs?: number
|
|
570
|
+
/**
|
|
571
|
+
* Upper bound on generated tokens, including reasoning tokens.
|
|
572
|
+
*
|
|
573
|
+
* Supersedes `max_tokens` on OpenAI reasoning models, which reject
|
|
574
|
+
* `max_tokens` outright.
|
|
575
|
+
*/
|
|
576
|
+
readonly maxCompletionTokens?: number
|
|
577
|
+
/**
|
|
578
|
+
* Latency tier to process the request under (e.g. `"auto"`, `"default"`,
|
|
579
|
+
* `"flex"`).
|
|
580
|
+
*/
|
|
581
|
+
readonly serviceTier?: string
|
|
582
|
+
/** Whether to store the completion for later retrieval by the provider. */
|
|
583
|
+
readonly store?: boolean
|
|
584
|
+
/** Developer-defined tags attached to the completion. */
|
|
585
|
+
readonly metadata?: Record<string, string>
|
|
586
|
+
/**
|
|
587
|
+
* Predicted output, for latency reduction when much of the response is
|
|
588
|
+
* known ahead of time.
|
|
589
|
+
*
|
|
590
|
+
* Untyped: the shape is provider-defined and still evolving, and the
|
|
591
|
+
* value is forwarded verbatim.
|
|
592
|
+
*/
|
|
593
|
+
readonly prediction?: JsonValue
|
|
594
|
+
/**
|
|
595
|
+
* Audio output parameters, required when `modalities` includes `audio`.
|
|
596
|
+
*
|
|
597
|
+
* Untyped for the same reason as `prediction`.
|
|
598
|
+
*/
|
|
599
|
+
readonly audio?: JsonValue
|
|
600
|
+
/**
|
|
601
|
+
* Web-search tool configuration for search-enabled models.
|
|
602
|
+
*
|
|
603
|
+
* Untyped for the same reason as `prediction`.
|
|
604
|
+
*/
|
|
605
|
+
readonly webSearchOptions?: JsonValue
|
|
512
606
|
/**
|
|
513
607
|
* Provider-specific extra parameters merged into the request body.
|
|
514
608
|
* Use for guardrails, safety settings, grounding config, etc.
|
|
@@ -551,21 +645,35 @@ export interface ChatCompletionTool {
|
|
|
551
645
|
export interface Choice {
|
|
552
646
|
/** Index of this choice in the choices array. */
|
|
553
647
|
readonly index?: number
|
|
554
|
-
/**
|
|
648
|
+
/**
|
|
649
|
+
* The assistant's message response.
|
|
650
|
+
*
|
|
651
|
+
* Serialized with an explicit `role: "assistant"`. The field is not stored
|
|
652
|
+
* on `AssistantMessage` because `Message` is an internally-tagged enum
|
|
653
|
+
* keyed on `role`, so a stored field would emit the key twice inside a
|
|
654
|
+
* request. OpenAI's response schema requires it here.
|
|
655
|
+
*/
|
|
555
656
|
readonly message?: AssistantMessage
|
|
556
657
|
/** Why the model stopped generating (stop, length, tool_calls, content_filter, etc.). */
|
|
557
658
|
readonly finishReason?: FinishReason
|
|
659
|
+
/**
|
|
660
|
+
* Per-token log probabilities, when the request asked for them.
|
|
661
|
+
*
|
|
662
|
+
* Required by OpenAI's response schema as an always-present, nullable key,
|
|
663
|
+
* so this is deliberately not `skip_serializing_if`.
|
|
664
|
+
*/
|
|
665
|
+
readonly logprobs?: JsonValue
|
|
558
666
|
}
|
|
559
667
|
|
|
560
668
|
/**
|
|
561
|
-
* A per-chunk transformation in the
|
|
669
|
+
* A per-chunk transformation in the `StreamPipeline`.
|
|
562
670
|
*
|
|
563
671
|
* Each middleware receives a typed chunk and returns `Ok(Some(chunk))`
|
|
564
672
|
* to pass it through (optionally modified), `Ok(None)` to drop the chunk,
|
|
565
673
|
* or `Err(e)` to propagate a stream error.
|
|
566
674
|
*
|
|
567
675
|
* The trait is object-safe so multiple middleware implementations can be
|
|
568
|
-
* chained inside
|
|
676
|
+
* chained inside `StreamPipeline`.
|
|
569
677
|
*/
|
|
570
678
|
export interface ChunkMiddleware {
|
|
571
679
|
/**
|
|
@@ -637,9 +745,35 @@ export interface CreateImageRequest {
|
|
|
637
745
|
readonly user?: string
|
|
638
746
|
}
|
|
639
747
|
|
|
640
|
-
/**
|
|
748
|
+
/**
|
|
749
|
+
* Request to create a response via the OpenAI Responses API (`POST /responses`).
|
|
750
|
+
*
|
|
751
|
+
* # Provider support
|
|
752
|
+
*
|
|
753
|
+
* **The Responses API path is OpenAI-only.** This type models the OpenAI Responses
|
|
754
|
+
* wire format, and unlike `ChatCompletionRequest` the body is sent to the provider
|
|
755
|
+
* verbatim: neither `Provider.transform_request` nor `Provider.transform_response`
|
|
756
|
+
* runs for Responses calls, and no provider in `schemas/providers.json` declares a
|
|
757
|
+
* `responses` endpoint.
|
|
758
|
+
*
|
|
759
|
+
* Pointing a Responses call at a provider that does not natively serve the OpenAI
|
|
760
|
+
* `/responses` contract (Anthropic, Vertex, Bedrock, Cohere, Google AI, Azure) is not
|
|
761
|
+
* supported: the request goes out unmodified, so the provider rejects it or returns a
|
|
762
|
+
* body that cannot be deserialized into `ResponseObject`. Use
|
|
763
|
+
* `ChatCompletionRequest` for cross-provider work — that path applies the
|
|
764
|
+
* per-provider request and response normalization this one does not.
|
|
765
|
+
*
|
|
766
|
+
* `ChatCompletionRequest`: crate.types.ChatCompletionRequest
|
|
767
|
+
*/
|
|
641
768
|
export interface CreateResponseRequest {
|
|
642
|
-
/**
|
|
769
|
+
/**
|
|
770
|
+
* Model ID, as named by the OpenAI Responses API (e.g. `"gpt-5"`).
|
|
771
|
+
*
|
|
772
|
+
* Sent to the wire exactly as given. The Responses path performs none of the
|
|
773
|
+
* chat path's model handling: a `provider/model` routing prefix is **not**
|
|
774
|
+
* stripped and does **not** re-route the request, which stays pinned to the
|
|
775
|
+
* provider the client was constructed with.
|
|
776
|
+
*/
|
|
643
777
|
readonly model?: string
|
|
644
778
|
/** Input data to process (e.g., a document to extract from). */
|
|
645
779
|
readonly input?: JsonValue
|
|
@@ -653,6 +787,27 @@ export interface CreateResponseRequest {
|
|
|
653
787
|
readonly maxOutputTokens?: number
|
|
654
788
|
/** Optional metadata. */
|
|
655
789
|
readonly metadata?: JsonValue
|
|
790
|
+
/**
|
|
791
|
+
* Extra top-level parameters shallow-merged into the request body, OpenAI-Python
|
|
792
|
+
* style (`{**body, **extra_body}`) — keys here override identically named fields
|
|
793
|
+
* above. Use it for OpenAI Responses fields this struct does not model directly,
|
|
794
|
+
* such as the top-level `reasoning.effort`.
|
|
795
|
+
*
|
|
796
|
+
* This is an OpenAI escape hatch, not a cross-provider one. On the chat path the
|
|
797
|
+
* providers that consume `extra_body` natively (Anthropic, Vertex, Bedrock) claim
|
|
798
|
+
* it inside their own `transform_request`; here no provider transform runs, so the
|
|
799
|
+
* merged keys always travel to the wire as literal OpenAI Responses fields.
|
|
800
|
+
*
|
|
801
|
+
* A non-object value cannot be merged into the body root and is dropped with a
|
|
802
|
+
* warning rather than sent.
|
|
803
|
+
*/
|
|
804
|
+
readonly extraBody?: JsonValue
|
|
805
|
+
/**
|
|
806
|
+
* Whether to stream the response.
|
|
807
|
+
*
|
|
808
|
+
* Managed by the client layer — do not set directly.
|
|
809
|
+
*/
|
|
810
|
+
readonly stream?: boolean
|
|
656
811
|
}
|
|
657
812
|
|
|
658
813
|
/** Request to generate speech audio from text. */
|
|
@@ -723,7 +878,7 @@ export interface DecodedDataUrl {
|
|
|
723
878
|
* provider is used as the fallback. This enables seamless migration between
|
|
724
879
|
* providers by changing only the model name.
|
|
725
880
|
*
|
|
726
|
-
* The provider is stored behind an
|
|
881
|
+
* The provider is stored behind an `Arc` so it can be shared cheaply into
|
|
727
882
|
* async closures and streaming tasks. Pre-computed auth headers and extra
|
|
728
883
|
* headers are cached at construction to avoid redundant encoding on every request.
|
|
729
884
|
*/
|
|
@@ -754,9 +909,9 @@ export declare class DefaultClient {
|
|
|
754
909
|
*
|
|
755
910
|
* Uses exponential backoff with configurable initial interval, maximum interval, and backoff multiplier.
|
|
756
911
|
* Optionally supports a timeout that aborts polling if exceeded.
|
|
757
|
-
* @throws Returns `BatchWaitError
|
|
758
|
-
* Returns `BatchWaitError
|
|
759
|
-
* Returns `BatchWaitError
|
|
912
|
+
* @throws Returns `BatchWaitError.Failed` if the batch reaches a failure terminal status.
|
|
913
|
+
* Returns `BatchWaitError.Timeout` if the configured timeout is exceeded.
|
|
914
|
+
* Returns `BatchWaitError.Client` for underlying client errors.
|
|
760
915
|
*/
|
|
761
916
|
waitForBatch(batchId: string, config?: WaitForBatchConfig | undefined | null): Promise<BatchObject>
|
|
762
917
|
createResponse(req?: CreateResponseRequest | undefined | null): Promise<ResponseObject>
|
|
@@ -799,12 +954,9 @@ export declare enum EmbeddingFormat {
|
|
|
799
954
|
}
|
|
800
955
|
|
|
801
956
|
/** Text or texts to embed. */
|
|
802
|
-
export
|
|
803
|
-
|
|
804
|
-
|
|
805
|
-
/** Multiple text strings (batch embedding). */
|
|
806
|
-
Multiple = "Multiple",
|
|
807
|
-
}
|
|
957
|
+
export type EmbeddingInput =
|
|
958
|
+
| string
|
|
959
|
+
| Array<string>
|
|
808
960
|
|
|
809
961
|
/** A single embedding vector. */
|
|
810
962
|
export interface EmbeddingObject {
|
|
@@ -859,11 +1011,11 @@ export interface EmbeddingResponse {
|
|
|
859
1011
|
export declare enum Enforcement {
|
|
860
1012
|
/**
|
|
861
1013
|
* Reject requests that would exceed the budget with
|
|
862
|
-
*
|
|
1014
|
+
* `LiterLlmError.BudgetExceeded`.
|
|
863
1015
|
*/
|
|
864
1016
|
Hard = "Hard",
|
|
865
1017
|
/**
|
|
866
|
-
* Allow requests through but emit a `tracing
|
|
1018
|
+
* Allow requests through but emit a `tracing.warn!` when the budget is
|
|
867
1019
|
* exceeded.
|
|
868
1020
|
*/
|
|
869
1021
|
Soft = "Soft",
|
|
@@ -943,7 +1095,7 @@ export declare enum FinishReason {
|
|
|
943
1095
|
export interface FunctionCall {
|
|
944
1096
|
/** Function name. */
|
|
945
1097
|
readonly name: string
|
|
946
|
-
/** Arguments as a JSON string (parse with serde_json
|
|
1098
|
+
/** Arguments as a JSON string (parse with serde_json.from_str). */
|
|
947
1099
|
readonly arguments: string
|
|
948
1100
|
}
|
|
949
1101
|
|
|
@@ -969,11 +1121,11 @@ export interface FunctionMessage {
|
|
|
969
1121
|
* Abstraction over a health probe strategy.
|
|
970
1122
|
*
|
|
971
1123
|
* Implementors issue a lightweight probe against `upstream` (typically a
|
|
972
|
-
* provider base URL or named identifier) and report
|
|
1124
|
+
* provider base URL or named identifier) and report `HealthStatus`.
|
|
973
1125
|
*/
|
|
974
1126
|
export interface HealthChecker {
|
|
975
1127
|
/**
|
|
976
|
-
* Probe `upstream` and return its current
|
|
1128
|
+
* Probe `upstream` and return its current `HealthStatus`.
|
|
977
1129
|
*
|
|
978
1130
|
* The parameter is taken by value (`String`) so that implementations can
|
|
979
1131
|
* move it into the returned future without a clone, making the
|
|
@@ -1018,7 +1170,15 @@ export interface ImagesResponse {
|
|
|
1018
1170
|
readonly data?: Array<Image>
|
|
1019
1171
|
}
|
|
1020
1172
|
|
|
1021
|
-
/**
|
|
1173
|
+
/**
|
|
1174
|
+
* An image URL reference with optional detail level for processing.
|
|
1175
|
+
*
|
|
1176
|
+
* No `deny_unknown_fields`: this type is shared with the response side
|
|
1177
|
+
* (see `AssistantPart.OutputImage`) where it is deserialized from
|
|
1178
|
+
* provider output, not just constructed as request input (see #51). A
|
|
1179
|
+
* provider adding a new field to its image-output object must not hard-fail
|
|
1180
|
+
* the whole response.
|
|
1181
|
+
*/
|
|
1022
1182
|
export interface ImageUrl {
|
|
1023
1183
|
/** URL of the image (data URI or HTTP/HTTPS URL). */
|
|
1024
1184
|
readonly url?: string
|
|
@@ -1048,14 +1208,119 @@ export interface JsonSchemaFormat {
|
|
|
1048
1208
|
readonly strict?: boolean
|
|
1049
1209
|
}
|
|
1050
1210
|
|
|
1211
|
+
/** Budget enforcement configuration. */
|
|
1212
|
+
export interface LlmBudgetConfig {
|
|
1213
|
+
/** Global spend limit in USD. */
|
|
1214
|
+
readonly globalLimit?: number
|
|
1215
|
+
/** Per-model spend limits in USD, keyed by model name. */
|
|
1216
|
+
readonly modelLimits?: Record<string, number>
|
|
1217
|
+
/** Enforcement mode: `"hard"` (reject over-budget requests) or `"soft"` (log only). */
|
|
1218
|
+
readonly enforcement?: string
|
|
1219
|
+
}
|
|
1220
|
+
|
|
1221
|
+
/** Response cache configuration. */
|
|
1222
|
+
export interface LlmCacheConfig {
|
|
1223
|
+
/** Maximum number of cached entries. */
|
|
1224
|
+
readonly maxEntries?: number
|
|
1225
|
+
/** Cache entry time-to-live, in seconds. */
|
|
1226
|
+
readonly ttlSeconds?: number
|
|
1227
|
+
/** Cache backend name (e.g. `"memory"`, or an `opendal` scheme). */
|
|
1228
|
+
readonly backend?: string
|
|
1229
|
+
/** Backend-specific configuration key/value pairs. */
|
|
1230
|
+
readonly backendConfig?: Record<string, string>
|
|
1231
|
+
}
|
|
1232
|
+
|
|
1233
|
+
/**
|
|
1234
|
+
* Canonical configuration for an LLM client.
|
|
1235
|
+
*
|
|
1236
|
+
* All fields except `model` are optional so that partially-specified
|
|
1237
|
+
* configs (e.g. from environment-driven defaults) round-trip cleanly.
|
|
1238
|
+
* Convert to a runtime client configuration via
|
|
1239
|
+
* `LlmConfig.into_client_builder`.
|
|
1240
|
+
*
|
|
1241
|
+
* `temperature` and `max_tokens` are request-time parameters rather than
|
|
1242
|
+
* client-level settings; they are carried on this struct for callers to
|
|
1243
|
+
* read when building individual requests, and are intentionally **not**
|
|
1244
|
+
* mapped by `LlmConfig.into_client_builder`.
|
|
1245
|
+
*
|
|
1246
|
+
* Implements `Debug` manually (see below) so `api_key` and header values are
|
|
1247
|
+
* redacted rather than printed in full.
|
|
1248
|
+
*/
|
|
1249
|
+
export interface LlmConfig {
|
|
1250
|
+
/** Model identifier (e.g. `"gpt-4o"`, `"bedrock/anthropic.claude-3-sonnet-20240229-v1:0"`). */
|
|
1251
|
+
readonly model?: string
|
|
1252
|
+
/** API key for authentication. */
|
|
1253
|
+
readonly apiKey?: string
|
|
1254
|
+
/**
|
|
1255
|
+
* Override base URL. When set, all requests go here and provider
|
|
1256
|
+
* auto-detection is skipped.
|
|
1257
|
+
*/
|
|
1258
|
+
readonly baseUrl?: string
|
|
1259
|
+
/** Request timeout, in seconds. */
|
|
1260
|
+
readonly timeoutSecs?: number
|
|
1261
|
+
/** Maximum number of retries on 429 / 5xx responses. */
|
|
1262
|
+
readonly maxRetries?: number
|
|
1263
|
+
/** Sampling temperature for requests built from this config. */
|
|
1264
|
+
readonly temperature?: number
|
|
1265
|
+
/** Maximum number of tokens to generate for requests built from this config. */
|
|
1266
|
+
readonly maxTokens?: number
|
|
1267
|
+
/**
|
|
1268
|
+
* Automatically load the API key from the provider's environment variable
|
|
1269
|
+
* when no explicit key is provided (default: `true`).
|
|
1270
|
+
*/
|
|
1271
|
+
readonly loadEnv?: boolean
|
|
1272
|
+
/** Extra headers sent on every request. */
|
|
1273
|
+
readonly headers?: Record<string, string>
|
|
1274
|
+
/** Custom provider configurations, in addition to the built-in providers. */
|
|
1275
|
+
readonly providers?: Array<LlmProviderConfig>
|
|
1276
|
+
/** Response cache configuration. */
|
|
1277
|
+
readonly cache?: LlmCacheConfig
|
|
1278
|
+
/** Budget enforcement configuration. */
|
|
1279
|
+
readonly budget?: LlmBudgetConfig
|
|
1280
|
+
/** Per-model rate limiting configuration. */
|
|
1281
|
+
readonly rateLimit?: LlmRateLimitConfig
|
|
1282
|
+
/** Enable per-request cost tracking. */
|
|
1283
|
+
readonly costTracking?: boolean
|
|
1284
|
+
/** Enable OpenTelemetry-compatible tracing spans. */
|
|
1285
|
+
readonly tracing?: boolean
|
|
1286
|
+
/** Cooldown duration after transient errors, in seconds. */
|
|
1287
|
+
readonly cooldownSecs?: number
|
|
1288
|
+
/** Background health check interval, in seconds. */
|
|
1289
|
+
readonly healthCheckSecs?: number
|
|
1290
|
+
/** AWS Bedrock configuration (region, credentials, cross-region routing). */
|
|
1291
|
+
readonly bedrock?: BedrockConfig
|
|
1292
|
+
}
|
|
1293
|
+
|
|
1294
|
+
/** A custom provider configuration entry. */
|
|
1295
|
+
export interface LlmProviderConfig {
|
|
1296
|
+
/** Provider name, used to key model prefix matching. */
|
|
1297
|
+
readonly name?: string
|
|
1298
|
+
/** Base URL for the provider's OpenAI-compatible API. */
|
|
1299
|
+
readonly baseUrl?: string
|
|
1300
|
+
/** Header name used to carry the API key (defaults to `Authorization` when unset). */
|
|
1301
|
+
readonly authHeader?: string
|
|
1302
|
+
/** Model name prefixes routed to this provider (e.g. `["my-provider/"]`). */
|
|
1303
|
+
readonly modelPrefixes?: Array<string>
|
|
1304
|
+
}
|
|
1305
|
+
|
|
1306
|
+
/** Per-model rate limiting configuration. */
|
|
1307
|
+
export interface LlmRateLimitConfig {
|
|
1308
|
+
/** Requests per minute limit. */
|
|
1309
|
+
readonly rpm?: number
|
|
1310
|
+
/** Tokens per minute limit. */
|
|
1311
|
+
readonly tpm?: number
|
|
1312
|
+
/** Rate limit window, in seconds. */
|
|
1313
|
+
readonly windowSeconds?: number
|
|
1314
|
+
}
|
|
1315
|
+
|
|
1051
1316
|
/** A chat message in a conversation. */
|
|
1052
1317
|
export type Message =
|
|
1053
|
-
| { role: 'system';
|
|
1054
|
-
| { role: 'user';
|
|
1055
|
-
| { role: 'assistant';
|
|
1056
|
-
| { role: 'tool';
|
|
1057
|
-
| { role: 'developer';
|
|
1058
|
-
| { role: 'function';
|
|
1318
|
+
| { role: 'system'; system: SystemMessage }
|
|
1319
|
+
| { role: 'user'; user: UserMessage }
|
|
1320
|
+
| { role: 'assistant'; assistant: AssistantMessage }
|
|
1321
|
+
| { role: 'tool'; tool: ToolMessage }
|
|
1322
|
+
| { role: 'developer'; developer: DeveloperMessage }
|
|
1323
|
+
| { role: 'function'; function: FunctionMessage }
|
|
1059
1324
|
|
|
1060
1325
|
/**
|
|
1061
1326
|
* Output modality requested from the model.
|
|
@@ -1074,11 +1339,11 @@ export declare enum Modality {
|
|
|
1074
1339
|
|
|
1075
1340
|
/**
|
|
1076
1341
|
* Public, FFI-friendly snapshot of a model's pricing and capability
|
|
1077
|
-
* metadata, projected from
|
|
1342
|
+
* metadata, projected from `ModelPricing`.
|
|
1078
1343
|
*
|
|
1079
|
-
* Unlike
|
|
1344
|
+
* Unlike `ModelPricing` (which is excluded from binding generation),
|
|
1080
1345
|
* `ModelInfo` is an owned plain-data DTO safe to hand across the FFI
|
|
1081
|
-
* boundary — see
|
|
1346
|
+
* boundary — see `model_info`.
|
|
1082
1347
|
*/
|
|
1083
1348
|
export interface ModelInfo {
|
|
1084
1349
|
/** Cost in USD per input (prompt) token. */
|
|
@@ -1162,7 +1427,7 @@ export interface ModelsListResponse {
|
|
|
1162
1427
|
|
|
1163
1428
|
/**
|
|
1164
1429
|
* Public, FFI-friendly snapshot of a single context-window pricing tier,
|
|
1165
|
-
* projected from
|
|
1430
|
+
* projected from `PricingTier`.
|
|
1166
1431
|
*/
|
|
1167
1432
|
export interface ModelTier {
|
|
1168
1433
|
/**
|
|
@@ -1239,12 +1504,9 @@ export interface ModerationCategoryScores {
|
|
|
1239
1504
|
}
|
|
1240
1505
|
|
|
1241
1506
|
/** Input to the moderation endpoint — a single string or multiple strings. */
|
|
1242
|
-
export
|
|
1243
|
-
|
|
1244
|
-
|
|
1245
|
-
/** Multiple text strings (batch moderation). */
|
|
1246
|
-
Multiple = "Multiple",
|
|
1247
|
-
}
|
|
1507
|
+
export type ModerationInput =
|
|
1508
|
+
| string
|
|
1509
|
+
| Array<string>
|
|
1248
1510
|
|
|
1249
1511
|
/** Request to classify content for policy violations. */
|
|
1250
1512
|
export interface ModerationRequest {
|
|
@@ -1332,7 +1594,7 @@ export interface PageDimensions {
|
|
|
1332
1594
|
/**
|
|
1333
1595
|
* Breakdown of tokens used in the prompt portion of a request.
|
|
1334
1596
|
*
|
|
1335
|
-
* `cached_tokens` is included in `Usage
|
|
1597
|
+
* `cached_tokens` is included in `Usage.prompt_tokens` — it is *not* an
|
|
1336
1598
|
* additional charge on top of the prompt token count. When pricing supports
|
|
1337
1599
|
* a `cache_read_input_token_cost`, the cached portion is billed at the
|
|
1338
1600
|
* discounted rate and the remainder at the regular input rate.
|
|
@@ -1355,19 +1617,7 @@ export interface PromptTokensDetails {
|
|
|
1355
1617
|
*
|
|
1356
1618
|
* All flags default to `false` so that newly added providers are safe.
|
|
1357
1619
|
*
|
|
1358
|
-
* Access via the crate-level
|
|
1359
|
-
*
|
|
1360
|
-
* ```rust
|
|
1361
|
-
* use liter_llm::capabilities;
|
|
1362
|
-
*
|
|
1363
|
-
* let caps = capabilities("openai");
|
|
1364
|
-
* assert!(caps.function_calling);
|
|
1365
|
-
* assert!(caps.vision);
|
|
1366
|
-
*
|
|
1367
|
-
* // Unknown providers return a default-all-false reference.
|
|
1368
|
-
* let unknown = capabilities("my-private-model");
|
|
1369
|
-
* assert!(!unknown.function_calling);
|
|
1370
|
-
* ```
|
|
1620
|
+
* Access via the crate-level `capabilities` function:
|
|
1371
1621
|
*/
|
|
1372
1622
|
export interface ProviderCapabilities {
|
|
1373
1623
|
/** The provider accepts image input in chat messages. */
|
|
@@ -1390,7 +1640,7 @@ export interface ProviderCapabilities {
|
|
|
1390
1640
|
* Static configuration for a single provider entry in providers.json.
|
|
1391
1641
|
*
|
|
1392
1642
|
* This struct deliberately does not include capability flags or streaming
|
|
1393
|
-
* format, which are accessed via the
|
|
1643
|
+
* format, which are accessed via the `capabilities` function.
|
|
1394
1644
|
*/
|
|
1395
1645
|
export interface ProviderConfig {
|
|
1396
1646
|
/** Provider identifier (matches the entry key in providers.json). */
|
|
@@ -1410,7 +1660,7 @@ export interface ProviderConfig {
|
|
|
1410
1660
|
*
|
|
1411
1661
|
* Each entry maps an OpenAI-spec field name (e.g. `"max_completion_tokens"`)
|
|
1412
1662
|
* to the name this provider expects (e.g. `"max_tokens"`). Applied
|
|
1413
|
-
* automatically by `ConfigDrivenProvider
|
|
1663
|
+
* automatically by `ConfigDrivenProvider.transform_request`.
|
|
1414
1664
|
*/
|
|
1415
1665
|
readonly paramMappings?: Record<string, string>
|
|
1416
1666
|
}
|
|
@@ -1430,9 +1680,11 @@ export declare enum ReasoningEffort {
|
|
|
1430
1680
|
Low = "low",
|
|
1431
1681
|
Medium = "medium",
|
|
1432
1682
|
High = "high",
|
|
1683
|
+
Minimal = "minimal",
|
|
1684
|
+
Max = "max",
|
|
1433
1685
|
}
|
|
1434
1686
|
|
|
1435
|
-
/** Result of a
|
|
1687
|
+
/** Result of a `refresh_catalog` call. */
|
|
1436
1688
|
export declare enum RefreshOutcome {
|
|
1437
1689
|
/**
|
|
1438
1690
|
* `config.enabled` was `false`; no network, filesystem, or overlay
|
|
@@ -1453,12 +1705,9 @@ export declare enum RefreshOutcome {
|
|
|
1453
1705
|
}
|
|
1454
1706
|
|
|
1455
1707
|
/** A document to be reranked — either a plain string or an object with a text field. */
|
|
1456
|
-
export
|
|
1457
|
-
|
|
1458
|
-
|
|
1459
|
-
/** Document with explicit text field (may include metadata). */
|
|
1460
|
-
Object = "Object",
|
|
1461
|
-
}
|
|
1708
|
+
export type RerankDocument =
|
|
1709
|
+
| string
|
|
1710
|
+
| { text: string }
|
|
1462
1711
|
|
|
1463
1712
|
/** Request to rerank documents by relevance to a query. */
|
|
1464
1713
|
export interface RerankRequest {
|
|
@@ -1627,12 +1876,9 @@ export interface SpecificToolChoice {
|
|
|
1627
1876
|
}
|
|
1628
1877
|
|
|
1629
1878
|
/** Stop sequence(s) that cause the model to stop generating. */
|
|
1630
|
-
export
|
|
1631
|
-
|
|
1632
|
-
|
|
1633
|
-
/** Multiple stop sequences. */
|
|
1634
|
-
Multiple = "Multiple",
|
|
1635
|
-
}
|
|
1879
|
+
export type StopSequence =
|
|
1880
|
+
| string
|
|
1881
|
+
| Array<string>
|
|
1636
1882
|
|
|
1637
1883
|
/** A streaming choice with incremental delta. */
|
|
1638
1884
|
export interface StreamChoice {
|
|
@@ -1666,7 +1912,7 @@ export interface StreamDelta {
|
|
|
1666
1912
|
* Most providers use standard Server-Sent Events (SSE). AWS Bedrock uses
|
|
1667
1913
|
* a proprietary binary EventStream framing.
|
|
1668
1914
|
*
|
|
1669
|
-
* Deserialized from the `streaming_format` JSON field via
|
|
1915
|
+
* Deserialized from the `streaming_format` JSON field via `serde`.
|
|
1670
1916
|
*/
|
|
1671
1917
|
export declare enum StreamFormat {
|
|
1672
1918
|
/** Standard Server-Sent Events (text/event-stream). */
|
|
@@ -1707,7 +1953,7 @@ export interface SystemMessage {
|
|
|
1707
1953
|
* Instructions or context that apply throughout the conversation.
|
|
1708
1954
|
*
|
|
1709
1955
|
* Accepts either a plain text string or an array of content parts,
|
|
1710
|
-
* mirroring
|
|
1956
|
+
* mirroring `UserContent` so that `Message.system_with_parts` works.
|
|
1711
1957
|
*/
|
|
1712
1958
|
readonly content?: UserContent
|
|
1713
1959
|
/** Optional name for the system message source. */
|
|
@@ -1725,12 +1971,9 @@ export interface ToolCall {
|
|
|
1725
1971
|
}
|
|
1726
1972
|
|
|
1727
1973
|
/** Tool usage mode or a specific tool to call. */
|
|
1728
|
-
export
|
|
1729
|
-
|
|
1730
|
-
|
|
1731
|
-
/** Force a specific tool to be called. */
|
|
1732
|
-
Specific = "Specific",
|
|
1733
|
-
}
|
|
1974
|
+
export type ToolChoice =
|
|
1975
|
+
| ToolChoiceMode
|
|
1976
|
+
| SpecificToolChoice
|
|
1734
1977
|
|
|
1735
1978
|
/** Tool choice mode. */
|
|
1736
1979
|
export declare enum ToolChoiceMode {
|
|
@@ -1744,8 +1987,15 @@ export declare enum ToolChoiceMode {
|
|
|
1744
1987
|
|
|
1745
1988
|
/** Tool execution result returned to the model. */
|
|
1746
1989
|
export interface ToolMessage {
|
|
1747
|
-
/**
|
|
1748
|
-
|
|
1990
|
+
/**
|
|
1991
|
+
* Result of the tool execution as plain text or an array of content parts
|
|
1992
|
+
* (text, images, documents, audio), mirroring `UserMessage.content`.
|
|
1993
|
+
*
|
|
1994
|
+
* `#[serde(untagged)]` on `UserContent` means a bare JSON string still
|
|
1995
|
+
* deserialises into `Text`, so tool results persisted before this field
|
|
1996
|
+
* carried structured content continue to round-trip.
|
|
1997
|
+
*/
|
|
1998
|
+
readonly content?: UserContent
|
|
1749
1999
|
/** ID of the tool call this result responds to. */
|
|
1750
2000
|
readonly toolCallId?: string
|
|
1751
2001
|
/** Optional tool/function name. */
|
|
@@ -1804,12 +2054,9 @@ export interface Usage {
|
|
|
1804
2054
|
}
|
|
1805
2055
|
|
|
1806
2056
|
/** User message content as either plain text or a list of multimodal parts. */
|
|
1807
|
-
export
|
|
1808
|
-
|
|
1809
|
-
|
|
1810
|
-
/** Array of content parts (text, images, documents, audio). */
|
|
1811
|
-
Parts = "Parts",
|
|
1812
|
-
}
|
|
2057
|
+
export type UserContent =
|
|
2058
|
+
| string
|
|
2059
|
+
| Array<ContentPart>
|
|
1813
2060
|
|
|
1814
2061
|
/** User message in the conversation. */
|
|
1815
2062
|
export interface UserMessage {
|
|
@@ -1841,30 +2088,39 @@ export interface WaitForBatchConfig {
|
|
|
1841
2088
|
*
|
|
1842
2089
|
* Returns `None` if the model is not present in the active pricing
|
|
1843
2090
|
* registry. Uses the same exact-match-then-prefix-fallback resolution as
|
|
1844
|
-
*
|
|
1845
|
-
*
|
|
2091
|
+
* `model_pricing`; unlike `model_pricing`, the result is an owned
|
|
2092
|
+
* `ModelInfo` value safe to hand across the FFI boundary.
|
|
1846
2093
|
*
|
|
1847
2094
|
* When a runtime catalog refresh has succeeded, this reflects the refreshed
|
|
1848
2095
|
* (overlay) catalog; otherwise it reflects the embedded catalog. See
|
|
1849
|
-
*
|
|
2096
|
+
* `model_pricing` for the embedded-only alternative.
|
|
1850
2097
|
*/
|
|
1851
2098
|
export declare function modelInfo(model: string): ModelInfo | null;
|
|
1852
2099
|
|
|
2100
|
+
/**
|
|
2101
|
+
* Record the estimated USD cost of a completion.
|
|
2102
|
+
*
|
|
2103
|
+
* Call from `CostTrackingService` once a
|
|
2104
|
+
* completion's cost has been computed. Emits `gen_ai.client.cost.usd`.
|
|
2105
|
+
* If the meter has not been initialized, this call is a no-op.
|
|
2106
|
+
*/
|
|
2107
|
+
export declare function recordCostUsd(system: string, model: string, operation: string, costUsd: number): void;
|
|
2108
|
+
|
|
1853
2109
|
/**
|
|
1854
2110
|
* Refresh the runtime catalog overlay per `config`.
|
|
1855
2111
|
*
|
|
1856
|
-
* - `config.enabled == false`: returns `Ok(
|
|
2112
|
+
* - `config.enabled == false`: returns `Ok(``RefreshOutcome.Disabled``)`
|
|
1857
2113
|
* immediately. No network, filesystem, or overlay activity.
|
|
1858
2114
|
* - A fresh on-disk cache (age < `config.ttl_seconds`) exists at the
|
|
1859
2115
|
* resolved cache path (`config.cache_path`, or a default under
|
|
1860
|
-
* `std
|
|
1861
|
-
* returning `Ok(
|
|
2116
|
+
* `std.env.temp_dir()`): read + flatten it and install the overlay,
|
|
2117
|
+
* returning `Ok(``RefreshOutcome.FromCache``)`. No network request is
|
|
1862
2118
|
* made.
|
|
1863
2119
|
* - Otherwise: validate `config.source_url` uses `https`
|
|
1864
|
-
* (
|
|
2120
|
+
* (`CatalogRefreshError.InsecureUrl` otherwise), fetch it, flatten it,
|
|
1865
2121
|
* install the overlay, best-effort write the raw JSON to the cache path
|
|
1866
2122
|
* (a cache write failure does not fail the refresh), and return
|
|
1867
|
-
* `Ok(
|
|
2123
|
+
* `Ok(``RefreshOutcome.Fetched``)`.
|
|
1868
2124
|
*
|
|
1869
2125
|
* On any error return, the overlay is left untouched: the previously
|
|
1870
2126
|
* active registry (a prior successful overlay, or the embedded catalog if
|
|
@@ -1900,6 +2156,7 @@ export declare class ChatStreamIterator {
|
|
|
1900
2156
|
}
|
|
1901
2157
|
|
|
1902
2158
|
export declare class LiterLlmErrorInfo {
|
|
2159
|
+
code(): number
|
|
1903
2160
|
statusCode(): number
|
|
1904
2161
|
isTransient(): boolean
|
|
1905
2162
|
errorType(): string
|