@xberg-io/liter-llm 1.16.0 → 1.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +7 -2
- package/index.d.ts +271 -131
- package/index.js +624 -69
- package/liter-llm-node.darwin-arm64.node +0 -0
- package/liter-llm-node.darwin-x64.node +0 -0
- package/liter-llm-node.linux-arm64-gnu.node +0 -0
- package/liter-llm-node.linux-x64-gnu.node +0 -0
- package/liter-llm-node.win32-arm64-msvc.node +0 -0
- package/liter-llm-node.win32-x64-msvc.node +0 -0
- package/package.json +13 -7
package/index.d.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
// This file is auto-generated by alef — DO NOT EDIT.
|
|
2
|
-
// alef:hash:
|
|
2
|
+
// alef:hash:5b669b2bae32edc478385d1c565e973c2b54c440018dc4506eccb2bec6078845
|
|
3
3
|
// To regenerate: alef generate
|
|
4
|
-
// To verify freshness: alef verify
|
|
4
|
+
// To verify freshness: alef verify
|
|
5
5
|
/* eslint-disable */
|
|
6
6
|
|
|
7
7
|
export type JsonValue = string | number | boolean | null | JsonValue[] | { [key: string]: JsonValue };
|
|
@@ -10,8 +10,8 @@ export type JsonValue = string | number | boolean | null | JsonValue[] | { [key:
|
|
|
10
10
|
* Return all provider configs from the registry.
|
|
11
11
|
*
|
|
12
12
|
* Useful for tooling, documentation generation, or runtime enumeration.
|
|
13
|
-
* Returns the public
|
|
14
|
-
* To query capability flags for a specific provider use
|
|
13
|
+
* Returns the public `ProviderConfig` slice (without capability flags).
|
|
14
|
+
* To query capability flags for a specific provider use `capabilities`.
|
|
15
15
|
*/
|
|
16
16
|
export declare function allProviders(): Array<ProviderConfig>;
|
|
17
17
|
|
|
@@ -31,8 +31,8 @@ export declare function capabilities(providerName: string): ProviderCapabilities
|
|
|
31
31
|
* Assert that `current_len + incoming` does not exceed `limit`.
|
|
32
32
|
*
|
|
33
33
|
* Call this before appending `incoming` bytes to any buffer that must
|
|
34
|
-
* stay below `limit`. Returns `Err(LiterLlmError
|
|
35
|
-
* and emits a `tracing
|
|
34
|
+
* stay below `limit`. Returns `Err(LiterLlmError.Streaming)` on overflow
|
|
35
|
+
* and emits a `tracing.warn!` with context.
|
|
36
36
|
*/
|
|
37
37
|
export declare function checkBound(context: string, currentLen: number, incoming: number, limit: number): void;
|
|
38
38
|
|
|
@@ -40,6 +40,9 @@ export declare function checkBound(context: string, currentLen: number, incoming
|
|
|
40
40
|
* Remove all guardrails from the global registry.
|
|
41
41
|
*
|
|
42
42
|
* Primarily useful in tests to reset state between test cases.
|
|
43
|
+
*
|
|
44
|
+
* If the lock was poisoned by a panicking guardrail on a previous access,
|
|
45
|
+
* the poisoned state is recovered rather than propagating the panic.
|
|
43
46
|
*/
|
|
44
47
|
export declare function clear(): void;
|
|
45
48
|
|
|
@@ -48,7 +51,7 @@ export declare function clear(): void;
|
|
|
48
51
|
* `completion_cost_with_cache`, and `model_info` to the
|
|
49
52
|
* embedded catalog.
|
|
50
53
|
*
|
|
51
|
-
* Primarily a test seam (see
|
|
54
|
+
* Primarily a test seam (see `install_catalog_overlay_from_str`); also
|
|
52
55
|
* usable by long-running processes that want to abandon a runtime refresh.
|
|
53
56
|
*/
|
|
54
57
|
export declare function clearCatalogOverlay(): void;
|
|
@@ -78,9 +81,9 @@ export declare function completionCost(model: string, promptTokens: number, comp
|
|
|
78
81
|
* input rate.
|
|
79
82
|
*
|
|
80
83
|
* Returns `None` if the model is not present in the embedded pricing
|
|
81
|
-
* registry, mirroring
|
|
84
|
+
* registry, mirroring `completion_cost`.
|
|
82
85
|
*
|
|
83
|
-
* When the model has
|
|
86
|
+
* When the model has `ModelPricing.tiers`, the tier whose
|
|
84
87
|
* `min_context_tokens` is the highest value `<= prompt_tokens` supplies the
|
|
85
88
|
* input/output/cache rates for the whole call; models without tiers (or
|
|
86
89
|
* when `prompt_tokens` is below every tier threshold) use the base rates
|
|
@@ -99,13 +102,13 @@ export declare function completionCostWithCache(model: string, promptTokens: num
|
|
|
99
102
|
export declare function complexProviderNames(): Array<string>;
|
|
100
103
|
|
|
101
104
|
/**
|
|
102
|
-
* Count tokens for a full
|
|
105
|
+
* Count tokens for a full `ChatCompletionRequest`.
|
|
103
106
|
*
|
|
104
107
|
* Sums tokens across all message text contents plus a per-message overhead
|
|
105
108
|
* of ~4 tokens (for role, separators, and formatting metadata). Tool
|
|
106
109
|
* definitions and multimodal content parts (images, audio, documents) are
|
|
107
110
|
* not counted — only textual content contributes to the token total.
|
|
108
|
-
* @throws Returns
|
|
111
|
+
* @throws Returns `LiterLlmError.BadRequest` if the tokenizer cannot be loaded or
|
|
109
112
|
* if tokenization fails for any message.
|
|
110
113
|
*/
|
|
111
114
|
export declare function countRequestTokens(model: string, req?: ChatCompletionRequest | undefined | null): number;
|
|
@@ -116,7 +119,7 @@ export declare function countRequestTokens(model: string, req?: ChatCompletionRe
|
|
|
116
119
|
* The tokenizer is resolved from the model name prefix (e.g. `"gpt-4o"` maps
|
|
117
120
|
* to the `Xenova/gpt-4o` HuggingFace tokenizer). Tokenizers are cached after
|
|
118
121
|
* first load.
|
|
119
|
-
* @throws Returns
|
|
122
|
+
* @throws Returns `LiterLlmError.BadRequest` if the tokenizer cannot be loaded
|
|
120
123
|
* (e.g. network failure on first use) or if tokenization itself fails.
|
|
121
124
|
*/
|
|
122
125
|
export declare function countTokens(model: string, text: string): number;
|
|
@@ -126,8 +129,8 @@ export declare function countTokens(model: string, text: string): number;
|
|
|
126
129
|
*
|
|
127
130
|
* This is the primary binding entry-point. All parameters except `api_key`
|
|
128
131
|
* are optional — omitting them uses the same defaults as
|
|
129
|
-
*
|
|
130
|
-
* @throws Returns
|
|
132
|
+
* `ClientConfigBuilder`.
|
|
133
|
+
* @throws Returns `LiterLlmError` if the underlying HTTP client cannot be
|
|
131
134
|
* constructed, or if the resolved provider configuration is invalid.
|
|
132
135
|
*/
|
|
133
136
|
export declare function createClient(apiKey: string, baseUrl?: string | undefined | null, timeoutSecs?: number | undefined | null, maxRetries?: number | undefined | null, modelHint?: string | undefined | null): DefaultClient;
|
|
@@ -136,13 +139,13 @@ export declare function createClient(apiKey: string, baseUrl?: string | undefine
|
|
|
136
139
|
* Create a new LLM client from a JSON string.
|
|
137
140
|
*
|
|
138
141
|
* The JSON object accepts the same fields as `liter-llm.toml` (snake_case).
|
|
139
|
-
* @throws Returns
|
|
142
|
+
* @throws Returns `LiterLlmError.BadRequest` if `json` is not valid JSON or
|
|
140
143
|
* contains unknown fields.
|
|
141
144
|
*/
|
|
142
145
|
export declare function createClientFromJson(json: string): DefaultClient;
|
|
143
146
|
|
|
144
147
|
/**
|
|
145
|
-
* Decode a base64 data URL into
|
|
148
|
+
* Decode a base64 data URL into `DecodedDataUrl`.
|
|
146
149
|
*
|
|
147
150
|
* Returns `None` for:
|
|
148
151
|
* - Non-data URLs (strings that do not start with `"data:"`).
|
|
@@ -157,7 +160,7 @@ export declare function decodeDataUrl(url: string): DecodedDataUrl | null;
|
|
|
157
160
|
/**
|
|
158
161
|
* Encode bytes as a base64 data URL: `data:<mime>;base64,<b64>`.
|
|
159
162
|
*
|
|
160
|
-
* `mime` defaults to
|
|
163
|
+
* `mime` defaults to `IMAGE_PNG` when `None`.
|
|
161
164
|
*/
|
|
162
165
|
export declare function encodeDataUrl(bytes: Uint8Array, mime?: string | undefined | null): string;
|
|
163
166
|
|
|
@@ -169,7 +172,7 @@ export declare function encodeDataUrl(bytes: Uint8Array, mime?: string | undefin
|
|
|
169
172
|
* another rustls crypto provider has already been installed is safe: the
|
|
170
173
|
* `Err` from `install_default()` is silently ignored.
|
|
171
174
|
*
|
|
172
|
-
* Called automatically by every internal `reqwest
|
|
175
|
+
* Called automatically by every internal `reqwest.Client` constructor
|
|
173
176
|
* (auth providers, default HTTP client). Bindings and downstream consumers
|
|
174
177
|
* reach those constructors transitively, so no manual init is required.
|
|
175
178
|
*
|
|
@@ -186,9 +189,9 @@ export declare function ensureCryptoProvider(): void;
|
|
|
186
189
|
* the network and disk cache entirely.
|
|
187
190
|
*
|
|
188
191
|
* Parses and flattens `catalog_json` with the same
|
|
189
|
-
*
|
|
192
|
+
* `registry_from_catalog_str` logic used for the embedded catalog and the
|
|
190
193
|
* network refresh path, then atomically swaps it in as the active overlay.
|
|
191
|
-
* A parse failure returns
|
|
194
|
+
* A parse failure returns `CatalogRefreshError.Parse` and leaves any
|
|
192
195
|
* existing overlay untouched.
|
|
193
196
|
*
|
|
194
197
|
* This is primarily a testable seam: it lets tests exercise overlay
|
|
@@ -202,16 +205,13 @@ export declare function installCatalogOverlayFromStr(catalogJson: string): void;
|
|
|
202
205
|
* Content shape for assistant messages.
|
|
203
206
|
*
|
|
204
207
|
* `#[serde(untagged)]` means providers returning a plain scalar string for the
|
|
205
|
-
* `content` field still deserialise correctly into `AssistantContent
|
|
208
|
+
* `content` field still deserialise correctly into `AssistantContent.Text(_)`.
|
|
206
209
|
* Providers returning an array of typed parts (e.g. after an image-generation
|
|
207
|
-
* or audio-synthesis request) deserialise into `AssistantContent
|
|
210
|
+
* or audio-synthesis request) deserialise into `AssistantContent.Parts(_)`.
|
|
208
211
|
*/
|
|
209
|
-
export
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
/** Structured parts — text, refusals, output images, output audio. */
|
|
213
|
-
Parts = "Parts",
|
|
214
|
-
}
|
|
212
|
+
export type AssistantContent =
|
|
213
|
+
| string
|
|
214
|
+
| Array<AssistantPart>
|
|
215
215
|
|
|
216
216
|
/** Assistant's response to a user message. */
|
|
217
217
|
export interface AssistantMessage {
|
|
@@ -225,7 +225,12 @@ export interface AssistantMessage {
|
|
|
225
225
|
readonly name?: string
|
|
226
226
|
/** Tool calls the model wants to execute, if any. */
|
|
227
227
|
readonly toolCalls?: Array<ToolCall>
|
|
228
|
-
/**
|
|
228
|
+
/**
|
|
229
|
+
* Refusal reason, if the model declined to respond per safety policies.
|
|
230
|
+
*
|
|
231
|
+
* OpenAI's response schema requires this key to be present even when null,
|
|
232
|
+
* so it is deliberately not `skip_serializing_if`.
|
|
233
|
+
*/
|
|
229
234
|
readonly refusal?: string
|
|
230
235
|
/** Deprecated legacy function_call field; retained for API compatibility. */
|
|
231
236
|
readonly functionCall?: FunctionCall
|
|
@@ -248,7 +253,12 @@ export type AssistantPart =
|
|
|
248
253
|
| { type: 'output_image'; imageUrl: ImageUrl }
|
|
249
254
|
| { type: 'output_audio'; audio: AudioContent }
|
|
250
255
|
|
|
251
|
-
/**
|
|
256
|
+
/**
|
|
257
|
+
* Audio content part for speech-capable models.
|
|
258
|
+
*
|
|
259
|
+
* No `deny_unknown_fields`: shared with the response side (see
|
|
260
|
+
* `AssistantPart.OutputAudio`), same rationale as `ImageUrl` (#51).
|
|
261
|
+
*/
|
|
252
262
|
export interface AudioContent {
|
|
253
263
|
/** Base64-encoded audio data. */
|
|
254
264
|
readonly data?: string
|
|
@@ -380,6 +390,9 @@ export declare enum BatchStatus {
|
|
|
380
390
|
* AWS environment variables (`AWS_DEFAULT_REGION` / `AWS_REGION`,
|
|
381
391
|
* `AWS_ACCESS_KEY_ID`, `AWS_SECRET_ACCESS_KEY`, `AWS_SESSION_TOKEN`,
|
|
382
392
|
* `BEDROCK_CROSS_REGION`).
|
|
393
|
+
*
|
|
394
|
+
* Implements `Debug` manually (see below) so the AWS credential fields are
|
|
395
|
+
* redacted rather than printed in full.
|
|
383
396
|
*/
|
|
384
397
|
export interface BedrockConfig {
|
|
385
398
|
/** AWS region (e.g. `"us-east-1"`). */
|
|
@@ -423,7 +436,7 @@ export interface CacheConfig {
|
|
|
423
436
|
}
|
|
424
437
|
|
|
425
438
|
/**
|
|
426
|
-
* Plain-data configuration for
|
|
439
|
+
* Plain-data configuration for `refresh_catalog`.
|
|
427
440
|
*
|
|
428
441
|
* Deliberately FFI/binding-friendly: no `Duration` or `PathBuf`, just
|
|
429
442
|
* primitives that translate directly across language boundaries.
|
|
@@ -431,14 +444,14 @@ export interface CacheConfig {
|
|
|
431
444
|
export interface CatalogRefreshConfig {
|
|
432
445
|
/**
|
|
433
446
|
* Runtime catalog refresh is entirely opt-in: when `false`,
|
|
434
|
-
*
|
|
435
|
-
* `Ok(
|
|
447
|
+
* `refresh_catalog` is a no-op that returns
|
|
448
|
+
* `Ok(``RefreshOutcome.Disabled``)` without touching the network,
|
|
436
449
|
* the filesystem, or the overlay registry.
|
|
437
450
|
*/
|
|
438
451
|
readonly enabled?: boolean
|
|
439
452
|
/**
|
|
440
453
|
* Source URL to fetch `catalog.json` from. Must be `https`. Defaults to
|
|
441
|
-
*
|
|
454
|
+
* `DEFAULT_CATALOG_URL`; configurable so self-hosted mirrors work.
|
|
442
455
|
*/
|
|
443
456
|
readonly sourceUrl?: string
|
|
444
457
|
/**
|
|
@@ -448,7 +461,7 @@ export interface CatalogRefreshConfig {
|
|
|
448
461
|
readonly ttlSeconds?: number
|
|
449
462
|
/**
|
|
450
463
|
* Filesystem path for the on-disk cache. `None` uses a default path
|
|
451
|
-
* under `std
|
|
464
|
+
* under `std.env.temp_dir()`.
|
|
452
465
|
*/
|
|
453
466
|
readonly cachePath?: string
|
|
454
467
|
}
|
|
@@ -482,9 +495,26 @@ export interface ChatCompletionRequest {
|
|
|
482
495
|
readonly model?: string
|
|
483
496
|
/** Conversation history from oldest to newest. */
|
|
484
497
|
readonly messages?: Array<Message>
|
|
485
|
-
/**
|
|
498
|
+
/**
|
|
499
|
+
* Sampling temperature. Higher increases randomness, lower is more deterministic.
|
|
500
|
+
* Defaults to 1.0.
|
|
501
|
+
*
|
|
502
|
+
* The accepted range depends on the provider the request is routed to. OpenAI-compatible
|
|
503
|
+
* providers accept `[0.0, 2.0]`; Anthropic and Amazon Bedrock both cap it at `1.0`, and
|
|
504
|
+
* for those two a value above the cap is rejected with a `BadRequest` error before the
|
|
505
|
+
* request is sent, rather than being silently clamped or left for the provider to reject.
|
|
506
|
+
*
|
|
507
|
+
* No range is enforced for providers whose own documentation does not state one — the
|
|
508
|
+
* value is forwarded and the provider decides. Consult the target provider's reference
|
|
509
|
+
* rather than assuming `[0.0, 2.0]` is portable.
|
|
510
|
+
*/
|
|
486
511
|
readonly temperature?: number
|
|
487
|
-
/**
|
|
512
|
+
/**
|
|
513
|
+
* Nucleus sampling parameter. Lower is more focused.
|
|
514
|
+
*
|
|
515
|
+
* Accepted ranges vary by provider (most document `[0.0, 1.0]`, but this is not
|
|
516
|
+
* universal — check the target provider's own documentation for its exact bounds).
|
|
517
|
+
*/
|
|
488
518
|
readonly topP?: number
|
|
489
519
|
/** Number of chat completions to generate. Defaults to 1. */
|
|
490
520
|
readonly n?: number
|
|
@@ -530,6 +560,49 @@ export interface ChatCompletionRequest {
|
|
|
530
560
|
* translates these to `generationConfig.responseModalities` (uppercase).
|
|
531
561
|
*/
|
|
532
562
|
readonly modalities?: Array<Modality>
|
|
563
|
+
/** Whether to return log probabilities of the output tokens. */
|
|
564
|
+
readonly logprobs?: boolean
|
|
565
|
+
/**
|
|
566
|
+
* Number of most-likely tokens to return log probabilities for, `0..=20`.
|
|
567
|
+
* Requires `logprobs` to be `true`.
|
|
568
|
+
*/
|
|
569
|
+
readonly topLogprobs?: number
|
|
570
|
+
/**
|
|
571
|
+
* Upper bound on generated tokens, including reasoning tokens.
|
|
572
|
+
*
|
|
573
|
+
* Supersedes `max_tokens` on OpenAI reasoning models, which reject
|
|
574
|
+
* `max_tokens` outright.
|
|
575
|
+
*/
|
|
576
|
+
readonly maxCompletionTokens?: number
|
|
577
|
+
/**
|
|
578
|
+
* Latency tier to process the request under (e.g. `"auto"`, `"default"`,
|
|
579
|
+
* `"flex"`).
|
|
580
|
+
*/
|
|
581
|
+
readonly serviceTier?: string
|
|
582
|
+
/** Whether to store the completion for later retrieval by the provider. */
|
|
583
|
+
readonly store?: boolean
|
|
584
|
+
/** Developer-defined tags attached to the completion. */
|
|
585
|
+
readonly metadata?: Record<string, string>
|
|
586
|
+
/**
|
|
587
|
+
* Predicted output, for latency reduction when much of the response is
|
|
588
|
+
* known ahead of time.
|
|
589
|
+
*
|
|
590
|
+
* Untyped: the shape is provider-defined and still evolving, and the
|
|
591
|
+
* value is forwarded verbatim.
|
|
592
|
+
*/
|
|
593
|
+
readonly prediction?: JsonValue
|
|
594
|
+
/**
|
|
595
|
+
* Audio output parameters, required when `modalities` includes `audio`.
|
|
596
|
+
*
|
|
597
|
+
* Untyped for the same reason as `prediction`.
|
|
598
|
+
*/
|
|
599
|
+
readonly audio?: JsonValue
|
|
600
|
+
/**
|
|
601
|
+
* Web-search tool configuration for search-enabled models.
|
|
602
|
+
*
|
|
603
|
+
* Untyped for the same reason as `prediction`.
|
|
604
|
+
*/
|
|
605
|
+
readonly webSearchOptions?: JsonValue
|
|
533
606
|
/**
|
|
534
607
|
* Provider-specific extra parameters merged into the request body.
|
|
535
608
|
* Use for guardrails, safety settings, grounding config, etc.
|
|
@@ -572,21 +645,35 @@ export interface ChatCompletionTool {
|
|
|
572
645
|
export interface Choice {
|
|
573
646
|
/** Index of this choice in the choices array. */
|
|
574
647
|
readonly index?: number
|
|
575
|
-
/**
|
|
648
|
+
/**
|
|
649
|
+
* The assistant's message response.
|
|
650
|
+
*
|
|
651
|
+
* Serialized with an explicit `role: "assistant"`. The field is not stored
|
|
652
|
+
* on `AssistantMessage` because `Message` is an internally-tagged enum
|
|
653
|
+
* keyed on `role`, so a stored field would emit the key twice inside a
|
|
654
|
+
* request. OpenAI's response schema requires it here.
|
|
655
|
+
*/
|
|
576
656
|
readonly message?: AssistantMessage
|
|
577
657
|
/** Why the model stopped generating (stop, length, tool_calls, content_filter, etc.). */
|
|
578
658
|
readonly finishReason?: FinishReason
|
|
659
|
+
/**
|
|
660
|
+
* Per-token log probabilities, when the request asked for them.
|
|
661
|
+
*
|
|
662
|
+
* Required by OpenAI's response schema as an always-present, nullable key,
|
|
663
|
+
* so this is deliberately not `skip_serializing_if`.
|
|
664
|
+
*/
|
|
665
|
+
readonly logprobs?: JsonValue
|
|
579
666
|
}
|
|
580
667
|
|
|
581
668
|
/**
|
|
582
|
-
* A per-chunk transformation in the
|
|
669
|
+
* A per-chunk transformation in the `StreamPipeline`.
|
|
583
670
|
*
|
|
584
671
|
* Each middleware receives a typed chunk and returns `Ok(Some(chunk))`
|
|
585
672
|
* to pass it through (optionally modified), `Ok(None)` to drop the chunk,
|
|
586
673
|
* or `Err(e)` to propagate a stream error.
|
|
587
674
|
*
|
|
588
675
|
* The trait is object-safe so multiple middleware implementations can be
|
|
589
|
-
* chained inside
|
|
676
|
+
* chained inside `StreamPipeline`.
|
|
590
677
|
*/
|
|
591
678
|
export interface ChunkMiddleware {
|
|
592
679
|
/**
|
|
@@ -658,9 +745,35 @@ export interface CreateImageRequest {
|
|
|
658
745
|
readonly user?: string
|
|
659
746
|
}
|
|
660
747
|
|
|
661
|
-
/**
|
|
748
|
+
/**
|
|
749
|
+
* Request to create a response via the OpenAI Responses API (`POST /responses`).
|
|
750
|
+
*
|
|
751
|
+
* # Provider support
|
|
752
|
+
*
|
|
753
|
+
* **The Responses API path is OpenAI-only.** This type models the OpenAI Responses
|
|
754
|
+
* wire format, and unlike `ChatCompletionRequest` the body is sent to the provider
|
|
755
|
+
* verbatim: neither `Provider.transform_request` nor `Provider.transform_response`
|
|
756
|
+
* runs for Responses calls, and no provider in `schemas/providers.json` declares a
|
|
757
|
+
* `responses` endpoint.
|
|
758
|
+
*
|
|
759
|
+
* Pointing a Responses call at a provider that does not natively serve the OpenAI
|
|
760
|
+
* `/responses` contract (Anthropic, Vertex, Bedrock, Cohere, Google AI, Azure) is not
|
|
761
|
+
* supported: the request goes out unmodified, so the provider rejects it or returns a
|
|
762
|
+
* body that cannot be deserialized into `ResponseObject`. Use
|
|
763
|
+
* `ChatCompletionRequest` for cross-provider work — that path applies the
|
|
764
|
+
* per-provider request and response normalization this one does not.
|
|
765
|
+
*
|
|
766
|
+
* `ChatCompletionRequest`: crate.types.ChatCompletionRequest
|
|
767
|
+
*/
|
|
662
768
|
export interface CreateResponseRequest {
|
|
663
|
-
/**
|
|
769
|
+
/**
|
|
770
|
+
* Model ID, as named by the OpenAI Responses API (e.g. `"gpt-5"`).
|
|
771
|
+
*
|
|
772
|
+
* Sent to the wire exactly as given. The Responses path performs none of the
|
|
773
|
+
* chat path's model handling: a `provider/model` routing prefix is **not**
|
|
774
|
+
* stripped and does **not** re-route the request, which stays pinned to the
|
|
775
|
+
* provider the client was constructed with.
|
|
776
|
+
*/
|
|
664
777
|
readonly model?: string
|
|
665
778
|
/** Input data to process (e.g., a document to extract from). */
|
|
666
779
|
readonly input?: JsonValue
|
|
@@ -674,6 +787,21 @@ export interface CreateResponseRequest {
|
|
|
674
787
|
readonly maxOutputTokens?: number
|
|
675
788
|
/** Optional metadata. */
|
|
676
789
|
readonly metadata?: JsonValue
|
|
790
|
+
/**
|
|
791
|
+
* Extra top-level parameters shallow-merged into the request body, OpenAI-Python
|
|
792
|
+
* style (`{**body, **extra_body}`) — keys here override identically named fields
|
|
793
|
+
* above. Use it for OpenAI Responses fields this struct does not model directly,
|
|
794
|
+
* such as the top-level `reasoning.effort`.
|
|
795
|
+
*
|
|
796
|
+
* This is an OpenAI escape hatch, not a cross-provider one. On the chat path the
|
|
797
|
+
* providers that consume `extra_body` natively (Anthropic, Vertex, Bedrock) claim
|
|
798
|
+
* it inside their own `transform_request`; here no provider transform runs, so the
|
|
799
|
+
* merged keys always travel to the wire as literal OpenAI Responses fields.
|
|
800
|
+
*
|
|
801
|
+
* A non-object value cannot be merged into the body root and is dropped with a
|
|
802
|
+
* warning rather than sent.
|
|
803
|
+
*/
|
|
804
|
+
readonly extraBody?: JsonValue
|
|
677
805
|
/**
|
|
678
806
|
* Whether to stream the response.
|
|
679
807
|
*
|
|
@@ -750,7 +878,7 @@ export interface DecodedDataUrl {
|
|
|
750
878
|
* provider is used as the fallback. This enables seamless migration between
|
|
751
879
|
* providers by changing only the model name.
|
|
752
880
|
*
|
|
753
|
-
* The provider is stored behind an
|
|
881
|
+
* The provider is stored behind an `Arc` so it can be shared cheaply into
|
|
754
882
|
* async closures and streaming tasks. Pre-computed auth headers and extra
|
|
755
883
|
* headers are cached at construction to avoid redundant encoding on every request.
|
|
756
884
|
*/
|
|
@@ -781,9 +909,9 @@ export declare class DefaultClient {
|
|
|
781
909
|
*
|
|
782
910
|
* Uses exponential backoff with configurable initial interval, maximum interval, and backoff multiplier.
|
|
783
911
|
* Optionally supports a timeout that aborts polling if exceeded.
|
|
784
|
-
* @throws Returns `BatchWaitError
|
|
785
|
-
* Returns `BatchWaitError
|
|
786
|
-
* Returns `BatchWaitError
|
|
912
|
+
* @throws Returns `BatchWaitError.Failed` if the batch reaches a failure terminal status.
|
|
913
|
+
* Returns `BatchWaitError.Timeout` if the configured timeout is exceeded.
|
|
914
|
+
* Returns `BatchWaitError.Client` for underlying client errors.
|
|
787
915
|
*/
|
|
788
916
|
waitForBatch(batchId: string, config?: WaitForBatchConfig | undefined | null): Promise<BatchObject>
|
|
789
917
|
createResponse(req?: CreateResponseRequest | undefined | null): Promise<ResponseObject>
|
|
@@ -817,6 +945,12 @@ export interface DocumentContent {
|
|
|
817
945
|
readonly mediaType?: string
|
|
818
946
|
}
|
|
819
947
|
|
|
948
|
+
/** A content part in a multimodal embedding input. */
|
|
949
|
+
export type EmbeddingContentPart =
|
|
950
|
+
| { type: 'text'; text: string }
|
|
951
|
+
| { type: 'image_url'; imageUrl: ImageUrl }
|
|
952
|
+
| { type: 'image_base64'; imageBase64: string }
|
|
953
|
+
|
|
820
954
|
/** The format in which the embedding vectors are returned. */
|
|
821
955
|
export declare enum EmbeddingFormat {
|
|
822
956
|
/** 32-bit floating-point numbers (default). */
|
|
@@ -825,13 +959,11 @@ export declare enum EmbeddingFormat {
|
|
|
825
959
|
Base64 = "base64",
|
|
826
960
|
}
|
|
827
961
|
|
|
828
|
-
/** Text or
|
|
829
|
-
export
|
|
830
|
-
|
|
831
|
-
|
|
832
|
-
|
|
833
|
-
Multiple = "Multiple",
|
|
834
|
-
}
|
|
962
|
+
/** Text, texts, or multimodal content to embed. */
|
|
963
|
+
export type EmbeddingInput =
|
|
964
|
+
| string
|
|
965
|
+
| Array<string>
|
|
966
|
+
| Array<EmbeddingContentPart>
|
|
835
967
|
|
|
836
968
|
/** A single embedding vector. */
|
|
837
969
|
export interface EmbeddingObject {
|
|
@@ -857,7 +989,7 @@ export interface EmbeddingObject {
|
|
|
857
989
|
export interface EmbeddingRequest {
|
|
858
990
|
/** Model ID (e.g., `"text-embedding-3-small"`). */
|
|
859
991
|
readonly model?: string
|
|
860
|
-
/** Text or
|
|
992
|
+
/** Text, texts, or multimodal content to embed. */
|
|
861
993
|
readonly input?: EmbeddingInput
|
|
862
994
|
/** Output format: float (native) or base64. */
|
|
863
995
|
readonly encodingFormat?: EmbeddingFormat
|
|
@@ -886,11 +1018,11 @@ export interface EmbeddingResponse {
|
|
|
886
1018
|
export declare enum Enforcement {
|
|
887
1019
|
/**
|
|
888
1020
|
* Reject requests that would exceed the budget with
|
|
889
|
-
*
|
|
1021
|
+
* `LiterLlmError.BudgetExceeded`.
|
|
890
1022
|
*/
|
|
891
1023
|
Hard = "Hard",
|
|
892
1024
|
/**
|
|
893
|
-
* Allow requests through but emit a `tracing
|
|
1025
|
+
* Allow requests through but emit a `tracing.warn!` when the budget is
|
|
894
1026
|
* exceeded.
|
|
895
1027
|
*/
|
|
896
1028
|
Soft = "Soft",
|
|
@@ -970,7 +1102,7 @@ export declare enum FinishReason {
|
|
|
970
1102
|
export interface FunctionCall {
|
|
971
1103
|
/** Function name. */
|
|
972
1104
|
readonly name: string
|
|
973
|
-
/** Arguments as a JSON string (parse with serde_json
|
|
1105
|
+
/** Arguments as a JSON string (parse with serde_json.from_str). */
|
|
974
1106
|
readonly arguments: string
|
|
975
1107
|
}
|
|
976
1108
|
|
|
@@ -996,11 +1128,11 @@ export interface FunctionMessage {
|
|
|
996
1128
|
* Abstraction over a health probe strategy.
|
|
997
1129
|
*
|
|
998
1130
|
* Implementors issue a lightweight probe against `upstream` (typically a
|
|
999
|
-
* provider base URL or named identifier) and report
|
|
1131
|
+
* provider base URL or named identifier) and report `HealthStatus`.
|
|
1000
1132
|
*/
|
|
1001
1133
|
export interface HealthChecker {
|
|
1002
1134
|
/**
|
|
1003
|
-
* Probe `upstream` and return its current
|
|
1135
|
+
* Probe `upstream` and return its current `HealthStatus`.
|
|
1004
1136
|
*
|
|
1005
1137
|
* The parameter is taken by value (`String`) so that implementations can
|
|
1006
1138
|
* move it into the returned future without a clone, making the
|
|
@@ -1045,7 +1177,15 @@ export interface ImagesResponse {
|
|
|
1045
1177
|
readonly data?: Array<Image>
|
|
1046
1178
|
}
|
|
1047
1179
|
|
|
1048
|
-
/**
|
|
1180
|
+
/**
|
|
1181
|
+
* An image URL reference with optional detail level for processing.
|
|
1182
|
+
*
|
|
1183
|
+
* No `deny_unknown_fields`: this type is shared with the response side
|
|
1184
|
+
* (see `AssistantPart.OutputImage`) where it is deserialized from
|
|
1185
|
+
* provider output, not just constructed as request input (see #51). A
|
|
1186
|
+
* provider adding a new field to its image-output object must not hard-fail
|
|
1187
|
+
* the whole response.
|
|
1188
|
+
*/
|
|
1049
1189
|
export interface ImageUrl {
|
|
1050
1190
|
/** URL of the image (data URI or HTTP/HTTPS URL). */
|
|
1051
1191
|
readonly url?: string
|
|
@@ -1053,6 +1193,12 @@ export interface ImageUrl {
|
|
|
1053
1193
|
readonly detail?: ImageDetail
|
|
1054
1194
|
}
|
|
1055
1195
|
|
|
1196
|
+
/** Configuration for the global per-client in-flight request limit. */
|
|
1197
|
+
export interface InFlightLimitConfig {
|
|
1198
|
+
/** Maximum simultaneously outstanding provider requests. `None` means unlimited. */
|
|
1199
|
+
readonly maxInFlight?: number
|
|
1200
|
+
}
|
|
1201
|
+
|
|
1056
1202
|
/** An intent prototype: `(intent_name, prototype_embedding, target_model_id)`. */
|
|
1057
1203
|
export interface IntentPrototype {
|
|
1058
1204
|
/** Human-readable name for the intent (used in logs/metrics). */
|
|
@@ -1103,12 +1249,15 @@ export interface LlmCacheConfig {
|
|
|
1103
1249
|
* All fields except `model` are optional so that partially-specified
|
|
1104
1250
|
* configs (e.g. from environment-driven defaults) round-trip cleanly.
|
|
1105
1251
|
* Convert to a runtime client configuration via
|
|
1106
|
-
*
|
|
1252
|
+
* `LlmConfig.into_client_builder`.
|
|
1107
1253
|
*
|
|
1108
1254
|
* `temperature` and `max_tokens` are request-time parameters rather than
|
|
1109
1255
|
* client-level settings; they are carried on this struct for callers to
|
|
1110
1256
|
* read when building individual requests, and are intentionally **not**
|
|
1111
|
-
* mapped by
|
|
1257
|
+
* mapped by `LlmConfig.into_client_builder`.
|
|
1258
|
+
*
|
|
1259
|
+
* Implements `Debug` manually (see below) so `api_key` and header values are
|
|
1260
|
+
* redacted rather than printed in full.
|
|
1112
1261
|
*/
|
|
1113
1262
|
export interface LlmConfig {
|
|
1114
1263
|
/** Model identifier (e.g. `"gpt-4o"`, `"bedrock/anthropic.claude-3-sonnet-20240229-v1:0"`). */
|
|
@@ -1143,6 +1292,8 @@ export interface LlmConfig {
|
|
|
1143
1292
|
readonly budget?: LlmBudgetConfig
|
|
1144
1293
|
/** Per-model rate limiting configuration. */
|
|
1145
1294
|
readonly rateLimit?: LlmRateLimitConfig
|
|
1295
|
+
/** Global per-client in-flight provider request limit. */
|
|
1296
|
+
readonly inFlightLimit?: LlmInFlightLimitConfig
|
|
1146
1297
|
/** Enable per-request cost tracking. */
|
|
1147
1298
|
readonly costTracking?: boolean
|
|
1148
1299
|
/** Enable OpenTelemetry-compatible tracing spans. */
|
|
@@ -1155,6 +1306,12 @@ export interface LlmConfig {
|
|
|
1155
1306
|
readonly bedrock?: BedrockConfig
|
|
1156
1307
|
}
|
|
1157
1308
|
|
|
1309
|
+
/** Global per-client in-flight provider request limit. */
|
|
1310
|
+
export interface LlmInFlightLimitConfig {
|
|
1311
|
+
/** Maximum simultaneously outstanding provider requests. `None` means unlimited. */
|
|
1312
|
+
readonly maxInFlight?: number
|
|
1313
|
+
}
|
|
1314
|
+
|
|
1158
1315
|
/** A custom provider configuration entry. */
|
|
1159
1316
|
export interface LlmProviderConfig {
|
|
1160
1317
|
/** Provider name, used to key model prefix matching. */
|
|
@@ -1179,12 +1336,12 @@ export interface LlmRateLimitConfig {
|
|
|
1179
1336
|
|
|
1180
1337
|
/** A chat message in a conversation. */
|
|
1181
1338
|
export type Message =
|
|
1182
|
-
| { role: 'system';
|
|
1183
|
-
| { role: 'user';
|
|
1184
|
-
| { role: 'assistant';
|
|
1185
|
-
| { role: 'tool';
|
|
1186
|
-
| { role: 'developer';
|
|
1187
|
-
| { role: 'function';
|
|
1339
|
+
| { role: 'system'; system: SystemMessage }
|
|
1340
|
+
| { role: 'user'; user: UserMessage }
|
|
1341
|
+
| { role: 'assistant'; assistant: AssistantMessage }
|
|
1342
|
+
| { role: 'tool'; tool: ToolMessage }
|
|
1343
|
+
| { role: 'developer'; developer: DeveloperMessage }
|
|
1344
|
+
| { role: 'function'; function: FunctionMessage }
|
|
1188
1345
|
|
|
1189
1346
|
/**
|
|
1190
1347
|
* Output modality requested from the model.
|
|
@@ -1203,11 +1360,11 @@ export declare enum Modality {
|
|
|
1203
1360
|
|
|
1204
1361
|
/**
|
|
1205
1362
|
* Public, FFI-friendly snapshot of a model's pricing and capability
|
|
1206
|
-
* metadata, projected from
|
|
1363
|
+
* metadata, projected from `ModelPricing`.
|
|
1207
1364
|
*
|
|
1208
|
-
* Unlike
|
|
1365
|
+
* Unlike `ModelPricing` (which is excluded from binding generation),
|
|
1209
1366
|
* `ModelInfo` is an owned plain-data DTO safe to hand across the FFI
|
|
1210
|
-
* boundary — see
|
|
1367
|
+
* boundary — see `model_info`.
|
|
1211
1368
|
*/
|
|
1212
1369
|
export interface ModelInfo {
|
|
1213
1370
|
/** Cost in USD per input (prompt) token. */
|
|
@@ -1291,7 +1448,7 @@ export interface ModelsListResponse {
|
|
|
1291
1448
|
|
|
1292
1449
|
/**
|
|
1293
1450
|
* Public, FFI-friendly snapshot of a single context-window pricing tier,
|
|
1294
|
-
* projected from
|
|
1451
|
+
* projected from `PricingTier`.
|
|
1295
1452
|
*/
|
|
1296
1453
|
export interface ModelTier {
|
|
1297
1454
|
/**
|
|
@@ -1368,12 +1525,9 @@ export interface ModerationCategoryScores {
|
|
|
1368
1525
|
}
|
|
1369
1526
|
|
|
1370
1527
|
/** Input to the moderation endpoint — a single string or multiple strings. */
|
|
1371
|
-
export
|
|
1372
|
-
|
|
1373
|
-
|
|
1374
|
-
/** Multiple text strings (batch moderation). */
|
|
1375
|
-
Multiple = "Multiple",
|
|
1376
|
-
}
|
|
1528
|
+
export type ModerationInput =
|
|
1529
|
+
| string
|
|
1530
|
+
| Array<string>
|
|
1377
1531
|
|
|
1378
1532
|
/** Request to classify content for policy violations. */
|
|
1379
1533
|
export interface ModerationRequest {
|
|
@@ -1461,7 +1615,7 @@ export interface PageDimensions {
|
|
|
1461
1615
|
/**
|
|
1462
1616
|
* Breakdown of tokens used in the prompt portion of a request.
|
|
1463
1617
|
*
|
|
1464
|
-
* `cached_tokens` is included in `Usage
|
|
1618
|
+
* `cached_tokens` is included in `Usage.prompt_tokens` — it is *not* an
|
|
1465
1619
|
* additional charge on top of the prompt token count. When pricing supports
|
|
1466
1620
|
* a `cache_read_input_token_cost`, the cached portion is billed at the
|
|
1467
1621
|
* discounted rate and the remainder at the regular input rate.
|
|
@@ -1484,19 +1638,7 @@ export interface PromptTokensDetails {
|
|
|
1484
1638
|
*
|
|
1485
1639
|
* All flags default to `false` so that newly added providers are safe.
|
|
1486
1640
|
*
|
|
1487
|
-
* Access via the crate-level
|
|
1488
|
-
*
|
|
1489
|
-
* ```rust
|
|
1490
|
-
* use liter_llm::capabilities;
|
|
1491
|
-
*
|
|
1492
|
-
* let caps = capabilities("openai");
|
|
1493
|
-
* assert!(caps.function_calling);
|
|
1494
|
-
* assert!(caps.vision);
|
|
1495
|
-
*
|
|
1496
|
-
* // Unknown providers return a default-all-false reference.
|
|
1497
|
-
* let unknown = capabilities("my-private-model");
|
|
1498
|
-
* assert!(!unknown.function_calling);
|
|
1499
|
-
* ```
|
|
1641
|
+
* Access via the crate-level `capabilities` function:
|
|
1500
1642
|
*/
|
|
1501
1643
|
export interface ProviderCapabilities {
|
|
1502
1644
|
/** The provider accepts image input in chat messages. */
|
|
@@ -1519,7 +1661,7 @@ export interface ProviderCapabilities {
|
|
|
1519
1661
|
* Static configuration for a single provider entry in providers.json.
|
|
1520
1662
|
*
|
|
1521
1663
|
* This struct deliberately does not include capability flags or streaming
|
|
1522
|
-
* format, which are accessed via the
|
|
1664
|
+
* format, which are accessed via the `capabilities` function.
|
|
1523
1665
|
*/
|
|
1524
1666
|
export interface ProviderConfig {
|
|
1525
1667
|
/** Provider identifier (matches the entry key in providers.json). */
|
|
@@ -1539,7 +1681,7 @@ export interface ProviderConfig {
|
|
|
1539
1681
|
*
|
|
1540
1682
|
* Each entry maps an OpenAI-spec field name (e.g. `"max_completion_tokens"`)
|
|
1541
1683
|
* to the name this provider expects (e.g. `"max_tokens"`). Applied
|
|
1542
|
-
* automatically by `ConfigDrivenProvider
|
|
1684
|
+
* automatically by `ConfigDrivenProvider.transform_request`.
|
|
1543
1685
|
*/
|
|
1544
1686
|
readonly paramMappings?: Record<string, string>
|
|
1545
1687
|
}
|
|
@@ -1563,7 +1705,7 @@ export declare enum ReasoningEffort {
|
|
|
1563
1705
|
Max = "max",
|
|
1564
1706
|
}
|
|
1565
1707
|
|
|
1566
|
-
/** Result of a
|
|
1708
|
+
/** Result of a `refresh_catalog` call. */
|
|
1567
1709
|
export declare enum RefreshOutcome {
|
|
1568
1710
|
/**
|
|
1569
1711
|
* `config.enabled` was `false`; no network, filesystem, or overlay
|
|
@@ -1584,12 +1726,9 @@ export declare enum RefreshOutcome {
|
|
|
1584
1726
|
}
|
|
1585
1727
|
|
|
1586
1728
|
/** A document to be reranked — either a plain string or an object with a text field. */
|
|
1587
|
-
export
|
|
1588
|
-
|
|
1589
|
-
|
|
1590
|
-
/** Document with explicit text field (may include metadata). */
|
|
1591
|
-
Object = "Object",
|
|
1592
|
-
}
|
|
1729
|
+
export type RerankDocument =
|
|
1730
|
+
| string
|
|
1731
|
+
| { text: string }
|
|
1593
1732
|
|
|
1594
1733
|
/** Request to rerank documents by relevance to a query. */
|
|
1595
1734
|
export interface RerankRequest {
|
|
@@ -1758,12 +1897,9 @@ export interface SpecificToolChoice {
|
|
|
1758
1897
|
}
|
|
1759
1898
|
|
|
1760
1899
|
/** Stop sequence(s) that cause the model to stop generating. */
|
|
1761
|
-
export
|
|
1762
|
-
|
|
1763
|
-
|
|
1764
|
-
/** Multiple stop sequences. */
|
|
1765
|
-
Multiple = "Multiple",
|
|
1766
|
-
}
|
|
1900
|
+
export type StopSequence =
|
|
1901
|
+
| string
|
|
1902
|
+
| Array<string>
|
|
1767
1903
|
|
|
1768
1904
|
/** A streaming choice with incremental delta. */
|
|
1769
1905
|
export interface StreamChoice {
|
|
@@ -1797,7 +1933,7 @@ export interface StreamDelta {
|
|
|
1797
1933
|
* Most providers use standard Server-Sent Events (SSE). AWS Bedrock uses
|
|
1798
1934
|
* a proprietary binary EventStream framing.
|
|
1799
1935
|
*
|
|
1800
|
-
* Deserialized from the `streaming_format` JSON field via
|
|
1936
|
+
* Deserialized from the `streaming_format` JSON field via `serde`.
|
|
1801
1937
|
*/
|
|
1802
1938
|
export declare enum StreamFormat {
|
|
1803
1939
|
/** Standard Server-Sent Events (text/event-stream). */
|
|
@@ -1838,7 +1974,7 @@ export interface SystemMessage {
|
|
|
1838
1974
|
* Instructions or context that apply throughout the conversation.
|
|
1839
1975
|
*
|
|
1840
1976
|
* Accepts either a plain text string or an array of content parts,
|
|
1841
|
-
* mirroring
|
|
1977
|
+
* mirroring `UserContent` so that `Message.system_with_parts` works.
|
|
1842
1978
|
*/
|
|
1843
1979
|
readonly content?: UserContent
|
|
1844
1980
|
/** Optional name for the system message source. */
|
|
@@ -1856,12 +1992,9 @@ export interface ToolCall {
|
|
|
1856
1992
|
}
|
|
1857
1993
|
|
|
1858
1994
|
/** Tool usage mode or a specific tool to call. */
|
|
1859
|
-
export
|
|
1860
|
-
|
|
1861
|
-
|
|
1862
|
-
/** Force a specific tool to be called. */
|
|
1863
|
-
Specific = "Specific",
|
|
1864
|
-
}
|
|
1995
|
+
export type ToolChoice =
|
|
1996
|
+
| ToolChoiceMode
|
|
1997
|
+
| SpecificToolChoice
|
|
1865
1998
|
|
|
1866
1999
|
/** Tool choice mode. */
|
|
1867
2000
|
export declare enum ToolChoiceMode {
|
|
@@ -1877,9 +2010,9 @@ export declare enum ToolChoiceMode {
|
|
|
1877
2010
|
export interface ToolMessage {
|
|
1878
2011
|
/**
|
|
1879
2012
|
* Result of the tool execution as plain text or an array of content parts
|
|
1880
|
-
* (text, images, documents, audio), mirroring
|
|
2013
|
+
* (text, images, documents, audio), mirroring `UserMessage.content`.
|
|
1881
2014
|
*
|
|
1882
|
-
* `#[serde(untagged)]` on
|
|
2015
|
+
* `#[serde(untagged)]` on `UserContent` means a bare JSON string still
|
|
1883
2016
|
* deserialises into `Text`, so tool results persisted before this field
|
|
1884
2017
|
* carried structured content continue to round-trip.
|
|
1885
2018
|
*/
|
|
@@ -1942,12 +2075,9 @@ export interface Usage {
|
|
|
1942
2075
|
}
|
|
1943
2076
|
|
|
1944
2077
|
/** User message content as either plain text or a list of multimodal parts. */
|
|
1945
|
-
export
|
|
1946
|
-
|
|
1947
|
-
|
|
1948
|
-
/** Array of content parts (text, images, documents, audio). */
|
|
1949
|
-
Parts = "Parts",
|
|
1950
|
-
}
|
|
2078
|
+
export type UserContent =
|
|
2079
|
+
| string
|
|
2080
|
+
| Array<ContentPart>
|
|
1951
2081
|
|
|
1952
2082
|
/** User message in the conversation. */
|
|
1953
2083
|
export interface UserMessage {
|
|
@@ -1979,30 +2109,39 @@ export interface WaitForBatchConfig {
|
|
|
1979
2109
|
*
|
|
1980
2110
|
* Returns `None` if the model is not present in the active pricing
|
|
1981
2111
|
* registry. Uses the same exact-match-then-prefix-fallback resolution as
|
|
1982
|
-
*
|
|
1983
|
-
*
|
|
2112
|
+
* `model_pricing`; unlike `model_pricing`, the result is an owned
|
|
2113
|
+
* `ModelInfo` value safe to hand across the FFI boundary.
|
|
1984
2114
|
*
|
|
1985
2115
|
* When a runtime catalog refresh has succeeded, this reflects the refreshed
|
|
1986
2116
|
* (overlay) catalog; otherwise it reflects the embedded catalog. See
|
|
1987
|
-
*
|
|
2117
|
+
* `model_pricing` for the embedded-only alternative.
|
|
1988
2118
|
*/
|
|
1989
2119
|
export declare function modelInfo(model: string): ModelInfo | null;
|
|
1990
2120
|
|
|
2121
|
+
/**
|
|
2122
|
+
* Record the estimated USD cost of a completion.
|
|
2123
|
+
*
|
|
2124
|
+
* Call from `CostTrackingService` once a
|
|
2125
|
+
* completion's cost has been computed. Emits `gen_ai.client.cost.usd`.
|
|
2126
|
+
* If the meter has not been initialized, this call is a no-op.
|
|
2127
|
+
*/
|
|
2128
|
+
export declare function recordCostUsd(system: string, model: string, operation: string, costUsd: number): void;
|
|
2129
|
+
|
|
1991
2130
|
/**
|
|
1992
2131
|
* Refresh the runtime catalog overlay per `config`.
|
|
1993
2132
|
*
|
|
1994
|
-
* - `config.enabled == false`: returns `Ok(
|
|
2133
|
+
* - `config.enabled == false`: returns `Ok(``RefreshOutcome.Disabled``)`
|
|
1995
2134
|
* immediately. No network, filesystem, or overlay activity.
|
|
1996
2135
|
* - A fresh on-disk cache (age < `config.ttl_seconds`) exists at the
|
|
1997
2136
|
* resolved cache path (`config.cache_path`, or a default under
|
|
1998
|
-
* `std
|
|
1999
|
-
* returning `Ok(
|
|
2137
|
+
* `std.env.temp_dir()`): read + flatten it and install the overlay,
|
|
2138
|
+
* returning `Ok(``RefreshOutcome.FromCache``)`. No network request is
|
|
2000
2139
|
* made.
|
|
2001
2140
|
* - Otherwise: validate `config.source_url` uses `https`
|
|
2002
|
-
* (
|
|
2141
|
+
* (`CatalogRefreshError.InsecureUrl` otherwise), fetch it, flatten it,
|
|
2003
2142
|
* install the overlay, best-effort write the raw JSON to the cache path
|
|
2004
2143
|
* (a cache write failure does not fail the refresh), and return
|
|
2005
|
-
* `Ok(
|
|
2144
|
+
* `Ok(``RefreshOutcome.Fetched``)`.
|
|
2006
2145
|
*
|
|
2007
2146
|
* On any error return, the overlay is left untouched: the previously
|
|
2008
2147
|
* active registry (a prior successful overlay, or the embedded catalog if
|
|
@@ -2038,6 +2177,7 @@ export declare class ChatStreamIterator {
|
|
|
2038
2177
|
}
|
|
2039
2178
|
|
|
2040
2179
|
export declare class LiterLlmErrorInfo {
|
|
2180
|
+
code(): number
|
|
2041
2181
|
statusCode(): number
|
|
2042
2182
|
isTransient(): boolean
|
|
2043
2183
|
errorType(): string
|