@xberg-io/liter-llm 1.16.0 → 1.17.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/index.d.ts CHANGED
@@ -1,7 +1,7 @@
1
1
  // This file is auto-generated by alef — DO NOT EDIT.
2
- // alef:hash:938b8ffd93c3e30d74875f875188c728addee9d4b80a97a7ac22716b3154c7c5
2
+ // alef:hash:6bae00d6b31bd5ec9e80699322018c3a78359cee1f06b9e2b7ffbdc6f1d4bc21
3
3
  // To regenerate: alef generate
4
- // To verify freshness: alef verify --exit-code
4
+ // To verify freshness: alef verify
5
5
  /* eslint-disable */
6
6
 
7
7
  export type JsonValue = string | number | boolean | null | JsonValue[] | { [key: string]: JsonValue };
@@ -10,8 +10,8 @@ export type JsonValue = string | number | boolean | null | JsonValue[] | { [key:
10
10
  * Return all provider configs from the registry.
11
11
  *
12
12
  * Useful for tooling, documentation generation, or runtime enumeration.
13
- * Returns the public [`ProviderConfig`] slice (without capability flags).
14
- * To query capability flags for a specific provider use [`capabilities`].
13
+ * Returns the public `ProviderConfig` slice (without capability flags).
14
+ * To query capability flags for a specific provider use `capabilities`.
15
15
  */
16
16
  export declare function allProviders(): Array<ProviderConfig>;
17
17
 
@@ -31,8 +31,8 @@ export declare function capabilities(providerName: string): ProviderCapabilities
31
31
  * Assert that `current_len + incoming` does not exceed `limit`.
32
32
  *
33
33
  * Call this before appending `incoming` bytes to any buffer that must
34
- * stay below `limit`. Returns `Err(LiterLlmError::Streaming)` on overflow
35
- * and emits a `tracing::warn!` with context.
34
+ * stay below `limit`. Returns `Err(LiterLlmError.Streaming)` on overflow
35
+ * and emits a `tracing.warn!` with context.
36
36
  */
37
37
  export declare function checkBound(context: string, currentLen: number, incoming: number, limit: number): void;
38
38
 
@@ -40,6 +40,9 @@ export declare function checkBound(context: string, currentLen: number, incoming
40
40
  * Remove all guardrails from the global registry.
41
41
  *
42
42
  * Primarily useful in tests to reset state between test cases.
43
+ *
44
+ * If the lock was poisoned by a panicking guardrail on a previous access,
45
+ * the poisoned state is recovered rather than propagating the panic.
43
46
  */
44
47
  export declare function clear(): void;
45
48
 
@@ -48,7 +51,7 @@ export declare function clear(): void;
48
51
  * `completion_cost_with_cache`, and `model_info` to the
49
52
  * embedded catalog.
50
53
  *
51
- * Primarily a test seam (see [`install_catalog_overlay_from_str`]); also
54
+ * Primarily a test seam (see `install_catalog_overlay_from_str`); also
52
55
  * usable by long-running processes that want to abandon a runtime refresh.
53
56
  */
54
57
  export declare function clearCatalogOverlay(): void;
@@ -78,9 +81,9 @@ export declare function completionCost(model: string, promptTokens: number, comp
78
81
  * input rate.
79
82
  *
80
83
  * Returns `None` if the model is not present in the embedded pricing
81
- * registry, mirroring [`completion_cost`].
84
+ * registry, mirroring `completion_cost`.
82
85
  *
83
- * When the model has [`ModelPricing::tiers`], the tier whose
86
+ * When the model has `ModelPricing.tiers`, the tier whose
84
87
  * `min_context_tokens` is the highest value `<= prompt_tokens` supplies the
85
88
  * input/output/cache rates for the whole call; models without tiers (or
86
89
  * when `prompt_tokens` is below every tier threshold) use the base rates
@@ -99,13 +102,13 @@ export declare function completionCostWithCache(model: string, promptTokens: num
99
102
  export declare function complexProviderNames(): Array<string>;
100
103
 
101
104
  /**
102
- * Count tokens for a full [`ChatCompletionRequest`].
105
+ * Count tokens for a full `ChatCompletionRequest`.
103
106
  *
104
107
  * Sums tokens across all message text contents plus a per-message overhead
105
108
  * of ~4 tokens (for role, separators, and formatting metadata). Tool
106
109
  * definitions and multimodal content parts (images, audio, documents) are
107
110
  * not counted — only textual content contributes to the token total.
108
- * @throws Returns [`LiterLlmError::BadRequest`] if the tokenizer cannot be loaded or
111
+ * @throws Returns `LiterLlmError.BadRequest` if the tokenizer cannot be loaded or
109
112
  * if tokenization fails for any message.
110
113
  */
111
114
  export declare function countRequestTokens(model: string, req?: ChatCompletionRequest | undefined | null): number;
@@ -116,7 +119,7 @@ export declare function countRequestTokens(model: string, req?: ChatCompletionRe
116
119
  * The tokenizer is resolved from the model name prefix (e.g. `"gpt-4o"` maps
117
120
  * to the `Xenova/gpt-4o` HuggingFace tokenizer). Tokenizers are cached after
118
121
  * first load.
119
- * @throws Returns [`LiterLlmError::BadRequest`] if the tokenizer cannot be loaded
122
+ * @throws Returns `LiterLlmError.BadRequest` if the tokenizer cannot be loaded
120
123
  * (e.g. network failure on first use) or if tokenization itself fails.
121
124
  */
122
125
  export declare function countTokens(model: string, text: string): number;
@@ -126,8 +129,8 @@ export declare function countTokens(model: string, text: string): number;
126
129
  *
127
130
  * This is the primary binding entry-point. All parameters except `api_key`
128
131
  * are optional — omitting them uses the same defaults as
129
- * [`ClientConfigBuilder`].
130
- * @throws Returns [`LiterLlmError`] if the underlying HTTP client cannot be
132
+ * `ClientConfigBuilder`.
133
+ * @throws Returns `LiterLlmError` if the underlying HTTP client cannot be
131
134
  * constructed, or if the resolved provider configuration is invalid.
132
135
  */
133
136
  export declare function createClient(apiKey: string, baseUrl?: string | undefined | null, timeoutSecs?: number | undefined | null, maxRetries?: number | undefined | null, modelHint?: string | undefined | null): DefaultClient;
@@ -136,13 +139,13 @@ export declare function createClient(apiKey: string, baseUrl?: string | undefine
136
139
  * Create a new LLM client from a JSON string.
137
140
  *
138
141
  * The JSON object accepts the same fields as `liter-llm.toml` (snake_case).
139
- * @throws Returns [`LiterLlmError::BadRequest`] if `json` is not valid JSON or
142
+ * @throws Returns `LiterLlmError.BadRequest` if `json` is not valid JSON or
140
143
  * contains unknown fields.
141
144
  */
142
145
  export declare function createClientFromJson(json: string): DefaultClient;
143
146
 
144
147
  /**
145
- * Decode a base64 data URL into [`DecodedDataUrl`].
148
+ * Decode a base64 data URL into `DecodedDataUrl`.
146
149
  *
147
150
  * Returns `None` for:
148
151
  * - Non-data URLs (strings that do not start with `"data:"`).
@@ -157,7 +160,7 @@ export declare function decodeDataUrl(url: string): DecodedDataUrl | null;
157
160
  /**
158
161
  * Encode bytes as a base64 data URL: `data:<mime>;base64,<b64>`.
159
162
  *
160
- * `mime` defaults to [`IMAGE_PNG`] when `None`.
163
+ * `mime` defaults to `IMAGE_PNG` when `None`.
161
164
  */
162
165
  export declare function encodeDataUrl(bytes: Uint8Array, mime?: string | undefined | null): string;
163
166
 
@@ -169,7 +172,7 @@ export declare function encodeDataUrl(bytes: Uint8Array, mime?: string | undefin
169
172
  * another rustls crypto provider has already been installed is safe: the
170
173
  * `Err` from `install_default()` is silently ignored.
171
174
  *
172
- * Called automatically by every internal `reqwest::Client` constructor
175
+ * Called automatically by every internal `reqwest.Client` constructor
173
176
  * (auth providers, default HTTP client). Bindings and downstream consumers
174
177
  * reach those constructors transitively, so no manual init is required.
175
178
  *
@@ -186,9 +189,9 @@ export declare function ensureCryptoProvider(): void;
186
189
  * the network and disk cache entirely.
187
190
  *
188
191
  * Parses and flattens `catalog_json` with the same
189
- * [`registry_from_catalog_str`] logic used for the embedded catalog and the
192
+ * `registry_from_catalog_str` logic used for the embedded catalog and the
190
193
  * network refresh path, then atomically swaps it in as the active overlay.
191
- * A parse failure returns [`CatalogRefreshError::Parse`] and leaves any
194
+ * A parse failure returns `CatalogRefreshError.Parse` and leaves any
192
195
  * existing overlay untouched.
193
196
  *
194
197
  * This is primarily a testable seam: it lets tests exercise overlay
@@ -202,16 +205,13 @@ export declare function installCatalogOverlayFromStr(catalogJson: string): void;
202
205
  * Content shape for assistant messages.
203
206
  *
204
207
  * `#[serde(untagged)]` means providers returning a plain scalar string for the
205
- * `content` field still deserialise correctly into `AssistantContent::Text(_)`.
208
+ * `content` field still deserialise correctly into `AssistantContent.Text(_)`.
206
209
  * Providers returning an array of typed parts (e.g. after an image-generation
207
- * or audio-synthesis request) deserialise into `AssistantContent::Parts(_)`.
210
+ * or audio-synthesis request) deserialise into `AssistantContent.Parts(_)`.
208
211
  */
209
- export declare enum AssistantContent {
210
- /** Plain text response (the common case for text-only models). */
211
- Text = "Text",
212
- /** Structured parts — text, refusals, output images, output audio. */
213
- Parts = "Parts",
214
- }
212
+ export type AssistantContent =
213
+ | string
214
+ | Array<AssistantPart>
215
215
 
216
216
  /** Assistant's response to a user message. */
217
217
  export interface AssistantMessage {
@@ -225,7 +225,12 @@ export interface AssistantMessage {
225
225
  readonly name?: string
226
226
  /** Tool calls the model wants to execute, if any. */
227
227
  readonly toolCalls?: Array<ToolCall>
228
- /** Refusal reason, if the model declined to respond per safety policies. */
228
+ /**
229
+ * Refusal reason, if the model declined to respond per safety policies.
230
+ *
231
+ * OpenAI's response schema requires this key to be present even when null,
232
+ * so it is deliberately not `skip_serializing_if`.
233
+ */
229
234
  readonly refusal?: string
230
235
  /** Deprecated legacy function_call field; retained for API compatibility. */
231
236
  readonly functionCall?: FunctionCall
@@ -248,7 +253,12 @@ export type AssistantPart =
248
253
  | { type: 'output_image'; imageUrl: ImageUrl }
249
254
  | { type: 'output_audio'; audio: AudioContent }
250
255
 
251
- /** Audio content part for speech-capable models. */
256
+ /**
257
+ * Audio content part for speech-capable models.
258
+ *
259
+ * No `deny_unknown_fields`: shared with the response side (see
260
+ * `AssistantPart.OutputAudio`), same rationale as `ImageUrl` (#51).
261
+ */
252
262
  export interface AudioContent {
253
263
  /** Base64-encoded audio data. */
254
264
  readonly data?: string
@@ -380,6 +390,9 @@ export declare enum BatchStatus {
380
390
  * AWS environment variables (`AWS_DEFAULT_REGION` / `AWS_REGION`,
381
391
  * `AWS_ACCESS_KEY_ID`, `AWS_SECRET_ACCESS_KEY`, `AWS_SESSION_TOKEN`,
382
392
  * `BEDROCK_CROSS_REGION`).
393
+ *
394
+ * Implements `Debug` manually (see below) so the AWS credential fields are
395
+ * redacted rather than printed in full.
383
396
  */
384
397
  export interface BedrockConfig {
385
398
  /** AWS region (e.g. `"us-east-1"`). */
@@ -423,7 +436,7 @@ export interface CacheConfig {
423
436
  }
424
437
 
425
438
  /**
426
- * Plain-data configuration for [`refresh_catalog`].
439
+ * Plain-data configuration for `refresh_catalog`.
427
440
  *
428
441
  * Deliberately FFI/binding-friendly: no `Duration` or `PathBuf`, just
429
442
  * primitives that translate directly across language boundaries.
@@ -431,14 +444,14 @@ export interface CacheConfig {
431
444
  export interface CatalogRefreshConfig {
432
445
  /**
433
446
  * Runtime catalog refresh is entirely opt-in: when `false`,
434
- * [`refresh_catalog`] is a no-op that returns
435
- * `Ok(`[`RefreshOutcome::Disabled`]`)` without touching the network,
447
+ * `refresh_catalog` is a no-op that returns
448
+ * `Ok(``RefreshOutcome.Disabled``)` without touching the network,
436
449
  * the filesystem, or the overlay registry.
437
450
  */
438
451
  readonly enabled?: boolean
439
452
  /**
440
453
  * Source URL to fetch `catalog.json` from. Must be `https`. Defaults to
441
- * [`DEFAULT_CATALOG_URL`]; configurable so self-hosted mirrors work.
454
+ * `DEFAULT_CATALOG_URL`; configurable so self-hosted mirrors work.
442
455
  */
443
456
  readonly sourceUrl?: string
444
457
  /**
@@ -448,7 +461,7 @@ export interface CatalogRefreshConfig {
448
461
  readonly ttlSeconds?: number
449
462
  /**
450
463
  * Filesystem path for the on-disk cache. `None` uses a default path
451
- * under `std::env::temp_dir()`.
464
+ * under `std.env.temp_dir()`.
452
465
  */
453
466
  readonly cachePath?: string
454
467
  }
@@ -482,9 +495,26 @@ export interface ChatCompletionRequest {
482
495
  readonly model?: string
483
496
  /** Conversation history from oldest to newest. */
484
497
  readonly messages?: Array<Message>
485
- /** Sampling temperature in `[0.0, 2.0]`. Higher increases randomness. Defaults to 1.0. */
498
+ /**
499
+ * Sampling temperature. Higher increases randomness, lower is more deterministic.
500
+ * Defaults to 1.0.
501
+ *
502
+ * The accepted range depends on the provider the request is routed to. OpenAI-compatible
503
+ * providers accept `[0.0, 2.0]`; Anthropic and Amazon Bedrock both cap it at `1.0`, and
504
+ * for those two a value above the cap is rejected with a `BadRequest` error before the
505
+ * request is sent, rather than being silently clamped or left for the provider to reject.
506
+ *
507
+ * No range is enforced for providers whose own documentation does not state one — the
508
+ * value is forwarded and the provider decides. Consult the target provider's reference
509
+ * rather than assuming `[0.0, 2.0]` is portable.
510
+ */
486
511
  readonly temperature?: number
487
- /** Nucleus sampling parameter in `[0.0, 1.0]`. Lower is more focused. */
512
+ /**
513
+ * Nucleus sampling parameter. Lower is more focused.
514
+ *
515
+ * Accepted ranges vary by provider (most document `[0.0, 1.0]`, but this is not
516
+ * universal — check the target provider's own documentation for its exact bounds).
517
+ */
488
518
  readonly topP?: number
489
519
  /** Number of chat completions to generate. Defaults to 1. */
490
520
  readonly n?: number
@@ -530,6 +560,49 @@ export interface ChatCompletionRequest {
530
560
  * translates these to `generationConfig.responseModalities` (uppercase).
531
561
  */
532
562
  readonly modalities?: Array<Modality>
563
+ /** Whether to return log probabilities of the output tokens. */
564
+ readonly logprobs?: boolean
565
+ /**
566
+ * Number of most-likely tokens to return log probabilities for, `0..=20`.
567
+ * Requires `logprobs` to be `true`.
568
+ */
569
+ readonly topLogprobs?: number
570
+ /**
571
+ * Upper bound on generated tokens, including reasoning tokens.
572
+ *
573
+ * Supersedes `max_tokens` on OpenAI reasoning models, which reject
574
+ * `max_tokens` outright.
575
+ */
576
+ readonly maxCompletionTokens?: number
577
+ /**
578
+ * Latency tier to process the request under (e.g. `"auto"`, `"default"`,
579
+ * `"flex"`).
580
+ */
581
+ readonly serviceTier?: string
582
+ /** Whether to store the completion for later retrieval by the provider. */
583
+ readonly store?: boolean
584
+ /** Developer-defined tags attached to the completion. */
585
+ readonly metadata?: Record<string, string>
586
+ /**
587
+ * Predicted output, for latency reduction when much of the response is
588
+ * known ahead of time.
589
+ *
590
+ * Untyped: the shape is provider-defined and still evolving, and the
591
+ * value is forwarded verbatim.
592
+ */
593
+ readonly prediction?: JsonValue
594
+ /**
595
+ * Audio output parameters, required when `modalities` includes `audio`.
596
+ *
597
+ * Untyped for the same reason as `prediction`.
598
+ */
599
+ readonly audio?: JsonValue
600
+ /**
601
+ * Web-search tool configuration for search-enabled models.
602
+ *
603
+ * Untyped for the same reason as `prediction`.
604
+ */
605
+ readonly webSearchOptions?: JsonValue
533
606
  /**
534
607
  * Provider-specific extra parameters merged into the request body.
535
608
  * Use for guardrails, safety settings, grounding config, etc.
@@ -572,21 +645,35 @@ export interface ChatCompletionTool {
572
645
  export interface Choice {
573
646
  /** Index of this choice in the choices array. */
574
647
  readonly index?: number
575
- /** The assistant's message response. */
648
+ /**
649
+ * The assistant's message response.
650
+ *
651
+ * Serialized with an explicit `role: "assistant"`. The field is not stored
652
+ * on `AssistantMessage` because `Message` is an internally-tagged enum
653
+ * keyed on `role`, so a stored field would emit the key twice inside a
654
+ * request. OpenAI's response schema requires it here.
655
+ */
576
656
  readonly message?: AssistantMessage
577
657
  /** Why the model stopped generating (stop, length, tool_calls, content_filter, etc.). */
578
658
  readonly finishReason?: FinishReason
659
+ /**
660
+ * Per-token log probabilities, when the request asked for them.
661
+ *
662
+ * Required by OpenAI's response schema as an always-present, nullable key,
663
+ * so this is deliberately not `skip_serializing_if`.
664
+ */
665
+ readonly logprobs?: JsonValue
579
666
  }
580
667
 
581
668
  /**
582
- * A per-chunk transformation in the [`StreamPipeline`].
669
+ * A per-chunk transformation in the `StreamPipeline`.
583
670
  *
584
671
  * Each middleware receives a typed chunk and returns `Ok(Some(chunk))`
585
672
  * to pass it through (optionally modified), `Ok(None)` to drop the chunk,
586
673
  * or `Err(e)` to propagate a stream error.
587
674
  *
588
675
  * The trait is object-safe so multiple middleware implementations can be
589
- * chained inside [`StreamPipeline`].
676
+ * chained inside `StreamPipeline`.
590
677
  */
591
678
  export interface ChunkMiddleware {
592
679
  /**
@@ -658,9 +745,35 @@ export interface CreateImageRequest {
658
745
  readonly user?: string
659
746
  }
660
747
 
661
- /** Request to create a structured response. */
748
+ /**
749
+ * Request to create a response via the OpenAI Responses API (`POST /responses`).
750
+ *
751
+ * # Provider support
752
+ *
753
+ * **The Responses API path is OpenAI-only.** This type models the OpenAI Responses
754
+ * wire format, and unlike `ChatCompletionRequest` the body is sent to the provider
755
+ * verbatim: neither `Provider.transform_request` nor `Provider.transform_response`
756
+ * runs for Responses calls, and no provider in `schemas/providers.json` declares a
757
+ * `responses` endpoint.
758
+ *
759
+ * Pointing a Responses call at a provider that does not natively serve the OpenAI
760
+ * `/responses` contract (Anthropic, Vertex, Bedrock, Cohere, Google AI, Azure) is not
761
+ * supported: the request goes out unmodified, so the provider rejects it or returns a
762
+ * body that cannot be deserialized into `ResponseObject`. Use
763
+ * `ChatCompletionRequest` for cross-provider work — that path applies the
764
+ * per-provider request and response normalization this one does not.
765
+ *
766
+ * `ChatCompletionRequest`: crate.types.ChatCompletionRequest
767
+ */
662
768
  export interface CreateResponseRequest {
663
- /** Model ID. */
769
+ /**
770
+ * Model ID, as named by the OpenAI Responses API (e.g. `"gpt-5"`).
771
+ *
772
+ * Sent to the wire exactly as given. The Responses path performs none of the
773
+ * chat path's model handling: a `provider/model` routing prefix is **not**
774
+ * stripped and does **not** re-route the request, which stays pinned to the
775
+ * provider the client was constructed with.
776
+ */
664
777
  readonly model?: string
665
778
  /** Input data to process (e.g., a document to extract from). */
666
779
  readonly input?: JsonValue
@@ -674,6 +787,21 @@ export interface CreateResponseRequest {
674
787
  readonly maxOutputTokens?: number
675
788
  /** Optional metadata. */
676
789
  readonly metadata?: JsonValue
790
+ /**
791
+ * Extra top-level parameters shallow-merged into the request body, OpenAI-Python
792
+ * style (`{**body, **extra_body}`) — keys here override identically named fields
793
+ * above. Use it for OpenAI Responses fields this struct does not model directly,
794
+ * such as the top-level `reasoning.effort`.
795
+ *
796
+ * This is an OpenAI escape hatch, not a cross-provider one. On the chat path the
797
+ * providers that consume `extra_body` natively (Anthropic, Vertex, Bedrock) claim
798
+ * it inside their own `transform_request`; here no provider transform runs, so the
799
+ * merged keys always travel to the wire as literal OpenAI Responses fields.
800
+ *
801
+ * A non-object value cannot be merged into the body root and is dropped with a
802
+ * warning rather than sent.
803
+ */
804
+ readonly extraBody?: JsonValue
677
805
  /**
678
806
  * Whether to stream the response.
679
807
  *
@@ -750,7 +878,7 @@ export interface DecodedDataUrl {
750
878
  * provider is used as the fallback. This enables seamless migration between
751
879
  * providers by changing only the model name.
752
880
  *
753
- * The provider is stored behind an [`Arc`] so it can be shared cheaply into
881
+ * The provider is stored behind an `Arc` so it can be shared cheaply into
754
882
  * async closures and streaming tasks. Pre-computed auth headers and extra
755
883
  * headers are cached at construction to avoid redundant encoding on every request.
756
884
  */
@@ -781,9 +909,9 @@ export declare class DefaultClient {
781
909
  *
782
910
  * Uses exponential backoff with configurable initial interval, maximum interval, and backoff multiplier.
783
911
  * Optionally supports a timeout that aborts polling if exceeded.
784
- * @throws Returns `BatchWaitError::Failed` if the batch reaches a failure terminal status.
785
- * Returns `BatchWaitError::Timeout` if the configured timeout is exceeded.
786
- * Returns `BatchWaitError::Client` for underlying client errors.
912
+ * @throws Returns `BatchWaitError.Failed` if the batch reaches a failure terminal status.
913
+ * Returns `BatchWaitError.Timeout` if the configured timeout is exceeded.
914
+ * Returns `BatchWaitError.Client` for underlying client errors.
787
915
  */
788
916
  waitForBatch(batchId: string, config?: WaitForBatchConfig | undefined | null): Promise<BatchObject>
789
917
  createResponse(req?: CreateResponseRequest | undefined | null): Promise<ResponseObject>
@@ -826,12 +954,9 @@ export declare enum EmbeddingFormat {
826
954
  }
827
955
 
828
956
  /** Text or texts to embed. */
829
- export declare enum EmbeddingInput {
830
- /** Single text string. */
831
- Single = "Single",
832
- /** Multiple text strings (batch embedding). */
833
- Multiple = "Multiple",
834
- }
957
+ export type EmbeddingInput =
958
+ | string
959
+ | Array<string>
835
960
 
836
961
  /** A single embedding vector. */
837
962
  export interface EmbeddingObject {
@@ -886,11 +1011,11 @@ export interface EmbeddingResponse {
886
1011
  export declare enum Enforcement {
887
1012
  /**
888
1013
  * Reject requests that would exceed the budget with
889
- * [`LiterLlmError::BudgetExceeded`].
1014
+ * `LiterLlmError.BudgetExceeded`.
890
1015
  */
891
1016
  Hard = "Hard",
892
1017
  /**
893
- * Allow requests through but emit a `tracing::warn!` when the budget is
1018
+ * Allow requests through but emit a `tracing.warn!` when the budget is
894
1019
  * exceeded.
895
1020
  */
896
1021
  Soft = "Soft",
@@ -970,7 +1095,7 @@ export declare enum FinishReason {
970
1095
  export interface FunctionCall {
971
1096
  /** Function name. */
972
1097
  readonly name: string
973
- /** Arguments as a JSON string (parse with serde_json::from_str). */
1098
+ /** Arguments as a JSON string (parse with serde_json.from_str). */
974
1099
  readonly arguments: string
975
1100
  }
976
1101
 
@@ -996,11 +1121,11 @@ export interface FunctionMessage {
996
1121
  * Abstraction over a health probe strategy.
997
1122
  *
998
1123
  * Implementors issue a lightweight probe against `upstream` (typically a
999
- * provider base URL or named identifier) and report [`HealthStatus`].
1124
+ * provider base URL or named identifier) and report `HealthStatus`.
1000
1125
  */
1001
1126
  export interface HealthChecker {
1002
1127
  /**
1003
- * Probe `upstream` and return its current [`HealthStatus`].
1128
+ * Probe `upstream` and return its current `HealthStatus`.
1004
1129
  *
1005
1130
  * The parameter is taken by value (`String`) so that implementations can
1006
1131
  * move it into the returned future without a clone, making the
@@ -1045,7 +1170,15 @@ export interface ImagesResponse {
1045
1170
  readonly data?: Array<Image>
1046
1171
  }
1047
1172
 
1048
- /** An image URL reference with optional detail level for processing. */
1173
+ /**
1174
+ * An image URL reference with optional detail level for processing.
1175
+ *
1176
+ * No `deny_unknown_fields`: this type is shared with the response side
1177
+ * (see `AssistantPart.OutputImage`) where it is deserialized from
1178
+ * provider output, not just constructed as request input (see #51). A
1179
+ * provider adding a new field to its image-output object must not hard-fail
1180
+ * the whole response.
1181
+ */
1049
1182
  export interface ImageUrl {
1050
1183
  /** URL of the image (data URI or HTTP/HTTPS URL). */
1051
1184
  readonly url?: string
@@ -1103,12 +1236,15 @@ export interface LlmCacheConfig {
1103
1236
  * All fields except `model` are optional so that partially-specified
1104
1237
  * configs (e.g. from environment-driven defaults) round-trip cleanly.
1105
1238
  * Convert to a runtime client configuration via
1106
- * [`LlmConfig::into_client_builder`].
1239
+ * `LlmConfig.into_client_builder`.
1107
1240
  *
1108
1241
  * `temperature` and `max_tokens` are request-time parameters rather than
1109
1242
  * client-level settings; they are carried on this struct for callers to
1110
1243
  * read when building individual requests, and are intentionally **not**
1111
- * mapped by [`LlmConfig::into_client_builder`].
1244
+ * mapped by `LlmConfig.into_client_builder`.
1245
+ *
1246
+ * Implements `Debug` manually (see below) so `api_key` and header values are
1247
+ * redacted rather than printed in full.
1112
1248
  */
1113
1249
  export interface LlmConfig {
1114
1250
  /** Model identifier (e.g. `"gpt-4o"`, `"bedrock/anthropic.claude-3-sonnet-20240229-v1:0"`). */
@@ -1179,12 +1315,12 @@ export interface LlmRateLimitConfig {
1179
1315
 
1180
1316
  /** A chat message in a conversation. */
1181
1317
  export type Message =
1182
- | { role: 'system'; 0: SystemMessage }
1183
- | { role: 'user'; 0: UserMessage }
1184
- | { role: 'assistant'; 0: AssistantMessage }
1185
- | { role: 'tool'; 0: ToolMessage }
1186
- | { role: 'developer'; 0: DeveloperMessage }
1187
- | { role: 'function'; 0: FunctionMessage }
1318
+ | { role: 'system'; system: SystemMessage }
1319
+ | { role: 'user'; user: UserMessage }
1320
+ | { role: 'assistant'; assistant: AssistantMessage }
1321
+ | { role: 'tool'; tool: ToolMessage }
1322
+ | { role: 'developer'; developer: DeveloperMessage }
1323
+ | { role: 'function'; function: FunctionMessage }
1188
1324
 
1189
1325
  /**
1190
1326
  * Output modality requested from the model.
@@ -1203,11 +1339,11 @@ export declare enum Modality {
1203
1339
 
1204
1340
  /**
1205
1341
  * Public, FFI-friendly snapshot of a model's pricing and capability
1206
- * metadata, projected from [`ModelPricing`].
1342
+ * metadata, projected from `ModelPricing`.
1207
1343
  *
1208
- * Unlike [`ModelPricing`] (which is excluded from binding generation),
1344
+ * Unlike `ModelPricing` (which is excluded from binding generation),
1209
1345
  * `ModelInfo` is an owned plain-data DTO safe to hand across the FFI
1210
- * boundary — see [`model_info`].
1346
+ * boundary — see `model_info`.
1211
1347
  */
1212
1348
  export interface ModelInfo {
1213
1349
  /** Cost in USD per input (prompt) token. */
@@ -1291,7 +1427,7 @@ export interface ModelsListResponse {
1291
1427
 
1292
1428
  /**
1293
1429
  * Public, FFI-friendly snapshot of a single context-window pricing tier,
1294
- * projected from [`PricingTier`].
1430
+ * projected from `PricingTier`.
1295
1431
  */
1296
1432
  export interface ModelTier {
1297
1433
  /**
@@ -1368,12 +1504,9 @@ export interface ModerationCategoryScores {
1368
1504
  }
1369
1505
 
1370
1506
  /** Input to the moderation endpoint — a single string or multiple strings. */
1371
- export declare enum ModerationInput {
1372
- /** Single text string. */
1373
- Single = "Single",
1374
- /** Multiple text strings (batch moderation). */
1375
- Multiple = "Multiple",
1376
- }
1507
+ export type ModerationInput =
1508
+ | string
1509
+ | Array<string>
1377
1510
 
1378
1511
  /** Request to classify content for policy violations. */
1379
1512
  export interface ModerationRequest {
@@ -1461,7 +1594,7 @@ export interface PageDimensions {
1461
1594
  /**
1462
1595
  * Breakdown of tokens used in the prompt portion of a request.
1463
1596
  *
1464
- * `cached_tokens` is included in `Usage::prompt_tokens` — it is *not* an
1597
+ * `cached_tokens` is included in `Usage.prompt_tokens` — it is *not* an
1465
1598
  * additional charge on top of the prompt token count. When pricing supports
1466
1599
  * a `cache_read_input_token_cost`, the cached portion is billed at the
1467
1600
  * discounted rate and the remainder at the regular input rate.
@@ -1484,19 +1617,7 @@ export interface PromptTokensDetails {
1484
1617
  *
1485
1618
  * All flags default to `false` so that newly added providers are safe.
1486
1619
  *
1487
- * Access via the crate-level [`capabilities`] function:
1488
- *
1489
- * ```rust
1490
- * use liter_llm::capabilities;
1491
- *
1492
- * let caps = capabilities("openai");
1493
- * assert!(caps.function_calling);
1494
- * assert!(caps.vision);
1495
- *
1496
- * // Unknown providers return a default-all-false reference.
1497
- * let unknown = capabilities("my-private-model");
1498
- * assert!(!unknown.function_calling);
1499
- * ```
1620
+ * Access via the crate-level `capabilities` function:
1500
1621
  */
1501
1622
  export interface ProviderCapabilities {
1502
1623
  /** The provider accepts image input in chat messages. */
@@ -1519,7 +1640,7 @@ export interface ProviderCapabilities {
1519
1640
  * Static configuration for a single provider entry in providers.json.
1520
1641
  *
1521
1642
  * This struct deliberately does not include capability flags or streaming
1522
- * format, which are accessed via the [`capabilities`] function.
1643
+ * format, which are accessed via the `capabilities` function.
1523
1644
  */
1524
1645
  export interface ProviderConfig {
1525
1646
  /** Provider identifier (matches the entry key in providers.json). */
@@ -1539,7 +1660,7 @@ export interface ProviderConfig {
1539
1660
  *
1540
1661
  * Each entry maps an OpenAI-spec field name (e.g. `"max_completion_tokens"`)
1541
1662
  * to the name this provider expects (e.g. `"max_tokens"`). Applied
1542
- * automatically by `ConfigDrivenProvider::transform_request`.
1663
+ * automatically by `ConfigDrivenProvider.transform_request`.
1543
1664
  */
1544
1665
  readonly paramMappings?: Record<string, string>
1545
1666
  }
@@ -1563,7 +1684,7 @@ export declare enum ReasoningEffort {
1563
1684
  Max = "max",
1564
1685
  }
1565
1686
 
1566
- /** Result of a [`refresh_catalog`] call. */
1687
+ /** Result of a `refresh_catalog` call. */
1567
1688
  export declare enum RefreshOutcome {
1568
1689
  /**
1569
1690
  * `config.enabled` was `false`; no network, filesystem, or overlay
@@ -1584,12 +1705,9 @@ export declare enum RefreshOutcome {
1584
1705
  }
1585
1706
 
1586
1707
  /** A document to be reranked — either a plain string or an object with a text field. */
1587
- export declare enum RerankDocument {
1588
- /** Plain text document content. */
1589
- Text = "Text",
1590
- /** Document with explicit text field (may include metadata). */
1591
- Object = "Object",
1592
- }
1708
+ export type RerankDocument =
1709
+ | string
1710
+ | { text: string }
1593
1711
 
1594
1712
  /** Request to rerank documents by relevance to a query. */
1595
1713
  export interface RerankRequest {
@@ -1758,12 +1876,9 @@ export interface SpecificToolChoice {
1758
1876
  }
1759
1877
 
1760
1878
  /** Stop sequence(s) that cause the model to stop generating. */
1761
- export declare enum StopSequence {
1762
- /** Single stop sequence. */
1763
- Single = "Single",
1764
- /** Multiple stop sequences. */
1765
- Multiple = "Multiple",
1766
- }
1879
+ export type StopSequence =
1880
+ | string
1881
+ | Array<string>
1767
1882
 
1768
1883
  /** A streaming choice with incremental delta. */
1769
1884
  export interface StreamChoice {
@@ -1797,7 +1912,7 @@ export interface StreamDelta {
1797
1912
  * Most providers use standard Server-Sent Events (SSE). AWS Bedrock uses
1798
1913
  * a proprietary binary EventStream framing.
1799
1914
  *
1800
- * Deserialized from the `streaming_format` JSON field via [`serde`].
1915
+ * Deserialized from the `streaming_format` JSON field via `serde`.
1801
1916
  */
1802
1917
  export declare enum StreamFormat {
1803
1918
  /** Standard Server-Sent Events (text/event-stream). */
@@ -1838,7 +1953,7 @@ export interface SystemMessage {
1838
1953
  * Instructions or context that apply throughout the conversation.
1839
1954
  *
1840
1955
  * Accepts either a plain text string or an array of content parts,
1841
- * mirroring [`UserContent`] so that `Message::system_with_parts` works.
1956
+ * mirroring `UserContent` so that `Message.system_with_parts` works.
1842
1957
  */
1843
1958
  readonly content?: UserContent
1844
1959
  /** Optional name for the system message source. */
@@ -1856,12 +1971,9 @@ export interface ToolCall {
1856
1971
  }
1857
1972
 
1858
1973
  /** Tool usage mode or a specific tool to call. */
1859
- export declare enum ToolChoice {
1860
- /** Predefined mode: auto, required, or none. */
1861
- Mode = "Mode",
1862
- /** Force a specific tool to be called. */
1863
- Specific = "Specific",
1864
- }
1974
+ export type ToolChoice =
1975
+ | ToolChoiceMode
1976
+ | SpecificToolChoice
1865
1977
 
1866
1978
  /** Tool choice mode. */
1867
1979
  export declare enum ToolChoiceMode {
@@ -1877,9 +1989,9 @@ export declare enum ToolChoiceMode {
1877
1989
  export interface ToolMessage {
1878
1990
  /**
1879
1991
  * Result of the tool execution as plain text or an array of content parts
1880
- * (text, images, documents, audio), mirroring [`UserMessage::content`].
1992
+ * (text, images, documents, audio), mirroring `UserMessage.content`.
1881
1993
  *
1882
- * `#[serde(untagged)]` on [`UserContent`] means a bare JSON string still
1994
+ * `#[serde(untagged)]` on `UserContent` means a bare JSON string still
1883
1995
  * deserialises into `Text`, so tool results persisted before this field
1884
1996
  * carried structured content continue to round-trip.
1885
1997
  */
@@ -1942,12 +2054,9 @@ export interface Usage {
1942
2054
  }
1943
2055
 
1944
2056
  /** User message content as either plain text or a list of multimodal parts. */
1945
- export declare enum UserContent {
1946
- /** Plain text content. */
1947
- Text = "Text",
1948
- /** Array of content parts (text, images, documents, audio). */
1949
- Parts = "Parts",
1950
- }
2057
+ export type UserContent =
2058
+ | string
2059
+ | Array<ContentPart>
1951
2060
 
1952
2061
  /** User message in the conversation. */
1953
2062
  export interface UserMessage {
@@ -1979,30 +2088,39 @@ export interface WaitForBatchConfig {
1979
2088
  *
1980
2089
  * Returns `None` if the model is not present in the active pricing
1981
2090
  * registry. Uses the same exact-match-then-prefix-fallback resolution as
1982
- * [`model_pricing`]; unlike `model_pricing`, the result is an owned
1983
- * [`ModelInfo`] value safe to hand across the FFI boundary.
2091
+ * `model_pricing`; unlike `model_pricing`, the result is an owned
2092
+ * `ModelInfo` value safe to hand across the FFI boundary.
1984
2093
  *
1985
2094
  * When a runtime catalog refresh has succeeded, this reflects the refreshed
1986
2095
  * (overlay) catalog; otherwise it reflects the embedded catalog. See
1987
- * [`model_pricing`] for the embedded-only alternative.
2096
+ * `model_pricing` for the embedded-only alternative.
1988
2097
  */
1989
2098
  export declare function modelInfo(model: string): ModelInfo | null;
1990
2099
 
2100
+ /**
2101
+ * Record the estimated USD cost of a completion.
2102
+ *
2103
+ * Call from `CostTrackingService` once a
2104
+ * completion's cost has been computed. Emits `gen_ai.client.cost.usd`.
2105
+ * If the meter has not been initialized, this call is a no-op.
2106
+ */
2107
+ export declare function recordCostUsd(system: string, model: string, operation: string, costUsd: number): void;
2108
+
1991
2109
  /**
1992
2110
  * Refresh the runtime catalog overlay per `config`.
1993
2111
  *
1994
- * - `config.enabled == false`: returns `Ok(`[`RefreshOutcome::Disabled`]`)`
2112
+ * - `config.enabled == false`: returns `Ok(``RefreshOutcome.Disabled``)`
1995
2113
  * immediately. No network, filesystem, or overlay activity.
1996
2114
  * - A fresh on-disk cache (age < `config.ttl_seconds`) exists at the
1997
2115
  * resolved cache path (`config.cache_path`, or a default under
1998
- * `std::env::temp_dir()`): read + flatten it and install the overlay,
1999
- * returning `Ok(`[`RefreshOutcome::FromCache`]`)`. No network request is
2116
+ * `std.env.temp_dir()`): read + flatten it and install the overlay,
2117
+ * returning `Ok(``RefreshOutcome.FromCache``)`. No network request is
2000
2118
  * made.
2001
2119
  * - Otherwise: validate `config.source_url` uses `https`
2002
- * ([`CatalogRefreshError::InsecureUrl`] otherwise), fetch it, flatten it,
2120
+ * (`CatalogRefreshError.InsecureUrl` otherwise), fetch it, flatten it,
2003
2121
  * install the overlay, best-effort write the raw JSON to the cache path
2004
2122
  * (a cache write failure does not fail the refresh), and return
2005
- * `Ok(`[`RefreshOutcome::Fetched`]`)`.
2123
+ * `Ok(``RefreshOutcome.Fetched``)`.
2006
2124
  *
2007
2125
  * On any error return, the overlay is left untouched: the previously
2008
2126
  * active registry (a prior successful overlay, or the embedded catalog if
@@ -2038,6 +2156,7 @@ export declare class ChatStreamIterator {
2038
2156
  }
2039
2157
 
2040
2158
  export declare class LiterLlmErrorInfo {
2159
+ code(): number
2041
2160
  statusCode(): number
2042
2161
  isTransient(): boolean
2043
2162
  errorType(): string