@oh-my-pi/pi-ai 18.1.5 → 18.1.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,45 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [18.1.8] - 2026-09-03
6
+
7
+ ### Added
8
+
9
+ - Added GPT-6 Astra support for preserving prompt caching when changing the thinking level during a conversation across the OpenAI and OpenAI Codex providers.
10
+
11
+ ### Changed
12
+
13
+ - Updated OpenAI Codex requests to improve routing by communicating the selected model and service tier across Responses, WebSocket, and remote-compaction requests.
14
+
15
+ ## [18.1.7] - 2026-09-03
16
+
17
+ ### Fixed
18
+
19
+ - Fixed DeepSeek-family Responses replay (e.g. opencode-go) rejecting a resumed thinking-mode turn with `400 The reasoning_text in the thinking mode must be passed back to the API` when compaction dropped the turn's reasoning; a non-empty placeholder is now synthesized instead of an empty `reasoning_text` ([#10690](https://github.com/can1357/oh-my-pi/issues/10690)).
20
+
21
+ ## [18.1.6] - 2026-09-03
22
+
23
+ ### Breaking Changes
24
+
25
+ - Renamed `claudeCodeSessionId` to `sessionId` in `AnthropicClientOptionsArgs`.
26
+ - Renamed `openAISessionId` to `sessionId` in `OpenAIRequestSetupOptions`.
27
+
28
+ ### Added
29
+
30
+ - Added Amazon Bedrock `requestMetadata` support for cost and usage attribution in AWS invocation logs.
31
+
32
+ ### Changed
33
+
34
+ - Codex GPT-5.6 requests now use full Responses by default, enabling independent tool calls to run in parallel; provider-native compaction continues to use catalog-selected Responses Lite.
35
+ - Inference requests now identify as omp by default while preserving explicit provider and OAuth User-Agent fingerprints. Amazon Bedrock requests use an `omp/<version>` User-Agent by default and honor configured `User-Agent` overrides.
36
+
37
+ ### Fixed
38
+
39
+ - Fixed Antigravity usage reporting to match the official client's five-hour and weekly quota buckets.
40
+ - Anthropic and OpenRouter credit-exhaustion errors now automatically switch to a sibling account instead of stopping the turn with a retry hint.
41
+ - Fixed OpenCode Go and Zen requests by including the required stable per-conversation session identification.
42
+ - Improved Anthropic prompt caching so explicit cache breakpoints preserve reusable tools and system prompts when the message tail changes.
43
+
5
44
  ## [18.1.5] - 2026-09-03
6
45
 
7
46
  ### Added
@@ -183,7 +222,7 @@
183
222
  - Fixed Codex continuations, retries, and compaction replacing or dropping the turn-scoped sticky-routing token ([#9277](https://github.com/can1357/oh-my-pi/issues/9277)).
184
223
  - Fixed Codex Responses append chains falling back to full-context replay when replay-sanitized assistant items differ only by output-only IDs or lifecycle status.
185
224
  - Fixed Cursor usage reporting “no usage data” for plans without a numeric legacy request cap.
186
- - Fixed DeepSeek models rejecting requests with HTTP 400 `unknown variant \`image_url\`, expected \`text\`` when screenshots or image-producing tool results are present in conversation history or when `model.input` claims vision capability; `convertMessages` in `openai-completions` now strips `image_url` content parts and injects non-vision image placeholders for all DeepSeek endpoints.
225
+ - Fixed DeepSeek models rejecting requests with HTTP 400 `unknown variant \`image_url\`, expected \`text\``when screenshots or image-producing tool results are present in conversation history or when`model.input`claims vision capability;`convertMessages`in`openai-completions`now strips`image_url` content parts and injects non-vision image placeholders for all DeepSeek endpoints.
187
226
  - Fixed `PI_PROXY` covering only provider streams: OAuth token refresh and login, usage probes, and model discovery went out through the bare global `fetch` and ignored it, so a region-blocked token endpoint answered `403 Request not allowed` (Anthropic `/v1/oauth/token`) and disabled the credential while the proxied stream itself worked. `installGlobalProxyFetch()` now routes the process-wide `fetch` through `PI_PROXY`; a per-request proxy such as `PI_PROXY_<PROVIDER>` still wins, and loopback / private-range / `NO_PROXY` targets stay direct.
188
227
  - Fixed Anthropic inference ignoring every proxy setting. `coworkFetch` runs on `node:https`, whose Bun shim discards both `agent.createConnection` and `options.createConnection`: the CONNECT tunnel to `PI_PROXY` was built, TLS-negotiated, then abandoned, and the request dialed `api.anthropic.com` on the default route (measured at the proxy: 581 bytes of handshake, zero request bytes). On a region-blocked egress that returned `403 {"type":"forbidden","message":"Request not allowed"}` with the proxy apparently configured. Proxied requests now go through Bun's own `fetch`, which honors `init.proxy`, trading the Cowork TLS/header profile for a proxy that actually carries the traffic; the dead tunnel plumbing is gone from the transport. `node:http2` (Cursor) does honor `createConnection` and is unaffected.
189
228
  - Fixed `cowork-fetch` capturing `globalThis.fetch` at module load, so a proxy wrapper installed later in startup was ignored on its fallback path.
@@ -253,7 +292,7 @@
253
292
 
254
293
  ### Changed
255
294
 
256
- - Fixed Gemini thought summaries occasionally leaking a raw `` ```thinking `` / `` ``````thinking `` fence delimiter into the reasoning block, so it no longer shows up as fence spam in the thinking display or persisted transcripts ([#8719](https://github.com/can1357/oh-my-pi/issues/8719)).
295
+ - Fixed Gemini thought summaries occasionally leaking a raw ` ```thinking ` / ` ``````thinking ` fence delimiter into the reasoning block, so it no longer shows up as fence spam in the thinking display or persisted transcripts ([#8719](https://github.com/can1357/oh-my-pi/issues/8719)).
257
296
  - Fixed the OpenCode Go login prompting for an "OpenCode Zen API key": the shared login flow now names the provider you selected, so connecting OpenCode Go asks for an OpenCode Go key (the `opencode.ai/auth` console is still shared, as documented upstream) ([#8738](https://github.com/can1357/oh-my-pi/issues/8738)).
258
297
  - Fixed Anthropic-compatible endpoints with strict prompt validation (e.g. Z.AI GLM `api.z.ai/api/anthropic`, which rejects the whole request with `400 code 1213 "The prompt parameter was not received normally"`) failing sessions once a tool returned empty output on a vision-capable model: empty successful `tool_result` blocks now encode as `content: ""` instead of `content: []`, which both the official API and strict compatible endpoints accept.
259
298
  - Fixed `retry.usageReservePct` (Reserve Margin) ignoring Claude Fable/Mythos weekly tier usage until it hit 100%, so a Fable model kept serving turns past the configured reserve; reserve health now honors the mapped tier row while credential-wide hard blocks still require confirmed exhaustion ([#8773](https://github.com/can1357/oh-my-pi/issues/8773)).
@@ -1305,7 +1344,7 @@
1305
1344
 
1306
1345
  - Added a third streaming thinking-loop detection heuristic to catch "progress-lexicon stalls" where models endlessly reshuffle motivational filler without introducing new vocabulary or concrete technical references
1307
1346
  - Added branded wordmark and logo animation to authentication flow
1308
- - Added a third streaming thinking-loop detection shape — a *progress-lexicon stall* — alongside verbatim tail repetition and near-duplicate (trigram) segments. It catches reasoning-summarizer loops that reshuffle the same motivational filler ("just doing it, pushing ahead, maintaining momentum") into fresh word order every paragraph: word-trigrams never cluster, but a run of substantial segments that recycle the recent vocabulary and introduce no *new* concrete reference (path / identifier / code-span) trips the guard. Summarizer title/heading lines (`**Bold Title**`, `## Heading`) are stripped before analysis so their ever-changing wording cannot mask the stall by inflating novelty. Calibrated against 537k real non-Gemini reasoning blocks (zero false positives at novelty floor 0.2 / run length 8; the real loop sustains runs of 10+).
1347
+ - Added a third streaming thinking-loop detection shape — a _progress-lexicon stall_ — alongside verbatim tail repetition and near-duplicate (trigram) segments. It catches reasoning-summarizer loops that reshuffle the same motivational filler ("just doing it, pushing ahead, maintaining momentum") into fresh word order every paragraph: word-trigrams never cluster, but a run of substantial segments that recycle the recent vocabulary and introduce no _new_ concrete reference (path / identifier / code-span) trips the guard. Summarizer title/heading lines (`**Bold Title**`, `## Heading`) are stripped before analysis so their ever-changing wording cannot mask the stall by inflating novelty. Calibrated against 537k real non-Gemini reasoning blocks (zero false positives at novelty floor 0.2 / run length 8; the real loop sustains runs of 10+).
1309
1348
  - Added CoreWeave Serverless Inference provider login support via `COREWEAVE_API_KEY` and `WANDB_API_KEY` fallback.
1310
1349
 
1311
1350
  ### Changed
@@ -1449,7 +1488,7 @@
1449
1488
  - Fixed tool call ID normalization for Anthropic-compatible models
1450
1489
  - Fixed Anthropic Messages replay sanitizing malformed tool-call IDs, including aborted native tool calls with empty IDs, so retries no longer send invalid `tool_use.id` / `tool_result.tool_use_id` pairs.
1451
1490
  - Fixed the Codex Responses WebSocket transport attributing a prior turn's output to the current one on a reused connection: a trailing/duplicate frame from a cleanly-completed previous response that slipped past the queue drain could be consumed as this request's terminal (ending the turn with empty output) or as a stale tool call. Frames are now keyed by `response.id` — a frame carrying the previous response's id is dropped, and one carrying a third id (or a regressed `sequence_number`) fails closed so the turn retries instead of mixing two responses' streams. Idless frames (deltas, the rate-limit/metadata preamble, `response.created`-less streams) still pass through, matching upstream codex-rs.
1452
- - Fixed `transformMessages` pulling an earlier, orphaned tool result onto a later tool call that reused the same id (left behind when compaction folded the originating `tool_use` into a summary). The pending-call flush now pairs each call with a result positioned *after* its assistant turn, so a reused id surfaces its own output rather than a prior turn's.
1491
+ - Fixed `transformMessages` pulling an earlier, orphaned tool result onto a later tool call that reused the same id (left behind when compaction folded the originating `tool_use` into a summary). The pending-call flush now pairs each call with a result positioned _after_ its assistant turn, so a reused id surfaces its own output rather than a prior turn's.
1453
1492
  - Fixed DashScope 429 rate-limit messages that mention authorization being classified as credential failures, preventing valid API keys from being invalidated after throttling. ([#3172](https://github.com/can1357/oh-my-pi/issues/3172))
1454
1493
  - Fixed OpenCode Go `401 Insufficient balance` quota errors being treated as unknown failures instead of usage-limit errors, restoring credential rotation and fallback chains. ([#3169](https://github.com/can1357/oh-my-pi/issues/3169))
1455
1494
 
@@ -1725,7 +1764,7 @@
1725
1764
  - Added `wrapInbandToolStream` function to process streaming responses with in-band tool call parsing
1726
1765
  - Added `ThinkingInbandScanner` for parsing thinking/reasoning blocks across dialects
1727
1766
  - Added `OwnedStream` class for managing dialect-aware streaming with tool call events
1728
- - Added in-band thinking channels to every dialect that was missing one: `gemini` (a ```` ```thinking ```` fence mirroring ```` ```tool_code ````), `gemma` (its native `<|channel>thought…<channel|>` reasoning channel), `kimi` (`<think>…</think>`), and `pi` (`<thinking>…</thinking>`). Each scanner now parses reasoning into thinking events instead of leaking chain-of-thought into the visible reply, and every dialect's `renderThinking` is a real channel that round-trips back through its scanner (no passthrough renderers).
1767
+ - Added in-band thinking channels to every dialect that was missing one: `gemini` (a ` ```thinking ` fence mirroring ` ```tool_code `), `gemma` (its native `<|channel>thought…<channel|>` reasoning channel), `kimi` (`<think>…</think>`), and `pi` (`<thinking>…</thinking>`). Each scanner now parses reasoning into thinking events instead of leaking chain-of-thought into the visible reply, and every dialect's `renderThinking` is a real channel that round-trips back through its scanner (no passthrough renderers).
1729
1768
 
1730
1769
  ### Changed
1731
1770
 
@@ -1749,7 +1788,7 @@
1749
1788
 
1750
1789
  ### Added
1751
1790
 
1752
- - Added the `gemini` in-band tool-call syntax with Python-style ```tool_code``` blocks and `default_api` invocations
1791
+ - Added the `gemini` in-band tool-call syntax with Python-style `tool_code` blocks and `default_api` invocations
1753
1792
  - Added the `gemma` token-delimited in-band tool-call syntax using `<|tool_call>` and `<|tool_response>` blocks
1754
1793
  - Added `gemini` and `gemma` to owned stream tool-result token detection so their tool responses are recognized
1755
1794
  - Fixed truncated Gemini and Gemma tool blocks from being emitted as plain text during streaming
@@ -1794,7 +1833,7 @@
1794
1833
  - Fixed Harmony leak handling support by adding `recoverHarmonyToolCall` plus leak-detection workflows for contaminated assistant messages so recoverable tool-call arguments can be safely truncated and retried
1795
1834
  - Fixed false-positive gating in Harmony leak heuristics using signal-based checks so unrelated text containing `to=functions...` is not treated as leaked tool-call markup
1796
1835
  - Routed Kimi, DeepSeek DSML, and plain thinking markup healing through the shared in-band scanners so provider leak repair and owned tool calling parse the same wire formats.
1797
- - Fixed Cursor provider (`cursor-agent` API) streaming dropping large MCP tool-call arguments — most visibly the built-in `task` tool's `tasks` array on multi-subagent dispatches, which failed downstream schema validation with `tasks: Invalid input: expected array, received undefined`. Two upstream behaviors were fighting the stream handler in `packages/ai/src/providers/cursor.ts`: (1) `args_text_delta` carries the *cumulative* args text so far per `agent.proto`, but the handler concatenated each snapshot onto the buffer, garbling the JSON; (2) `tool_call_completed` carries an `McpArgs` map that omits oversized parameters entirely and downgrades unparsable values to their raw string fallback, but the handler unconditionally overwrote the streamed args with that map. The handler now strips the already-buffered prefix from each `args_text_delta` snapshot (falling back to append when the snapshot doesn't extend the buffer) and merges the decoded `McpArgs` map into the streamed args — preserving streamed keys the completion frame omits and the structured value when the completion frame downgrades to a string. ([#2615](https://github.com/can1357/oh-my-pi/issues/2615))
1836
+ - Fixed Cursor provider (`cursor-agent` API) streaming dropping large MCP tool-call arguments — most visibly the built-in `task` tool's `tasks` array on multi-subagent dispatches, which failed downstream schema validation with `tasks: Invalid input: expected array, received undefined`. Two upstream behaviors were fighting the stream handler in `packages/ai/src/providers/cursor.ts`: (1) `args_text_delta` carries the _cumulative_ args text so far per `agent.proto`, but the handler concatenated each snapshot onto the buffer, garbling the JSON; (2) `tool_call_completed` carries an `McpArgs` map that omits oversized parameters entirely and downgrades unparsable values to their raw string fallback, but the handler unconditionally overwrote the streamed args with that map. The handler now strips the already-buffered prefix from each `args_text_delta` snapshot (falling back to append when the snapshot doesn't extend the buffer) and merges the decoded `McpArgs` map into the streamed args — preserving streamed keys the completion frame omits and the structured value when the completion frame downgrades to a string. ([#2615](https://github.com/can1357/oh-my-pi/issues/2615))
1798
1837
  - Fixed Codex Responses stream mis-routing interleaved `function_call_arguments.delta` events when more than one tool call was open concurrently. The runtime tracked a singleton `currentItem`/`currentBlock`, so every delta — regardless of `item_id` — was appended to whichever item was most recently added, and `output_item.done` for the earlier call then overwrote a sibling's stored arguments (visible as `tasks: Invalid input: expected array, received undefined` on the `task` tool). Open items are now keyed by `item_id` with `output_index` fallback; deltas/done events route to the matching block, late deltas whose item already closed are dropped instead of corrupting a sibling, and `toolcall_*` stream events emit the right `contentIndex` per call ([#2619](https://github.com/can1357/oh-my-pi/issues/2619)).
1799
1838
 
1800
1839
  ## [15.13.1] - 2026-06-15
@@ -2003,7 +2042,7 @@
2003
2042
 
2004
2043
  ### Breaking Changes
2005
2044
 
2006
- - The model catalog moved to the new `@oh-my-pi/pi-catalog` package. Deep subpath exports `@oh-my-pi/pi-ai/models.json`, `/models`, `/model-cache`, `/model-manager`, `/model-thinking`, `/effort`, `/provider-models*`, `/utils/discovery*`, `/providers/openai-codex/constants`, `/providers/google-gemini-headers`, and `/providers/openai-completions-compat` are gone — import the `@oh-my-pi/pi-catalog` equivalents (`/models.json`, `/models`, `/model-cache`, `/model-manager`, `/model-thinking`, `/effort`, `/provider-models*`, `/discovery*`, `/wire/codex`, `/wire/gemini-headers`, `/compat/openai`). The pi-ai root barrel re-exports only the model/effort *types* its own signatures use (`Model`, `Api`, `ThinkingConfig`, `Effort`, `Usage`, compat interfaces) — catalog *values* (`getBundledModel(s)`, `calculateCost`, `modelsAreEqual`, `clampThinkingLevelForModel`, `DEFAULT_MODEL_PER_PROVIDER`, …) must be imported from `@oh-my-pi/pi-catalog`.
2045
+ - The model catalog moved to the new `@oh-my-pi/pi-catalog` package. Deep subpath exports `@oh-my-pi/pi-ai/models.json`, `/models`, `/model-cache`, `/model-manager`, `/model-thinking`, `/effort`, `/provider-models*`, `/utils/discovery*`, `/providers/openai-codex/constants`, `/providers/google-gemini-headers`, and `/providers/openai-completions-compat` are gone — import the `@oh-my-pi/pi-catalog` equivalents (`/models.json`, `/models`, `/model-cache`, `/model-manager`, `/model-thinking`, `/effort`, `/provider-models*`, `/discovery*`, `/wire/codex`, `/wire/gemini-headers`, `/compat/openai`). The pi-ai root barrel re-exports only the model/effort _types_ its own signatures use (`Model`, `Api`, `ThinkingConfig`, `Effort`, `Usage`, compat interfaces) — catalog _values_ (`getBundledModel(s)`, `calculateCost`, `modelsAreEqual`, `clampThinkingLevelForModel`, `DEFAULT_MODEL_PER_PROVIDER`, …) must be imported from `@oh-my-pi/pi-catalog`.
2007
2046
  - `ProviderDefinition` is now auth-only: `defaultModel`, `createModelManagerOptions`, `catalogDiscovery`, `dynamicModelsAuthoritative`, `allowUnauthenticated`, and `specialModelManager` moved to pi-catalog's `CATALOG_PROVIDERS` table, and `KnownProviderId` was replaced by pi-catalog's `KnownProvider` (registry completeness is enforced by a compile-time check against that union). The pure GitHub Copilot key/endpoint helpers moved from `registry/oauth/github-copilot` to `@oh-my-pi/pi-catalog/wire/github-copilot`.
2008
2047
 
2009
2048
  ### Added
@@ -2092,7 +2131,7 @@
2092
2131
  - Fixed `AnthropicMessagesClient` spreading `fetchOptions` after the core request fields, letting a caller-supplied `signal`/`method`/`body` silently disconnect the timeout controller or corrupt the request. Transport extras (TLS) still pass through; core fields now always win.
2093
2132
  - Fixed Foundry mTLS/CA material being cached for the process lifetime when the env vars point at files: the cache key now folds in the file mtime so on-disk certificate rotation takes effect.
2094
2133
  - Fixed the Claude Code fingerprint version drifting across surfaces: the usage endpoint (`claude-cli/2.1.160`) and OAuth bootstrap (`claude-code/2.1.160`) pinned a stale version while `/v1/messages` reported 2.1.165; both now derive from `claudeCodeVersion`.
2095
- - Fixed a system prompt that merely *mentions* `x-anthropic-billing-header:` mid-text suppressing the entire Claude Code system-block injection (billing header, instruction, and cch attestation); the resumed-session guard now anchors with `startsWith`.
2134
+ - Fixed a system prompt that merely _mentions_ `x-anthropic-billing-header:` mid-text suppressing the entire Claude Code system-block injection (billing header, instruction, and cch attestation); the resumed-session guard now anchors with `startsWith`.
2096
2135
  - Fixed lone surrogates in cross-API tool-call arguments reaching Anthropic's strict UTF-8 validation: replayed OpenAI/Google-origin `tool_use.input` string leaves are now deep-sanitized with `toWellFormed()`, while same-API Anthropic arguments stay byte-identical to keep prompt-cache prefixes stable.
2097
2136
  - Bounded the many-image resize fan-out to 4 concurrent decodes (it previously decoded every oversized image at once, two encode pipelines each — multi-GB transient memory at the 20+-image threshold that activates the feature).
2098
2137
  - Fixed `mergeHeaders` merging case-sensitively on the Copilot/client-options path, where a miscased user-configured header (e.g. `authorization` next to the synthesized `Authorization`) survived as two keys that the `Headers` constructor joins comma-separated on the wire.
@@ -47,5 +47,12 @@ export interface BedrockOptions extends StreamOptions {
47
47
  * we omit it for them.
48
48
  */
49
49
  thinkingDisplay?: BedrockThinkingDisplay;
50
+ /**
51
+ * Per-request Bedrock invocation-log tags. Merged over `model.requestMetadata`
52
+ * (per-call entries win on key collision). AWS caps the result at 16 entries;
53
+ * keys 1-256 chars, values 0-256 chars, both limited to
54
+ * `[a-zA-Z0-9\s:_@$#=/+,-.]`. Entries outside those limits are dropped.
55
+ */
56
+ requestMetadata?: Record<string, string>;
50
57
  }
51
58
  export declare const streamBedrock: StreamFunction<"bedrock-converse-stream">;
@@ -164,7 +164,7 @@ export type AnthropicClientOptionsArgs = {
164
164
  disableStrictTools?: boolean;
165
165
  fetch?: FetchImpl;
166
166
  maxRetryDelayMs?: number;
167
- claudeCodeSessionId?: string;
167
+ sessionId?: string;
168
168
  };
169
169
  export type AnthropicClientOptionsResult = {
170
170
  isOAuthToken: boolean;
@@ -0,0 +1,24 @@
1
+ /** Shared inference request identity headers. */
2
+ import type { FetchImpl } from "../types.js";
3
+ /** Options controlling provider and protocol inference headers. */
4
+ export interface InferenceHeaderOptions {
5
+ provider: string;
6
+ protocol: "anthropic" | "google" | "openai";
7
+ sessionId?: string;
8
+ }
9
+ /** Set a header unless the map already contains that field under any casing. */
10
+ export declare function setHeaderIfAbsent(headers: Record<string, string>, name: string, value: string): void;
11
+ /**
12
+ * Project omp's identity and authoritative conversation id onto the headers
13
+ * understood by the active inference protocol and host.
14
+ */
15
+ export declare function applyInferenceHeaders(headers: Record<string, string>, options: InferenceHeaderOptions): void;
16
+ /**
17
+ * Apply omp's process-wide inference User-Agent default. Any explicit header,
18
+ * including Anthropic and Codex OAuth fingerprints, remains authoritative.
19
+ *
20
+ * Plain-object headers stay plain objects: custom `fetch` implementations
21
+ * (proxies, tests) index `init.headers` by name and must not be handed a
22
+ * `Headers` instance instead.
23
+ */
24
+ export declare function withInferenceUserAgent(fetchImpl: FetchImpl): FetchImpl;
@@ -21,10 +21,8 @@ export interface CodexRequestOptions {
21
21
  textVerbosity?: "low" | "medium" | "high";
22
22
  include?: string[];
23
23
  /**
24
- * Responses Lite transport override; defaults to the model's
25
- * `useResponsesLite`. Lite moves instructions/tools into input items,
26
- * strips image detail, and disables parallel tool calling (codex-rs
27
- * `use_responses_lite`).
24
+ * Responses Lite transport opt-in. Normal inference defaults to full
25
+ * Responses so the model can emit independent tool calls in parallel.
28
26
  */
29
27
  responsesLite?: boolean;
30
28
  }
@@ -70,13 +68,12 @@ export interface RequestBody {
70
68
  [key: string]: unknown;
71
69
  }
72
70
  /**
73
- * Resolve whether a Codex request uses the Responses Lite transport: an
74
- * explicit option wins, then the `PI_CODEX_RESPONSES_LITE` env override
75
- * (`1`/`true` forces Lite, `0`/`false` forces the full Responses body),
76
- * otherwise the model's catalog flag (codex-rs `model_info.use_responses_lite`)
77
- * decides.
71
+ * Resolve whether a Codex request explicitly opts into Responses Lite.
72
+ *
73
+ * Provider-native compaction passes the model's `useResponsesLite` flag as an
74
+ * explicit option; normal inference defaults to the full Responses contract.
78
75
  */
79
- export declare function resolveCodexResponsesLite(model: Model<"openai-codex-responses">, requested: boolean | undefined): boolean;
76
+ export declare function resolveCodexResponsesLite(requested: boolean | undefined): boolean;
80
77
  /**
81
78
  * Structural view of a Responses-style body mutated by the Lite rewrite.
82
79
  * Loose (`unknown`) property types let the turn transformer (`RequestBody`)
@@ -12,13 +12,9 @@ export interface OpenAICodexResponsesOptions extends StreamOptions {
12
12
  preferWebsockets?: boolean;
13
13
  serviceTier?: ServiceTier;
14
14
  /**
15
- * Responses Lite transport override; defaults to the model's catalog
16
- * `useResponsesLite` flag (codex-rs `use_responses_lite`). Sends
17
- * `x-openai-internal-codex-responses-lite: true` on HTTP requests and on the
18
- * WebSocket upgrade (the marker is connection-scoped there, so lite and
19
- * non-lite turns never share a pooled socket), moves instructions/tools
20
- * into input items, strips image detail, and disables parallel tool
21
- * calling — mirroring codex-rs.
15
+ * Responses Lite transport opt-in. Normal inference defaults to full
16
+ * Responses; provider-native compaction explicitly follows the model's
17
+ * `useResponsesLite` flag.
22
18
  */
23
19
  responsesLite?: boolean;
24
20
  /**
@@ -0,0 +1,64 @@
1
+ /**
2
+ * Mid-conversation reasoning effort via `configuration_update` input items
3
+ * (GPT-6 Astra; `model.compat.supportsConfigurationUpdate`).
4
+ *
5
+ * The request-level `reasoning.effort` is pinned to the value of the session's
6
+ * first request so the cached prompt prefix survives an effort change. Each
7
+ * later change is carried as a `configuration_update` item inserted at the
8
+ * tail of the transcript — before the user message it takes effect on, or
9
+ * after the latest tool result when the level changes inside a tool loop — and
10
+ * replayed at that position on every subsequent request until another update
11
+ * overrides it. Mirrors the Anthropic provider's stable `output_config.effort`
12
+ * planning.
13
+ *
14
+ * Used by both the platform Responses provider and the Codex provider; the
15
+ * state lives in each provider's session state, keyed per conversation.
16
+ *
17
+ * Wire constraints (verified against the Codex backend): only `gpt-6-astra`
18
+ * accepts the item type, consecutive updates are rejected, and
19
+ * `/responses/compact` rejects histories containing them — compaction
20
+ * requests are built outside this planner and never carry the items.
21
+ */
22
+ /** `configuration_update` input item; only `reasoning.effort` is updatable. */
23
+ export interface ConfigurationUpdateItem {
24
+ type: "configuration_update";
25
+ reasoning: {
26
+ effort: string;
27
+ };
28
+ }
29
+ interface EffortTransition<TEffort extends string> {
30
+ /** Input-array position the item is spliced into (before `input[index]`). */
31
+ index: number;
32
+ /** Fingerprint of `input[index - 1]` at record time; a mismatch means the history was rewritten. */
33
+ anchor: string;
34
+ effort: TEffort;
35
+ }
36
+ /** Per-conversation effort baseline and recorded transitions. */
37
+ export interface OpenAIEffortControlState<TEffort extends string = string> {
38
+ baseEffort?: TEffort;
39
+ currentEffort?: TEffort;
40
+ transitions: EffortTransition<TEffort>[];
41
+ }
42
+ export declare function createOpenAIEffortControlState<TEffort extends string>(): OpenAIEffortControlState<TEffort>;
43
+ /**
44
+ * Fetch (or create) the control state for one conversation from a provider's
45
+ * bounded per-session map, refreshing its LRU slot.
46
+ */
47
+ export declare function getOpenAIEffortControlState<TEffort extends string>(states: Map<string, OpenAIEffortControlState<TEffort>>, key: string): OpenAIEffortControlState<TEffort>;
48
+ interface AnchorableItem {
49
+ type?: string | null;
50
+ role?: string;
51
+ id?: string | null;
52
+ status?: string | null;
53
+ }
54
+ /**
55
+ * Pin the request-level effort to the session baseline and splice pending
56
+ * `configuration_update` items into `input` (mutated in place).
57
+ *
58
+ * `input` is the freshly built transcript for this request, without any
59
+ * `configuration_update` items. `requested` is the wire effort the caller
60
+ * would otherwise send at the request level. Returns the effort to send at the
61
+ * request level (`requested` on the first request, the baseline afterwards).
62
+ */
63
+ export declare function planStableOpenAIEffort<TItem extends AnchorableItem, TEffort extends string>(state: OpenAIEffortControlState<TEffort>, input: Array<TItem | ConfigurationUpdateItem>, requested: TEffort): TEffort;
64
+ export {};
@@ -2918,7 +2918,7 @@ export interface ResponseInputImageContent {
2918
2918
  * `assistant` role are presumed to have been generated by the model in previous
2919
2919
  * interactions.
2920
2920
  */
2921
- export type ResponseInputItem = EasyInputMessage | ResponseInputItem.Message | ResponseOutputMessage | ResponseFileSearchToolCall | ResponseComputerToolCall | ResponseInputItem.ComputerCallOutput | ResponseFunctionWebSearch | ResponseFunctionToolCall | ResponseInputItem.FunctionCallOutput | ResponseInputItem.ToolSearchCall | ResponseToolSearchOutputItemParam | ResponseInputItem.AdditionalTools | ResponseReasoningItem | ResponseCompactionItemParam | ResponseInputItem.ImageGenerationCall | ResponseCodeInterpreterToolCall | ResponseInputItem.LocalShellCall | ResponseInputItem.LocalShellCallOutput | ResponseInputItem.ShellCall | ResponseInputItem.ShellCallOutput | ResponseInputItem.ApplyPatchCall | ResponseInputItem.ApplyPatchCallOutput | ResponseInputItem.McpListTools | ResponseInputItem.McpApprovalRequest | ResponseInputItem.McpApprovalResponse | ResponseInputItem.McpCall | ResponseCustomToolCallOutput | ResponseCustomToolCall | ResponseInputItem.CompactionTrigger | ResponseInputItem.ItemReference;
2921
+ export type ResponseInputItem = EasyInputMessage | ResponseInputItem.Message | ResponseOutputMessage | ResponseFileSearchToolCall | ResponseComputerToolCall | ResponseInputItem.ComputerCallOutput | ResponseFunctionWebSearch | ResponseFunctionToolCall | ResponseInputItem.FunctionCallOutput | ResponseInputItem.ToolSearchCall | ResponseToolSearchOutputItemParam | ResponseInputItem.AdditionalTools | ResponseReasoningItem | ResponseCompactionItemParam | ResponseInputItem.ImageGenerationCall | ResponseCodeInterpreterToolCall | ResponseInputItem.LocalShellCall | ResponseInputItem.LocalShellCallOutput | ResponseInputItem.ShellCall | ResponseInputItem.ShellCallOutput | ResponseInputItem.ApplyPatchCall | ResponseInputItem.ApplyPatchCallOutput | ResponseInputItem.McpListTools | ResponseInputItem.McpApprovalRequest | ResponseInputItem.McpApprovalResponse | ResponseInputItem.McpCall | ResponseCustomToolCallOutput | ResponseCustomToolCall | ResponseInputItem.CompactionTrigger | ResponseInputItem.ConfigurationUpdate | ResponseInputItem.ItemReference;
2922
2922
  export declare namespace ResponseInputItem {
2923
2923
  /**
2924
2924
  * A message input to the model with a role indicating instruction following
@@ -3504,6 +3504,20 @@ export declare namespace ResponseInputItem {
3504
3504
  */
3505
3505
  type: "compaction_trigger";
3506
3506
  }
3507
+ /**
3508
+ * Changes reasoning effort for subsequent responses without touching the
3509
+ * request-level `reasoning.effort` (GPT-6 Astra). Must not be adjacent to
3510
+ * another `configuration_update`.
3511
+ */
3512
+ interface ConfigurationUpdate {
3513
+ /**
3514
+ * The type of the item. Always `configuration_update`.
3515
+ */
3516
+ type: "configuration_update";
3517
+ reasoning: {
3518
+ effort: string;
3519
+ };
3520
+ }
3507
3521
  /**
3508
3522
  * An internal identifier for an item to reference.
3509
3523
  */
@@ -1,7 +1,8 @@
1
1
  import type { Context, Model, OpenAICompat, ProviderSessionState, ServiceTier, StreamFunction, StreamOptions, Tool, ToolChoice } from "../types.js";
2
2
  import { type OpenAIResponsesToolChoice } from "../utils/tool-choice.js";
3
+ import { type OpenAIEffortControlState } from "./openai-configuration-update.js";
3
4
  import { type OpenAIReasoningEffortFallbackState } from "./openai-reasoning-fallback.js";
4
- import type { Tool as OpenAITool, ResponseCreateParamsStreaming, ResponseInput } from "./openai-responses-wire.js";
5
+ import type { Tool as OpenAITool, ReasoningEffort, ResponseCreateParamsStreaming, ResponseInput } from "./openai-responses-wire.js";
5
6
  import { type OpenAIPromptCacheOptions, type OpenAIStrictToolsScope, type OpenAIStrictToolsState } from "./openai-shared.js";
6
7
  export interface OpenAIResponsesOptions extends StreamOptions {
7
8
  reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
@@ -63,7 +64,11 @@ interface OpenAIResponsesProviderSessionState extends ProviderSessionState, Open
63
64
  nativeHistoryReplayWarmed: boolean;
64
65
  /** Stateful `previous_response_id` chain baselines, keyed by baseUrl/model/session. */
65
66
  chains: Map<string, OpenAIResponsesChainState>;
67
+ /** `configuration_update` effort baselines, keyed by baseUrl/model/session. */
68
+ effortControls: Map<string, OpenAIEffortControlState<ResponsesStableEffort>>;
66
69
  }
70
+ /** Wire efforts a `configuration_update` can carry: every real tier, never `none`/null. */
71
+ type ResponsesStableEffort = Exclude<ReasoningEffort, "none" | null>;
67
72
  interface OpenAIResponsesChainState {
68
73
  /**
69
74
  * Wire params of the last successful turn; never carries
@@ -56,7 +56,7 @@ export interface OpenAIRequestSetupOptions {
56
56
  apiVersion: string;
57
57
  deploymentName: string;
58
58
  };
59
- openAISessionId?: string;
59
+ sessionId?: string;
60
60
  promptCacheSessionId?: string;
61
61
  }
62
62
  export interface OpenAIRequestSetup {
@@ -493,6 +493,18 @@ export interface BuildResponsesInputOptions<TApi extends Api> {
493
493
  */
494
494
  export declare function escapeReplayedControlTokens(items: ResponseInput): ResponseInput;
495
495
  export declare function buildResponsesInput<TApi extends Api>(options: BuildResponsesInputOptions<TApi>): ResponseInput;
496
+ /**
497
+ * Non-empty `reasoning_text` shipped for a synthesized reasoning item when no
498
+ * thinking text survived history reconstruction. DeepSeek-family Responses
499
+ * targets (e.g. opencode-go) reject BOTH a missing reasoning item and one whose
500
+ * `reasoning_text` is empty — "The reasoning_text in the thinking mode must be
501
+ * passed back to the API" (#8248 covered the missing case, #10690 the empty
502
+ * one). The item's presence plus a non-empty payload is what satisfies the
503
+ * contract; the exact text is immaterial once the source turn's reasoning is
504
+ * gone. Kept out of `reasoning_content="."`-territory since DeepSeek rejects the
505
+ * bare-dot synthetic placeholder on the chat-completions path.
506
+ */
507
+ export declare const SYNTHETIC_REASONING_REPLAY_PLACEHOLDER = "reasoning unavailable";
496
508
  export declare function convertResponsesAssistantMessage<TApi extends Api>(assistantMsg: AssistantMessage, model: Model<TApi>, msgIndex: number, knownCallIds: Set<string>, includeThinkingSignatures?: boolean, customCallIds?: Set<string>, preserveMessageIds?: boolean, supportsCustomToolCalls?: boolean, customToolWireNameMap?: ReadonlyMap<string, string>, computerCallIds?: Set<string>, requiresReasoningReplayForAllTurns?: boolean, requiresReasoningReplayForToolCalls?: boolean): ResponseInput;
497
509
  /**
498
510
  * Responses wire output for a tool result plus its text-only fallback.
@@ -463,6 +463,13 @@ export interface SimpleStreamOptions extends Omit<StreamOptions, "apiKey"> {
463
463
  guardrailIdentifier?: string;
464
464
  guardrailVersion?: string;
465
465
  guardrailTrace?: "enabled" | "disabled" | "enabled_full";
466
+ /**
467
+ * Bedrock invocation-log tags forwarded through transports that do not dispatch
468
+ * directly to the Bedrock provider. Unlike the guardrail fields above, these
469
+ * MERGE per key with the model's own `requestMetadata` (these win) rather than
470
+ * replacing it wholesale — they are independent attribution tags, not one value.
471
+ */
472
+ requestMetadata?: Record<string, string>;
466
473
  /** Optional tool choice override for compatible providers */
467
474
  toolChoice?: ToolChoice;
468
475
  /** OpenAI service tier for processing priority/cost control. Ignored by non-OpenAI providers. */
@@ -1,6 +1,13 @@
1
1
  import type { ResponseInput } from "./providers/openai-responses-wire.js";
2
2
  import type { CacheRetention, OpenAIResponsesHistoryPayload, ProviderPayload } from "./types.js";
3
3
  export { isRecord } from "@oh-my-pi/pi-utils";
4
+ /**
5
+ * Read a header value ignoring key casing. HTTP header names are
6
+ * case-insensitive, but `Record<string, string>` header bags are not, so a
7
+ * config-authored `User-Agent` and a caller-authored `user-agent` are the same
8
+ * header to every provider that lowercases before merging.
9
+ */
10
+ export declare function getHeaderCaseInsensitive(headers: Record<string, string> | undefined, headerName: string): string | undefined;
4
11
  export declare function normalizeSystemPrompts(systemPrompt: readonly string[] | string | undefined | null): string[];
5
12
  export declare function normalizeToolCallId(id: string): string;
6
13
  type ResponsesToolItemIdPrefix = "fc" | "ctc";
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@oh-my-pi/pi-ai",
4
- "version": "18.1.5",
4
+ "version": "18.1.8",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://omp.sh",
7
7
  "author": "Stencil Labs, Inc.",
@@ -37,10 +37,10 @@
37
37
  "fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
38
38
  },
39
39
  "dependencies": {
40
- "@oh-my-pi/omptype": "18.1.5",
41
- "@oh-my-pi/pi-catalog": "18.1.5",
42
- "@oh-my-pi/pi-utils": "18.1.5",
43
- "@oh-my-pi/pi-wire": "18.1.5"
40
+ "@oh-my-pi/omptype": "18.1.8",
41
+ "@oh-my-pi/pi-catalog": "18.1.8",
42
+ "@oh-my-pi/pi-utils": "18.1.8",
43
+ "@oh-my-pi/pi-wire": "18.1.8"
44
44
  },
45
45
  "devDependencies": {
46
46
  "@types/bun": "^1.3.14"
@@ -24,6 +24,12 @@ const SERVER_ERROR_BACKOFF_MS = 20 * 1000; // 20s
24
24
  const ACCOUNT_RATE_LIMIT_PATTERN =
25
25
  /\baccount(?:'s)?\b[^\n]{0,80}\brate.?limit\b|\brate.?limit\b[^\n]{0,80}\baccount\b/i;
26
26
  const INSUFFICIENT_BALANCE_PATTERN = /insufficient.?balance/i;
27
+ // Prepaid-credit exhaustion phrased around the credit balance rather than a
28
+ // quota: Anthropic "This request would exceed your available credits given
29
+ // your current in-flight requests" (402), OpenRouter "Insufficient credits",
30
+ // "credits exhausted". Account-local, so rotate to a sibling credential.
31
+ const CREDITS_EXHAUSTED_PATTERN =
32
+ /\b(?:exceed\w*|insufficient|not enough)\b[^\n]{0,40}\bcredits?\b|\bcredits?\b[^\n]{0,40}\b(?:exhausted|depleted)\b/i;
27
33
  const SPEND_LIMIT_PATTERN = /spend.?limit/i;
28
34
  const SUBSCRIPTION_CAP_PATTERN =
29
35
  /\b(?:subscription|plan|membership)\b[^\n]{0,80}\b(?:rate.?limits?|quota|cap)\b|\b(?:rate.?limits?|quota|cap)\b[^\n]{0,80}\b(?:subscription|plan|membership)\b/i;
@@ -247,7 +253,8 @@ export function parseRateLimitReason(errorMessage: string): RateLimitReason {
247
253
  lower.includes("out of credits") ||
248
254
  lower.includes("spending-limit") ||
249
255
  lower.includes("spending limit") ||
250
- INSUFFICIENT_BALANCE_PATTERN.test(errorMessage)
256
+ INSUFFICIENT_BALANCE_PATTERN.test(errorMessage) ||
257
+ CREDITS_EXHAUSTED_PATTERN.test(errorMessage)
251
258
  ) {
252
259
  return "QUOTA_EXHAUSTED";
253
260
  }
@@ -390,6 +397,7 @@ export function matchesUsageLimitText(errorMessage: string): boolean {
390
397
  if (isDashScopeTokenLimitText(errorMessage)) return false;
391
398
  return (
392
399
  USAGE_LIMIT_PATTERN.test(errorMessage) ||
400
+ CREDITS_EXHAUSTED_PATTERN.test(errorMessage) ||
393
401
  (CN_QUOTA_EXHAUSTED_PATTERN.test(errorMessage) && !CN_TRANSIENT_CAP_PATTERN.test(errorMessage)) ||
394
402
  SPEND_LIMIT_PATTERN.test(errorMessage) ||
395
403
  ACCOUNT_RATE_LIMIT_PATTERN.test(errorMessage) ||
@@ -10,7 +10,14 @@
10
10
  import type { Effort } from "@oh-my-pi/pi-catalog/effort";
11
11
  import { mapEffortToAnthropicAdaptiveEffort, requireSupportedEffort } from "@oh-my-pi/pi-catalog/model-thinking";
12
12
  import { calculateCost } from "@oh-my-pi/pi-catalog/models";
13
- import { $flag, fetchWithRetry, logger, parseStreamingJson, parseStreamingJsonThrottled } from "@oh-my-pi/pi-utils";
13
+ import {
14
+ $flag,
15
+ fetchWithRetry,
16
+ logger,
17
+ parseStreamingJson,
18
+ parseStreamingJsonThrottled,
19
+ USER_AGENT,
20
+ } from "@oh-my-pi/pi-utils";
14
21
  import { renderDemotedThinking } from "../dialect/demotion";
15
22
  import * as AIError from "../error";
16
23
  import { resolveAwsBearerToken } from "../registry/aws";
@@ -102,6 +109,13 @@ export interface BedrockOptions extends StreamOptions {
102
109
  * we omit it for them.
103
110
  */
104
111
  thinkingDisplay?: BedrockThinkingDisplay;
112
+ /**
113
+ * Per-request Bedrock invocation-log tags. Merged over `model.requestMetadata`
114
+ * (per-call entries win on key collision). AWS caps the result at 16 entries;
115
+ * keys 1-256 chars, values 0-256 chars, both limited to
116
+ * `[a-zA-Z0-9\s:_@$#=/+,-.]`. Entries outside those limits are dropped.
117
+ */
118
+ requestMetadata?: Record<string, string>;
105
119
  }
106
120
 
107
121
  function resolveBearerToken(options: BedrockOptions): string | undefined {
@@ -283,6 +297,7 @@ interface ConverseStreamRequest {
283
297
  toolConfig?: WireToolConfig;
284
298
  guardrailConfig?: WireGuardrailConfig;
285
299
  additionalModelRequestFields?: Record<string, unknown>;
300
+ requestMetadata?: Record<string, string>;
286
301
  additionalModelResponseFieldPaths?: string[];
287
302
  }
288
303
 
@@ -319,6 +334,41 @@ interface MetadataEvent {
319
334
  };
320
335
  }
321
336
 
337
+ const REQUEST_METADATA_PATTERN = /^[a-zA-Z0-9\s:_@$#=/+,\-.]*$/;
338
+ const REQUEST_METADATA_MAX_ENTRIES = 16;
339
+ const REQUEST_METADATA_MAX_LENGTH = 256;
340
+
341
+ /**
342
+ * Bedrock rejects the whole invocation on a malformed `requestMetadata` entry.
343
+ * Attribution tags must never cost a turn, so invalid and excess entries are
344
+ * dropped with a warning instead of failing the request. Returns `undefined`
345
+ * for an empty result so the field is omitted from the body entirely.
346
+ */
347
+ function sanitizeRequestMetadata(raw: unknown): Record<string, string> | undefined {
348
+ if (!isRecord(raw)) return undefined;
349
+ const out: Record<string, string> = {};
350
+ const dropped: string[] = [];
351
+ let kept = 0;
352
+ for (const [key, value] of Object.entries(raw)) {
353
+ if (
354
+ typeof value !== "string" ||
355
+ key.length < 1 ||
356
+ key.length > REQUEST_METADATA_MAX_LENGTH ||
357
+ !REQUEST_METADATA_PATTERN.test(key) ||
358
+ value.length > REQUEST_METADATA_MAX_LENGTH ||
359
+ !REQUEST_METADATA_PATTERN.test(value) ||
360
+ kept >= REQUEST_METADATA_MAX_ENTRIES
361
+ ) {
362
+ dropped.push(key);
363
+ continue;
364
+ }
365
+ out[key] = value;
366
+ kept++;
367
+ }
368
+ if (dropped.length > 0) logger.warn("Bedrock requestMetadata entries dropped", { keys: dropped });
369
+ return kept > 0 ? out : undefined;
370
+ }
371
+
322
372
  export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
323
373
  model: Model<"bedrock-converse-stream">,
324
374
  context: Context,
@@ -391,10 +441,17 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
391
441
  toolConfig,
392
442
  guardrailConfig: buildGuardrailConfig(options),
393
443
  additionalModelRequestFields,
444
+ requestMetadata:
445
+ model.requestMetadata || options.requestMetadata
446
+ ? { ...model.requestMetadata, ...options.requestMetadata }
447
+ : undefined,
394
448
  ...(prefixMismatchBehavior ? { additionalModelResponseFieldPaths: ["/input_transformations"] } : {}),
395
449
  };
396
450
  const replacementInput = await options?.onPayload?.(commandInput, model);
397
451
  if (replacementInput !== undefined) commandInput = replacementInput as ConverseStreamRequest;
452
+ // After the hook so extension-injected tags are validated too, and before the
453
+ // raw dump so the inspector shows exactly what was sent.
454
+ commandInput = { ...commandInput, requestMetadata: sanitizeRequestMetadata(commandInput.requestMetadata) };
398
455
 
399
456
  const host = `bedrock-runtime.${region}.amazonaws.com`;
400
457
  const url = `https://${host}/model/${encodeURIComponent(model.id)}/converse-stream`;
@@ -429,11 +486,21 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
429
486
  // comma-joined wire header, so AWS validates different bytes than were
430
487
  // signed and rejects the request.
431
488
  const callerHeaders: Record<string, string> = {};
432
- for (const [name, value] of Object.entries(options?.headers ?? {})) {
433
- const field = name.toLowerCase();
434
- if (SIGNER_OWNED_HEADERS.has(field) || BEDROCK_RESERVED_HEADERS.has(field)) continue;
435
- callerHeaders[field] = value;
489
+ // `model.headers` first, `options.headers` second: `StreamOptions.headers` is
490
+ // documented (types.ts:431-435) as merged ON TOP of model-defined headers.
491
+ // Both pass the same filter, so a config-authored `Host`/`Content-Type`
492
+ // cannot desync the signature either.
493
+ for (const source of [model.headers, options?.headers]) {
494
+ for (const [name, value] of Object.entries(source ?? {})) {
495
+ const field = name.toLowerCase();
496
+ if (SIGNER_OWNED_HEADERS.has(field) || BEDROCK_RESERVED_HEADERS.has(field)) continue;
497
+ callerHeaders[field] = value;
498
+ }
436
499
  }
500
+ // SigV4 never signs `user-agent` (UNSIGNABLE in aws-sigv4.ts:42-58), so this
501
+ // default cannot break the signature. Without it Bun's fetch sends
502
+ // `Bun/<version>`, and that is what CloudTrail records for every request.
503
+ callerHeaders["user-agent"] ??= USER_AGENT;
437
504
  const baseHeaders: Record<string, string> = {
438
505
  ...callerHeaders,
439
506
  "content-type": "application/json",