@oh-my-pi/pi-ai 18.1.5 → 18.1.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,29 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [18.1.6] - 2026-09-03
6
+
7
+ ### Breaking Changes
8
+
9
+ - Renamed `claudeCodeSessionId` to `sessionId` in `AnthropicClientOptionsArgs`.
10
+ - Renamed `openAISessionId` to `sessionId` in `OpenAIRequestSetupOptions`.
11
+
12
+ ### Added
13
+
14
+ - Added Amazon Bedrock `requestMetadata` support for cost and usage attribution in AWS invocation logs.
15
+
16
+ ### Changed
17
+
18
+ - Codex GPT-5.6 requests now use full Responses by default, enabling independent tool calls to run in parallel; provider-native compaction continues to use catalog-selected Responses Lite.
19
+ - Inference requests now identify as omp by default while preserving explicit provider and OAuth User-Agent fingerprints. Amazon Bedrock requests use an `omp/<version>` User-Agent by default and honor configured `User-Agent` overrides.
20
+
21
+ ### Fixed
22
+
23
+ - Fixed Antigravity usage reporting to match the official client's five-hour and weekly quota buckets.
24
+ - Anthropic and OpenRouter credit-exhaustion errors now automatically switch to a sibling account instead of stopping the turn with a retry hint.
25
+ - Fixed OpenCode Go and Zen requests by including the required stable per-conversation session identification.
26
+ - Improved Anthropic prompt caching so explicit cache breakpoints preserve reusable tools and system prompts when the message tail changes.
27
+
5
28
  ## [18.1.5] - 2026-09-03
6
29
 
7
30
  ### Added
@@ -183,7 +206,7 @@
183
206
  - Fixed Codex continuations, retries, and compaction replacing or dropping the turn-scoped sticky-routing token ([#9277](https://github.com/can1357/oh-my-pi/issues/9277)).
184
207
  - Fixed Codex Responses append chains falling back to full-context replay when replay-sanitized assistant items differ only by output-only IDs or lifecycle status.
185
208
  - Fixed Cursor usage reporting “no usage data” for plans without a numeric legacy request cap.
186
- - Fixed DeepSeek models rejecting requests with HTTP 400 `unknown variant \`image_url\`, expected \`text\`` when screenshots or image-producing tool results are present in conversation history or when `model.input` claims vision capability; `convertMessages` in `openai-completions` now strips `image_url` content parts and injects non-vision image placeholders for all DeepSeek endpoints.
209
+ - Fixed DeepSeek models rejecting requests with HTTP 400 `unknown variant \`image_url\`, expected \`text\``when screenshots or image-producing tool results are present in conversation history or when`model.input`claims vision capability;`convertMessages`in`openai-completions`now strips`image_url` content parts and injects non-vision image placeholders for all DeepSeek endpoints.
187
210
  - Fixed `PI_PROXY` covering only provider streams: OAuth token refresh and login, usage probes, and model discovery went out through the bare global `fetch` and ignored it, so a region-blocked token endpoint answered `403 Request not allowed` (Anthropic `/v1/oauth/token`) and disabled the credential while the proxied stream itself worked. `installGlobalProxyFetch()` now routes the process-wide `fetch` through `PI_PROXY`; a per-request proxy such as `PI_PROXY_<PROVIDER>` still wins, and loopback / private-range / `NO_PROXY` targets stay direct.
188
211
  - Fixed Anthropic inference ignoring every proxy setting. `coworkFetch` runs on `node:https`, whose Bun shim discards both `agent.createConnection` and `options.createConnection`: the CONNECT tunnel to `PI_PROXY` was built, TLS-negotiated, then abandoned, and the request dialed `api.anthropic.com` on the default route (measured at the proxy: 581 bytes of handshake, zero request bytes). On a region-blocked egress that returned `403 {"type":"forbidden","message":"Request not allowed"}` with the proxy apparently configured. Proxied requests now go through Bun's own `fetch`, which honors `init.proxy`, trading the Cowork TLS/header profile for a proxy that actually carries the traffic; the dead tunnel plumbing is gone from the transport. `node:http2` (Cursor) does honor `createConnection` and is unaffected.
189
212
  - Fixed `cowork-fetch` capturing `globalThis.fetch` at module load, so a proxy wrapper installed later in startup was ignored on its fallback path.
@@ -253,7 +276,7 @@
253
276
 
254
277
  ### Changed
255
278
 
256
- - Fixed Gemini thought summaries occasionally leaking a raw `` ```thinking `` / `` ``````thinking `` fence delimiter into the reasoning block, so it no longer shows up as fence spam in the thinking display or persisted transcripts ([#8719](https://github.com/can1357/oh-my-pi/issues/8719)).
279
+ - Fixed Gemini thought summaries occasionally leaking a raw ` ```thinking ` / ` ``````thinking ` fence delimiter into the reasoning block, so it no longer shows up as fence spam in the thinking display or persisted transcripts ([#8719](https://github.com/can1357/oh-my-pi/issues/8719)).
257
280
  - Fixed the OpenCode Go login prompting for an "OpenCode Zen API key": the shared login flow now names the provider you selected, so connecting OpenCode Go asks for an OpenCode Go key (the `opencode.ai/auth` console is still shared, as documented upstream) ([#8738](https://github.com/can1357/oh-my-pi/issues/8738)).
258
281
  - Fixed Anthropic-compatible endpoints with strict prompt validation (e.g. Z.AI GLM `api.z.ai/api/anthropic`, which rejects the whole request with `400 code 1213 "The prompt parameter was not received normally"`) failing sessions once a tool returned empty output on a vision-capable model: empty successful `tool_result` blocks now encode as `content: ""` instead of `content: []`, which both the official API and strict compatible endpoints accept.
259
282
  - Fixed `retry.usageReservePct` (Reserve Margin) ignoring Claude Fable/Mythos weekly tier usage until it hit 100%, so a Fable model kept serving turns past the configured reserve; reserve health now honors the mapped tier row while credential-wide hard blocks still require confirmed exhaustion ([#8773](https://github.com/can1357/oh-my-pi/issues/8773)).
@@ -1305,7 +1328,7 @@
1305
1328
 
1306
1329
  - Added a third streaming thinking-loop detection heuristic to catch "progress-lexicon stalls" where models endlessly reshuffle motivational filler without introducing new vocabulary or concrete technical references
1307
1330
  - Added branded wordmark and logo animation to authentication flow
1308
- - Added a third streaming thinking-loop detection shape — a *progress-lexicon stall* — alongside verbatim tail repetition and near-duplicate (trigram) segments. It catches reasoning-summarizer loops that reshuffle the same motivational filler ("just doing it, pushing ahead, maintaining momentum") into fresh word order every paragraph: word-trigrams never cluster, but a run of substantial segments that recycle the recent vocabulary and introduce no *new* concrete reference (path / identifier / code-span) trips the guard. Summarizer title/heading lines (`**Bold Title**`, `## Heading`) are stripped before analysis so their ever-changing wording cannot mask the stall by inflating novelty. Calibrated against 537k real non-Gemini reasoning blocks (zero false positives at novelty floor 0.2 / run length 8; the real loop sustains runs of 10+).
1331
+ - Added a third streaming thinking-loop detection shape — a _progress-lexicon stall_ — alongside verbatim tail repetition and near-duplicate (trigram) segments. It catches reasoning-summarizer loops that reshuffle the same motivational filler ("just doing it, pushing ahead, maintaining momentum") into fresh word order every paragraph: word-trigrams never cluster, but a run of substantial segments that recycle the recent vocabulary and introduce no _new_ concrete reference (path / identifier / code-span) trips the guard. Summarizer title/heading lines (`**Bold Title**`, `## Heading`) are stripped before analysis so their ever-changing wording cannot mask the stall by inflating novelty. Calibrated against 537k real non-Gemini reasoning blocks (zero false positives at novelty floor 0.2 / run length 8; the real loop sustains runs of 10+).
1309
1332
  - Added CoreWeave Serverless Inference provider login support via `COREWEAVE_API_KEY` and `WANDB_API_KEY` fallback.
1310
1333
 
1311
1334
  ### Changed
@@ -1449,7 +1472,7 @@
1449
1472
  - Fixed tool call ID normalization for Anthropic-compatible models
1450
1473
  - Fixed Anthropic Messages replay sanitizing malformed tool-call IDs, including aborted native tool calls with empty IDs, so retries no longer send invalid `tool_use.id` / `tool_result.tool_use_id` pairs.
1451
1474
  - Fixed the Codex Responses WebSocket transport attributing a prior turn's output to the current one on a reused connection: a trailing/duplicate frame from a cleanly-completed previous response that slipped past the queue drain could be consumed as this request's terminal (ending the turn with empty output) or as a stale tool call. Frames are now keyed by `response.id` — a frame carrying the previous response's id is dropped, and one carrying a third id (or a regressed `sequence_number`) fails closed so the turn retries instead of mixing two responses' streams. Idless frames (deltas, the rate-limit/metadata preamble, `response.created`-less streams) still pass through, matching upstream codex-rs.
1452
- - Fixed `transformMessages` pulling an earlier, orphaned tool result onto a later tool call that reused the same id (left behind when compaction folded the originating `tool_use` into a summary). The pending-call flush now pairs each call with a result positioned *after* its assistant turn, so a reused id surfaces its own output rather than a prior turn's.
1475
+ - Fixed `transformMessages` pulling an earlier, orphaned tool result onto a later tool call that reused the same id (left behind when compaction folded the originating `tool_use` into a summary). The pending-call flush now pairs each call with a result positioned _after_ its assistant turn, so a reused id surfaces its own output rather than a prior turn's.
1453
1476
  - Fixed DashScope 429 rate-limit messages that mention authorization being classified as credential failures, preventing valid API keys from being invalidated after throttling. ([#3172](https://github.com/can1357/oh-my-pi/issues/3172))
1454
1477
  - Fixed OpenCode Go `401 Insufficient balance` quota errors being treated as unknown failures instead of usage-limit errors, restoring credential rotation and fallback chains. ([#3169](https://github.com/can1357/oh-my-pi/issues/3169))
1455
1478
 
@@ -1725,7 +1748,7 @@
1725
1748
  - Added `wrapInbandToolStream` function to process streaming responses with in-band tool call parsing
1726
1749
  - Added `ThinkingInbandScanner` for parsing thinking/reasoning blocks across dialects
1727
1750
  - Added `OwnedStream` class for managing dialect-aware streaming with tool call events
1728
- - Added in-band thinking channels to every dialect that was missing one: `gemini` (a ```` ```thinking ```` fence mirroring ```` ```tool_code ````), `gemma` (its native `<|channel>thought…<channel|>` reasoning channel), `kimi` (`<think>…</think>`), and `pi` (`<thinking>…</thinking>`). Each scanner now parses reasoning into thinking events instead of leaking chain-of-thought into the visible reply, and every dialect's `renderThinking` is a real channel that round-trips back through its scanner (no passthrough renderers).
1751
+ - Added in-band thinking channels to every dialect that was missing one: `gemini` (a ` ```thinking ` fence mirroring ` ```tool_code `), `gemma` (its native `<|channel>thought…<channel|>` reasoning channel), `kimi` (`<think>…</think>`), and `pi` (`<thinking>…</thinking>`). Each scanner now parses reasoning into thinking events instead of leaking chain-of-thought into the visible reply, and every dialect's `renderThinking` is a real channel that round-trips back through its scanner (no passthrough renderers).
1729
1752
 
1730
1753
  ### Changed
1731
1754
 
@@ -1749,7 +1772,7 @@
1749
1772
 
1750
1773
  ### Added
1751
1774
 
1752
- - Added the `gemini` in-band tool-call syntax with Python-style ```tool_code``` blocks and `default_api` invocations
1775
+ - Added the `gemini` in-band tool-call syntax with Python-style `tool_code` blocks and `default_api` invocations
1753
1776
  - Added the `gemma` token-delimited in-band tool-call syntax using `<|tool_call>` and `<|tool_response>` blocks
1754
1777
  - Added `gemini` and `gemma` to owned stream tool-result token detection so their tool responses are recognized
1755
1778
  - Fixed truncated Gemini and Gemma tool blocks from being emitted as plain text during streaming
@@ -1794,7 +1817,7 @@
1794
1817
  - Fixed Harmony leak handling support by adding `recoverHarmonyToolCall` plus leak-detection workflows for contaminated assistant messages so recoverable tool-call arguments can be safely truncated and retried
1795
1818
  - Fixed false-positive gating in Harmony leak heuristics using signal-based checks so unrelated text containing `to=functions...` is not treated as leaked tool-call markup
1796
1819
  - Routed Kimi, DeepSeek DSML, and plain thinking markup healing through the shared in-band scanners so provider leak repair and owned tool calling parse the same wire formats.
1797
- - Fixed Cursor provider (`cursor-agent` API) streaming dropping large MCP tool-call arguments — most visibly the built-in `task` tool's `tasks` array on multi-subagent dispatches, which failed downstream schema validation with `tasks: Invalid input: expected array, received undefined`. Two upstream behaviors were fighting the stream handler in `packages/ai/src/providers/cursor.ts`: (1) `args_text_delta` carries the *cumulative* args text so far per `agent.proto`, but the handler concatenated each snapshot onto the buffer, garbling the JSON; (2) `tool_call_completed` carries an `McpArgs` map that omits oversized parameters entirely and downgrades unparsable values to their raw string fallback, but the handler unconditionally overwrote the streamed args with that map. The handler now strips the already-buffered prefix from each `args_text_delta` snapshot (falling back to append when the snapshot doesn't extend the buffer) and merges the decoded `McpArgs` map into the streamed args — preserving streamed keys the completion frame omits and the structured value when the completion frame downgrades to a string. ([#2615](https://github.com/can1357/oh-my-pi/issues/2615))
1820
+ - Fixed Cursor provider (`cursor-agent` API) streaming dropping large MCP tool-call arguments — most visibly the built-in `task` tool's `tasks` array on multi-subagent dispatches, which failed downstream schema validation with `tasks: Invalid input: expected array, received undefined`. Two upstream behaviors were fighting the stream handler in `packages/ai/src/providers/cursor.ts`: (1) `args_text_delta` carries the _cumulative_ args text so far per `agent.proto`, but the handler concatenated each snapshot onto the buffer, garbling the JSON; (2) `tool_call_completed` carries an `McpArgs` map that omits oversized parameters entirely and downgrades unparsable values to their raw string fallback, but the handler unconditionally overwrote the streamed args with that map. The handler now strips the already-buffered prefix from each `args_text_delta` snapshot (falling back to append when the snapshot doesn't extend the buffer) and merges the decoded `McpArgs` map into the streamed args — preserving streamed keys the completion frame omits and the structured value when the completion frame downgrades to a string. ([#2615](https://github.com/can1357/oh-my-pi/issues/2615))
1798
1821
  - Fixed Codex Responses stream mis-routing interleaved `function_call_arguments.delta` events when more than one tool call was open concurrently. The runtime tracked a singleton `currentItem`/`currentBlock`, so every delta — regardless of `item_id` — was appended to whichever item was most recently added, and `output_item.done` for the earlier call then overwrote a sibling's stored arguments (visible as `tasks: Invalid input: expected array, received undefined` on the `task` tool). Open items are now keyed by `item_id` with `output_index` fallback; deltas/done events route to the matching block, late deltas whose item already closed are dropped instead of corrupting a sibling, and `toolcall_*` stream events emit the right `contentIndex` per call ([#2619](https://github.com/can1357/oh-my-pi/issues/2619)).
1799
1822
 
1800
1823
  ## [15.13.1] - 2026-06-15
@@ -2003,7 +2026,7 @@
2003
2026
 
2004
2027
  ### Breaking Changes
2005
2028
 
2006
- - The model catalog moved to the new `@oh-my-pi/pi-catalog` package. Deep subpath exports `@oh-my-pi/pi-ai/models.json`, `/models`, `/model-cache`, `/model-manager`, `/model-thinking`, `/effort`, `/provider-models*`, `/utils/discovery*`, `/providers/openai-codex/constants`, `/providers/google-gemini-headers`, and `/providers/openai-completions-compat` are gone — import the `@oh-my-pi/pi-catalog` equivalents (`/models.json`, `/models`, `/model-cache`, `/model-manager`, `/model-thinking`, `/effort`, `/provider-models*`, `/discovery*`, `/wire/codex`, `/wire/gemini-headers`, `/compat/openai`). The pi-ai root barrel re-exports only the model/effort *types* its own signatures use (`Model`, `Api`, `ThinkingConfig`, `Effort`, `Usage`, compat interfaces) — catalog *values* (`getBundledModel(s)`, `calculateCost`, `modelsAreEqual`, `clampThinkingLevelForModel`, `DEFAULT_MODEL_PER_PROVIDER`, …) must be imported from `@oh-my-pi/pi-catalog`.
2029
+ - The model catalog moved to the new `@oh-my-pi/pi-catalog` package. Deep subpath exports `@oh-my-pi/pi-ai/models.json`, `/models`, `/model-cache`, `/model-manager`, `/model-thinking`, `/effort`, `/provider-models*`, `/utils/discovery*`, `/providers/openai-codex/constants`, `/providers/google-gemini-headers`, and `/providers/openai-completions-compat` are gone — import the `@oh-my-pi/pi-catalog` equivalents (`/models.json`, `/models`, `/model-cache`, `/model-manager`, `/model-thinking`, `/effort`, `/provider-models*`, `/discovery*`, `/wire/codex`, `/wire/gemini-headers`, `/compat/openai`). The pi-ai root barrel re-exports only the model/effort _types_ its own signatures use (`Model`, `Api`, `ThinkingConfig`, `Effort`, `Usage`, compat interfaces) — catalog _values_ (`getBundledModel(s)`, `calculateCost`, `modelsAreEqual`, `clampThinkingLevelForModel`, `DEFAULT_MODEL_PER_PROVIDER`, …) must be imported from `@oh-my-pi/pi-catalog`.
2007
2030
  - `ProviderDefinition` is now auth-only: `defaultModel`, `createModelManagerOptions`, `catalogDiscovery`, `dynamicModelsAuthoritative`, `allowUnauthenticated`, and `specialModelManager` moved to pi-catalog's `CATALOG_PROVIDERS` table, and `KnownProviderId` was replaced by pi-catalog's `KnownProvider` (registry completeness is enforced by a compile-time check against that union). The pure GitHub Copilot key/endpoint helpers moved from `registry/oauth/github-copilot` to `@oh-my-pi/pi-catalog/wire/github-copilot`.
2008
2031
 
2009
2032
  ### Added
@@ -2092,7 +2115,7 @@
2092
2115
  - Fixed `AnthropicMessagesClient` spreading `fetchOptions` after the core request fields, letting a caller-supplied `signal`/`method`/`body` silently disconnect the timeout controller or corrupt the request. Transport extras (TLS) still pass through; core fields now always win.
2093
2116
  - Fixed Foundry mTLS/CA material being cached for the process lifetime when the env vars point at files: the cache key now folds in the file mtime so on-disk certificate rotation takes effect.
2094
2117
  - Fixed the Claude Code fingerprint version drifting across surfaces: the usage endpoint (`claude-cli/2.1.160`) and OAuth bootstrap (`claude-code/2.1.160`) pinned a stale version while `/v1/messages` reported 2.1.165; both now derive from `claudeCodeVersion`.
2095
- - Fixed a system prompt that merely *mentions* `x-anthropic-billing-header:` mid-text suppressing the entire Claude Code system-block injection (billing header, instruction, and cch attestation); the resumed-session guard now anchors with `startsWith`.
2118
+ - Fixed a system prompt that merely _mentions_ `x-anthropic-billing-header:` mid-text suppressing the entire Claude Code system-block injection (billing header, instruction, and cch attestation); the resumed-session guard now anchors with `startsWith`.
2096
2119
  - Fixed lone surrogates in cross-API tool-call arguments reaching Anthropic's strict UTF-8 validation: replayed OpenAI/Google-origin `tool_use.input` string leaves are now deep-sanitized with `toWellFormed()`, while same-API Anthropic arguments stay byte-identical to keep prompt-cache prefixes stable.
2097
2120
  - Bounded the many-image resize fan-out to 4 concurrent decodes (it previously decoded every oversized image at once, two encode pipelines each — multi-GB transient memory at the 20+-image threshold that activates the feature).
2098
2121
  - Fixed `mergeHeaders` merging case-sensitively on the Copilot/client-options path, where a miscased user-configured header (e.g. `authorization` next to the synthesized `Authorization`) survived as two keys that the `Headers` constructor joins comma-separated on the wire.
@@ -47,5 +47,12 @@ export interface BedrockOptions extends StreamOptions {
47
47
  * we omit it for them.
48
48
  */
49
49
  thinkingDisplay?: BedrockThinkingDisplay;
50
+ /**
51
+ * Per-request Bedrock invocation-log tags. Merged over `model.requestMetadata`
52
+ * (per-call entries win on key collision). AWS caps the result at 16 entries;
53
+ * keys 1-256 chars, values 0-256 chars, both limited to
54
+ * `[a-zA-Z0-9\s:_@$#=/+,-.]`. Entries outside those limits are dropped.
55
+ */
56
+ requestMetadata?: Record<string, string>;
50
57
  }
51
58
  export declare const streamBedrock: StreamFunction<"bedrock-converse-stream">;
@@ -164,7 +164,7 @@ export type AnthropicClientOptionsArgs = {
164
164
  disableStrictTools?: boolean;
165
165
  fetch?: FetchImpl;
166
166
  maxRetryDelayMs?: number;
167
- claudeCodeSessionId?: string;
167
+ sessionId?: string;
168
168
  };
169
169
  export type AnthropicClientOptionsResult = {
170
170
  isOAuthToken: boolean;
@@ -0,0 +1,24 @@
1
+ /** Shared inference request identity headers. */
2
+ import type { FetchImpl } from "../types.js";
3
+ /** Options controlling provider and protocol inference headers. */
4
+ export interface InferenceHeaderOptions {
5
+ provider: string;
6
+ protocol: "anthropic" | "google" | "openai";
7
+ sessionId?: string;
8
+ }
9
+ /** Set a header unless the map already contains that field under any casing. */
10
+ export declare function setHeaderIfAbsent(headers: Record<string, string>, name: string, value: string): void;
11
+ /**
12
+ * Project omp's identity and authoritative conversation id onto the headers
13
+ * understood by the active inference protocol and host.
14
+ */
15
+ export declare function applyInferenceHeaders(headers: Record<string, string>, options: InferenceHeaderOptions): void;
16
+ /**
17
+ * Apply omp's process-wide inference User-Agent default. Any explicit header,
18
+ * including Anthropic and Codex OAuth fingerprints, remains authoritative.
19
+ *
20
+ * Plain-object headers stay plain objects: custom `fetch` implementations
21
+ * (proxies, tests) index `init.headers` by name and must not be handed a
22
+ * `Headers` instance instead.
23
+ */
24
+ export declare function withInferenceUserAgent(fetchImpl: FetchImpl): FetchImpl;
@@ -21,10 +21,8 @@ export interface CodexRequestOptions {
21
21
  textVerbosity?: "low" | "medium" | "high";
22
22
  include?: string[];
23
23
  /**
24
- * Responses Lite transport override; defaults to the model's
25
- * `useResponsesLite`. Lite moves instructions/tools into input items,
26
- * strips image detail, and disables parallel tool calling (codex-rs
27
- * `use_responses_lite`).
24
+ * Responses Lite transport opt-in. Normal inference defaults to full
25
+ * Responses so the model can emit independent tool calls in parallel.
28
26
  */
29
27
  responsesLite?: boolean;
30
28
  }
@@ -70,13 +68,12 @@ export interface RequestBody {
70
68
  [key: string]: unknown;
71
69
  }
72
70
  /**
73
- * Resolve whether a Codex request uses the Responses Lite transport: an
74
- * explicit option wins, then the `PI_CODEX_RESPONSES_LITE` env override
75
- * (`1`/`true` forces Lite, `0`/`false` forces the full Responses body),
76
- * otherwise the model's catalog flag (codex-rs `model_info.use_responses_lite`)
77
- * decides.
71
+ * Resolve whether a Codex request explicitly opts into Responses Lite.
72
+ *
73
+ * Provider-native compaction passes the model's `useResponsesLite` flag as an
74
+ * explicit option; normal inference defaults to the full Responses contract.
78
75
  */
79
- export declare function resolveCodexResponsesLite(model: Model<"openai-codex-responses">, requested: boolean | undefined): boolean;
76
+ export declare function resolveCodexResponsesLite(requested: boolean | undefined): boolean;
80
77
  /**
81
78
  * Structural view of a Responses-style body mutated by the Lite rewrite.
82
79
  * Loose (`unknown`) property types let the turn transformer (`RequestBody`)
@@ -12,13 +12,9 @@ export interface OpenAICodexResponsesOptions extends StreamOptions {
12
12
  preferWebsockets?: boolean;
13
13
  serviceTier?: ServiceTier;
14
14
  /**
15
- * Responses Lite transport override; defaults to the model's catalog
16
- * `useResponsesLite` flag (codex-rs `use_responses_lite`). Sends
17
- * `x-openai-internal-codex-responses-lite: true` on HTTP requests and on the
18
- * WebSocket upgrade (the marker is connection-scoped there, so lite and
19
- * non-lite turns never share a pooled socket), moves instructions/tools
20
- * into input items, strips image detail, and disables parallel tool
21
- * calling — mirroring codex-rs.
15
+ * Responses Lite transport opt-in. Normal inference defaults to full
16
+ * Responses; provider-native compaction explicitly follows the model's
17
+ * `useResponsesLite` flag.
22
18
  */
23
19
  responsesLite?: boolean;
24
20
  /**
@@ -56,7 +56,7 @@ export interface OpenAIRequestSetupOptions {
56
56
  apiVersion: string;
57
57
  deploymentName: string;
58
58
  };
59
- openAISessionId?: string;
59
+ sessionId?: string;
60
60
  promptCacheSessionId?: string;
61
61
  }
62
62
  export interface OpenAIRequestSetup {
@@ -463,6 +463,13 @@ export interface SimpleStreamOptions extends Omit<StreamOptions, "apiKey"> {
463
463
  guardrailIdentifier?: string;
464
464
  guardrailVersion?: string;
465
465
  guardrailTrace?: "enabled" | "disabled" | "enabled_full";
466
+ /**
467
+ * Bedrock invocation-log tags forwarded through transports that do not dispatch
468
+ * directly to the Bedrock provider. Unlike the guardrail fields above, these
469
+ * MERGE per key with the model's own `requestMetadata` (these win) rather than
470
+ * replacing it wholesale — they are independent attribution tags, not one value.
471
+ */
472
+ requestMetadata?: Record<string, string>;
466
473
  /** Optional tool choice override for compatible providers */
467
474
  toolChoice?: ToolChoice;
468
475
  /** OpenAI service tier for processing priority/cost control. Ignored by non-OpenAI providers. */
@@ -1,6 +1,13 @@
1
1
  import type { ResponseInput } from "./providers/openai-responses-wire.js";
2
2
  import type { CacheRetention, OpenAIResponsesHistoryPayload, ProviderPayload } from "./types.js";
3
3
  export { isRecord } from "@oh-my-pi/pi-utils";
4
+ /**
5
+ * Read a header value ignoring key casing. HTTP header names are
6
+ * case-insensitive, but `Record<string, string>` header bags are not, so a
7
+ * config-authored `User-Agent` and a caller-authored `user-agent` are the same
8
+ * header to every provider that lowercases before merging.
9
+ */
10
+ export declare function getHeaderCaseInsensitive(headers: Record<string, string> | undefined, headerName: string): string | undefined;
4
11
  export declare function normalizeSystemPrompts(systemPrompt: readonly string[] | string | undefined | null): string[];
5
12
  export declare function normalizeToolCallId(id: string): string;
6
13
  type ResponsesToolItemIdPrefix = "fc" | "ctc";
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@oh-my-pi/pi-ai",
4
- "version": "18.1.5",
4
+ "version": "18.1.6",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://omp.sh",
7
7
  "author": "Stencil Labs, Inc.",
@@ -37,10 +37,10 @@
37
37
  "fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
38
38
  },
39
39
  "dependencies": {
40
- "@oh-my-pi/omptype": "18.1.5",
41
- "@oh-my-pi/pi-catalog": "18.1.5",
42
- "@oh-my-pi/pi-utils": "18.1.5",
43
- "@oh-my-pi/pi-wire": "18.1.5"
40
+ "@oh-my-pi/omptype": "18.1.6",
41
+ "@oh-my-pi/pi-catalog": "18.1.6",
42
+ "@oh-my-pi/pi-utils": "18.1.6",
43
+ "@oh-my-pi/pi-wire": "18.1.6"
44
44
  },
45
45
  "devDependencies": {
46
46
  "@types/bun": "^1.3.14"
@@ -24,6 +24,12 @@ const SERVER_ERROR_BACKOFF_MS = 20 * 1000; // 20s
24
24
  const ACCOUNT_RATE_LIMIT_PATTERN =
25
25
  /\baccount(?:'s)?\b[^\n]{0,80}\brate.?limit\b|\brate.?limit\b[^\n]{0,80}\baccount\b/i;
26
26
  const INSUFFICIENT_BALANCE_PATTERN = /insufficient.?balance/i;
27
+ // Prepaid-credit exhaustion phrased around the credit balance rather than a
28
+ // quota: Anthropic "This request would exceed your available credits given
29
+ // your current in-flight requests" (402), OpenRouter "Insufficient credits",
30
+ // "credits exhausted". Account-local, so rotate to a sibling credential.
31
+ const CREDITS_EXHAUSTED_PATTERN =
32
+ /\b(?:exceed\w*|insufficient|not enough)\b[^\n]{0,40}\bcredits?\b|\bcredits?\b[^\n]{0,40}\b(?:exhausted|depleted)\b/i;
27
33
  const SPEND_LIMIT_PATTERN = /spend.?limit/i;
28
34
  const SUBSCRIPTION_CAP_PATTERN =
29
35
  /\b(?:subscription|plan|membership)\b[^\n]{0,80}\b(?:rate.?limits?|quota|cap)\b|\b(?:rate.?limits?|quota|cap)\b[^\n]{0,80}\b(?:subscription|plan|membership)\b/i;
@@ -247,7 +253,8 @@ export function parseRateLimitReason(errorMessage: string): RateLimitReason {
247
253
  lower.includes("out of credits") ||
248
254
  lower.includes("spending-limit") ||
249
255
  lower.includes("spending limit") ||
250
- INSUFFICIENT_BALANCE_PATTERN.test(errorMessage)
256
+ INSUFFICIENT_BALANCE_PATTERN.test(errorMessage) ||
257
+ CREDITS_EXHAUSTED_PATTERN.test(errorMessage)
251
258
  ) {
252
259
  return "QUOTA_EXHAUSTED";
253
260
  }
@@ -390,6 +397,7 @@ export function matchesUsageLimitText(errorMessage: string): boolean {
390
397
  if (isDashScopeTokenLimitText(errorMessage)) return false;
391
398
  return (
392
399
  USAGE_LIMIT_PATTERN.test(errorMessage) ||
400
+ CREDITS_EXHAUSTED_PATTERN.test(errorMessage) ||
393
401
  (CN_QUOTA_EXHAUSTED_PATTERN.test(errorMessage) && !CN_TRANSIENT_CAP_PATTERN.test(errorMessage)) ||
394
402
  SPEND_LIMIT_PATTERN.test(errorMessage) ||
395
403
  ACCOUNT_RATE_LIMIT_PATTERN.test(errorMessage) ||
@@ -10,7 +10,14 @@
10
10
  import type { Effort } from "@oh-my-pi/pi-catalog/effort";
11
11
  import { mapEffortToAnthropicAdaptiveEffort, requireSupportedEffort } from "@oh-my-pi/pi-catalog/model-thinking";
12
12
  import { calculateCost } from "@oh-my-pi/pi-catalog/models";
13
- import { $flag, fetchWithRetry, logger, parseStreamingJson, parseStreamingJsonThrottled } from "@oh-my-pi/pi-utils";
13
+ import {
14
+ $flag,
15
+ fetchWithRetry,
16
+ logger,
17
+ parseStreamingJson,
18
+ parseStreamingJsonThrottled,
19
+ USER_AGENT,
20
+ } from "@oh-my-pi/pi-utils";
14
21
  import { renderDemotedThinking } from "../dialect/demotion";
15
22
  import * as AIError from "../error";
16
23
  import { resolveAwsBearerToken } from "../registry/aws";
@@ -102,6 +109,13 @@ export interface BedrockOptions extends StreamOptions {
102
109
  * we omit it for them.
103
110
  */
104
111
  thinkingDisplay?: BedrockThinkingDisplay;
112
+ /**
113
+ * Per-request Bedrock invocation-log tags. Merged over `model.requestMetadata`
114
+ * (per-call entries win on key collision). AWS caps the result at 16 entries;
115
+ * keys 1-256 chars, values 0-256 chars, both limited to
116
+ * `[a-zA-Z0-9\s:_@$#=/+,-.]`. Entries outside those limits are dropped.
117
+ */
118
+ requestMetadata?: Record<string, string>;
105
119
  }
106
120
 
107
121
  function resolveBearerToken(options: BedrockOptions): string | undefined {
@@ -283,6 +297,7 @@ interface ConverseStreamRequest {
283
297
  toolConfig?: WireToolConfig;
284
298
  guardrailConfig?: WireGuardrailConfig;
285
299
  additionalModelRequestFields?: Record<string, unknown>;
300
+ requestMetadata?: Record<string, string>;
286
301
  additionalModelResponseFieldPaths?: string[];
287
302
  }
288
303
 
@@ -319,6 +334,41 @@ interface MetadataEvent {
319
334
  };
320
335
  }
321
336
 
337
+ const REQUEST_METADATA_PATTERN = /^[a-zA-Z0-9\s:_@$#=/+,\-.]*$/;
338
+ const REQUEST_METADATA_MAX_ENTRIES = 16;
339
+ const REQUEST_METADATA_MAX_LENGTH = 256;
340
+
341
+ /**
342
+ * Bedrock rejects the whole invocation on a malformed `requestMetadata` entry.
343
+ * Attribution tags must never cost a turn, so invalid and excess entries are
344
+ * dropped with a warning instead of failing the request. Returns `undefined`
345
+ * for an empty result so the field is omitted from the body entirely.
346
+ */
347
+ function sanitizeRequestMetadata(raw: unknown): Record<string, string> | undefined {
348
+ if (!isRecord(raw)) return undefined;
349
+ const out: Record<string, string> = {};
350
+ const dropped: string[] = [];
351
+ let kept = 0;
352
+ for (const [key, value] of Object.entries(raw)) {
353
+ if (
354
+ typeof value !== "string" ||
355
+ key.length < 1 ||
356
+ key.length > REQUEST_METADATA_MAX_LENGTH ||
357
+ !REQUEST_METADATA_PATTERN.test(key) ||
358
+ value.length > REQUEST_METADATA_MAX_LENGTH ||
359
+ !REQUEST_METADATA_PATTERN.test(value) ||
360
+ kept >= REQUEST_METADATA_MAX_ENTRIES
361
+ ) {
362
+ dropped.push(key);
363
+ continue;
364
+ }
365
+ out[key] = value;
366
+ kept++;
367
+ }
368
+ if (dropped.length > 0) logger.warn("Bedrock requestMetadata entries dropped", { keys: dropped });
369
+ return kept > 0 ? out : undefined;
370
+ }
371
+
322
372
  export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
323
373
  model: Model<"bedrock-converse-stream">,
324
374
  context: Context,
@@ -391,10 +441,17 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
391
441
  toolConfig,
392
442
  guardrailConfig: buildGuardrailConfig(options),
393
443
  additionalModelRequestFields,
444
+ requestMetadata:
445
+ model.requestMetadata || options.requestMetadata
446
+ ? { ...model.requestMetadata, ...options.requestMetadata }
447
+ : undefined,
394
448
  ...(prefixMismatchBehavior ? { additionalModelResponseFieldPaths: ["/input_transformations"] } : {}),
395
449
  };
396
450
  const replacementInput = await options?.onPayload?.(commandInput, model);
397
451
  if (replacementInput !== undefined) commandInput = replacementInput as ConverseStreamRequest;
452
+ // After the hook so extension-injected tags are validated too, and before the
453
+ // raw dump so the inspector shows exactly what was sent.
454
+ commandInput = { ...commandInput, requestMetadata: sanitizeRequestMetadata(commandInput.requestMetadata) };
398
455
 
399
456
  const host = `bedrock-runtime.${region}.amazonaws.com`;
400
457
  const url = `https://${host}/model/${encodeURIComponent(model.id)}/converse-stream`;
@@ -429,11 +486,21 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
429
486
  // comma-joined wire header, so AWS validates different bytes than were
430
487
  // signed and rejects the request.
431
488
  const callerHeaders: Record<string, string> = {};
432
- for (const [name, value] of Object.entries(options?.headers ?? {})) {
433
- const field = name.toLowerCase();
434
- if (SIGNER_OWNED_HEADERS.has(field) || BEDROCK_RESERVED_HEADERS.has(field)) continue;
435
- callerHeaders[field] = value;
489
+ // `model.headers` first, `options.headers` second: `StreamOptions.headers` is
490
+ // documented (types.ts:431-435) as merged ON TOP of model-defined headers.
491
+ // Both pass the same filter, so a config-authored `Host`/`Content-Type`
492
+ // cannot desync the signature either.
493
+ for (const source of [model.headers, options?.headers]) {
494
+ for (const [name, value] of Object.entries(source ?? {})) {
495
+ const field = name.toLowerCase();
496
+ if (SIGNER_OWNED_HEADERS.has(field) || BEDROCK_RESERVED_HEADERS.has(field)) continue;
497
+ callerHeaders[field] = value;
498
+ }
436
499
  }
500
+ // SigV4 never signs `user-agent` (UNSIGNABLE in aws-sigv4.ts:42-58), so this
501
+ // default cannot break the signature. Without it Bun's fetch sends
502
+ // `Bun/<version>`, and that is what CloudTrail records for every request.
503
+ callerHeaders["user-agent"] ??= USER_AGENT;
437
504
  const baseHeaders: Record<string, string> = {
438
505
  ...callerHeaders,
439
506
  "content-type": "application/json",
@@ -48,7 +48,13 @@ import type {
48
48
  ToolResultMessage,
49
49
  Usage,
50
50
  } from "../types";
51
- import { isRecord, normalizeSystemPrompts, normalizeToolCallId, resolveCacheRetention } from "../utils";
51
+ import {
52
+ getHeaderCaseInsensitive,
53
+ isRecord,
54
+ normalizeSystemPrompts,
55
+ normalizeToolCallId,
56
+ resolveCacheRetention,
57
+ } from "../utils";
52
58
  import { createAbortSourceTracker } from "../utils/abort";
53
59
  import {
54
60
  clearStreamingPartialJson,
@@ -104,6 +110,7 @@ import {
104
110
  resolveGitHubCopilotBaseUrl,
105
111
  } from "./github-copilot-headers";
106
112
  import { getOpenAIPromptCacheKey } from "./openai-shared";
113
+ import { applyInferenceHeaders } from "./inference-headers";
107
114
  import { transformMessages } from "./transform-messages";
108
115
  import { NON_VISION_IMAGE_PLACEHOLDER } from "./vision-guard";
109
116
 
@@ -234,15 +241,6 @@ function buildClaudeCodeBetas({
234
241
  return betas;
235
242
  }
236
243
 
237
- function getHeaderCaseInsensitive(headers: Record<string, string> | undefined, headerName: string): string | undefined {
238
- if (!headers) return undefined;
239
- const normalizedName = headerName.toLowerCase();
240
- for (const [key, value] of Object.entries(headers)) {
241
- if (key.toLowerCase() === normalizedName) return value;
242
- }
243
- return undefined;
244
- }
245
-
246
244
  function isClaudeCodeClientUserAgent(userAgent: string | undefined): userAgent is string {
247
245
  if (!userAgent) return false;
248
246
  return userAgent.toLowerCase().startsWith("claude-cli");
@@ -1238,7 +1236,7 @@ export type AnthropicClientOptionsArgs = {
1238
1236
  disableStrictTools?: boolean;
1239
1237
  fetch?: FetchImpl;
1240
1238
  maxRetryDelayMs?: number;
1241
- claudeCodeSessionId?: string;
1239
+ sessionId?: string;
1242
1240
  };
1243
1241
 
1244
1242
  export type AnthropicClientOptionsResult = {
@@ -2177,7 +2175,10 @@ const streamAnthropicOnce = (
2177
2175
  thinkingDisplay: options?.thinkingDisplay,
2178
2176
  fetch: options?.fetch,
2179
2177
  maxRetryDelayMs: options?.maxRetryDelayMs,
2180
- claudeCodeSessionId: options?.sessionId ?? extractClaudeMetadataSessionId(options?.metadata?.user_id),
2178
+ sessionId:
2179
+ options?.sessionId ??
2180
+ extractClaudeMetadataSessionId(options?.metadata?.user_id) ??
2181
+ options?.promptCacheKey,
2181
2182
  disableStrictTools,
2182
2183
  });
2183
2184
  client = created.client;
@@ -3203,7 +3204,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
3203
3204
  thinkingEnabled = false,
3204
3205
  isOAuth,
3205
3206
  maxRetryDelayMs,
3206
- claudeCodeSessionId,
3207
+ sessionId,
3207
3208
  disableStrictTools: disableStrictToolsOverride,
3208
3209
  } = args;
3209
3210
  const compat = model.compat;
@@ -3266,6 +3267,11 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
3266
3267
  dynamicHeaders,
3267
3268
  headers,
3268
3269
  );
3270
+ applyInferenceHeaders(defaultHeaders, {
3271
+ provider: model.provider,
3272
+ protocol: "anthropic",
3273
+ sessionId,
3274
+ });
3269
3275
 
3270
3276
  return {
3271
3277
  isOAuthToken: false,
@@ -3288,22 +3294,23 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
3288
3294
  betaFeatures.push(interleavedThinkingBeta);
3289
3295
  }
3290
3296
 
3297
+ const requestModelHeaders = mergeHeaders(
3298
+ model.headers,
3299
+ foundryCustomHeaders,
3300
+ getUmansWebSearchHeader(model, mergeHeaders(model.headers, headers)),
3301
+ headers,
3302
+ dynamicHeaders,
3303
+ );
3291
3304
  const defaultHeaders = buildAnthropicHeaders({
3292
3305
  apiKey,
3293
3306
  baseUrl,
3294
3307
  isOAuth: oauthToken,
3295
3308
  extraBetas: betaFeatures,
3296
3309
  stream,
3297
- modelHeaders: mergeHeaders(
3298
- model.headers,
3299
- foundryCustomHeaders,
3300
- getUmansWebSearchHeader(model, mergeHeaders(model.headers, headers)),
3301
- headers,
3302
- dynamicHeaders,
3303
- ),
3310
+ modelHeaders: requestModelHeaders,
3304
3311
  isCloudflareAiGateway: model.provider === "cloudflare-ai-gateway",
3305
3312
  allowAnthropicHeaderOverrides: model.compat.allowAnthropicHeaderOverrides,
3306
- claudeCodeSessionId,
3313
+ claudeCodeSessionId: sessionId,
3307
3314
  claudeCodeBetas: oauthToken
3308
3315
  ? buildClaudeCodeBetas({
3309
3316
  agentRequest: hasTools || thinkingEnabled,
@@ -3313,6 +3320,11 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
3313
3320
  })
3314
3321
  : [],
3315
3322
  });
3323
+ applyInferenceHeaders(defaultHeaders, {
3324
+ provider: model.provider,
3325
+ protocol: "anthropic",
3326
+ sessionId,
3327
+ });
3316
3328
 
3317
3329
  if (model.provider === "cloudflare-ai-gateway") {
3318
3330
  return {
@@ -3491,6 +3503,64 @@ function applyPromptCaching(params: MessageCreateParamsStreaming, cacheControl?:
3491
3503
  }
3492
3504
  }
3493
3505
 
3506
+ /**
3507
+ * Anchor cache_control on the stable request head — the last (non-deferred)
3508
+ * tool definition and the last system block. The canonical cache order is
3509
+ * tools → system → messages, so a breakpoint on the final system block caches
3510
+ * the entire tools+system prefix, and the extra tool breakpoint keeps the tool
3511
+ * definitions cached even when the system text changes. This guarantees the
3512
+ * large, unchanging head is a cache hit on every turn regardless of how the
3513
+ * message tail churns — the breakpoint placement first-party Anthropic clients
3514
+ * (Claude Code, Pi) use. Without it, the general API-key path anchors only the
3515
+ * moving message tail, so tail churn re-writes the whole head uncached.
3516
+ *
3517
+ * Anthropic allows at most 4 cache breakpoints per request. At most one is
3518
+ * spent on tools and one on system here, leaving two for the message tail in
3519
+ * `applyPromptCaching`. Head caching is skipped entirely when the head is
3520
+ * already anchored — the OAuth Claude Code path caches its own instruction
3521
+ * block at buildAnthropicSystemBlocks, and via the canonical tools → system
3522
+ * order that single system breakpoint already caches every preceding tool. Re-
3523
+ * anchoring there would be redundant, would change the OAuth wire, and could
3524
+ * push a tool-heavy request over the 4-breakpoint budget, so the general
3525
+ * API-key path (nothing cached upstream) is the only one decorated here.
3526
+ *
3527
+ * Runs after the byte-stability plane (planStableAnthropicSystem /
3528
+ * planStableAnthropicTools), which hands back fresh block/tool copies each turn
3529
+ * and keys tool identity off a fingerprint that excludes cache_control — so
3530
+ * decorating here is byte-stable across turns and never forces a re-baseline.
3531
+ */
3532
+ function applyHeadCaching(
3533
+ systemBlocks: AnthropicSystemBlock[] | undefined,
3534
+ tools: AnthropicWireTool[] | undefined,
3535
+ cacheControl?: AnthropicCacheControl,
3536
+ ): void {
3537
+ if (!cacheControl) return;
3538
+
3539
+ // If anything in the head already carries a breakpoint, the head is already
3540
+ // cached (OAuth anchors its identity system block, which — canonical order
3541
+ // tools → system — caches all tools too). Leave it untouched.
3542
+ const headAlreadyCached =
3543
+ (systemBlocks?.some(block => block.cache_control != null) ?? false) ||
3544
+ (tools?.some(tool => tool.cache_control != null) ?? false);
3545
+ if (headAlreadyCached) return;
3546
+
3547
+ if (tools && tools.length > 0) {
3548
+ // Deferred tools are not part of the checked prefix until referenced, so
3549
+ // anchor the last tool that actually sits in the stable prefix.
3550
+ for (let index = tools.length - 1; index >= 0; index--) {
3551
+ const tool = tools[index];
3552
+ if (!tool || tool.defer_loading) continue;
3553
+ tool.cache_control = cloneAnthropicCacheControl(cacheControl);
3554
+ break;
3555
+ }
3556
+ }
3557
+
3558
+ if (systemBlocks && systemBlocks.length > 0) {
3559
+ const lastBlock = systemBlocks[systemBlocks.length - 1];
3560
+ if (lastBlock) lastBlock.cache_control = cloneAnthropicCacheControl(cacheControl);
3561
+ }
3562
+ }
3563
+
3494
3564
  function usesAdaptiveThinkingTagOnly(model: Model<"anthropic-messages">): boolean {
3495
3565
  const thinking = model.thinking;
3496
3566
  if (thinking?.mode !== "anthropic-adaptive") return false;
@@ -3966,6 +4036,9 @@ function buildParams(
3966
4036
  if (controlState) syncAnthropicControlState(controlState, wireMessages);
3967
4037
  systemBlocks = planStableAnthropicSystem(systemBlocks, controlState, model.compat.supportsMidConversationSystem);
3968
4038
  tools = planStableAnthropicTools(tools, wireMessages, controlState, model.compat.supportsMidConversationToolChanges);
4039
+ // Anchor the stable tools+system head so it stays cached across turns; the
4040
+ // moving message tail is anchored separately in applyPromptCaching below.
4041
+ applyHeadCaching(systemBlocks, tools, cacheControl);
3969
4042
  const topLevelEffort = planStableAnthropicEffort(
3970
4043
  outputConfigEffort,
3971
4044
  wireMessages,