@oh-my-pi/pi-ai 18.1.5 → 18.1.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +32 -9
- package/dist/types/providers/amazon-bedrock.d.ts +7 -0
- package/dist/types/providers/anthropic.d.ts +1 -1
- package/dist/types/providers/inference-headers.d.ts +24 -0
- package/dist/types/providers/openai-codex/request-transformer.d.ts +7 -10
- package/dist/types/providers/openai-codex-responses.d.ts +3 -7
- package/dist/types/providers/openai-shared.d.ts +1 -1
- package/dist/types/types.d.ts +7 -0
- package/dist/types/utils.d.ts +7 -0
- package/package.json +5 -5
- package/src/error/rate-limit.ts +9 -1
- package/src/providers/amazon-bedrock.ts +72 -5
- package/src/providers/anthropic.ts +94 -21
- package/src/providers/google.ts +6 -0
- package/src/providers/inference-headers.ts +79 -0
- package/src/providers/openai-codex/request-transformer.ts +9 -15
- package/src/providers/openai-codex-responses.ts +5 -9
- package/src/providers/openai-completions.ts +3 -0
- package/src/providers/openai-responses.ts +1 -1
- package/src/providers/openai-shared.ts +8 -13
- package/src/providers/pi-native-server.ts +1 -0
- package/src/stream.ts +51 -8
- package/src/types.ts +7 -0
- package/src/usage/google-antigravity.ts +236 -42
- package/src/utils.ts +18 -0
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,29 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [18.1.6] - 2026-09-03
|
|
6
|
+
|
|
7
|
+
### Breaking Changes
|
|
8
|
+
|
|
9
|
+
- Renamed `claudeCodeSessionId` to `sessionId` in `AnthropicClientOptionsArgs`.
|
|
10
|
+
- Renamed `openAISessionId` to `sessionId` in `OpenAIRequestSetupOptions`.
|
|
11
|
+
|
|
12
|
+
### Added
|
|
13
|
+
|
|
14
|
+
- Added Amazon Bedrock `requestMetadata` support for cost and usage attribution in AWS invocation logs.
|
|
15
|
+
|
|
16
|
+
### Changed
|
|
17
|
+
|
|
18
|
+
- Codex GPT-5.6 requests now use full Responses by default, enabling independent tool calls to run in parallel; provider-native compaction continues to use catalog-selected Responses Lite.
|
|
19
|
+
- Inference requests now identify as omp by default while preserving explicit provider and OAuth User-Agent fingerprints. Amazon Bedrock requests use an `omp/<version>` User-Agent by default and honor configured `User-Agent` overrides.
|
|
20
|
+
|
|
21
|
+
### Fixed
|
|
22
|
+
|
|
23
|
+
- Fixed Antigravity usage reporting to match the official client's five-hour and weekly quota buckets.
|
|
24
|
+
- Anthropic and OpenRouter credit-exhaustion errors now automatically switch to a sibling account instead of stopping the turn with a retry hint.
|
|
25
|
+
- Fixed OpenCode Go and Zen requests by including the required stable per-conversation session identification.
|
|
26
|
+
- Improved Anthropic prompt caching so explicit cache breakpoints preserve reusable tools and system prompts when the message tail changes.
|
|
27
|
+
|
|
5
28
|
## [18.1.5] - 2026-09-03
|
|
6
29
|
|
|
7
30
|
### Added
|
|
@@ -183,7 +206,7 @@
|
|
|
183
206
|
- Fixed Codex continuations, retries, and compaction replacing or dropping the turn-scoped sticky-routing token ([#9277](https://github.com/can1357/oh-my-pi/issues/9277)).
|
|
184
207
|
- Fixed Codex Responses append chains falling back to full-context replay when replay-sanitized assistant items differ only by output-only IDs or lifecycle status.
|
|
185
208
|
- Fixed Cursor usage reporting “no usage data” for plans without a numeric legacy request cap.
|
|
186
|
-
- Fixed DeepSeek models rejecting requests with HTTP 400 `unknown variant \`image_url\`, expected \`text\``
|
|
209
|
+
- Fixed DeepSeek models rejecting requests with HTTP 400 `unknown variant \`image_url\`, expected \`text\``when screenshots or image-producing tool results are present in conversation history or when`model.input`claims vision capability;`convertMessages`in`openai-completions`now strips`image_url` content parts and injects non-vision image placeholders for all DeepSeek endpoints.
|
|
187
210
|
- Fixed `PI_PROXY` covering only provider streams: OAuth token refresh and login, usage probes, and model discovery went out through the bare global `fetch` and ignored it, so a region-blocked token endpoint answered `403 Request not allowed` (Anthropic `/v1/oauth/token`) and disabled the credential while the proxied stream itself worked. `installGlobalProxyFetch()` now routes the process-wide `fetch` through `PI_PROXY`; a per-request proxy such as `PI_PROXY_<PROVIDER>` still wins, and loopback / private-range / `NO_PROXY` targets stay direct.
|
|
188
211
|
- Fixed Anthropic inference ignoring every proxy setting. `coworkFetch` runs on `node:https`, whose Bun shim discards both `agent.createConnection` and `options.createConnection`: the CONNECT tunnel to `PI_PROXY` was built, TLS-negotiated, then abandoned, and the request dialed `api.anthropic.com` on the default route (measured at the proxy: 581 bytes of handshake, zero request bytes). On a region-blocked egress that returned `403 {"type":"forbidden","message":"Request not allowed"}` with the proxy apparently configured. Proxied requests now go through Bun's own `fetch`, which honors `init.proxy`, trading the Cowork TLS/header profile for a proxy that actually carries the traffic; the dead tunnel plumbing is gone from the transport. `node:http2` (Cursor) does honor `createConnection` and is unaffected.
|
|
189
212
|
- Fixed `cowork-fetch` capturing `globalThis.fetch` at module load, so a proxy wrapper installed later in startup was ignored on its fallback path.
|
|
@@ -253,7 +276,7 @@
|
|
|
253
276
|
|
|
254
277
|
### Changed
|
|
255
278
|
|
|
256
|
-
- Fixed Gemini thought summaries occasionally leaking a raw
|
|
279
|
+
- Fixed Gemini thought summaries occasionally leaking a raw ` ```thinking ` / ` ``````thinking ` fence delimiter into the reasoning block, so it no longer shows up as fence spam in the thinking display or persisted transcripts ([#8719](https://github.com/can1357/oh-my-pi/issues/8719)).
|
|
257
280
|
- Fixed the OpenCode Go login prompting for an "OpenCode Zen API key": the shared login flow now names the provider you selected, so connecting OpenCode Go asks for an OpenCode Go key (the `opencode.ai/auth` console is still shared, as documented upstream) ([#8738](https://github.com/can1357/oh-my-pi/issues/8738)).
|
|
258
281
|
- Fixed Anthropic-compatible endpoints with strict prompt validation (e.g. Z.AI GLM `api.z.ai/api/anthropic`, which rejects the whole request with `400 code 1213 "The prompt parameter was not received normally"`) failing sessions once a tool returned empty output on a vision-capable model: empty successful `tool_result` blocks now encode as `content: ""` instead of `content: []`, which both the official API and strict compatible endpoints accept.
|
|
259
282
|
- Fixed `retry.usageReservePct` (Reserve Margin) ignoring Claude Fable/Mythos weekly tier usage until it hit 100%, so a Fable model kept serving turns past the configured reserve; reserve health now honors the mapped tier row while credential-wide hard blocks still require confirmed exhaustion ([#8773](https://github.com/can1357/oh-my-pi/issues/8773)).
|
|
@@ -1305,7 +1328,7 @@
|
|
|
1305
1328
|
|
|
1306
1329
|
- Added a third streaming thinking-loop detection heuristic to catch "progress-lexicon stalls" where models endlessly reshuffle motivational filler without introducing new vocabulary or concrete technical references
|
|
1307
1330
|
- Added branded wordmark and logo animation to authentication flow
|
|
1308
|
-
- Added a third streaming thinking-loop detection shape — a
|
|
1331
|
+
- Added a third streaming thinking-loop detection shape — a _progress-lexicon stall_ — alongside verbatim tail repetition and near-duplicate (trigram) segments. It catches reasoning-summarizer loops that reshuffle the same motivational filler ("just doing it, pushing ahead, maintaining momentum") into fresh word order every paragraph: word-trigrams never cluster, but a run of substantial segments that recycle the recent vocabulary and introduce no _new_ concrete reference (path / identifier / code-span) trips the guard. Summarizer title/heading lines (`**Bold Title**`, `## Heading`) are stripped before analysis so their ever-changing wording cannot mask the stall by inflating novelty. Calibrated against 537k real non-Gemini reasoning blocks (zero false positives at novelty floor 0.2 / run length 8; the real loop sustains runs of 10+).
|
|
1309
1332
|
- Added CoreWeave Serverless Inference provider login support via `COREWEAVE_API_KEY` and `WANDB_API_KEY` fallback.
|
|
1310
1333
|
|
|
1311
1334
|
### Changed
|
|
@@ -1449,7 +1472,7 @@
|
|
|
1449
1472
|
- Fixed tool call ID normalization for Anthropic-compatible models
|
|
1450
1473
|
- Fixed Anthropic Messages replay sanitizing malformed tool-call IDs, including aborted native tool calls with empty IDs, so retries no longer send invalid `tool_use.id` / `tool_result.tool_use_id` pairs.
|
|
1451
1474
|
- Fixed the Codex Responses WebSocket transport attributing a prior turn's output to the current one on a reused connection: a trailing/duplicate frame from a cleanly-completed previous response that slipped past the queue drain could be consumed as this request's terminal (ending the turn with empty output) or as a stale tool call. Frames are now keyed by `response.id` — a frame carrying the previous response's id is dropped, and one carrying a third id (or a regressed `sequence_number`) fails closed so the turn retries instead of mixing two responses' streams. Idless frames (deltas, the rate-limit/metadata preamble, `response.created`-less streams) still pass through, matching upstream codex-rs.
|
|
1452
|
-
- Fixed `transformMessages` pulling an earlier, orphaned tool result onto a later tool call that reused the same id (left behind when compaction folded the originating `tool_use` into a summary). The pending-call flush now pairs each call with a result positioned
|
|
1475
|
+
- Fixed `transformMessages` pulling an earlier, orphaned tool result onto a later tool call that reused the same id (left behind when compaction folded the originating `tool_use` into a summary). The pending-call flush now pairs each call with a result positioned _after_ its assistant turn, so a reused id surfaces its own output rather than a prior turn's.
|
|
1453
1476
|
- Fixed DashScope 429 rate-limit messages that mention authorization being classified as credential failures, preventing valid API keys from being invalidated after throttling. ([#3172](https://github.com/can1357/oh-my-pi/issues/3172))
|
|
1454
1477
|
- Fixed OpenCode Go `401 Insufficient balance` quota errors being treated as unknown failures instead of usage-limit errors, restoring credential rotation and fallback chains. ([#3169](https://github.com/can1357/oh-my-pi/issues/3169))
|
|
1455
1478
|
|
|
@@ -1725,7 +1748,7 @@
|
|
|
1725
1748
|
- Added `wrapInbandToolStream` function to process streaming responses with in-band tool call parsing
|
|
1726
1749
|
- Added `ThinkingInbandScanner` for parsing thinking/reasoning blocks across dialects
|
|
1727
1750
|
- Added `OwnedStream` class for managing dialect-aware streaming with tool call events
|
|
1728
|
-
- Added in-band thinking channels to every dialect that was missing one: `gemini` (a
|
|
1751
|
+
- Added in-band thinking channels to every dialect that was missing one: `gemini` (a ` ```thinking ` fence mirroring ` ```tool_code `), `gemma` (its native `<|channel>thought…<channel|>` reasoning channel), `kimi` (`<think>…</think>`), and `pi` (`<thinking>…</thinking>`). Each scanner now parses reasoning into thinking events instead of leaking chain-of-thought into the visible reply, and every dialect's `renderThinking` is a real channel that round-trips back through its scanner (no passthrough renderers).
|
|
1729
1752
|
|
|
1730
1753
|
### Changed
|
|
1731
1754
|
|
|
@@ -1749,7 +1772,7 @@
|
|
|
1749
1772
|
|
|
1750
1773
|
### Added
|
|
1751
1774
|
|
|
1752
|
-
- Added the `gemini` in-band tool-call syntax with Python-style
|
|
1775
|
+
- Added the `gemini` in-band tool-call syntax with Python-style `tool_code` blocks and `default_api` invocations
|
|
1753
1776
|
- Added the `gemma` token-delimited in-band tool-call syntax using `<|tool_call>` and `<|tool_response>` blocks
|
|
1754
1777
|
- Added `gemini` and `gemma` to owned stream tool-result token detection so their tool responses are recognized
|
|
1755
1778
|
- Fixed truncated Gemini and Gemma tool blocks from being emitted as plain text during streaming
|
|
@@ -1794,7 +1817,7 @@
|
|
|
1794
1817
|
- Fixed Harmony leak handling support by adding `recoverHarmonyToolCall` plus leak-detection workflows for contaminated assistant messages so recoverable tool-call arguments can be safely truncated and retried
|
|
1795
1818
|
- Fixed false-positive gating in Harmony leak heuristics using signal-based checks so unrelated text containing `to=functions...` is not treated as leaked tool-call markup
|
|
1796
1819
|
- Routed Kimi, DeepSeek DSML, and plain thinking markup healing through the shared in-band scanners so provider leak repair and owned tool calling parse the same wire formats.
|
|
1797
|
-
- Fixed Cursor provider (`cursor-agent` API) streaming dropping large MCP tool-call arguments — most visibly the built-in `task` tool's `tasks` array on multi-subagent dispatches, which failed downstream schema validation with `tasks: Invalid input: expected array, received undefined`. Two upstream behaviors were fighting the stream handler in `packages/ai/src/providers/cursor.ts`: (1) `args_text_delta` carries the
|
|
1820
|
+
- Fixed Cursor provider (`cursor-agent` API) streaming dropping large MCP tool-call arguments — most visibly the built-in `task` tool's `tasks` array on multi-subagent dispatches, which failed downstream schema validation with `tasks: Invalid input: expected array, received undefined`. Two upstream behaviors were fighting the stream handler in `packages/ai/src/providers/cursor.ts`: (1) `args_text_delta` carries the _cumulative_ args text so far per `agent.proto`, but the handler concatenated each snapshot onto the buffer, garbling the JSON; (2) `tool_call_completed` carries an `McpArgs` map that omits oversized parameters entirely and downgrades unparsable values to their raw string fallback, but the handler unconditionally overwrote the streamed args with that map. The handler now strips the already-buffered prefix from each `args_text_delta` snapshot (falling back to append when the snapshot doesn't extend the buffer) and merges the decoded `McpArgs` map into the streamed args — preserving streamed keys the completion frame omits and the structured value when the completion frame downgrades to a string. ([#2615](https://github.com/can1357/oh-my-pi/issues/2615))
|
|
1798
1821
|
- Fixed Codex Responses stream mis-routing interleaved `function_call_arguments.delta` events when more than one tool call was open concurrently. The runtime tracked a singleton `currentItem`/`currentBlock`, so every delta — regardless of `item_id` — was appended to whichever item was most recently added, and `output_item.done` for the earlier call then overwrote a sibling's stored arguments (visible as `tasks: Invalid input: expected array, received undefined` on the `task` tool). Open items are now keyed by `item_id` with `output_index` fallback; deltas/done events route to the matching block, late deltas whose item already closed are dropped instead of corrupting a sibling, and `toolcall_*` stream events emit the right `contentIndex` per call ([#2619](https://github.com/can1357/oh-my-pi/issues/2619)).
|
|
1799
1822
|
|
|
1800
1823
|
## [15.13.1] - 2026-06-15
|
|
@@ -2003,7 +2026,7 @@
|
|
|
2003
2026
|
|
|
2004
2027
|
### Breaking Changes
|
|
2005
2028
|
|
|
2006
|
-
- The model catalog moved to the new `@oh-my-pi/pi-catalog` package. Deep subpath exports `@oh-my-pi/pi-ai/models.json`, `/models`, `/model-cache`, `/model-manager`, `/model-thinking`, `/effort`, `/provider-models*`, `/utils/discovery*`, `/providers/openai-codex/constants`, `/providers/google-gemini-headers`, and `/providers/openai-completions-compat` are gone — import the `@oh-my-pi/pi-catalog` equivalents (`/models.json`, `/models`, `/model-cache`, `/model-manager`, `/model-thinking`, `/effort`, `/provider-models*`, `/discovery*`, `/wire/codex`, `/wire/gemini-headers`, `/compat/openai`). The pi-ai root barrel re-exports only the model/effort
|
|
2029
|
+
- The model catalog moved to the new `@oh-my-pi/pi-catalog` package. Deep subpath exports `@oh-my-pi/pi-ai/models.json`, `/models`, `/model-cache`, `/model-manager`, `/model-thinking`, `/effort`, `/provider-models*`, `/utils/discovery*`, `/providers/openai-codex/constants`, `/providers/google-gemini-headers`, and `/providers/openai-completions-compat` are gone — import the `@oh-my-pi/pi-catalog` equivalents (`/models.json`, `/models`, `/model-cache`, `/model-manager`, `/model-thinking`, `/effort`, `/provider-models*`, `/discovery*`, `/wire/codex`, `/wire/gemini-headers`, `/compat/openai`). The pi-ai root barrel re-exports only the model/effort _types_ its own signatures use (`Model`, `Api`, `ThinkingConfig`, `Effort`, `Usage`, compat interfaces) — catalog _values_ (`getBundledModel(s)`, `calculateCost`, `modelsAreEqual`, `clampThinkingLevelForModel`, `DEFAULT_MODEL_PER_PROVIDER`, …) must be imported from `@oh-my-pi/pi-catalog`.
|
|
2007
2030
|
- `ProviderDefinition` is now auth-only: `defaultModel`, `createModelManagerOptions`, `catalogDiscovery`, `dynamicModelsAuthoritative`, `allowUnauthenticated`, and `specialModelManager` moved to pi-catalog's `CATALOG_PROVIDERS` table, and `KnownProviderId` was replaced by pi-catalog's `KnownProvider` (registry completeness is enforced by a compile-time check against that union). The pure GitHub Copilot key/endpoint helpers moved from `registry/oauth/github-copilot` to `@oh-my-pi/pi-catalog/wire/github-copilot`.
|
|
2008
2031
|
|
|
2009
2032
|
### Added
|
|
@@ -2092,7 +2115,7 @@
|
|
|
2092
2115
|
- Fixed `AnthropicMessagesClient` spreading `fetchOptions` after the core request fields, letting a caller-supplied `signal`/`method`/`body` silently disconnect the timeout controller or corrupt the request. Transport extras (TLS) still pass through; core fields now always win.
|
|
2093
2116
|
- Fixed Foundry mTLS/CA material being cached for the process lifetime when the env vars point at files: the cache key now folds in the file mtime so on-disk certificate rotation takes effect.
|
|
2094
2117
|
- Fixed the Claude Code fingerprint version drifting across surfaces: the usage endpoint (`claude-cli/2.1.160`) and OAuth bootstrap (`claude-code/2.1.160`) pinned a stale version while `/v1/messages` reported 2.1.165; both now derive from `claudeCodeVersion`.
|
|
2095
|
-
- Fixed a system prompt that merely
|
|
2118
|
+
- Fixed a system prompt that merely _mentions_ `x-anthropic-billing-header:` mid-text suppressing the entire Claude Code system-block injection (billing header, instruction, and cch attestation); the resumed-session guard now anchors with `startsWith`.
|
|
2096
2119
|
- Fixed lone surrogates in cross-API tool-call arguments reaching Anthropic's strict UTF-8 validation: replayed OpenAI/Google-origin `tool_use.input` string leaves are now deep-sanitized with `toWellFormed()`, while same-API Anthropic arguments stay byte-identical to keep prompt-cache prefixes stable.
|
|
2097
2120
|
- Bounded the many-image resize fan-out to 4 concurrent decodes (it previously decoded every oversized image at once, two encode pipelines each — multi-GB transient memory at the 20+-image threshold that activates the feature).
|
|
2098
2121
|
- Fixed `mergeHeaders` merging case-sensitively on the Copilot/client-options path, where a miscased user-configured header (e.g. `authorization` next to the synthesized `Authorization`) survived as two keys that the `Headers` constructor joins comma-separated on the wire.
|
|
@@ -47,5 +47,12 @@ export interface BedrockOptions extends StreamOptions {
|
|
|
47
47
|
* we omit it for them.
|
|
48
48
|
*/
|
|
49
49
|
thinkingDisplay?: BedrockThinkingDisplay;
|
|
50
|
+
/**
|
|
51
|
+
* Per-request Bedrock invocation-log tags. Merged over `model.requestMetadata`
|
|
52
|
+
* (per-call entries win on key collision). AWS caps the result at 16 entries;
|
|
53
|
+
* keys 1-256 chars, values 0-256 chars, both limited to
|
|
54
|
+
* `[a-zA-Z0-9\s:_@$#=/+,-.]`. Entries outside those limits are dropped.
|
|
55
|
+
*/
|
|
56
|
+
requestMetadata?: Record<string, string>;
|
|
50
57
|
}
|
|
51
58
|
export declare const streamBedrock: StreamFunction<"bedrock-converse-stream">;
|
|
@@ -164,7 +164,7 @@ export type AnthropicClientOptionsArgs = {
|
|
|
164
164
|
disableStrictTools?: boolean;
|
|
165
165
|
fetch?: FetchImpl;
|
|
166
166
|
maxRetryDelayMs?: number;
|
|
167
|
-
|
|
167
|
+
sessionId?: string;
|
|
168
168
|
};
|
|
169
169
|
export type AnthropicClientOptionsResult = {
|
|
170
170
|
isOAuthToken: boolean;
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
/** Shared inference request identity headers. */
|
|
2
|
+
import type { FetchImpl } from "../types.js";
|
|
3
|
+
/** Options controlling provider and protocol inference headers. */
|
|
4
|
+
export interface InferenceHeaderOptions {
|
|
5
|
+
provider: string;
|
|
6
|
+
protocol: "anthropic" | "google" | "openai";
|
|
7
|
+
sessionId?: string;
|
|
8
|
+
}
|
|
9
|
+
/** Set a header unless the map already contains that field under any casing. */
|
|
10
|
+
export declare function setHeaderIfAbsent(headers: Record<string, string>, name: string, value: string): void;
|
|
11
|
+
/**
|
|
12
|
+
* Project omp's identity and authoritative conversation id onto the headers
|
|
13
|
+
* understood by the active inference protocol and host.
|
|
14
|
+
*/
|
|
15
|
+
export declare function applyInferenceHeaders(headers: Record<string, string>, options: InferenceHeaderOptions): void;
|
|
16
|
+
/**
|
|
17
|
+
* Apply omp's process-wide inference User-Agent default. Any explicit header,
|
|
18
|
+
* including Anthropic and Codex OAuth fingerprints, remains authoritative.
|
|
19
|
+
*
|
|
20
|
+
* Plain-object headers stay plain objects: custom `fetch` implementations
|
|
21
|
+
* (proxies, tests) index `init.headers` by name and must not be handed a
|
|
22
|
+
* `Headers` instance instead.
|
|
23
|
+
*/
|
|
24
|
+
export declare function withInferenceUserAgent(fetchImpl: FetchImpl): FetchImpl;
|
|
@@ -21,10 +21,8 @@ export interface CodexRequestOptions {
|
|
|
21
21
|
textVerbosity?: "low" | "medium" | "high";
|
|
22
22
|
include?: string[];
|
|
23
23
|
/**
|
|
24
|
-
* Responses Lite transport
|
|
25
|
-
*
|
|
26
|
-
* strips image detail, and disables parallel tool calling (codex-rs
|
|
27
|
-
* `use_responses_lite`).
|
|
24
|
+
* Responses Lite transport opt-in. Normal inference defaults to full
|
|
25
|
+
* Responses so the model can emit independent tool calls in parallel.
|
|
28
26
|
*/
|
|
29
27
|
responsesLite?: boolean;
|
|
30
28
|
}
|
|
@@ -70,13 +68,12 @@ export interface RequestBody {
|
|
|
70
68
|
[key: string]: unknown;
|
|
71
69
|
}
|
|
72
70
|
/**
|
|
73
|
-
* Resolve whether a Codex request
|
|
74
|
-
*
|
|
75
|
-
*
|
|
76
|
-
*
|
|
77
|
-
* decides.
|
|
71
|
+
* Resolve whether a Codex request explicitly opts into Responses Lite.
|
|
72
|
+
*
|
|
73
|
+
* Provider-native compaction passes the model's `useResponsesLite` flag as an
|
|
74
|
+
* explicit option; normal inference defaults to the full Responses contract.
|
|
78
75
|
*/
|
|
79
|
-
export declare function resolveCodexResponsesLite(
|
|
76
|
+
export declare function resolveCodexResponsesLite(requested: boolean | undefined): boolean;
|
|
80
77
|
/**
|
|
81
78
|
* Structural view of a Responses-style body mutated by the Lite rewrite.
|
|
82
79
|
* Loose (`unknown`) property types let the turn transformer (`RequestBody`)
|
|
@@ -12,13 +12,9 @@ export interface OpenAICodexResponsesOptions extends StreamOptions {
|
|
|
12
12
|
preferWebsockets?: boolean;
|
|
13
13
|
serviceTier?: ServiceTier;
|
|
14
14
|
/**
|
|
15
|
-
* Responses Lite transport
|
|
16
|
-
*
|
|
17
|
-
* `
|
|
18
|
-
* WebSocket upgrade (the marker is connection-scoped there, so lite and
|
|
19
|
-
* non-lite turns never share a pooled socket), moves instructions/tools
|
|
20
|
-
* into input items, strips image detail, and disables parallel tool
|
|
21
|
-
* calling — mirroring codex-rs.
|
|
15
|
+
* Responses Lite transport opt-in. Normal inference defaults to full
|
|
16
|
+
* Responses; provider-native compaction explicitly follows the model's
|
|
17
|
+
* `useResponsesLite` flag.
|
|
22
18
|
*/
|
|
23
19
|
responsesLite?: boolean;
|
|
24
20
|
/**
|
package/dist/types/types.d.ts
CHANGED
|
@@ -463,6 +463,13 @@ export interface SimpleStreamOptions extends Omit<StreamOptions, "apiKey"> {
|
|
|
463
463
|
guardrailIdentifier?: string;
|
|
464
464
|
guardrailVersion?: string;
|
|
465
465
|
guardrailTrace?: "enabled" | "disabled" | "enabled_full";
|
|
466
|
+
/**
|
|
467
|
+
* Bedrock invocation-log tags forwarded through transports that do not dispatch
|
|
468
|
+
* directly to the Bedrock provider. Unlike the guardrail fields above, these
|
|
469
|
+
* MERGE per key with the model's own `requestMetadata` (these win) rather than
|
|
470
|
+
* replacing it wholesale — they are independent attribution tags, not one value.
|
|
471
|
+
*/
|
|
472
|
+
requestMetadata?: Record<string, string>;
|
|
466
473
|
/** Optional tool choice override for compatible providers */
|
|
467
474
|
toolChoice?: ToolChoice;
|
|
468
475
|
/** OpenAI service tier for processing priority/cost control. Ignored by non-OpenAI providers. */
|
package/dist/types/utils.d.ts
CHANGED
|
@@ -1,6 +1,13 @@
|
|
|
1
1
|
import type { ResponseInput } from "./providers/openai-responses-wire.js";
|
|
2
2
|
import type { CacheRetention, OpenAIResponsesHistoryPayload, ProviderPayload } from "./types.js";
|
|
3
3
|
export { isRecord } from "@oh-my-pi/pi-utils";
|
|
4
|
+
/**
|
|
5
|
+
* Read a header value ignoring key casing. HTTP header names are
|
|
6
|
+
* case-insensitive, but `Record<string, string>` header bags are not, so a
|
|
7
|
+
* config-authored `User-Agent` and a caller-authored `user-agent` are the same
|
|
8
|
+
* header to every provider that lowercases before merging.
|
|
9
|
+
*/
|
|
10
|
+
export declare function getHeaderCaseInsensitive(headers: Record<string, string> | undefined, headerName: string): string | undefined;
|
|
4
11
|
export declare function normalizeSystemPrompts(systemPrompt: readonly string[] | string | undefined | null): string[];
|
|
5
12
|
export declare function normalizeToolCallId(id: string): string;
|
|
6
13
|
type ResponsesToolItemIdPrefix = "fc" | "ctc";
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@oh-my-pi/pi-ai",
|
|
4
|
-
"version": "18.1.
|
|
4
|
+
"version": "18.1.6",
|
|
5
5
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
6
6
|
"homepage": "https://omp.sh",
|
|
7
7
|
"author": "Stencil Labs, Inc.",
|
|
@@ -37,10 +37,10 @@
|
|
|
37
37
|
"fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
|
|
38
38
|
},
|
|
39
39
|
"dependencies": {
|
|
40
|
-
"@oh-my-pi/omptype": "18.1.
|
|
41
|
-
"@oh-my-pi/pi-catalog": "18.1.
|
|
42
|
-
"@oh-my-pi/pi-utils": "18.1.
|
|
43
|
-
"@oh-my-pi/pi-wire": "18.1.
|
|
40
|
+
"@oh-my-pi/omptype": "18.1.6",
|
|
41
|
+
"@oh-my-pi/pi-catalog": "18.1.6",
|
|
42
|
+
"@oh-my-pi/pi-utils": "18.1.6",
|
|
43
|
+
"@oh-my-pi/pi-wire": "18.1.6"
|
|
44
44
|
},
|
|
45
45
|
"devDependencies": {
|
|
46
46
|
"@types/bun": "^1.3.14"
|
package/src/error/rate-limit.ts
CHANGED
|
@@ -24,6 +24,12 @@ const SERVER_ERROR_BACKOFF_MS = 20 * 1000; // 20s
|
|
|
24
24
|
const ACCOUNT_RATE_LIMIT_PATTERN =
|
|
25
25
|
/\baccount(?:'s)?\b[^\n]{0,80}\brate.?limit\b|\brate.?limit\b[^\n]{0,80}\baccount\b/i;
|
|
26
26
|
const INSUFFICIENT_BALANCE_PATTERN = /insufficient.?balance/i;
|
|
27
|
+
// Prepaid-credit exhaustion phrased around the credit balance rather than a
|
|
28
|
+
// quota: Anthropic "This request would exceed your available credits given
|
|
29
|
+
// your current in-flight requests" (402), OpenRouter "Insufficient credits",
|
|
30
|
+
// "credits exhausted". Account-local, so rotate to a sibling credential.
|
|
31
|
+
const CREDITS_EXHAUSTED_PATTERN =
|
|
32
|
+
/\b(?:exceed\w*|insufficient|not enough)\b[^\n]{0,40}\bcredits?\b|\bcredits?\b[^\n]{0,40}\b(?:exhausted|depleted)\b/i;
|
|
27
33
|
const SPEND_LIMIT_PATTERN = /spend.?limit/i;
|
|
28
34
|
const SUBSCRIPTION_CAP_PATTERN =
|
|
29
35
|
/\b(?:subscription|plan|membership)\b[^\n]{0,80}\b(?:rate.?limits?|quota|cap)\b|\b(?:rate.?limits?|quota|cap)\b[^\n]{0,80}\b(?:subscription|plan|membership)\b/i;
|
|
@@ -247,7 +253,8 @@ export function parseRateLimitReason(errorMessage: string): RateLimitReason {
|
|
|
247
253
|
lower.includes("out of credits") ||
|
|
248
254
|
lower.includes("spending-limit") ||
|
|
249
255
|
lower.includes("spending limit") ||
|
|
250
|
-
INSUFFICIENT_BALANCE_PATTERN.test(errorMessage)
|
|
256
|
+
INSUFFICIENT_BALANCE_PATTERN.test(errorMessage) ||
|
|
257
|
+
CREDITS_EXHAUSTED_PATTERN.test(errorMessage)
|
|
251
258
|
) {
|
|
252
259
|
return "QUOTA_EXHAUSTED";
|
|
253
260
|
}
|
|
@@ -390,6 +397,7 @@ export function matchesUsageLimitText(errorMessage: string): boolean {
|
|
|
390
397
|
if (isDashScopeTokenLimitText(errorMessage)) return false;
|
|
391
398
|
return (
|
|
392
399
|
USAGE_LIMIT_PATTERN.test(errorMessage) ||
|
|
400
|
+
CREDITS_EXHAUSTED_PATTERN.test(errorMessage) ||
|
|
393
401
|
(CN_QUOTA_EXHAUSTED_PATTERN.test(errorMessage) && !CN_TRANSIENT_CAP_PATTERN.test(errorMessage)) ||
|
|
394
402
|
SPEND_LIMIT_PATTERN.test(errorMessage) ||
|
|
395
403
|
ACCOUNT_RATE_LIMIT_PATTERN.test(errorMessage) ||
|
|
@@ -10,7 +10,14 @@
|
|
|
10
10
|
import type { Effort } from "@oh-my-pi/pi-catalog/effort";
|
|
11
11
|
import { mapEffortToAnthropicAdaptiveEffort, requireSupportedEffort } from "@oh-my-pi/pi-catalog/model-thinking";
|
|
12
12
|
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
|
|
13
|
-
import {
|
|
13
|
+
import {
|
|
14
|
+
$flag,
|
|
15
|
+
fetchWithRetry,
|
|
16
|
+
logger,
|
|
17
|
+
parseStreamingJson,
|
|
18
|
+
parseStreamingJsonThrottled,
|
|
19
|
+
USER_AGENT,
|
|
20
|
+
} from "@oh-my-pi/pi-utils";
|
|
14
21
|
import { renderDemotedThinking } from "../dialect/demotion";
|
|
15
22
|
import * as AIError from "../error";
|
|
16
23
|
import { resolveAwsBearerToken } from "../registry/aws";
|
|
@@ -102,6 +109,13 @@ export interface BedrockOptions extends StreamOptions {
|
|
|
102
109
|
* we omit it for them.
|
|
103
110
|
*/
|
|
104
111
|
thinkingDisplay?: BedrockThinkingDisplay;
|
|
112
|
+
/**
|
|
113
|
+
* Per-request Bedrock invocation-log tags. Merged over `model.requestMetadata`
|
|
114
|
+
* (per-call entries win on key collision). AWS caps the result at 16 entries;
|
|
115
|
+
* keys 1-256 chars, values 0-256 chars, both limited to
|
|
116
|
+
* `[a-zA-Z0-9\s:_@$#=/+,-.]`. Entries outside those limits are dropped.
|
|
117
|
+
*/
|
|
118
|
+
requestMetadata?: Record<string, string>;
|
|
105
119
|
}
|
|
106
120
|
|
|
107
121
|
function resolveBearerToken(options: BedrockOptions): string | undefined {
|
|
@@ -283,6 +297,7 @@ interface ConverseStreamRequest {
|
|
|
283
297
|
toolConfig?: WireToolConfig;
|
|
284
298
|
guardrailConfig?: WireGuardrailConfig;
|
|
285
299
|
additionalModelRequestFields?: Record<string, unknown>;
|
|
300
|
+
requestMetadata?: Record<string, string>;
|
|
286
301
|
additionalModelResponseFieldPaths?: string[];
|
|
287
302
|
}
|
|
288
303
|
|
|
@@ -319,6 +334,41 @@ interface MetadataEvent {
|
|
|
319
334
|
};
|
|
320
335
|
}
|
|
321
336
|
|
|
337
|
+
const REQUEST_METADATA_PATTERN = /^[a-zA-Z0-9\s:_@$#=/+,\-.]*$/;
|
|
338
|
+
const REQUEST_METADATA_MAX_ENTRIES = 16;
|
|
339
|
+
const REQUEST_METADATA_MAX_LENGTH = 256;
|
|
340
|
+
|
|
341
|
+
/**
|
|
342
|
+
* Bedrock rejects the whole invocation on a malformed `requestMetadata` entry.
|
|
343
|
+
* Attribution tags must never cost a turn, so invalid and excess entries are
|
|
344
|
+
* dropped with a warning instead of failing the request. Returns `undefined`
|
|
345
|
+
* for an empty result so the field is omitted from the body entirely.
|
|
346
|
+
*/
|
|
347
|
+
function sanitizeRequestMetadata(raw: unknown): Record<string, string> | undefined {
|
|
348
|
+
if (!isRecord(raw)) return undefined;
|
|
349
|
+
const out: Record<string, string> = {};
|
|
350
|
+
const dropped: string[] = [];
|
|
351
|
+
let kept = 0;
|
|
352
|
+
for (const [key, value] of Object.entries(raw)) {
|
|
353
|
+
if (
|
|
354
|
+
typeof value !== "string" ||
|
|
355
|
+
key.length < 1 ||
|
|
356
|
+
key.length > REQUEST_METADATA_MAX_LENGTH ||
|
|
357
|
+
!REQUEST_METADATA_PATTERN.test(key) ||
|
|
358
|
+
value.length > REQUEST_METADATA_MAX_LENGTH ||
|
|
359
|
+
!REQUEST_METADATA_PATTERN.test(value) ||
|
|
360
|
+
kept >= REQUEST_METADATA_MAX_ENTRIES
|
|
361
|
+
) {
|
|
362
|
+
dropped.push(key);
|
|
363
|
+
continue;
|
|
364
|
+
}
|
|
365
|
+
out[key] = value;
|
|
366
|
+
kept++;
|
|
367
|
+
}
|
|
368
|
+
if (dropped.length > 0) logger.warn("Bedrock requestMetadata entries dropped", { keys: dropped });
|
|
369
|
+
return kept > 0 ? out : undefined;
|
|
370
|
+
}
|
|
371
|
+
|
|
322
372
|
export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
|
|
323
373
|
model: Model<"bedrock-converse-stream">,
|
|
324
374
|
context: Context,
|
|
@@ -391,10 +441,17 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
|
|
|
391
441
|
toolConfig,
|
|
392
442
|
guardrailConfig: buildGuardrailConfig(options),
|
|
393
443
|
additionalModelRequestFields,
|
|
444
|
+
requestMetadata:
|
|
445
|
+
model.requestMetadata || options.requestMetadata
|
|
446
|
+
? { ...model.requestMetadata, ...options.requestMetadata }
|
|
447
|
+
: undefined,
|
|
394
448
|
...(prefixMismatchBehavior ? { additionalModelResponseFieldPaths: ["/input_transformations"] } : {}),
|
|
395
449
|
};
|
|
396
450
|
const replacementInput = await options?.onPayload?.(commandInput, model);
|
|
397
451
|
if (replacementInput !== undefined) commandInput = replacementInput as ConverseStreamRequest;
|
|
452
|
+
// After the hook so extension-injected tags are validated too, and before the
|
|
453
|
+
// raw dump so the inspector shows exactly what was sent.
|
|
454
|
+
commandInput = { ...commandInput, requestMetadata: sanitizeRequestMetadata(commandInput.requestMetadata) };
|
|
398
455
|
|
|
399
456
|
const host = `bedrock-runtime.${region}.amazonaws.com`;
|
|
400
457
|
const url = `https://${host}/model/${encodeURIComponent(model.id)}/converse-stream`;
|
|
@@ -429,11 +486,21 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
|
|
|
429
486
|
// comma-joined wire header, so AWS validates different bytes than were
|
|
430
487
|
// signed and rejects the request.
|
|
431
488
|
const callerHeaders: Record<string, string> = {};
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
489
|
+
// `model.headers` first, `options.headers` second: `StreamOptions.headers` is
|
|
490
|
+
// documented (types.ts:431-435) as merged ON TOP of model-defined headers.
|
|
491
|
+
// Both pass the same filter, so a config-authored `Host`/`Content-Type`
|
|
492
|
+
// cannot desync the signature either.
|
|
493
|
+
for (const source of [model.headers, options?.headers]) {
|
|
494
|
+
for (const [name, value] of Object.entries(source ?? {})) {
|
|
495
|
+
const field = name.toLowerCase();
|
|
496
|
+
if (SIGNER_OWNED_HEADERS.has(field) || BEDROCK_RESERVED_HEADERS.has(field)) continue;
|
|
497
|
+
callerHeaders[field] = value;
|
|
498
|
+
}
|
|
436
499
|
}
|
|
500
|
+
// SigV4 never signs `user-agent` (UNSIGNABLE in aws-sigv4.ts:42-58), so this
|
|
501
|
+
// default cannot break the signature. Without it Bun's fetch sends
|
|
502
|
+
// `Bun/<version>`, and that is what CloudTrail records for every request.
|
|
503
|
+
callerHeaders["user-agent"] ??= USER_AGENT;
|
|
437
504
|
const baseHeaders: Record<string, string> = {
|
|
438
505
|
...callerHeaders,
|
|
439
506
|
"content-type": "application/json",
|
|
@@ -48,7 +48,13 @@ import type {
|
|
|
48
48
|
ToolResultMessage,
|
|
49
49
|
Usage,
|
|
50
50
|
} from "../types";
|
|
51
|
-
import {
|
|
51
|
+
import {
|
|
52
|
+
getHeaderCaseInsensitive,
|
|
53
|
+
isRecord,
|
|
54
|
+
normalizeSystemPrompts,
|
|
55
|
+
normalizeToolCallId,
|
|
56
|
+
resolveCacheRetention,
|
|
57
|
+
} from "../utils";
|
|
52
58
|
import { createAbortSourceTracker } from "../utils/abort";
|
|
53
59
|
import {
|
|
54
60
|
clearStreamingPartialJson,
|
|
@@ -104,6 +110,7 @@ import {
|
|
|
104
110
|
resolveGitHubCopilotBaseUrl,
|
|
105
111
|
} from "./github-copilot-headers";
|
|
106
112
|
import { getOpenAIPromptCacheKey } from "./openai-shared";
|
|
113
|
+
import { applyInferenceHeaders } from "./inference-headers";
|
|
107
114
|
import { transformMessages } from "./transform-messages";
|
|
108
115
|
import { NON_VISION_IMAGE_PLACEHOLDER } from "./vision-guard";
|
|
109
116
|
|
|
@@ -234,15 +241,6 @@ function buildClaudeCodeBetas({
|
|
|
234
241
|
return betas;
|
|
235
242
|
}
|
|
236
243
|
|
|
237
|
-
function getHeaderCaseInsensitive(headers: Record<string, string> | undefined, headerName: string): string | undefined {
|
|
238
|
-
if (!headers) return undefined;
|
|
239
|
-
const normalizedName = headerName.toLowerCase();
|
|
240
|
-
for (const [key, value] of Object.entries(headers)) {
|
|
241
|
-
if (key.toLowerCase() === normalizedName) return value;
|
|
242
|
-
}
|
|
243
|
-
return undefined;
|
|
244
|
-
}
|
|
245
|
-
|
|
246
244
|
function isClaudeCodeClientUserAgent(userAgent: string | undefined): userAgent is string {
|
|
247
245
|
if (!userAgent) return false;
|
|
248
246
|
return userAgent.toLowerCase().startsWith("claude-cli");
|
|
@@ -1238,7 +1236,7 @@ export type AnthropicClientOptionsArgs = {
|
|
|
1238
1236
|
disableStrictTools?: boolean;
|
|
1239
1237
|
fetch?: FetchImpl;
|
|
1240
1238
|
maxRetryDelayMs?: number;
|
|
1241
|
-
|
|
1239
|
+
sessionId?: string;
|
|
1242
1240
|
};
|
|
1243
1241
|
|
|
1244
1242
|
export type AnthropicClientOptionsResult = {
|
|
@@ -2177,7 +2175,10 @@ const streamAnthropicOnce = (
|
|
|
2177
2175
|
thinkingDisplay: options?.thinkingDisplay,
|
|
2178
2176
|
fetch: options?.fetch,
|
|
2179
2177
|
maxRetryDelayMs: options?.maxRetryDelayMs,
|
|
2180
|
-
|
|
2178
|
+
sessionId:
|
|
2179
|
+
options?.sessionId ??
|
|
2180
|
+
extractClaudeMetadataSessionId(options?.metadata?.user_id) ??
|
|
2181
|
+
options?.promptCacheKey,
|
|
2181
2182
|
disableStrictTools,
|
|
2182
2183
|
});
|
|
2183
2184
|
client = created.client;
|
|
@@ -3203,7 +3204,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
|
|
|
3203
3204
|
thinkingEnabled = false,
|
|
3204
3205
|
isOAuth,
|
|
3205
3206
|
maxRetryDelayMs,
|
|
3206
|
-
|
|
3207
|
+
sessionId,
|
|
3207
3208
|
disableStrictTools: disableStrictToolsOverride,
|
|
3208
3209
|
} = args;
|
|
3209
3210
|
const compat = model.compat;
|
|
@@ -3266,6 +3267,11 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
|
|
|
3266
3267
|
dynamicHeaders,
|
|
3267
3268
|
headers,
|
|
3268
3269
|
);
|
|
3270
|
+
applyInferenceHeaders(defaultHeaders, {
|
|
3271
|
+
provider: model.provider,
|
|
3272
|
+
protocol: "anthropic",
|
|
3273
|
+
sessionId,
|
|
3274
|
+
});
|
|
3269
3275
|
|
|
3270
3276
|
return {
|
|
3271
3277
|
isOAuthToken: false,
|
|
@@ -3288,22 +3294,23 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
|
|
|
3288
3294
|
betaFeatures.push(interleavedThinkingBeta);
|
|
3289
3295
|
}
|
|
3290
3296
|
|
|
3297
|
+
const requestModelHeaders = mergeHeaders(
|
|
3298
|
+
model.headers,
|
|
3299
|
+
foundryCustomHeaders,
|
|
3300
|
+
getUmansWebSearchHeader(model, mergeHeaders(model.headers, headers)),
|
|
3301
|
+
headers,
|
|
3302
|
+
dynamicHeaders,
|
|
3303
|
+
);
|
|
3291
3304
|
const defaultHeaders = buildAnthropicHeaders({
|
|
3292
3305
|
apiKey,
|
|
3293
3306
|
baseUrl,
|
|
3294
3307
|
isOAuth: oauthToken,
|
|
3295
3308
|
extraBetas: betaFeatures,
|
|
3296
3309
|
stream,
|
|
3297
|
-
modelHeaders:
|
|
3298
|
-
model.headers,
|
|
3299
|
-
foundryCustomHeaders,
|
|
3300
|
-
getUmansWebSearchHeader(model, mergeHeaders(model.headers, headers)),
|
|
3301
|
-
headers,
|
|
3302
|
-
dynamicHeaders,
|
|
3303
|
-
),
|
|
3310
|
+
modelHeaders: requestModelHeaders,
|
|
3304
3311
|
isCloudflareAiGateway: model.provider === "cloudflare-ai-gateway",
|
|
3305
3312
|
allowAnthropicHeaderOverrides: model.compat.allowAnthropicHeaderOverrides,
|
|
3306
|
-
claudeCodeSessionId,
|
|
3313
|
+
claudeCodeSessionId: sessionId,
|
|
3307
3314
|
claudeCodeBetas: oauthToken
|
|
3308
3315
|
? buildClaudeCodeBetas({
|
|
3309
3316
|
agentRequest: hasTools || thinkingEnabled,
|
|
@@ -3313,6 +3320,11 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
|
|
|
3313
3320
|
})
|
|
3314
3321
|
: [],
|
|
3315
3322
|
});
|
|
3323
|
+
applyInferenceHeaders(defaultHeaders, {
|
|
3324
|
+
provider: model.provider,
|
|
3325
|
+
protocol: "anthropic",
|
|
3326
|
+
sessionId,
|
|
3327
|
+
});
|
|
3316
3328
|
|
|
3317
3329
|
if (model.provider === "cloudflare-ai-gateway") {
|
|
3318
3330
|
return {
|
|
@@ -3491,6 +3503,64 @@ function applyPromptCaching(params: MessageCreateParamsStreaming, cacheControl?:
|
|
|
3491
3503
|
}
|
|
3492
3504
|
}
|
|
3493
3505
|
|
|
3506
|
+
/**
|
|
3507
|
+
* Anchor cache_control on the stable request head — the last (non-deferred)
|
|
3508
|
+
* tool definition and the last system block. The canonical cache order is
|
|
3509
|
+
* tools → system → messages, so a breakpoint on the final system block caches
|
|
3510
|
+
* the entire tools+system prefix, and the extra tool breakpoint keeps the tool
|
|
3511
|
+
* definitions cached even when the system text changes. This guarantees the
|
|
3512
|
+
* large, unchanging head is a cache hit on every turn regardless of how the
|
|
3513
|
+
* message tail churns — the breakpoint placement first-party Anthropic clients
|
|
3514
|
+
* (Claude Code, Pi) use. Without it, the general API-key path anchors only the
|
|
3515
|
+
* moving message tail, so tail churn re-writes the whole head uncached.
|
|
3516
|
+
*
|
|
3517
|
+
* Anthropic allows at most 4 cache breakpoints per request. At most one is
|
|
3518
|
+
* spent on tools and one on system here, leaving two for the message tail in
|
|
3519
|
+
* `applyPromptCaching`. Head caching is skipped entirely when the head is
|
|
3520
|
+
* already anchored — the OAuth Claude Code path caches its own instruction
|
|
3521
|
+
* block at buildAnthropicSystemBlocks, and via the canonical tools → system
|
|
3522
|
+
* order that single system breakpoint already caches every preceding tool. Re-
|
|
3523
|
+
* anchoring there would be redundant, would change the OAuth wire, and could
|
|
3524
|
+
* push a tool-heavy request over the 4-breakpoint budget, so the general
|
|
3525
|
+
* API-key path (nothing cached upstream) is the only one decorated here.
|
|
3526
|
+
*
|
|
3527
|
+
* Runs after the byte-stability plane (planStableAnthropicSystem /
|
|
3528
|
+
* planStableAnthropicTools), which hands back fresh block/tool copies each turn
|
|
3529
|
+
* and keys tool identity off a fingerprint that excludes cache_control — so
|
|
3530
|
+
* decorating here is byte-stable across turns and never forces a re-baseline.
|
|
3531
|
+
*/
|
|
3532
|
+
function applyHeadCaching(
|
|
3533
|
+
systemBlocks: AnthropicSystemBlock[] | undefined,
|
|
3534
|
+
tools: AnthropicWireTool[] | undefined,
|
|
3535
|
+
cacheControl?: AnthropicCacheControl,
|
|
3536
|
+
): void {
|
|
3537
|
+
if (!cacheControl) return;
|
|
3538
|
+
|
|
3539
|
+
// If anything in the head already carries a breakpoint, the head is already
|
|
3540
|
+
// cached (OAuth anchors its identity system block, which — canonical order
|
|
3541
|
+
// tools → system — caches all tools too). Leave it untouched.
|
|
3542
|
+
const headAlreadyCached =
|
|
3543
|
+
(systemBlocks?.some(block => block.cache_control != null) ?? false) ||
|
|
3544
|
+
(tools?.some(tool => tool.cache_control != null) ?? false);
|
|
3545
|
+
if (headAlreadyCached) return;
|
|
3546
|
+
|
|
3547
|
+
if (tools && tools.length > 0) {
|
|
3548
|
+
// Deferred tools are not part of the checked prefix until referenced, so
|
|
3549
|
+
// anchor the last tool that actually sits in the stable prefix.
|
|
3550
|
+
for (let index = tools.length - 1; index >= 0; index--) {
|
|
3551
|
+
const tool = tools[index];
|
|
3552
|
+
if (!tool || tool.defer_loading) continue;
|
|
3553
|
+
tool.cache_control = cloneAnthropicCacheControl(cacheControl);
|
|
3554
|
+
break;
|
|
3555
|
+
}
|
|
3556
|
+
}
|
|
3557
|
+
|
|
3558
|
+
if (systemBlocks && systemBlocks.length > 0) {
|
|
3559
|
+
const lastBlock = systemBlocks[systemBlocks.length - 1];
|
|
3560
|
+
if (lastBlock) lastBlock.cache_control = cloneAnthropicCacheControl(cacheControl);
|
|
3561
|
+
}
|
|
3562
|
+
}
|
|
3563
|
+
|
|
3494
3564
|
function usesAdaptiveThinkingTagOnly(model: Model<"anthropic-messages">): boolean {
|
|
3495
3565
|
const thinking = model.thinking;
|
|
3496
3566
|
if (thinking?.mode !== "anthropic-adaptive") return false;
|
|
@@ -3966,6 +4036,9 @@ function buildParams(
|
|
|
3966
4036
|
if (controlState) syncAnthropicControlState(controlState, wireMessages);
|
|
3967
4037
|
systemBlocks = planStableAnthropicSystem(systemBlocks, controlState, model.compat.supportsMidConversationSystem);
|
|
3968
4038
|
tools = planStableAnthropicTools(tools, wireMessages, controlState, model.compat.supportsMidConversationToolChanges);
|
|
4039
|
+
// Anchor the stable tools+system head so it stays cached across turns; the
|
|
4040
|
+
// moving message tail is anchored separately in applyPromptCaching below.
|
|
4041
|
+
applyHeadCaching(systemBlocks, tools, cacheControl);
|
|
3969
4042
|
const topLevelEffort = planStableAnthropicEffort(
|
|
3970
4043
|
outputConfigEffort,
|
|
3971
4044
|
wireMessages,
|