@oh-my-pi/pi-ai 18.1.6 → 18.1.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/CHANGELOG.md +27 -104
  2. package/README.md +66 -66
  3. package/dist/types/providers/cursor.d.ts +2 -2
  4. package/dist/types/providers/inference-headers.d.ts +4 -4
  5. package/dist/types/providers/openai-configuration-update.d.ts +64 -0
  6. package/dist/types/providers/openai-responses-wire.d.ts +15 -1
  7. package/dist/types/providers/openai-responses.d.ts +6 -1
  8. package/dist/types/providers/openai-shared.d.ts +12 -0
  9. package/dist/types/registry/engine/oauth-code.d.ts +1 -1
  10. package/dist/types/registry/oauth/callback-server.d.ts +2 -0
  11. package/dist/types/registry/oauth/native-scheme-callback.d.ts +13 -0
  12. package/dist/types/registry/oauth/types.d.ts +2 -1
  13. package/dist/types/utils/proxy.d.ts +7 -0
  14. package/dist/types/utils/request-debug.d.ts +6 -5
  15. package/dist/types/utils/transport-fetch.d.ts +18 -0
  16. package/dist/types/utils.d.ts +1 -0
  17. package/package.json +38 -37
  18. package/src/auth-storage.ts +10 -9
  19. package/src/providers/cursor.ts +16 -19
  20. package/src/providers/inference-headers.ts +17 -16
  21. package/src/providers/openai-codex-responses.ts +48 -2
  22. package/src/providers/openai-configuration-update.ts +170 -0
  23. package/src/providers/openai-responses-wire.ts +15 -0
  24. package/src/providers/openai-responses.ts +41 -0
  25. package/src/providers/openai-shared.ts +22 -6
  26. package/src/providers/transform-messages.ts +10 -6
  27. package/src/registry/engine/oauth-code.ts +3 -1
  28. package/src/registry/oauth/callback-server.ts +152 -24
  29. package/src/registry/oauth/native-scheme-callback.ts +62 -0
  30. package/src/registry/oauth/types.ts +2 -1
  31. package/src/stream.ts +4 -24
  32. package/src/utils/proxy.ts +37 -31
  33. package/src/utils/request-debug.ts +6 -29
  34. package/src/utils/transport-fetch.ts +50 -0
  35. package/src/utils.ts +3 -24
package/CHANGELOG.md CHANGED
@@ -2,6 +2,32 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [18.1.9] - 2026-09-04
6
+
7
+ ### Added
8
+
9
+ - Added recoverable native custom-scheme OAuth callbacks for macOS, Linux desktops, and Windows, with a manual fallback for unavailable or remote sessions.
10
+
11
+ ### Fixed
12
+
13
+ - Fixed Gemini tool continuations through custom Anthropic Messages proxies and OpenAI Responses relays, preserving tool-call and result associations across multi-turn requests.
14
+
15
+ ## [18.1.8] - 2026-09-03
16
+
17
+ ### Added
18
+
19
+ - Added GPT-6 Astra support for preserving prompt caching when changing the thinking level during a conversation across the OpenAI and OpenAI Codex providers.
20
+
21
+ ### Changed
22
+
23
+ - Updated OpenAI Codex requests to improve routing by communicating the selected model and service tier across Responses, WebSocket, and remote-compaction requests.
24
+
25
+ ## [18.1.7] - 2026-09-03
26
+
27
+ ### Fixed
28
+
29
+ - Fixed DeepSeek-family Responses replay (e.g. opencode-go) rejecting a resumed thinking-mode turn with `400 The reasoning_text in the thinking mode must be passed back to the API` when compaction dropped the turn's reasoning; a non-empty placeholder is now synthesized instead of an empty `reasoning_text` ([#10690](https://github.com/can1357/oh-my-pi/issues/10690)).
30
+
5
31
  ## [18.1.6] - 2026-09-03
6
32
 
7
33
  ### Breaking Changes
@@ -2022,107 +2048,4 @@
2022
2048
  - Scoped Antigravity usage blocking and ranking by model family (`gemini-*`/`gemma-*` → Google, `claude-*` → Anthropic, `gpt-*`/`openai/*` → OpenAI), so an exhausted Gemini counter no longer makes a healthy Claude/OpenAI Antigravity credential unavailable until reset. ([#2198](https://github.com/can1357/oh-my-pi/issues/2198))
2023
2049
  - Fixed no-model Antigravity credential lookups (e.g. image-provider discovery) inheriting provider-wide exhaustion: `scopeLimits` now returns no limits without a concrete backend counter, and `blockScope` always returns a counter scope so missing model context can never fall through to AuthStorage's provider-wide block bucket. ([#2198](https://github.com/can1357/oh-my-pi/issues/2198))
2024
2050
 
2025
- ## [15.10.11] - 2026-06-10
2026
-
2027
- ### Breaking Changes
2028
-
2029
- - The model catalog moved to the new `@oh-my-pi/pi-catalog` package. Deep subpath exports `@oh-my-pi/pi-ai/models.json`, `/models`, `/model-cache`, `/model-manager`, `/model-thinking`, `/effort`, `/provider-models*`, `/utils/discovery*`, `/providers/openai-codex/constants`, `/providers/google-gemini-headers`, and `/providers/openai-completions-compat` are gone — import the `@oh-my-pi/pi-catalog` equivalents (`/models.json`, `/models`, `/model-cache`, `/model-manager`, `/model-thinking`, `/effort`, `/provider-models*`, `/discovery*`, `/wire/codex`, `/wire/gemini-headers`, `/compat/openai`). The pi-ai root barrel re-exports only the model/effort _types_ its own signatures use (`Model`, `Api`, `ThinkingConfig`, `Effort`, `Usage`, compat interfaces) — catalog _values_ (`getBundledModel(s)`, `calculateCost`, `modelsAreEqual`, `clampThinkingLevelForModel`, `DEFAULT_MODEL_PER_PROVIDER`, …) must be imported from `@oh-my-pi/pi-catalog`.
2030
- - `ProviderDefinition` is now auth-only: `defaultModel`, `createModelManagerOptions`, `catalogDiscovery`, `dynamicModelsAuthoritative`, `allowUnauthenticated`, and `specialModelManager` moved to pi-catalog's `CATALOG_PROVIDERS` table, and `KnownProviderId` was replaced by pi-catalog's `KnownProvider` (registry completeness is enforced by a compile-time check against that union). The pure GitHub Copilot key/endpoint helpers moved from `registry/oauth/github-copilot` to `@oh-my-pi/pi-catalog/wire/github-copilot`.
2031
-
2032
- ### Added
2033
-
2034
- - Exported `wrapFetchForCch` so non-streaming OAuth callers (e.g. the web-search provider) can patch the Claude Code billing-header `cch` attestation into their request bodies instead of shipping the `cch=00000` placeholder.
2035
-
2036
- ### Changed
2037
-
2038
- - Reduced idle-watchdog churn on the token hot path: the abort promise/listener is created once per stream instead of per yielded item, the deadline uses a persistent re-armed timer instead of a `setTimeout` create/destroy pair per delta, and the persistent race promises are re-minted every 1024 items so per-race reaction records cannot accumulate for the stream's whole life.
2039
- - Memoized Anthropic many-image downscaling by content-block identity, so long sessions with stable message objects no longer re-decode and re-encode every oversized image on each request and retry.
2040
- - Tool-argument validation errors now truncate embedded argument strings at 256 chars per field — a failed `write`-class call no longer echoes hundreds of KB of payload back to the model as the error message.
2041
- - Auth storage no longer issues per-boot no-op writes: the schema-version row is only rewritten when the recorded version actually changes, and the credential identity-key backfill skips rows whose derived identity is null — reopening a current-schema database now performs zero write transactions
2042
- - Plain provider env-var names moved to the catalog table: registry defs dropped their 48 `envKeys` literals (including the pure `$pickenv` pickers for `huggingface`/`qwen-portal`/`xai-oauth`), `getEnvApiKey` now derives those fallbacks from `CATALOG_PROVIDERS[].envVars`, and `envKeys` remains only for computed resolvers (Anthropic Foundry, Vertex ADC, Bedrock credential chains) and non-catalog providers (`kagi`, `tavily`, `parallel`, `perplexity`)
2043
- - Protocol handlers are now pure `model.compat` readers — the per-request `resolve*Compat`/`detect*Compat` calls (anthropic ×11, responses ×3, completions wrappers), inline `strictResponsesPairing` host detection, the OpenCode `reasoning_content` mutation block, and all `resolvedBaseUrl` threading are gone. Compat is materialized once at model build time (`@oh-my-pi/pi-catalog` `buildModel`); the OpenCode thinking-mode quirk is a precomputed `compat.whenThinking` pointer swap, and request-time base-URL overrides only feed the HTTP client. Behavior is unchanged (the Anthropic `supportsLongCacheRetention` official-endpoint gate is folded into detection).
2044
- - Providers now read baked thinking/wire metadata instead of re-parsing model ids per request: the Anthropic handler gates sampling params on `model.compat.supportsSamplingParams` and adaptive `display` on `model.thinking.supportsDisplay` (Bedrock too), adaptive effort tiers come from the baked `thinking.effortMap`, the Google `thinkingLevel` map is static, and effort-dial-less reasoners (`thinking: undefined`, e.g. `xai-oauth/grok-build`) short-circuit `resolveOpenAiReasoningEffort` without the removed `modelOmitsReasoningEffort` predicate.
2045
- - Anthropic streaming retries now use a 10-retry budget with the Anthropic-compatible 0.5s exponential backoff capped at 8s with jitter; server `retry-after` hints still win, and retryable pre-content failures such as 502s no longer stop after three tries.
2046
-
2047
- ### Fixed
2048
-
2049
- - Fixed Ollama chat requests honoring `omitMaxOutputTokens`, sending `think: false` when reasoning is explicitly disabled, and preserving HTTP 400 response bodies in surfaced errors.
2050
- - Fixed `AuthStorage.markUsageLimitReached` collapsing "every sibling is momentarily blocked" into "no sibling exists": it now returns `UsageLimitMarkResult` with the earliest sibling block expiry (`retryAtMs`), so retry layers can wait out a short-lived block (60s post-401, 5-min usage-probe) instead of adopting the provider's multi-hour retry-after. `rotateSessionCredential` and the auth-gateway adapt to the new shape.
2051
- - Fixed Gemini streaming silently presenting truncated or blocked output as a successful `stop`: in-band `{"error":{...}}` events and `promptFeedback.blockReason` chunks were never inspected, and a stream ending without any `finishReason` kept the initialized `stop` — all three now surface as errors (both the API-key and gemini-cli/Antigravity consumers), and the `toolUse` stop-reason override no longer masks `SAFETY`/`MALFORMED_FUNCTION_CALL` finishes that arrive after a valid tool call.
2052
- - Fixed Gemini/Bedrock error finishes reporting "An unknown error occurred": the raw finish/stop reason (`MALFORMED_FUNCTION_CALL`, `RECITATION`, `guardrail_intervened`, …) is now recorded into the surfaced error message.
2053
- - Fixed the Anthropic provider retry loop ignoring server `retry-after` on 429/529 — it now waits `max(headerDelay, backoff)` instead of hammering a rate-limited endpoint three times within ~14s of guaranteed failures.
2054
- - Fixed in-stream Anthropic SSE `error` events being thrown as raw JSON envelopes; the structured `error.type`/`message` is parsed out, keeping retry classification on the typed token instead of accidental regex hits.
2055
- - Fixed transparent-reconnect tolerance duplicating content behind replaying proxies: after a duplicate `message_start`, replayed `content_block_start` events for already-closed indexes are now consumed silently instead of appending duplicate text/tool calls.
2056
- - Fixed the Anthropic gateway accepting malformed known-type content blocks (e.g. `{type:"text", text:123}`) through the unknown-block catch-all, corrupting history and surfacing later as an opaque TypeError — they now fail validation with a clean 400. The gateway's encode stream also emits `ping` keepalives every 15s and a complete `message_start`/`message_delta`/`message_stop` envelope when the inner stream ends without a terminal event, so strict clients no longer classify slow or empty streams as protocol errors.
2057
- - Fixed dotted-version Claude ids (`claude-opus-4.7`/`4.8` on GitHub Copilot, Vercel AI Gateway, Zenmux) missing adaptive thinking `display` support — streamed reasoning stayed hidden on those entries because the display predicate only matched dash-form ids (same failure class as #1373).
2058
- - Fixed the Mistral `requiresThinkingAsText` replay path calling `.unshift()` on string assistant content — an unconditional TypeError that failed any same-model history turn carrying both thinking and text.
2059
- - Fixed the Responses gateway stripping `encrypted_content` from inbound reasoning items (strip-mode schema), which broke codex-style stateless replay; the schema is now loose, restoring the symmetry the outbound encoder already preserved. Composite internal `callId|itemId` ids are also split before hitting the wire so third-party clients that validate `call_id` charsets no longer reject them.
2060
- - Ported the shared unfinished-tool-call sweep to the codex `response.completed` handler, so a lost `output_item.done` can no longer persist a tool call with stale `{}` arguments and transient parser fields into session history.
2061
- - Fixed live text freezing until item completion when a lossy proxy drops `content_part.added`: the missing part is now synthesized on the first `output_text`/`refusal` delta (shared and codex decoders).
2062
- - Fixed interleaved `content`/`tool_calls` deltas fragmenting a tool call into a truncated call plus a nameless phantom: text/thinking transitions no longer finish open tool-call blocks, so index-only continuation deltas re-find them.
2063
- - Fixed the Azure chat-completions path ignoring `AZURE_OPENAI_DEPLOYMENT_NAME_MAP` (only the Responses provider honored it), producing opaque 404s when deployment names differ from catalog model ids.
2064
- - Fixed the chat gateway discarding inbound assistant `reasoning_content`, which fed DeepSeek/Kimi exact-replay upstreams a placeholder instead of the model's actual reasoning; it now round-trips as a thinking block, and `toolcall_end` emits a corrective id/name chunk when the streamed start carried empty values.
2065
- - Fixed the auth retry loop minting OAuth tokens and firing a doomed request after the caller aborted, and stopped masking resolver failures (broker/network/refresh errors) as "No API key" — the actual cause is preserved.
2066
- - Fixed `EventStream.end()` without a terminal result leaving `.result()` pending forever (reachable via extension streams and the lazy wrapper); it now rejects with a synthesized error.
2067
- - Fixed the Copilot retry wrapper blind-retrying every retryable error with fixed 400ms delays: 429/5xx now honor `Retry-After` (capped at 30s) and other statuses are not retried, while status-less transport blips keep the linear retry.
2068
- - Fixed the OpenAI completions error path ending the stream without closing open text/thinking/tool-call blocks, leaving consumers with orphaned block lifecycles on every stream error or idle-timeout abort.
2069
- - Fixed DSML hold-back freezing display on any bare `<` in model output for up to 256 chars: idle-state holding now only triggers on a strict DSML section-open prefix, and blowing the 1MB parameter cap no longer leaks the closing envelope tags as visible text; a capped parameter value also carries an explicit `…[parameter truncated]` marker instead of executing the tool with silently corrupted input.
2070
- - Fixed schema normalization blanking DAG-shared subtrees to `{}`: the visited-set cycle guard treated a subschema object reused across two properties as a cycle; path-tracking `enter`/`exit` now allows sharing while still short-circuiting true cycles, frozen input schemas no longer throw, and the path counter no longer leaks depth on the cycle branch (which made every later normalization of the same object misreport a cycle).
2071
- - Fixed shared in-flight Google token refreshes being bound to the first caller's `AbortSignal`, failing every concurrent waiter when one parallel Vertex call was cancelled; callers now race their own signal against a detached refresh, which is bounded by its own 30s timeout so a hung fetch cannot pin the in-flight slot until process restart.
2072
- - Fixed Gemini <3 multimodal tool results breaking the single-function-response-turn invariant for parallel tool calls (image turns are buffered and flushed after the merged functionResponse turn), and the gemini-cli consumer now defaults missing `functionCall.args` to `{}` like the shared consumer.
2073
- - Fixed Bedrock dropping `toolConfig` entirely when `toolChoice` is `"none"` while history still contains tool blocks — the Converse API rejects such requests, so tool specs are kept and only the choice is omitted.
2074
- - Fixed AWS credential handling serving expired credentials until process restart: cache entries are invalidated on 401/403, file-sourced session-token credentials get a 5-minute TTL, and concurrent first requests single-flight instead of spawning duplicate `credential_process`/SSO fetches — the shared resolution is detached from the first caller's abort signal (one cancelled request no longer fails every waiter) and bounded by its own 30s timeout. The eventstream reader also cancels the response body on abnormal exit instead of leaving the HTTP connection draining.
2075
- - Fixed an unbounded, zero-backoff Codex WebSocket reconnect loop on `websocket_connection_limit_reached`: the no-content reconnect path never consulted the retry budget and never waited, hammering the endpoint forever when the limit is account-scoped. Reconnects are now budgeted and delayed like every other WS retry path, falling back to a single SSE replay when exhausted.
2076
- - Fixed the Codex whitespace-loop breaker not observing degenerate frames that arrive after their item closed (or before it opened) — those frames count as stream progress, so the idle watchdogs never fired and the turn hung forever, which is exactly the failure mode the breaker exists for. Whitespace-loop recovery now also refuses to replay the turn once a `toolcall_end` was delivered, surfacing the error instead of re-emitting the same tool calls.
2077
- - Fixed the two remaining Codex retry paths (WS mid-stream reconnect and the empty-content SSE fallback) leaking blockless native output items (e.g. `web_search_call`) from the failed attempt into the replayed turn's `providerPayload` and append baseline.
2078
- - Fixed Codex WebSocket failure handling closing whatever connection currently occupies the session slot — including a concurrent caller's in-flight CONNECTING handshake, whose rejection (`websocket closed before open`) is classified fatal and disabled WebSockets for the whole session. Failure cleanup now skips CONNECTING sockets and the pool re-joins replacement handshakes (bounded).
2079
- - Fixed the Codex request transformer not repairing orphan `custom_tool_call_output` items (only `function_call_output` was folded into an assistant note) — a compaction splice that dropped an `apply_patch` call while keeping its result produced a hard 400 on the default GPT-5 Codex toolset.
2080
- - Fixed `processResponsesStream` finalizing reasoning items via a bare `itemId` content scan instead of the routed entry: with id-less reasoning items (local hosts), every `output_item.done` matched the FIRST thinking block — the second item's text clobbered it and the second block was never finalized or signed.
2081
- - Fixed `processResponsesStream` dropping tool calls and message text whose `output_item.added` event was lost (lossy proxies): `toolcall_end` was emitted with a dangling contentIndex while the call never entered `message.content`, so the agent loop silently never executed it. The done handler now synthesizes the missing block; still-open tool-call blocks are also final-parsed at `response.completed` so the `toolUse` override cannot hand the agent stale `{}` arguments.
2082
- - Fixed `response.incomplete` with `incomplete_details.reason: "content_filter"` being reported as a token-cap truncation (`stopReason: "length"`) — the agent loop's length recovery then asked the model to "shorten" a filtered prompt. Content-filtered turns now surface as errors; usage is also populated from `response.failed` events, and an unknown terminal status degrades to `"stop"` with a logged anomaly instead of throwing away a fully-streamed response.
2083
- - Fixed Copilot `premiumRequests` accounting being dropped from failed/cancelled responses: `populateResponsesUsageFromResponse` replaced `usage` wholesale and the error path threw before the success-path re-apply. The populate now preserves the field.
2084
- - Fixed `deduplicateToolCallIds` suffixing the whole composite Responses id (`callId|itemId`) — `normalizeResponsesToolCallId` extracts the first segment as the wire `call_id` at encode time, so both copies collapsed back onto one `call_id` and the request carried duplicate call/output pairs. The suffix and length budget now apply per segment.
2085
- - Gated native history payload replay on api + model id in both Responses providers: after a mid-session model switch, reasoning items carrying encrypted content minted by the previous model were replayed verbatim under the new model. Replay now falls back to block re-encode (which already strips foreign signatures), matching `transformMessages`' same-model trust rule.
2086
- - Fixed Azure OpenAI Responses requests omitting `store: false` while requesting `reasoning.encrypted_content` (stateless-only per OpenAI), replaying custom tool calls paired with mismatched `function_call_output` items (customCallIds was never threaded through), letting the SDK's internal retries (maxRetries 5) silently re-POST inside the explicit first-event deadline, and sending a `prompt_cache_key` when the caller opted out via `cacheRetention: "none"`.
2087
- - Fixed strict-pairing Responses backends (Azure, Copilot) silently discarding tool results whose call is absent from history — the result is now folded into an assistant note (same shape as orphan-output repair) so the model keeps the information.
2088
- - Fixed the OpenAI Responses first-event watchdog staying armed across the `onResponse` notification callback (a slow callback aborted an already-connected stream), Copilot transient-model retries re-attempting on an already-aborted signal (instant dead retry surfacing the scheduler's AbortError), Codex `reasoningSummary: null` being coerced to `"auto"` (the documented omit-summary contract was unreachable), nested Codex error codes (`response.error.code`) being invisible to the connection-limit/previous-response recovery matchers, and the session id leaking unredacted into `PI_CODEX_DEBUG` logs via the `x-client-request-id` header.
2089
- - Fixed `processResponsesStream` (shared by `openai-responses` and `azure-openai-responses`) ignoring the terminal `response.incomplete` event: a max-output-tokens-truncated response ended with `stopReason: "stop"`, zero usage, and no cost instead of `"length"` with the reported token counts. `response.incomplete` is now handled alongside `response.completed` and counts as stream progress for the idle watchdogs.
2090
- - Fixed custom tool-call content blocks keeping the transient `partialJson` accumulation buffer (and a potentially stale `arguments.input`) after `response.output_item.done` in the shared Responses stream processor — the function_call branch already cleaned these up.
2091
- - Fixed two OpenAI Codex stream-retry paths (whitespace-loop recovery and retryable provider errors) leaking native output items from the abandoned attempt into the replayed turn's `providerPayload` — stale reasoning items completed before the failure were re-sent as history input on subsequent requests alongside the retry's own items.
2092
- - Fixed the Codex WebSocket queue wiping already-received frames when a transport error arrived: a `response.completed` queued just before an eager server close was discarded, turning a finished response into a spurious `websocket closed` failure and a full request replay. Errors now append behind pending data frames.
2093
- - Fixed concurrent `getOrCreateCodexWebSocketConnection` callers (prewarm racing the first request) tearing down each other's in-flight handshake — closing a CONNECTING socket rejected the other caller with a fatal `websocket closed before open`, disabling WebSockets for the entire session. Callers now join the pending handshake.
2094
- - Stopped the Codex connection-limit recovery from replaying a turn over SSE after a `toolcall_end` had already been delivered to the consumer (`canSafelyReplayWebsocketOverSse` guard was bypassed, re-emitting the same tool calls); the error now surfaces instead.
2095
- - Extended the Codex whitespace-only argument-delta circuit breaker to `custom_tool_call_input.delta` frames, which counted as stream progress and could keep a degenerate response alive forever with no cap on buffer growth.
2096
- - Fixed Codex stream failures during transport open reporting a synthetic request dump (empty URL/body) instead of the real request, and a `response.created` event resetting the recorded time-to-first-token.
2097
- - Fixed the Codex WebSocket connect watchdog timer leaking (pinning the event loop for up to 10s) when the request signal aborted before or during the handshake.
2098
- - Fixed OpenRouter-hosted Anthropic adaptive reasoning models (Claude Fable/Mythos 5 and Opus 4.6+) so the catalog exposes `xhigh`; Fable/Mythos and Opus 4.7+ requests now map user `high`/`xhigh` onto OpenRouter's Anthropic `xhigh`/`max` effort scale.
2099
- - Fixed an unknown Anthropic `stop_reason` failing the whole turn after the response had fully streamed. `mapStopReason` threw on unrecognized values, and since the reason arrives on the trailing `message_delta` the error was unretryable — the live `model_context_window_exceeded` stop reason (default on Sonnet 4.5+) hit this path. It now maps to `length`, and any future unknown reason degrades to a logged anomaly plus a normal `stop` instead of an error.
2100
- - Stopped clamping API-key Anthropic requests to Claude Code's 64k output cap. The `CLAUDE_CODE_MAX_OUTPUT_TOKENS` clamp exists to match the OAuth wire fingerprint, but `buildParams` applied it unconditionally, silently halving the output budget of 128k-output models (e.g. Opus 4.8) for API-key callers. OAuth requests keep the clamp.
2101
- - Stopped a successful strict-tools fallback from shipping `errorMessage` on a `stopReason: "stop"` assistant message. After a grammar-too-large 400 triggered the non-strict retry, the original 400 text was kept on the final message even when the retry succeeded — consumers that treat `errorMessage` presence as failure (e.g. balance probes) misclassified the turn, and the stale text suppressed later refusal explanations. The fallback is now logged instead.
2102
- - Fixed model-supplied `User-Agent` headers being silently dropped on non-OAuth Anthropic requests. `enforcedHeaderKeys` filtered the header out of `modelHeaders` in every branch but only the OAuth branch set one back; the Cloudflare-gateway, bearer-gateway, and `X-Api-Key` branches now forward the caller's value verbatim.
2103
- - Stopped sending the `fast-mode-2026-02-01` beta header once a session has learned the endpoint+model rejects fast mode (`fastModeDisabled` provider state), matching the already-dropped `speed` param.
2104
- - Stopped `buildAnthropicHeaders` defaulting API-key requests onto the full Claude Code OAuth beta list (`oauth-2025-04-20`, `claude-code-20250219`, …). The `claudeCodeBetas` default is now OAuth-gated, matching the streaming path — the web-search header builder was the only caller hitting the default, so API-key search requests now carry just their own betas (e.g. `web-search-2025-03-05`). An empty `anthropic-beta` header is omitted entirely instead of being sent as an empty string.
2105
- - Fixed image-bearing `developer` messages being upgraded to mid-conversation `system` turns on Opus 4.8+/Fable/Mythos 5. System content is text-only on the wire, so a developer turn carrying image blocks in an upgrade-eligible position produced a 400; it now stays a `user` message.
2106
- - Fixed a spliced reconnect's second envelope overwriting the completed Anthropic message: `message_delta` was not gated by the terminal-stop flag (content events and duplicate `message_start` were), so the splice's `stop_reason`/usage replaced the finished turn's — a `tool_use` turn could be relabeled `stop`, and the harness then never executed the streamed tool calls. Post-terminal deltas are now logged as envelope anomalies and skipped.
2107
- - Fixed a `ping` arriving before `message_start` consuming the Anthropic first-event watchdog: the stall was then classified as a terminal mid-stream idle timeout instead of a retryable first-event timeout. Pings no longer count as the first item but still refresh the idle deadline once content is flowing.
2108
- - Fixed Anthropic-compatible proxies that omit `usage`/`delta` objects from `message_start`/`message_delta`/`content_block_*` envelopes crashing the turn with an unretryable `TypeError`; the missing payloads now degrade to logged envelope anomalies like every other malformed-frame case.
2109
- - Fixed `applyPromptCaching` placing `cache_control` on `thinking`/`redacted_thinking` blocks — Anthropic rejects that with a 400. A thinking-only assistant turn inside the trailing cache window (e.g. followed by the synthetic `Continue.` pad) no longer receives a breakpoint.
2110
- - Fixed consecutive `assistant` params reaching the wire when an empty user/developer turn between two assistant turns was dropped by the converter (e.g. an empty "nudge" submission after a length-truncated reply); Anthropic 400s on non-alternating assistant turns, and the broken triple replayed on every subsequent request. A `user: "Continue."` separator is now inserted, mirroring the trailing-prefill fallback.
2111
- - Fixed adaptive-display classification misparsing bare dated Opus ids: `claude-opus-4-20250514` (Opus 4.0) parsed as minor `20250514` ≥ 4.7, which silently dropped the `interleaved-thinking-2025-05-14` beta for API-key Opus 4.0 requests.
2112
- - Fixed `output_config.effort` shipping without the `effort-2025-11-24` beta on thinking-off requests against adaptive-only Claude models (the effort:"low" pin), and the mid-conversation `system` role shipping without `mid-conversation-system-2026-04-07` on API-key and OAuth-utility requests; both betas are now added whenever the request can carry the corresponding field.
2113
- - Fixed GitHub Copilot anthropic-messages requests going out with no `Content-Type` and no `anthropic-version` header — the copilot branch builds its headers from scratch and Bun's fetch does not default `Content-Type` for string bodies. Both headers are now pinned to match every other branch.
2114
- - Fixed Anthropic client/provider retry multiplication: with the first-event watchdog disabled (`PI_STREAM_FIRST_EVENT_TIMEOUT_MS=0`), the client's internal `maxRetries: 5` reactivated and stacked with the provider loop's 3 retries — up to 24 wire attempts with double backoff. The provider now pins per-request `maxRetries: 0` unconditionally.
2115
- - Fixed `AnthropicMessagesClient` spreading `fetchOptions` after the core request fields, letting a caller-supplied `signal`/`method`/`body` silently disconnect the timeout controller or corrupt the request. Transport extras (TLS) still pass through; core fields now always win.
2116
- - Fixed Foundry mTLS/CA material being cached for the process lifetime when the env vars point at files: the cache key now folds in the file mtime so on-disk certificate rotation takes effect.
2117
- - Fixed the Claude Code fingerprint version drifting across surfaces: the usage endpoint (`claude-cli/2.1.160`) and OAuth bootstrap (`claude-code/2.1.160`) pinned a stale version while `/v1/messages` reported 2.1.165; both now derive from `claudeCodeVersion`.
2118
- - Fixed a system prompt that merely _mentions_ `x-anthropic-billing-header:` mid-text suppressing the entire Claude Code system-block injection (billing header, instruction, and cch attestation); the resumed-session guard now anchors with `startsWith`.
2119
- - Fixed lone surrogates in cross-API tool-call arguments reaching Anthropic's strict UTF-8 validation: replayed OpenAI/Google-origin `tool_use.input` string leaves are now deep-sanitized with `toWellFormed()`, while same-API Anthropic arguments stay byte-identical to keep prompt-cache prefixes stable.
2120
- - Bounded the many-image resize fan-out to 4 concurrent decodes (it previously decoded every oversized image at once, two encode pipelines each — multi-GB transient memory at the 20+-image threshold that activates the feature).
2121
- - Fixed `mergeHeaders` merging case-sensitively on the Copilot/client-options path, where a miscased user-configured header (e.g. `authorization` next to the synthesized `Authorization`) survived as two keys that the `Headers` constructor joins comma-separated on the wire.
2122
- - Hardened the Anthropic stream lifecycle: prologue failures (e.g. a malformed Copilot credential in `buildCopilotDynamicHeaders`) and error-finalization failures now surface as an `error` event instead of an unhandled rejection that left `stream.result()` hanging forever; the spurious "cch billing placeholder not patched" warning no longer fires when the placeholder only appears in user content.
2123
-
2124
- ### Removed
2125
-
2126
- - Removed the dead `iterateUntilAbort` helper (superseded by `iterateWithIdleTimeout`); it leaked the upstream iterator when the consumer abandoned mid-yield and had no production call sites.
2127
-
2128
- Older entries are archived in [packages/ai/CHANGELOG.md@c821261d1018](https://github.com/can1357/oh-my-pi/blob/c821261d10180d60bd96c1b7334227691c9e14f6/packages/ai/CHANGELOG.md).
2051
+ Older entries are archived in [packages/ai/CHANGELOG.md@8a9097246135](https://github.com/can1357/oh-my-pi/blob/8a9097246135bd572ff96fb552121fe1194d2906/packages/ai/CHANGELOG.md).
package/README.md CHANGED
@@ -10,38 +10,38 @@ Unified LLM API with automatic model discovery, provider configuration, token an
10
10
  - [Installation](#installation)
11
11
  - [Quick Start](#quick-start)
12
12
  - [Tools](#tools)
13
- - [Defining Tools](#defining-tools)
14
- - [Handling Tool Calls](#handling-tool-calls)
15
- - [Streaming Tool Calls with Partial JSON](#streaming-tool-calls-with-partial-json)
16
- - [Validating Tool Arguments](#validating-tool-arguments)
17
- - [Complete Event Reference](#complete-event-reference)
13
+ - [Defining Tools](#defining-tools)
14
+ - [Handling Tool Calls](#handling-tool-calls)
15
+ - [Streaming Tool Calls with Partial JSON](#streaming-tool-calls-with-partial-json)
16
+ - [Validating Tool Arguments](#validating-tool-arguments)
17
+ - [Complete Event Reference](#complete-event-reference)
18
18
  - [Image Input](#image-input)
19
19
  - [Thinking/Reasoning](#thinkingreasoning)
20
- - [Unified Interface](#unified-interface-streamsimplecompletesimple)
21
- - [Provider-Specific Options](#provider-specific-options-streamcomplete)
22
- - [Streaming Thinking Content](#streaming-thinking-content)
20
+ - [Unified Interface](#unified-interface-streamsimplecompletesimple)
21
+ - [Provider-Specific Options](#provider-specific-options-streamcomplete)
22
+ - [Streaming Thinking Content](#streaming-thinking-content)
23
23
  - [Stop Reasons](#stop-reasons)
24
24
  - [Error Handling](#error-handling)
25
- - [Aborting Requests](#aborting-requests)
26
- - [Continuing After Abort](#continuing-after-abort)
25
+ - [Aborting Requests](#aborting-requests)
26
+ - [Continuing After Abort](#continuing-after-abort)
27
27
  - [APIs, Models, and Providers](#apis-models-and-providers)
28
- - [Providers and Models](#providers-and-models)
29
- - [Querying Providers and Models](#querying-providers-and-models)
30
- - [Custom Models](#custom-models)
31
- - [OpenAI Compatibility Settings](#openai-compatibility-settings)
32
- - [Type Safety](#type-safety)
28
+ - [Providers and Models](#providers-and-models)
29
+ - [Querying Providers and Models](#querying-providers-and-models)
30
+ - [Custom Models](#custom-models)
31
+ - [OpenAI Compatibility Settings](#openai-compatibility-settings)
32
+ - [Type Safety](#type-safety)
33
33
  - [Cross-Provider Handoffs](#cross-provider-handoffs)
34
34
  - [Context Serialization](#context-serialization)
35
35
  - [Browser Usage](#browser-usage)
36
- - [Environment Variables](#environment-variables-nodejs-only)
37
- - [Checking Environment Variables](#checking-environment-variables)
36
+ - [Environment Variables](#environment-variables-nodejs-only)
37
+ - [Checking Environment Variables](#checking-environment-variables)
38
38
  - [OAuth Providers](#oauth-providers)
39
- - [Vertex AI (ADC)](#vertex-ai-adc)
40
- - [CLI Login](#cli-login)
41
- - [Programmatic OAuth](#programmatic-oauth)
42
- - [Login Flow Example](#login-flow-example)
43
- - [Using OAuth Tokens](#using-oauth-tokens)
44
- - [Provider Notes](#provider-notes)
39
+ - [Vertex AI (ADC)](#vertex-ai-adc)
40
+ - [CLI Login](#cli-login)
41
+ - [Programmatic OAuth](#programmatic-oauth)
42
+ - [Login Flow Example](#login-flow-example)
43
+ - [Using OAuth Tokens](#using-oauth-tokens)
44
+ - [Provider Notes](#provider-notes)
45
45
  - [License](#license)
46
46
 
47
47
  ## Supported Providers
@@ -172,7 +172,7 @@ const finalMessage = await s.result();
172
172
  context.messages.push(finalMessage);
173
173
 
174
174
  // Handle tool calls if any
175
- const toolCalls = finalMessage.content.filter((b) => b.type === "toolCall");
175
+ const toolCalls = finalMessage.content.filter(b => b.type === "toolCall");
176
176
  for (const call of toolCalls) {
177
177
  // Execute the tool
178
178
  const result =
@@ -464,7 +464,7 @@ const response = await completeSimple(
464
464
  },
465
465
  {
466
466
  reasoning: "medium", // 'minimal' | 'low' | 'medium' | 'high' | 'xhigh' (xhigh maps to high on non-OpenAI providers)
467
- }
467
+ },
468
468
  );
469
469
 
470
470
  // Access thinking and text blocks
@@ -583,7 +583,7 @@ const s = stream(
583
583
  },
584
584
  {
585
585
  signal,
586
- }
586
+ },
587
587
  );
588
588
 
589
589
  for await (const event of s) {
@@ -643,7 +643,7 @@ Example:
643
643
  const response = await complete(model, context, {
644
644
  apiKey: "sk-live",
645
645
  headers: { "X-Debug-Trace": "true" },
646
- onPayload: (payload) => {
646
+ onPayload: payload => {
647
647
  console.log("request payload", payload);
648
648
  },
649
649
  });
@@ -918,7 +918,7 @@ const response = await complete(
918
918
  },
919
919
  {
920
920
  apiKey: "your-api-key",
921
- }
921
+ },
922
922
  );
923
923
  ```
924
924
 
@@ -928,40 +928,40 @@ const response = await complete(
928
928
 
929
929
  In Node.js environments, you can set environment variables to avoid passing API keys:
930
930
 
931
- | Provider | Environment Variable(s) |
932
- | -------------- | ---------------------------------------------------------------------------- |
933
- | OpenAI | `OPENAI_API_KEY` |
934
- | Anthropic | `ANTHROPIC_API_KEY` or `ANTHROPIC_OAUTH_TOKEN` (or `ANTHROPIC_FOUNDRY_API_KEY` when `CLAUDE_CODE_USE_FOUNDRY=true`) |
935
- | Google | `GEMINI_API_KEY` |
936
- | Vertex AI | `GOOGLE_CLOUD_PROJECT` (or `GCLOUD_PROJECT`) + `GOOGLE_CLOUD_LOCATION` + ADC |
937
- | Mistral | `MISTRAL_API_KEY` |
938
- | Groq | `GROQ_API_KEY` |
939
- | Cerebras | `CEREBRAS_API_KEY` |
940
- | Together | `TOGETHER_API_KEY` |
941
- | Qianfan | `QIANFAN_API_KEY` |
942
- | Hugging Face | `HUGGINGFACE_HUB_TOKEN` or `HF_TOKEN` |
943
- | Synthetic | `SYNTHETIC_API_KEY` |
944
- | NVIDIA | `NVIDIA_API_KEY` |
945
- | NanoGPT | `NANO_GPT_API_KEY` |
946
- | Novita | `NOVITA_API_KEY` |
947
- | DeepInfra | `DEEPINFRA_API_KEY` |
948
- | Venice | `VENICE_API_KEY` |
949
- | Moonshot | `MOONSHOT_API_KEY` |
950
- | xAI | `XAI_API_KEY` |
951
- | OpenRouter | `OPENROUTER_API_KEY` |
952
- | LiteLLM | `LITELLM_API_KEY` |
953
- | Ollama | `OLLAMA_API_KEY` (optional for local deployments) |
954
- | Ollama Cloud | `OLLAMA_CLOUD_API_KEY` |
955
- | Qwen Portal | `QWEN_OAUTH_TOKEN` or `QWEN_PORTAL_API_KEY` |
956
- | QwenCloud Token Plan | `ALIBABA_TOKEN_PLAN_API_KEY` or `BAILIAN_TOKEN_PLAN_API_KEY` |
957
- | zAI | `ZAI_API_KEY` |
958
- | Umans AI Coding Plan | `UMANS_AI_CODING_PLAN_API_KEY` |
959
- | MiniMax Code | `MINIMAX_CODE_API_KEY` (international) or `MINIMAX_CODE_CN_API_KEY` (China) |
960
- | Xiaomi MiMo | `XIAOMI_API_KEY` |
961
- | ZenMux | `ZENMUX_API_KEY` |
962
- | vLLM | `VLLM_API_KEY` |
963
- | Cloudflare AI Gateway | `CLOUDFLARE_AI_GATEWAY_API_KEY` + `CLOUDFLARE_ACCOUNT_ID` + `CLOUDFLARE_GATEWAY_ID` |
964
- | GitHub Copilot | `COPILOT_GITHUB_TOKEN` or `GH_TOKEN` or `GITHUB_TOKEN` |
931
+ | Provider | Environment Variable(s) |
932
+ | --------------------- | ------------------------------------------------------------------------------------------------------------------- |
933
+ | OpenAI | `OPENAI_API_KEY` |
934
+ | Anthropic | `ANTHROPIC_API_KEY` or `ANTHROPIC_OAUTH_TOKEN` (or `ANTHROPIC_FOUNDRY_API_KEY` when `CLAUDE_CODE_USE_FOUNDRY=true`) |
935
+ | Google | `GEMINI_API_KEY` |
936
+ | Vertex AI | `GOOGLE_CLOUD_PROJECT` (or `GCLOUD_PROJECT`) + `GOOGLE_CLOUD_LOCATION` + ADC |
937
+ | Mistral | `MISTRAL_API_KEY` |
938
+ | Groq | `GROQ_API_KEY` |
939
+ | Cerebras | `CEREBRAS_API_KEY` |
940
+ | Together | `TOGETHER_API_KEY` |
941
+ | Qianfan | `QIANFAN_API_KEY` |
942
+ | Hugging Face | `HUGGINGFACE_HUB_TOKEN` or `HF_TOKEN` |
943
+ | Synthetic | `SYNTHETIC_API_KEY` |
944
+ | NVIDIA | `NVIDIA_API_KEY` |
945
+ | NanoGPT | `NANO_GPT_API_KEY` |
946
+ | Novita | `NOVITA_API_KEY` |
947
+ | DeepInfra | `DEEPINFRA_API_KEY` |
948
+ | Venice | `VENICE_API_KEY` |
949
+ | Moonshot | `MOONSHOT_API_KEY` |
950
+ | xAI | `XAI_API_KEY` |
951
+ | OpenRouter | `OPENROUTER_API_KEY` |
952
+ | LiteLLM | `LITELLM_API_KEY` |
953
+ | Ollama | `OLLAMA_API_KEY` (optional for local deployments) |
954
+ | Ollama Cloud | `OLLAMA_CLOUD_API_KEY` |
955
+ | Qwen Portal | `QWEN_OAUTH_TOKEN` or `QWEN_PORTAL_API_KEY` |
956
+ | QwenCloud Token Plan | `ALIBABA_TOKEN_PLAN_API_KEY` or `BAILIAN_TOKEN_PLAN_API_KEY` |
957
+ | zAI | `ZAI_API_KEY` |
958
+ | Umans AI Coding Plan | `UMANS_AI_CODING_PLAN_API_KEY` |
959
+ | MiniMax Code | `MINIMAX_CODE_API_KEY` (international) or `MINIMAX_CODE_CN_API_KEY` (China) |
960
+ | Xiaomi MiMo | `XIAOMI_API_KEY` |
961
+ | ZenMux | `ZENMUX_API_KEY` |
962
+ | vLLM | `VLLM_API_KEY` |
963
+ | Cloudflare AI Gateway | `CLOUDFLARE_AI_GATEWAY_API_KEY` + `CLOUDFLARE_ACCOUNT_ID` + `CLOUDFLARE_GATEWAY_ID` |
964
+ | GitHub Copilot | `COPILOT_GITHUB_TOKEN` or `GH_TOKEN` or `GITHUB_TOKEN` |
965
965
 
966
966
  `/login cloudflare-ai-gateway` collects and stores the gateway token, account ID, and gateway ID. For environment configuration, set all three Cloudflare values above. OMP derives provider endpoints from the account and gateway IDs.
967
967
 
@@ -997,7 +997,7 @@ Provider endpoint defaults for the current OpenAI-compatible integrations:
997
997
  - LiteLLM: `http://localhost:4000/v1`
998
998
  - Cloudflare AI Gateway: native Anthropic, OpenAI, and Workers AI routes under `https://gateway.ai.cloudflare.com/v1/<account>/<gateway>`
999
999
  - Qwen Portal: `https://portal.qwen.ai/v1`
1000
- When set, the library automatically uses these keys:
1000
+ When set, the library automatically uses these keys:
1001
1001
 
1002
1002
  ```typescript
1003
1003
  // Uses OPENAI_API_KEY from environment
@@ -1118,10 +1118,10 @@ const credentials = await getProviderDefinition("github-copilot")?.login?.({
1118
1118
  console.log(`Open: ${url}`);
1119
1119
  if (instructions) console.log(instructions);
1120
1120
  },
1121
- onPrompt: async (prompt) => {
1121
+ onPrompt: async prompt => {
1122
1122
  return await getUserInput(prompt.message);
1123
1123
  },
1124
- onProgress: (message) => console.log(message),
1124
+ onProgress: message => console.log(message),
1125
1125
  });
1126
1126
 
1127
1127
  // Store credentials yourself
@@ -1155,7 +1155,7 @@ const response = await complete(
1155
1155
  {
1156
1156
  messages: [{ role: "user", content: "Hello!" }],
1157
1157
  },
1158
- { apiKey: result.apiKey }
1158
+ { apiKey: result.apiKey },
1159
1159
  );
1160
1160
  ```
1161
1161
 
@@ -122,8 +122,8 @@ export declare function handleServerMessage(msg: AgentServerMessage, output: Ass
122
122
  * and nullable for the one caller whose block is NOT pre-resolved: MCP without
123
123
  * an `mcp` handler, which `agent-loop.ts` runs locally and pairs itself.
124
124
  */
125
- export declare function resolveExecHandler<TArgs, TResult>(args: TArgs, handler: ((args: TArgs) => Promise<CursorExecHandlerResult<TResult>>) | undefined, onToolResult: CursorToolResultHandler | undefined, buildFromToolResult: (toolResult: ToolResultMessage) => TResult, buildRejected: (reason: string) => TResult, buildError: (error: string) => TResult, pairing: CursorExecPairing | null): Promise<{
126
- execResult: TResult;
125
+ export declare function resolveExecHandler<TArgs, R>(args: TArgs, handler: ((args: TArgs) => Promise<CursorExecHandlerResult<R>>) | undefined, onToolResult: CursorToolResultHandler | undefined, buildFromToolResult: (toolResult: ToolResultMessage) => R, buildRejected: (reason: string) => R, buildError: (error: string) => R, pairing: CursorExecPairing | null): Promise<{
126
+ execResult: R;
127
127
  toolResult?: ToolResultMessage;
128
128
  }>;
129
129
  /**
@@ -1,5 +1,4 @@
1
1
  /** Shared inference request identity headers. */
2
- import type { FetchImpl } from "../types.js";
3
2
  /** Options controlling provider and protocol inference headers. */
4
3
  export interface InferenceHeaderOptions {
5
4
  provider: string;
@@ -14,11 +13,12 @@ export declare function setHeaderIfAbsent(headers: Record<string, string>, name:
14
13
  */
15
14
  export declare function applyInferenceHeaders(headers: Record<string, string>, options: InferenceHeaderOptions): void;
16
15
  /**
17
- * Apply omp's process-wide inference User-Agent default. Any explicit header,
18
- * including Anthropic and Codex OAuth fingerprints, remains authoritative.
16
+ * Return `init` with omp's process-wide inference User-Agent default applied.
17
+ * Any explicit header, including Anthropic and Codex OAuth fingerprints,
18
+ * remains authoritative. Called per request by `transportFetch`.
19
19
  *
20
20
  * Plain-object headers stay plain objects: custom `fetch` implementations
21
21
  * (proxies, tests) index `init.headers` by name and must not be handed a
22
22
  * `Headers` instance instead.
23
23
  */
24
- export declare function withInferenceUserAgent(fetchImpl: FetchImpl): FetchImpl;
24
+ export declare function withInferenceUserAgent(input: string | URL | Request, init: RequestInit | undefined): RequestInit | undefined;
@@ -0,0 +1,64 @@
1
+ /**
2
+ * Mid-conversation reasoning effort via `configuration_update` input items
3
+ * (GPT-6 Astra; `model.compat.supportsConfigurationUpdate`).
4
+ *
5
+ * The request-level `reasoning.effort` is pinned to the value of the session's
6
+ * first request so the cached prompt prefix survives an effort change. Each
7
+ * later change is carried as a `configuration_update` item inserted at the
8
+ * tail of the transcript — before the user message it takes effect on, or
9
+ * after the latest tool result when the level changes inside a tool loop — and
10
+ * replayed at that position on every subsequent request until another update
11
+ * overrides it. Mirrors the Anthropic provider's stable `output_config.effort`
12
+ * planning.
13
+ *
14
+ * Used by both the platform Responses provider and the Codex provider; the
15
+ * state lives in each provider's session state, keyed per conversation.
16
+ *
17
+ * Wire constraints (verified against the Codex backend): only `gpt-6-astra`
18
+ * accepts the item type, consecutive updates are rejected, and
19
+ * `/responses/compact` rejects histories containing them — compaction
20
+ * requests are built outside this planner and never carry the items.
21
+ */
22
+ /** `configuration_update` input item; only `reasoning.effort` is updatable. */
23
+ export interface ConfigurationUpdateItem {
24
+ type: "configuration_update";
25
+ reasoning: {
26
+ effort: string;
27
+ };
28
+ }
29
+ interface EffortTransition<TEffort extends string> {
30
+ /** Input-array position the item is spliced into (before `input[index]`). */
31
+ index: number;
32
+ /** Fingerprint of `input[index - 1]` at record time; a mismatch means the history was rewritten. */
33
+ anchor: string;
34
+ effort: TEffort;
35
+ }
36
+ /** Per-conversation effort baseline and recorded transitions. */
37
+ export interface OpenAIEffortControlState<TEffort extends string = string> {
38
+ baseEffort?: TEffort;
39
+ currentEffort?: TEffort;
40
+ transitions: EffortTransition<TEffort>[];
41
+ }
42
+ export declare function createOpenAIEffortControlState<TEffort extends string>(): OpenAIEffortControlState<TEffort>;
43
+ /**
44
+ * Fetch (or create) the control state for one conversation from a provider's
45
+ * bounded per-session map, refreshing its LRU slot.
46
+ */
47
+ export declare function getOpenAIEffortControlState<TEffort extends string>(states: Map<string, OpenAIEffortControlState<TEffort>>, key: string): OpenAIEffortControlState<TEffort>;
48
+ interface AnchorableItem {
49
+ type?: string | null;
50
+ role?: string;
51
+ id?: string | null;
52
+ status?: string | null;
53
+ }
54
+ /**
55
+ * Pin the request-level effort to the session baseline and splice pending
56
+ * `configuration_update` items into `input` (mutated in place).
57
+ *
58
+ * `input` is the freshly built transcript for this request, without any
59
+ * `configuration_update` items. `requested` is the wire effort the caller
60
+ * would otherwise send at the request level. Returns the effort to send at the
61
+ * request level (`requested` on the first request, the baseline afterwards).
62
+ */
63
+ export declare function planStableOpenAIEffort<TItem extends AnchorableItem, TEffort extends string>(state: OpenAIEffortControlState<TEffort>, input: Array<TItem | ConfigurationUpdateItem>, requested: TEffort): TEffort;
64
+ export {};
@@ -2918,7 +2918,7 @@ export interface ResponseInputImageContent {
2918
2918
  * `assistant` role are presumed to have been generated by the model in previous
2919
2919
  * interactions.
2920
2920
  */
2921
- export type ResponseInputItem = EasyInputMessage | ResponseInputItem.Message | ResponseOutputMessage | ResponseFileSearchToolCall | ResponseComputerToolCall | ResponseInputItem.ComputerCallOutput | ResponseFunctionWebSearch | ResponseFunctionToolCall | ResponseInputItem.FunctionCallOutput | ResponseInputItem.ToolSearchCall | ResponseToolSearchOutputItemParam | ResponseInputItem.AdditionalTools | ResponseReasoningItem | ResponseCompactionItemParam | ResponseInputItem.ImageGenerationCall | ResponseCodeInterpreterToolCall | ResponseInputItem.LocalShellCall | ResponseInputItem.LocalShellCallOutput | ResponseInputItem.ShellCall | ResponseInputItem.ShellCallOutput | ResponseInputItem.ApplyPatchCall | ResponseInputItem.ApplyPatchCallOutput | ResponseInputItem.McpListTools | ResponseInputItem.McpApprovalRequest | ResponseInputItem.McpApprovalResponse | ResponseInputItem.McpCall | ResponseCustomToolCallOutput | ResponseCustomToolCall | ResponseInputItem.CompactionTrigger | ResponseInputItem.ItemReference;
2921
+ export type ResponseInputItem = EasyInputMessage | ResponseInputItem.Message | ResponseOutputMessage | ResponseFileSearchToolCall | ResponseComputerToolCall | ResponseInputItem.ComputerCallOutput | ResponseFunctionWebSearch | ResponseFunctionToolCall | ResponseInputItem.FunctionCallOutput | ResponseInputItem.ToolSearchCall | ResponseToolSearchOutputItemParam | ResponseInputItem.AdditionalTools | ResponseReasoningItem | ResponseCompactionItemParam | ResponseInputItem.ImageGenerationCall | ResponseCodeInterpreterToolCall | ResponseInputItem.LocalShellCall | ResponseInputItem.LocalShellCallOutput | ResponseInputItem.ShellCall | ResponseInputItem.ShellCallOutput | ResponseInputItem.ApplyPatchCall | ResponseInputItem.ApplyPatchCallOutput | ResponseInputItem.McpListTools | ResponseInputItem.McpApprovalRequest | ResponseInputItem.McpApprovalResponse | ResponseInputItem.McpCall | ResponseCustomToolCallOutput | ResponseCustomToolCall | ResponseInputItem.CompactionTrigger | ResponseInputItem.ConfigurationUpdate | ResponseInputItem.ItemReference;
2922
2922
  export declare namespace ResponseInputItem {
2923
2923
  /**
2924
2924
  * A message input to the model with a role indicating instruction following
@@ -3504,6 +3504,20 @@ export declare namespace ResponseInputItem {
3504
3504
  */
3505
3505
  type: "compaction_trigger";
3506
3506
  }
3507
+ /**
3508
+ * Changes reasoning effort for subsequent responses without touching the
3509
+ * request-level `reasoning.effort` (GPT-6 Astra). Must not be adjacent to
3510
+ * another `configuration_update`.
3511
+ */
3512
+ interface ConfigurationUpdate {
3513
+ /**
3514
+ * The type of the item. Always `configuration_update`.
3515
+ */
3516
+ type: "configuration_update";
3517
+ reasoning: {
3518
+ effort: string;
3519
+ };
3520
+ }
3507
3521
  /**
3508
3522
  * An internal identifier for an item to reference.
3509
3523
  */
@@ -1,7 +1,8 @@
1
1
  import type { Context, Model, OpenAICompat, ProviderSessionState, ServiceTier, StreamFunction, StreamOptions, Tool, ToolChoice } from "../types.js";
2
2
  import { type OpenAIResponsesToolChoice } from "../utils/tool-choice.js";
3
+ import { type OpenAIEffortControlState } from "./openai-configuration-update.js";
3
4
  import { type OpenAIReasoningEffortFallbackState } from "./openai-reasoning-fallback.js";
4
- import type { Tool as OpenAITool, ResponseCreateParamsStreaming, ResponseInput } from "./openai-responses-wire.js";
5
+ import type { Tool as OpenAITool, ReasoningEffort, ResponseCreateParamsStreaming, ResponseInput } from "./openai-responses-wire.js";
5
6
  import { type OpenAIPromptCacheOptions, type OpenAIStrictToolsScope, type OpenAIStrictToolsState } from "./openai-shared.js";
6
7
  export interface OpenAIResponsesOptions extends StreamOptions {
7
8
  reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
@@ -63,7 +64,11 @@ interface OpenAIResponsesProviderSessionState extends ProviderSessionState, Open
63
64
  nativeHistoryReplayWarmed: boolean;
64
65
  /** Stateful `previous_response_id` chain baselines, keyed by baseUrl/model/session. */
65
66
  chains: Map<string, OpenAIResponsesChainState>;
67
+ /** `configuration_update` effort baselines, keyed by baseUrl/model/session. */
68
+ effortControls: Map<string, OpenAIEffortControlState<ResponsesStableEffort>>;
66
69
  }
70
+ /** Wire efforts a `configuration_update` can carry: every real tier, never `none`/null. */
71
+ type ResponsesStableEffort = Exclude<ReasoningEffort, "none" | null>;
67
72
  interface OpenAIResponsesChainState {
68
73
  /**
69
74
  * Wire params of the last successful turn; never carries
@@ -493,6 +493,18 @@ export interface BuildResponsesInputOptions<TApi extends Api> {
493
493
  */
494
494
  export declare function escapeReplayedControlTokens(items: ResponseInput): ResponseInput;
495
495
  export declare function buildResponsesInput<TApi extends Api>(options: BuildResponsesInputOptions<TApi>): ResponseInput;
496
+ /**
497
+ * Non-empty `reasoning_text` shipped for a synthesized reasoning item when no
498
+ * thinking text survived history reconstruction. DeepSeek-family Responses
499
+ * targets (e.g. opencode-go) reject BOTH a missing reasoning item and one whose
500
+ * `reasoning_text` is empty — "The reasoning_text in the thinking mode must be
501
+ * passed back to the API" (#8248 covered the missing case, #10690 the empty
502
+ * one). The item's presence plus a non-empty payload is what satisfies the
503
+ * contract; the exact text is immaterial once the source turn's reasoning is
504
+ * gone. Kept out of `reasoning_content="."`-territory since DeepSeek rejects the
505
+ * bare-dot synthetic placeholder on the chat-completions path.
506
+ */
507
+ export declare const SYNTHETIC_REASONING_REPLAY_PLACEHOLDER = "reasoning unavailable";
496
508
  export declare function convertResponsesAssistantMessage<TApi extends Api>(assistantMsg: AssistantMessage, model: Model<TApi>, msgIndex: number, knownCallIds: Set<string>, includeThinkingSignatures?: boolean, customCallIds?: Set<string>, preserveMessageIds?: boolean, supportsCustomToolCalls?: boolean, customToolWireNameMap?: ReadonlyMap<string, string>, computerCallIds?: Set<string>, requiresReasoningReplayForAllTurns?: boolean, requiresReasoningReplayForToolCalls?: boolean): ResponseInput;
497
509
  /**
498
510
  * Responses wire output for a tool result plus its text-only fallback.
@@ -1,6 +1,6 @@
1
1
  /**
2
2
  * `login "oauth-code"` engine: authorization-code grant (optionally PKCE)
3
- * through the loopback callback server, followed by the declared token
3
+ * through the configured callback transport, followed by the declared token
4
4
  * exchange, credential projection, userinfo enrichment and after-exchange hook.
5
5
  */
6
6
  import type { CompiledAuthProvider, CompiledCallback, CompiledOAuthCodeLogin } from "@oh-my-pi/pi-catalog/compat/types";
@@ -25,6 +25,8 @@ export interface OAuthCallbackFlowOptions {
25
25
  allowPortFallback?: boolean;
26
26
  /** Skip the local callback server entirely; the user pastes the code or redirect URL back. */
27
27
  manualInputOnly?: boolean;
28
+ /** Receive a custom-scheme redirect through the native OS handler when supported. */
29
+ nativeScheme?: boolean;
28
30
  }
29
31
  /**
30
32
  * Abstract base class for OAuth flows with local callback servers.
@@ -0,0 +1,13 @@
1
+ /** Native callback lifetime exposed to the provider-independent OAuth flow. */
2
+ export interface NativeSchemeCallbackReceiver {
3
+ /** Restore owned settings; native recovery data remains intact on failure. */
4
+ dispose(): Promise<void>;
5
+ /** Wait for a complete URL while forwarding cancellation to native execution. */
6
+ waitForCallback(signal?: AbortSignal, timeoutMs?: number): Promise<string>;
7
+ }
8
+ /** Cancellation for native callback registration and its active lifetime. */
9
+ export interface NativeSchemeCallbackOptions {
10
+ signal?: AbortSignal;
11
+ }
12
+ /** Connect OAuth's AbortSignals to the native, recoverable desktop callback receiver. */
13
+ export declare function createNativeSchemeCallbackReceiver(scheme: string, options?: NativeSchemeCallbackOptions): Promise<NativeSchemeCallbackReceiver | undefined>;
@@ -66,7 +66,8 @@ export interface OAuthProviderInfo {
66
66
  export interface OAuthController {
67
67
  onAuth?(info: OAuthAuthInfo): void;
68
68
  onProgress?(message: string): void;
69
- onManualCodeInput?(): Promise<string>;
69
+ /** Request pasted callback input; stop any visible prompt when `signal` aborts. */
70
+ onManualCodeInput?(signal?: AbortSignal): Promise<string>;
70
71
  onPrompt?(prompt: OAuthPrompt): Promise<string>;
71
72
  signal?: AbortSignal;
72
73
  fetch?: FetchImpl;