@oh-my-pi/pi-ai 18.1.16 → 18.1.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,15 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [18.1.17] - 2026-09-10
6
+
7
+ ### Fixed
8
+
9
+ - Fixed transient Python HTTP/2 stream resets and HTTP/1.1 chunked response interruptions being treated as terminal errors when forwarded by a proxy ([#11160](https://github.com/can1357/oh-my-pi/pull/11160) by [@cyriusweng](https://github.com/cyriusweng)).
10
+ - Ollama cache hits now populate cached-token usage: `prompt_eval_cached_count` from the `/api/chat` done chunk maps to `cacheRead`, with `input` reduced to the uncached portion, so status-line `cache_turn`/`cache_hit` segments and cache-prefix audits report real hit rates instead of false misses.
11
+ - Fixed requests that run across a price change being costed at the newer rate; peak/off-peak estimates now use the rate in effect when the request started.
12
+ - Fixed GitHub Copilot Business seats getting HTTP 403 on every model while the same token succeeds with a Chat client identity: chat and model-policy requests now identify as `copilot-chat`, denied requests retry once as the Copilot CLI (`copilot-developer-cli`), and `COPILOT_INTEGRATION_ID` pins the `Copilot-Integration-Id` header up front; model discovery keeps the CLI identity and the 403 message names the identity and the remedies ([#11372](https://github.com/can1357/oh-my-pi/issues/11372)).
13
+
5
14
  ## [18.1.16] - 2026-09-09
6
15
 
7
16
  ### Fixed
@@ -28,6 +28,10 @@ export declare const Flag: {
28
28
  };
29
29
  export type Flag = (typeof Flag)[keyof typeof Flag];
30
30
  export declare const STREAM_READ_ERROR_PATTERN: RegExp;
31
+ /** Python h2/httpx diagnostics forwarded through provider or proxy error events. */
32
+ export declare const PYTHON_HTTP2_STREAM_RESET_PATTERN: RegExp;
33
+ /** Python h11/httpx EOF while reading an HTTP/1.1 chunked response body. */
34
+ export declare const PYTHON_HTTP_INCOMPLETE_CHUNK_PATTERN: RegExp;
31
35
  export declare const TRANSIENT_TRANSPORT_PATTERN: RegExp;
32
36
  /**
33
37
  * Local llama.cpp / Ollama deterministic tool-call argument JSON parse failure.
@@ -1,4 +1,4 @@
1
- import type { Message } from "../types.js";
1
+ import type { FetchImpl, Message } from "../types.js";
2
2
  /**
3
3
  * Infer whether the current request to Copilot is user-initiated or agent-initiated.
4
4
  * Accepts `unknown[]` because providers may pass pre-converted message shapes.
@@ -11,6 +11,32 @@ export type CopilotDynamicHeaders = {
11
11
  premiumRequests: CopilotPremiumRequests;
12
12
  };
13
13
  export declare function resolveGitHubCopilotBaseUrl(baseUrl: string | undefined, apiKey: string | undefined): string | undefined;
14
+ /**
15
+ * Opt-in `Copilot-Integration-Id` override for chat and model-policy requests.
16
+ * Reads `COPILOT_INTEGRATION_ID`; unset/invalid keeps the chat-surface default
17
+ * (`COPILOT_CHAT_INTEGRATION_ID`). Model discovery keeps the CLI identity: it
18
+ * unlocks enterprise/experimental models and listing models is not
19
+ * policy-gated the way chat completions are (#11372).
20
+ */
21
+ export declare function resolveCopilotIntegrationIdOverride(env?: Record<string, string | undefined>): string | undefined;
22
+ /**
23
+ * Effective identity before the chat-surface default: explicit value, then
24
+ * request headers, then `COPILOT_INTEGRATION_ID`. Pure given its inputs, so
25
+ * tests inject literals instead of mutating process state.
26
+ */
27
+ export declare function resolveCopilotRequestIdentity(headers?: Record<string, string>, explicit?: unknown, env?: Record<string, string | undefined>): string | undefined;
28
+ /**
29
+ * Reissue chat-surface Copilot 403s once as the Copilot CLI.
30
+ *
31
+ * Chat is the default surface (`COPILOT_CHAT_INTEGRATION_ID`) because Business
32
+ * organizations that gate premium models per client surface commonly allow
33
+ * chat while blocking CLI/agentic clients (issue #11372). The retry fires
34
+ * only for requests carrying the chat default and only when the caller
35
+ * resolved no explicit identity — an explicit choice is never second-guessed.
36
+ * The denied body is drained before reissuing, and the retry carries the CLI
37
+ * identity so the guard passes it through: at most two requests, never a loop.
38
+ */
39
+ export declare function wrapFetchForCopilotFallback(base: FetchImpl | undefined, enabled: boolean, integrationId?: unknown): FetchImpl;
14
40
  export declare function inferCopilotInitiator(messages: unknown[]): CopilotInitiator;
15
41
  /** Check whether any message in the conversation contains image content. */
16
42
  export declare function hasCopilotVisionInput(messages: Message[]): boolean;
@@ -37,4 +63,6 @@ export declare function buildCopilotDynamicHeaders(params: {
37
63
  headers?: Record<string, string>;
38
64
  initiatorOverride?: CopilotInitiator;
39
65
  planTier?: string;
66
+ /** Raw explicit identity; validated here, chat default when absent/invalid. */
67
+ integrationId?: unknown;
40
68
  }): CopilotDynamicHeaders;
@@ -42,5 +42,5 @@ export interface OpenAICompletionsOptions extends StreamOptions {
42
42
  * assistant output commits the attempt.
43
43
  */
44
44
  export declare const streamOpenAICompletions: StreamFunction<"openai-completions">;
45
- export declare function parseChunkUsage(rawUsage: object, model: Model<"openai-completions">, premiumRequests: number | undefined): AssistantMessage["usage"];
45
+ export declare function parseChunkUsage(rawUsage: object, model: Model<"openai-completions">, premiumRequests: number | undefined, timestamp?: number): AssistantMessage["usage"];
46
46
  export declare function convertMessages(model: Model<"openai-completions">, context: Context, compat: ResolvedOpenAICompat): ChatCompletionMessageParam[];
@@ -1,4 +1,4 @@
1
- import type { ImageContent, Model, TextContent } from "../types.js";
1
+ import type { Api, ImageContent, Model, TextContent } from "../types.js";
2
2
  export declare const NON_VISION_IMAGE_PLACEHOLDER = "[image omitted: model does not support vision]";
3
3
  export declare function partitionVisionContent(content: ReadonlyArray<TextContent | ImageContent>, supportsImages: boolean): {
4
4
  textBlocks: TextContent[];
@@ -12,4 +12,18 @@ export declare function joinTextWithImagePlaceholder(text: string, omittedImages
12
12
  * misconfigured provider descriptors or user model entries (e.g. text-only
13
13
  * DashScope Qwen SKUs, DeepSeek models) whose endpoints reject `image_url`.
14
14
  */
15
- export declare function isOpenAICompletionsVisionSupported(model: Model<"openai-completions">): boolean;
15
+ export declare function isOpenAICompletionsVisionSupported(model: Model<"openai-completions" | "openrouter">): boolean;
16
+ /**
17
+ * Whether the transport that will carry `model` sends image content on the wire.
18
+ *
19
+ * The `pi-native` transport forwards the original context (images included) to
20
+ * the gateway, which resolves its own model server-side, so the Chat
21
+ * Completions guard below never runs client-side and the declared input
22
+ * applies. Otherwise the OpenAI Chat Completions path applies the text-only
23
+ * guard, as does the OpenRouter chat fallback (`PI_OPENROUTER_RESPONSES=0`,
24
+ * which dispatches `openrouter` models through `streamOpenAICompletions`);
25
+ * every other API ships the modalities the model declares. Callers that report
26
+ * or gate on the wire (for example the `omp models` table) read this
27
+ * predicate; declared capability reads `model.input`.
28
+ */
29
+ export declare function sendsImageInputOnWire(model: Model<Api>): boolean;
@@ -8,6 +8,7 @@ type GitHubCopilotLoginOptions = {
8
8
  allowEmpty?: boolean;
9
9
  }) => Promise<string>;
10
10
  onProgress?: (message: string) => void;
11
+ copilotIntegrationId?: unknown;
11
12
  signal?: AbortSignal;
12
13
  pollIntervalFloorMs?: number;
13
14
  pollIntervalScaleMs?: number;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@oh-my-pi/pi-ai",
3
- "version": "18.1.16",
3
+ "version": "18.1.17",
4
4
  "description": "Unified LLM API with automatic model discovery and provider configuration",
5
5
  "keywords": [
6
6
  "ai",
@@ -124,11 +124,11 @@
124
124
  "fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
125
125
  },
126
126
  "dependencies": {
127
- "@oh-my-pi/omptype": "18.1.16",
128
- "@oh-my-pi/pi-catalog": "18.1.16",
129
- "@oh-my-pi/pi-natives": "18.1.16",
130
- "@oh-my-pi/pi-utils": "18.1.16",
131
- "@oh-my-pi/pi-wire": "18.1.16"
127
+ "@oh-my-pi/omptype": "18.1.17",
128
+ "@oh-my-pi/pi-catalog": "18.1.17",
129
+ "@oh-my-pi/pi-natives": "18.1.17",
130
+ "@oh-my-pi/pi-utils": "18.1.17",
131
+ "@oh-my-pi/pi-wire": "18.1.17"
132
132
  },
133
133
  "devDependencies": {
134
134
  "@types/bun": "^1.3.14"
@@ -156,6 +156,11 @@ const TIMEOUT_PATTERN = /\b(?:operation\s+)?timed?\s*out\b|\btimeout\b|\bstream
156
156
  const TRANSIENT_ENVELOPE_PATTERN = /anthropic stream envelope error:/i;
157
157
  const TRANSIENT_ENVELOPE_TRUNCATION_PATTERN = /before message_(?:start|stop)/i;
158
158
  export const STREAM_READ_ERROR_PATTERN = /stream[_ -]?read[_ -]?error/i;
159
+ /** Python h2/httpx diagnostics forwarded through provider or proxy error events. */
160
+ export const PYTHON_HTTP2_STREAM_RESET_PATTERN = /<StreamReset stream_id:\d+, error_code:(?:2|7), remote_reset:True>/;
161
+ /** Python h11/httpx EOF while reading an HTTP/1.1 chunked response body. */
162
+ export const PYTHON_HTTP_INCOMPLETE_CHUNK_PATTERN =
163
+ /peer closed connection without sending complete message body \(incomplete chunked read\)/;
159
164
  export const TRANSIENT_TRANSPORT_PATTERN =
160
165
  /\b(?:no[_ -]?capacity|(?:high|peak)[ _-]?demand|(?:at|over|insufficient)[ _-]?capacity|capacity[ _-]?(?:exceeded|exhausted)|peak[ _-]?load)\b|overloaded|provider.?returned.?error|rate.?limit|too many requests|auth-gateway\s+5\d{2}(?=[:\s]|$)|\b(?:429|500|502|503|504)\b|service.?unavailable|server.?error|internal.?error|retry your request|network.?error|connection.?error|connection.?refused|unable.?to.?connect\.\s*is the computer able to access the url\?|other side closed|fetch failed|upstream.?connect|upstream.?request.?failed|reset before headers|socket hang up|timed? out|timeout|terminated|retry delay|stream stall|no error details in response|HTTP2(?:StreamReset|RefusedStream|EnhanceYourCalm)|nghttp2_(?:internal_error|refused_stream)|stream closed with error code nghttp2_(?:internal_error|refused_stream)|malformed.?function.?call/i;
161
166
  const AUTH_FAILURE_PATTERN =
@@ -396,6 +401,8 @@ function isTransientErrorText(text: string): boolean {
396
401
  return (
397
402
  isUnexpectedSocketCloseMessage(text) ||
398
403
  isStreamReadErrorText(text) ||
404
+ PYTHON_HTTP2_STREAM_RESET_PATTERN.test(text) ||
405
+ PYTHON_HTTP_INCOMPLETE_CHUNK_PATTERN.test(text) ||
399
406
  (TRANSIENT_ENVELOPE_PATTERN.test(text) && TRANSIENT_ENVELOPE_TRUNCATION_PATTERN.test(text)) ||
400
407
  TRANSIENT_TRANSPORT_PATTERN.test(text)
401
408
  );
@@ -836,7 +836,7 @@ function handleMetadata(event: MetadataEvent, model: Model<"bedrock-converse-str
836
836
  output.usage.cacheRead = event.usage.cacheReadInputTokens || 0;
837
837
  output.usage.cacheWrite = event.usage.cacheWriteInputTokens || 0;
838
838
  output.usage.totalTokens = event.usage.totalTokens || output.usage.input + output.usage.output;
839
- calculateCost(model, output.usage);
839
+ calculateCost(model, output.usage, output.timestamp);
840
840
  }
841
841
  }
842
842
 
@@ -107,7 +107,9 @@ import {
107
107
  import {
108
108
  buildCopilotDynamicHeaders,
109
109
  hasCopilotVisionInput,
110
+ resolveCopilotRequestIdentity,
110
111
  resolveGitHubCopilotBaseUrl,
112
+ wrapFetchForCopilotFallback,
111
113
  } from "./github-copilot-headers";
112
114
  import { getOpenAIPromptCacheKey } from "./openai-shared";
113
115
  import { applyInferenceHeaders } from "./inference-headers";
@@ -1805,6 +1807,7 @@ function calculateFallbackTurnCost(
1805
1807
  requestModel: Model<"anthropic-messages">,
1806
1808
  usage: Usage,
1807
1809
  source: AnthropicWireUsage,
1810
+ timestamp: number,
1808
1811
  ): boolean {
1809
1812
  const iterations = source.iterations ?? [];
1810
1813
  if (iterations.length === 0) return false;
@@ -1830,7 +1833,7 @@ function calculateFallbackTurnCost(
1830
1833
  iterationUsage.cacheWrite = cacheWriteTokens;
1831
1834
  iterationUsage.totalTokens =
1832
1835
  iterationUsage.input + iterationUsage.output + iterationUsage.cacheRead + iterationUsage.cacheWrite;
1833
- calculateCost(resolveIterationModel(requestModel, iteration.model), iterationUsage);
1836
+ calculateCost(resolveIterationModel(requestModel, iteration.model), iterationUsage, timestamp);
1834
1837
  cost.input += iterationUsage.cost.input;
1835
1838
  cost.output += iterationUsage.cost.output;
1836
1839
  cost.cacheRead += iterationUsage.cost.cacheRead;
@@ -2025,6 +2028,7 @@ const streamAnthropicOnce = (
2025
2028
  hasImages: hasCopilotVisionInput(context.messages),
2026
2029
  premiumMultiplier: model.premiumMultiplier,
2027
2030
  headers: { ...model.headers, ...options?.headers },
2031
+ integrationId: resolveCopilotRequestIdentity(options?.headers),
2028
2032
  initiatorOverride: options?.initiatorOverride,
2029
2033
  })
2030
2034
  : undefined;
@@ -2274,7 +2278,7 @@ const streamAnthropicOnce = (
2274
2278
  applyAnthropicUsageExtras(output.usage, wireUsage);
2275
2279
  output.usage.totalTokens =
2276
2280
  output.usage.input + output.usage.output + output.usage.cacheRead + output.usage.cacheWrite;
2277
- calculateCost(model, output.usage);
2281
+ calculateCost(model, output.usage, output.timestamp);
2278
2282
  output.duration = performance.now() - startTime;
2279
2283
  stream.push({ type: "start", partial: output });
2280
2284
  stream.push({ type: "done", reason: "stop", message: output });
@@ -2528,11 +2532,11 @@ const streamAnthropicOnce = (
2528
2532
  if (serverSideFallback) {
2529
2533
  const served = fallbackServedModelFromUsage(startUsage);
2530
2534
  if (served) output.model = served;
2531
- if (!calculateFallbackTurnCost(model, output.usage, startUsage)) {
2532
- calculateCost(model, output.usage);
2535
+ if (!calculateFallbackTurnCost(model, output.usage, startUsage, output.timestamp)) {
2536
+ calculateCost(model, output.usage, output.timestamp);
2533
2537
  }
2534
2538
  } else {
2535
- calculateCost(model, output.usage);
2539
+ calculateCost(model, output.usage, output.timestamp);
2536
2540
  }
2537
2541
  } else {
2538
2542
  reportAnthropicEnvelopeAnomaly("message_start missing usage");
@@ -2860,11 +2864,11 @@ const streamAnthropicOnce = (
2860
2864
  if (serverSideFallback) {
2861
2865
  const served = fallbackServedModelFromUsage(deltaUsage);
2862
2866
  if (served) output.model = served;
2863
- if (!calculateFallbackTurnCost(model, output.usage, deltaUsage)) {
2864
- calculateCost(model, output.usage);
2867
+ if (!calculateFallbackTurnCost(model, output.usage, deltaUsage, output.timestamp)) {
2868
+ calculateCost(model, output.usage, output.timestamp);
2865
2869
  }
2866
2870
  } else {
2867
- calculateCost(model, output.usage);
2871
+ calculateCost(model, output.usage, output.timestamp);
2868
2872
  }
2869
2873
  }
2870
2874
  } else if (event.type === "message_stop") {
@@ -3281,7 +3285,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
3281
3285
  maxRetries: 5,
3282
3286
  maxRetryDelayMs,
3283
3287
  defaultHeaders,
3284
- fetch: cchFetch,
3288
+ fetch: wrapFetchForCopilotFallback(cchFetch, true, resolveCopilotRequestIdentity(headers)),
3285
3289
  fetchOptions,
3286
3290
  };
3287
3291
  }
@@ -957,7 +957,7 @@ function streamCursorWithWireMode(
957
957
  endCurrentThinkingBlock(output, stream, state);
958
958
  flushOpenToolCalls(output, stream, state);
959
959
 
960
- calculateCost(model, output.usage);
960
+ calculateCost(model, output.usage, output.timestamp);
961
961
 
962
962
  output.duration = performance.now() - startTime;
963
963
  if (firstTokenTime) output.ttft = firstTokenTime - startTime;
@@ -450,7 +450,7 @@ export const streamDevin: StreamFunction<"devin-agent"> = (
450
450
  toolBlocks.size > 0 ? "toolUse" : latestStopReason === StopReason.MAX_TOKENS ? "length" : "stop";
451
451
  output.stopReason = doneReason;
452
452
 
453
- calculateCost(model, output.usage);
453
+ calculateCost(model, output.usage, output.timestamp);
454
454
  output.duration = performance.now() - startTime;
455
455
  if (firstTokenTime) output.ttft = firstTokenTime - startTime;
456
456
 
@@ -1,9 +1,12 @@
1
1
  import {
2
2
  COPILOT_CAPI_IDENTITY_HEADERS,
3
+ COPILOT_CHAT_INTEGRATION_ID,
3
4
  getGitHubCopilotBaseUrl,
5
+ normalizeCopilotIntegrationId,
4
6
  parseGitHubCopilotApiKey,
5
7
  } from "@oh-my-pi/pi-catalog/wire/github-copilot";
6
- import type { Message } from "../types";
8
+ import { $env, logger } from "@oh-my-pi/pi-utils";
9
+ import type { FetchImpl, Message } from "../types";
7
10
  /**
8
11
  * Infer whether the current request to Copilot is user-initiated or agent-initiated.
9
12
  * Accepts `unknown[]` because providers may pass pre-converted message shapes.
@@ -26,6 +29,86 @@ export function resolveGitHubCopilotBaseUrl(
26
29
  if (baseUrl && !baseUrl.includes("githubcopilot.com")) return baseUrl;
27
30
  return getGitHubCopilotBaseUrl(enterpriseUrl);
28
31
  }
32
+
33
+ /**
34
+ * Opt-in `Copilot-Integration-Id` override for chat and model-policy requests.
35
+ * Reads `COPILOT_INTEGRATION_ID`; unset/invalid keeps the chat-surface default
36
+ * (`COPILOT_CHAT_INTEGRATION_ID`). Model discovery keeps the CLI identity: it
37
+ * unlocks enterprise/experimental models and listing models is not
38
+ * policy-gated the way chat completions are (#11372).
39
+ */
40
+ export function resolveCopilotIntegrationIdOverride(
41
+ env: Record<string, string | undefined> = $env,
42
+ ): string | undefined {
43
+ return normalizeCopilotIntegrationId(env.COPILOT_INTEGRATION_ID);
44
+ }
45
+
46
+ /**
47
+ * Explicit caller-supplied `Copilot-Integration-Id`, matched case-insensitively.
48
+ * Takes only caller layers (`extraHeaders` / `options.headers`) — never model
49
+ * catalog headers — so a catalog default can never masquerade as a choice.
50
+ */
51
+ function explicitCopilotIntegrationId(headers: Record<string, string> | undefined): unknown {
52
+ if (!headers) return undefined;
53
+ for (const name of Object.keys(headers)) {
54
+ if (name.toLowerCase() === "copilot-integration-id") return headers[name];
55
+ }
56
+ return undefined;
57
+ }
58
+
59
+ /**
60
+ * Effective identity before the chat-surface default: explicit value, then
61
+ * request headers, then `COPILOT_INTEGRATION_ID`. Pure given its inputs, so
62
+ * tests inject literals instead of mutating process state.
63
+ */
64
+ export function resolveCopilotRequestIdentity(
65
+ headers?: Record<string, string>,
66
+ explicit?: unknown,
67
+ env: Record<string, string | undefined> = $env,
68
+ ): string | undefined {
69
+ return (
70
+ normalizeCopilotIntegrationId(explicit) ??
71
+ normalizeCopilotIntegrationId(explicitCopilotIntegrationId(headers)) ??
72
+ resolveCopilotIntegrationIdOverride(env)
73
+ );
74
+ }
75
+
76
+ /**
77
+ * Reissue chat-surface Copilot 403s once as the Copilot CLI.
78
+ *
79
+ * Chat is the default surface (`COPILOT_CHAT_INTEGRATION_ID`) because Business
80
+ * organizations that gate premium models per client surface commonly allow
81
+ * chat while blocking CLI/agentic clients (issue #11372). The retry fires
82
+ * only for requests carrying the chat default and only when the caller
83
+ * resolved no explicit identity — an explicit choice is never second-guessed.
84
+ * The denied body is drained before reissuing, and the retry carries the CLI
85
+ * identity so the guard passes it through: at most two requests, never a loop.
86
+ */
87
+ export function wrapFetchForCopilotFallback(
88
+ base: FetchImpl | undefined,
89
+ enabled: boolean,
90
+ integrationId?: unknown,
91
+ ): FetchImpl {
92
+ const inner = base ?? fetch;
93
+ if (!enabled) return inner;
94
+ return async (input, init) => {
95
+ const response = await inner(input, init);
96
+ if (response.status !== 403) return response;
97
+ if (input instanceof Request) return response;
98
+ if (normalizeCopilotIntegrationId(integrationId) !== undefined) return response;
99
+ const outgoing = new Headers(init?.headers);
100
+ if (outgoing.get("Copilot-Integration-Id") !== COPILOT_CHAT_INTEGRATION_ID) {
101
+ return response;
102
+ }
103
+ try {
104
+ await response.arrayBuffer();
105
+ } catch {}
106
+ logger.warn("GitHub Copilot chat identity denied (HTTP 403); retrying once as the Copilot CLI");
107
+ const retryHeaders = new Headers(outgoing);
108
+ retryHeaders.set("Copilot-Integration-Id", COPILOT_CAPI_IDENTITY_HEADERS["Copilot-Integration-Id"]);
109
+ return inner(input, { ...init, headers: retryHeaders });
110
+ };
111
+ }
29
112
  export function inferCopilotInitiator(messages: unknown[]): CopilotInitiator {
30
113
  if (messages.length === 0) return "user";
31
114
 
@@ -121,6 +204,8 @@ export function buildCopilotDynamicHeaders(params: {
121
204
  headers?: Record<string, string>;
122
205
  initiatorOverride?: CopilotInitiator;
123
206
  planTier?: string;
207
+ /** Raw explicit identity; validated here, chat default when absent/invalid. */
208
+ integrationId?: unknown;
124
209
  }): CopilotDynamicHeaders {
125
210
  const initiator =
126
211
  params.initiatorOverride ?? getCopilotInitiatorOverride(params.headers) ?? inferCopilotInitiator(params.messages);
@@ -129,6 +214,8 @@ export function buildCopilotDynamicHeaders(params: {
129
214
  "X-Initiator": initiator,
130
215
  "X-Interaction-Type": `conversation-${initiator}`,
131
216
  };
217
+ headers["Copilot-Integration-Id"] =
218
+ normalizeCopilotIntegrationId(params.integrationId) ?? COPILOT_CHAT_INTEGRATION_ID;
132
219
 
133
220
  if (params.hasImages) {
134
221
  headers["Copilot-Vision-Request"] = "true";
@@ -894,7 +894,7 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
894
894
  total: 0,
895
895
  },
896
896
  };
897
- calculateCost(model, output.usage);
897
+ calculateCost(model, output.usage, output.timestamp);
898
898
  }
899
899
  }
900
900
 
@@ -755,7 +755,7 @@ export async function consumeGoogleStream<T extends GoogleApiType>(args: {
755
755
  total: 0,
756
756
  },
757
757
  };
758
- calculateCost(model, output.usage);
758
+ calculateCost(model, output.usage, output.timestamp);
759
759
  }
760
760
  }
761
761
 
@@ -82,6 +82,7 @@ type OllamaChatChunk = {
82
82
  done?: boolean;
83
83
  done_reason?: string;
84
84
  prompt_eval_count?: number;
85
+ prompt_eval_cached_count?: number;
85
86
  eval_count?: number;
86
87
  };
87
88
 
@@ -698,9 +699,17 @@ const streamOllamaOnce = (
698
699
  if (healedToolCallEmitted && output.stopReason === "stop") {
699
700
  output.stopReason = "toolUse";
700
701
  }
701
- output.usage.input = chunk.prompt_eval_count ?? 0;
702
+ // Ollama reports prompt cache splits as prompt_eval_cached_count
703
+ // (cached) vs prompt_eval_count (total; cached + uncached =
704
+ // total). Map cached → cacheRead, uncached → input so
705
+ // cache_turn/cache_hit status segments and the cache-prefix
706
+ // audit see real hit rates. Local Ollama omits the cached
707
+ // field; the ?? 0 fallbacks keep local behavior
708
+ // byte-identical to before.
709
+ output.usage.cacheRead = chunk.prompt_eval_cached_count ?? 0;
710
+ output.usage.input = (chunk.prompt_eval_count ?? 0) - output.usage.cacheRead;
702
711
  output.usage.output = chunk.eval_count ?? 0;
703
- output.usage.totalTokens = output.usage.input + output.usage.output;
712
+ output.usage.totalTokens = output.usage.input + output.usage.output + output.usage.cacheRead;
704
713
  }
705
714
  }
706
715
  if (streamMarkupHealing) {
@@ -2572,7 +2572,7 @@ class CodexStreamProcessor {
2572
2572
  hasExecutableIncompleteResponsesToolCalls(output);
2573
2573
  finalizePendingResponsesToolCalls(output);
2574
2574
 
2575
- calculateCost(model, output.usage);
2575
+ calculateCost(model, output.usage, output.timestamp);
2576
2576
  applyCodexServiceTierPricing(model, output.usage, serviceTier, runtime.requestBodyForState.service_tier);
2577
2577
  output.stopReason = mapOpenAIResponsesStopReason(status);
2578
2578
  promoteResponsesToolUseStopReason(
@@ -4837,7 +4837,12 @@ export function isRetryableCodexFailureEvent(rawEvent: Record<string, unknown>):
4837
4837
  return true;
4838
4838
  }
4839
4839
  const message = error?.message ?? event.message ?? event.response?.message;
4840
- return !!message && CODEX_RETRYABLE_EVENT_MESSAGE.test(message);
4840
+ return (
4841
+ !!message &&
4842
+ (CODEX_RETRYABLE_EVENT_MESSAGE.test(message) ||
4843
+ AIError.PYTHON_HTTP2_STREAM_RESET_PATTERN.test(message) ||
4844
+ AIError.PYTHON_HTTP_INCOMPLETE_CHUNK_PATTERN.test(message))
4845
+ );
4841
4846
  }
4842
4847
 
4843
4848
  export function createCodexProviderStreamError(rawEvent: Record<string, unknown>): CodexProviderStreamError {
@@ -78,6 +78,7 @@ import {
78
78
  rememberOpenAIReasoningEffortFallback,
79
79
  resolveOpenAIReasoningEffortFallback,
80
80
  } from "./openai-reasoning-fallback";
81
+ import { resolveCopilotRequestIdentity, wrapFetchForCopilotFallback } from "./github-copilot-headers";
81
82
  import {
82
83
  applyChatCompletionsReasoningParams,
83
84
  applyChatCompletionsToolStream,
@@ -796,7 +797,11 @@ const streamOpenAICompletionsOnce = (
796
797
  headers: headersWithTimeout,
797
798
  body: params,
798
799
  signal: requestSignal,
799
- fetch: options?.fetch,
800
+ fetch: wrapFetchForCopilotFallback(
801
+ options?.fetch,
802
+ model.provider === "github-copilot",
803
+ resolveCopilotRequestIdentity(options?.headers),
804
+ ),
800
805
  // Transient 408/429/5xx get Retry-After-aware transport retries.
801
806
  // The first-event watchdog above aborts `requestSignal`, which
802
807
  // bounds every attempt and backoff sleep — retries cannot
@@ -1122,7 +1127,7 @@ const streamOpenAICompletionsOnce = (
1122
1127
  let sawUsagePayload = false;
1123
1128
  let awaitTrailingUsageDetails = false;
1124
1129
  const applyUsagePayload = (rawUsage: object): void => {
1125
- output.usage = parseChunkUsage(rawUsage, model, premiumRequestsTotal);
1130
+ output.usage = parseChunkUsage(rawUsage, model, premiumRequestsTotal, output.timestamp);
1126
1131
  sawUsagePayload = true;
1127
1132
  awaitTrailingUsageDetails = !hasPositiveCacheReadTokenField(rawUsage);
1128
1133
  };
@@ -1854,6 +1859,7 @@ export function parseChunkUsage(
1854
1859
  rawUsage: object,
1855
1860
  model: Model<"openai-completions">,
1856
1861
  premiumRequests: number | undefined,
1862
+ timestamp?: number,
1857
1863
  ): AssistantMessage["usage"] {
1858
1864
  const usageLike = rawUsage as OpenAICompletionsUsageLike;
1859
1865
  const rawPromptTokenDetails = usageLike.prompt_tokens_details;
@@ -1895,7 +1901,7 @@ export function parseChunkUsage(
1895
1901
  cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
1896
1902
  ...(premiumRequests !== undefined ? { premiumRequests } : {}),
1897
1903
  };
1898
- calculateCost(model, usage);
1904
+ calculateCost(model, usage, timestamp);
1899
1905
  applyProviderReportedCost(model, usage, rawUsage);
1900
1906
  return usage;
1901
1907
  }
@@ -64,6 +64,7 @@ import {
64
64
  rememberOpenAIReasoningEffortFallback,
65
65
  resolveOpenAIReasoningEffortFallback,
66
66
  } from "./openai-reasoning-fallback";
67
+ import { resolveCopilotRequestIdentity, wrapFetchForCopilotFallback } from "./github-copilot-headers";
67
68
  import type {
68
69
  Tool as OpenAITool,
69
70
  ReasoningEffort,
@@ -567,7 +568,11 @@ const streamOpenAIResponsesOnce = (
567
568
  headers: headersWithTimeout,
568
569
  body: requestParams,
569
570
  signal: requestSignal,
570
- fetch: options?.fetch,
571
+ fetch: wrapFetchForCopilotFallback(
572
+ options?.fetch,
573
+ model.provider === "github-copilot",
574
+ resolveCopilotRequestIdentity(options?.headers),
575
+ ),
571
576
  // Transient 408/429/5xx get Retry-After-aware transport
572
577
  // retries; the first-event watchdog aborts `requestSignal`,
573
578
  // so retries cannot extend the caller's deadline.
@@ -59,6 +59,7 @@ import {
59
59
  type ToolResultMessage,
60
60
  type Usage,
61
61
  } from "../types";
62
+ import { resolveCopilotRequestIdentity } from "./github-copilot-headers";
62
63
 
63
64
  export type { OpenAIPromptCacheOptions } from "../types";
64
65
 
@@ -259,6 +260,7 @@ export function resolveOpenAIRequestSetup(
259
260
  premiumMultiplier: model.premiumMultiplier,
260
261
  headers,
261
262
  initiatorOverride: options.initiatorOverride,
263
+ integrationId: resolveCopilotRequestIdentity(options.extraHeaders),
262
264
  });
263
265
  Object.assign(headers, copilot.headers);
264
266
  copilotPremiumRequests = copilot.premiumRequests;
@@ -3350,7 +3352,7 @@ export async function processResponsesStream<TApi extends Api>(
3350
3352
  output.responseId = response.id;
3351
3353
  }
3352
3354
  populateResponsesUsageFromResponse(output, response?.usage);
3353
- calculateCost(model, output.usage);
3355
+ calculateCost(model, output.usage, output.timestamp);
3354
3356
  applyProviderReportedCost(model, output.usage, response?.usage);
3355
3357
  applyOpenAIResponsesServiceTierCost(
3356
3358
  model,
@@ -1,4 +1,5 @@
1
- import type { ImageContent, Model, TextContent } from "../types";
1
+ import { $env } from "@oh-my-pi/pi-utils";
2
+ import type { Api, ImageContent, Model, TextContent } from "../types";
2
3
 
3
4
  export const NON_VISION_IMAGE_PLACEHOLDER = "[image omitted: model does not support vision]";
4
5
  export function partitionVisionContent(
@@ -35,8 +36,33 @@ export function joinTextWithImagePlaceholder(text: string, omittedImages: boolea
35
36
  * misconfigured provider descriptors or user model entries (e.g. text-only
36
37
  * DashScope Qwen SKUs, DeepSeek models) whose endpoints reject `image_url`.
37
38
  */
38
- export function isOpenAICompletionsVisionSupported(model: Model<"openai-completions">): boolean {
39
+ export function isOpenAICompletionsVisionSupported(model: Model<"openai-completions" | "openrouter">): boolean {
39
40
  if (!model.input.includes("image")) return false;
40
41
  if (model.compat.stripImageInput) return false;
41
42
  return true;
42
43
  }
44
+
45
+ /**
46
+ * Whether the transport that will carry `model` sends image content on the wire.
47
+ *
48
+ * The `pi-native` transport forwards the original context (images included) to
49
+ * the gateway, which resolves its own model server-side, so the Chat
50
+ * Completions guard below never runs client-side and the declared input
51
+ * applies. Otherwise the OpenAI Chat Completions path applies the text-only
52
+ * guard, as does the OpenRouter chat fallback (`PI_OPENROUTER_RESPONSES=0`,
53
+ * which dispatches `openrouter` models through `streamOpenAICompletions`);
54
+ * every other API ships the modalities the model declares. Callers that report
55
+ * or gate on the wire (for example the `omp models` table) read this
56
+ * predicate; declared capability reads `model.input`.
57
+ */
58
+ export function sendsImageInputOnWire(model: Model<Api>): boolean {
59
+ if (model.transport === "pi-native") return model.input.includes("image");
60
+ if (isGuardedCompletionsTransport(model)) return isOpenAICompletionsVisionSupported(model);
61
+ return model.input.includes("image");
62
+ }
63
+
64
+ /** True for the transports that encode through the Chat Completions guard. */
65
+ function isGuardedCompletionsTransport(model: Model<Api>): model is Model<"openai-completions" | "openrouter"> {
66
+ if (model.api === "openai-completions") return true;
67
+ return model.api === "openrouter" && $env.PI_OPENROUTER_RESPONSES === "0";
68
+ }
@@ -15,12 +15,19 @@ import { scheduler } from "node:timers/promises";
15
15
  import { getBundledModels } from "@oh-my-pi/pi-catalog/models";
16
16
  import {
17
17
  COPILOT_API_HEADERS,
18
+ COPILOT_CHAT_INTEGRATION_ID,
18
19
  discoverGitHubCopilotApiEndpoint,
19
20
  getGitHubCopilotBaseUrl,
20
21
  isPublicGitHubHost,
22
+ normalizeCopilotIntegrationId,
21
23
  normalizeDomain,
22
24
  normalizeGitHubCopilotEnterpriseDomain,
23
25
  } from "@oh-my-pi/pi-catalog/wire/github-copilot";
26
+ import { $env } from "@oh-my-pi/pi-utils";
27
+ import {
28
+ resolveCopilotIntegrationIdOverride,
29
+ wrapFetchForCopilotFallback,
30
+ } from "../../providers/github-copilot-headers";
24
31
  import * as AIError from "../../error";
25
32
  import type { FetchImpl } from "../../types";
26
33
  import type { OAuthController, OAuthCredentials } from "./types";
@@ -51,6 +58,7 @@ type GitHubCopilotLoginOptions = {
51
58
  onAuth: (url: string, instructions?: string) => void;
52
59
  onPrompt: (prompt: { message: string; placeholder?: string; allowEmpty?: boolean }) => Promise<string>;
53
60
  onProgress?: (message: string) => void;
61
+ copilotIntegrationId?: unknown;
54
62
  signal?: AbortSignal;
55
63
  pollIntervalFloorMs?: number;
56
64
  pollIntervalScaleMs?: number;
@@ -259,6 +267,7 @@ async function enableGitHubCopilotModel(
259
267
  fetchImpl: FetchImpl,
260
268
  enterpriseDomain: string | undefined,
261
269
  apiEndpoint: string | undefined,
270
+ integrationId?: string,
262
271
  ): Promise<boolean> {
263
272
  const baseUrl = apiEndpoint ?? getGitHubCopilotBaseUrl(enterpriseDomain);
264
273
  const url = `${baseUrl}/models/${modelId}/policy`;
@@ -270,6 +279,7 @@ async function enableGitHubCopilotModel(
270
279
  "Content-Type": "application/json",
271
280
  Authorization: `Bearer ${token}`,
272
281
  ...COPILOT_API_HEADERS,
282
+ "Copilot-Integration-Id": integrationId ?? COPILOT_CHAT_INTEGRATION_ID,
273
283
  "Openai-Intent": "chat-policy",
274
284
  "X-Initiator": "user",
275
285
  "X-Interaction-Type": "chat-policy",
@@ -292,16 +302,24 @@ async function enableAllGitHubCopilotModels(
292
302
  apiEndpoint: string | undefined,
293
303
  fetchImpl: FetchImpl,
294
304
  onProgress?: (model: string, success: boolean) => void,
305
+ integrationId?: unknown,
295
306
  ): Promise<void> {
296
- // Synthesized catalog variants (Copilot long-context `-1m` entries) share
297
- // the upstream model id; enable each wire id exactly once.
298
307
  const wireModelIds = [...new Set(getBundledModels("github-copilot").map(model => model.requestModelId ?? model.id))];
308
+ const resolvedId = normalizeCopilotIntegrationId(integrationId) ?? resolveCopilotIntegrationIdOverride();
309
+ const copilotFetch = wrapFetchForCopilotFallback(fetchImpl, true, resolvedId);
299
310
  const BATCH_SIZE = 5;
300
311
  for (let i = 0; i < wireModelIds.length; i += BATCH_SIZE) {
301
312
  const batch = wireModelIds.slice(i, i + BATCH_SIZE);
302
313
  await Promise.all(
303
314
  batch.map(async modelId => {
304
- const success = await enableGitHubCopilotModel(token, modelId, fetchImpl, enterpriseDomain, apiEndpoint);
315
+ const success = await enableGitHubCopilotModel(
316
+ token,
317
+ modelId,
318
+ copilotFetch,
319
+ enterpriseDomain,
320
+ apiEndpoint,
321
+ resolvedId,
322
+ );
305
323
  onProgress?.(modelId, success);
306
324
  }),
307
325
  );
@@ -368,7 +386,14 @@ export async function loginGitHubCopilot(options: GitHubCopilotLoginOptions): Pr
368
386
 
369
387
  // Enable all models after successful login
370
388
  options.onProgress?.("Enabling models...");
371
- await enableAllGitHubCopilotModels(githubAccessToken, enterpriseDomain ?? undefined, apiEndpoint, fetchImpl);
389
+ await enableAllGitHubCopilotModels(
390
+ githubAccessToken,
391
+ enterpriseDomain ?? undefined,
392
+ apiEndpoint,
393
+ fetchImpl,
394
+ undefined,
395
+ options.copilotIntegrationId,
396
+ );
372
397
  return credentials;
373
398
  }
374
399
 
@@ -108,7 +108,7 @@ export function rewriteCopilotError(errorMessage: string, error: unknown, provid
108
108
  return `GitHub Copilot authentication failed (HTTP 401). Your token may have been revoked. Please re-login with /login github-copilot`;
109
109
  }
110
110
  if (status === 403) {
111
- return `GitHub Copilot access denied (HTTP 403). Your account may not have access to this model or feature. Check your Copilot plan or model policy settings.`;
111
+ return `GitHub Copilot access denied (HTTP 403). Your token is valid but the account may not have access to this model or feature. Check your Copilot plan or model policy settings. Business organizations can also restrict which clients may call the API: omp sends Copilot-Integration-Id copilot-chat by default (COPILOT_INTEGRATION_ID overrides it) and retries a denied default-identity request once as the Copilot CLI (copilot-developer-cli). If both identities are denied, ask your org admin to allow one of them; if you pinned an identity, try the other.`;
112
112
  }
113
113
  return errorMessage;
114
114
  }