@linxiraos/pi-ai 1.1.14 → 1.1.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/CHANGELOG.md +2 -74
  2. package/dist/types/error/flags.d.ts +6 -0
  3. package/dist/types/providers/anthropic-wire.d.ts +44 -7
  4. package/dist/types/providers/anthropic.d.ts +42 -0
  5. package/dist/types/providers/github-copilot-headers.d.ts +71 -1
  6. package/dist/types/providers/openai-completions.d.ts +1 -1
  7. package/dist/types/providers/openai-shared.d.ts +8 -0
  8. package/dist/types/providers/vision-guard.d.ts +16 -2
  9. package/dist/types/registry/oauth/github-copilot.d.ts +1 -0
  10. package/dist/types/registry/oauth/muse-code.d.ts +4 -4
  11. package/dist/types/types.d.ts +47 -1
  12. package/dist/types/usage/charm-hyper.d.ts +2 -0
  13. package/dist/types/utils/block-symbols.d.ts +41 -0
  14. package/package.json +6 -6
  15. package/src/auth-storage.ts +2 -0
  16. package/src/error/finalize.ts +9 -2
  17. package/src/error/flags.ts +46 -1
  18. package/src/error/rate-limit.ts +14 -1
  19. package/src/error/retryable.ts +2 -0
  20. package/src/providers/amazon-bedrock.ts +35 -25
  21. package/src/providers/anthropic-wire.ts +42 -9
  22. package/src/providers/anthropic.ts +648 -82
  23. package/src/providers/cursor.ts +1 -1
  24. package/src/providers/devin.ts +1 -1
  25. package/src/providers/github-copilot-headers.ts +242 -1
  26. package/src/providers/google-gemini-cli.ts +1 -1
  27. package/src/providers/google-shared.ts +1 -1
  28. package/src/providers/ollama.ts +11 -2
  29. package/src/providers/openai-codex-responses.ts +7 -2
  30. package/src/providers/openai-completions.ts +23 -4
  31. package/src/providers/openai-responses.ts +17 -9
  32. package/src/providers/openai-shared.ts +25 -4
  33. package/src/providers/pi-native-server.ts +3 -0
  34. package/src/providers/transform-messages.ts +7 -3
  35. package/src/providers/vision-guard.ts +28 -2
  36. package/src/registry/oauth/github-copilot.ts +28 -4
  37. package/src/registry/oauth/muse-code.ts +2 -2
  38. package/src/registry/oauth/openai-codex.ts +7 -8
  39. package/src/stream.ts +1 -0
  40. package/src/types.ts +49 -1
  41. package/src/usage/charm-hyper.ts +95 -0
  42. package/src/usage/claude.ts +106 -12
  43. package/src/usage/kimi.ts +11 -1
  44. package/src/utils/block-symbols.ts +47 -0
  45. package/src/utils/empty-completion-retry.ts +7 -4
  46. package/src/utils/http-inspector.ts +1 -1
  47. package/src/utils/proxy.ts +4 -3
package/CHANGELOG.md CHANGED
@@ -2,79 +2,7 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
- ## [1.1.14] - 2026-09-12
5
+ ## [1.1.15] - 2026-09-16
6
6
 
7
- - 随 1.1.14 版本线发布:bazel 构建面(crates/*/BUILD.bazel)版本号纳入一致性检查,CI 原生构建与桌面冒烟守卫修复。
7
+ - 上游 v18.1.17–v18.1.21 同步:provider 修复与补全回退链强化(保序、预算、同模型区分)随同步带过。
8
8
 
9
- ## [1.1.13] - 2026-09-10
10
-
11
- - 上游 v18.1.16 同步:AuthStorage 合并封锁契约——凭据级 `blockedUntilMs` 与 `providerTimed` 时间线合并,先到的更长封锁在短提示到来时保持有效;GitHub Copilot OAuth 拆分公共 GitHub / GHE 双 client-id;Codex WebSocket 传输 abort 携带 cause 链。
12
-
13
- ## [1.1.12] - 2026-09-10
14
-
15
- - 品牌与合并工具链维护版本;无本包用户可见变更。
16
-
17
- ## [1.1.11] - 2026-09-08
18
-
19
- ### Added
20
-
21
- - Muse Code subscription sign-in, credential refresh, inference, and quota reporting in `/usage`, with durable rate-limit backoff so quota refresh recovers instead of repeatedly retrying.
22
-
23
- ## [1.1.10-omp18.1.12] - 2026-09-06
24
-
25
- ### Added
26
-
27
- - Added Muse Code subscription sign-in, credential refresh, inference, and quota reporting in `/usage`, with durable rate-limit backoff so quota refresh recovers instead of repeatedly retrying.
28
-
29
- ## [1.1.10-omp18.1.11] - 2026-09-05
30
-
31
- ### Fixed
32
-
33
- - Fixed OpenCode Go usage polls (`GET /zen/go/v1/usage`) missing `x-opencode-session` and omp's `User-Agent`: background polls now attribute with the stable install id so the requests OpenCode flags as `Bun fetch` carry the required session header.
34
- - GitHub Copilot sign-in now requests only basic profile access, restoring login for Enterprise organizations that reject repository, gist, and Codespaces permissions ([#10656](https://github.com/can1357/oh-my-pi/issues/10656)).
35
-
36
- ## [1.1.10] - 2026-09-07
37
-
38
- - GitHub Copilot sign-in now requests only basic profile access, restoring login for Enterprise organizations that reject repository, gist, and Codespaces permissions.
39
- - Transient gateway stream failures are now retried instead of surfacing as session errors.
40
-
41
- ## [1.1.9] - 2026-09-05
42
-
43
- - Z.ai OAuth key name sends zeta (merge restored the upstream oh-my-pi literal in tests); xAI/OpenAI-compatible requests send the zeta User-Agent again.
44
-
45
- ## [1.1.6] - 2026-08-30
46
-
47
- - 同步上游 OMP v18.0.9(`cc14e04f075d`)。
48
-
49
- ## [1.1.5] - 2026-08-26
50
-
51
- - 同步上游 OMP v18.0.5 / v18.0.6:新增 Yolo-Auto / OpenRouter 浏览器登录与 DeepInfra 支持,空补全重试重构(withReplaySafeStreamRetry)。
52
-
53
- ## [1.1.2] - 2026-08-25
54
-
55
- ### Fixed
56
-
57
- - Republished as 1.1.2 to reset the `latest` tag after the broken 1.1.0 (no functional change over 1.1.1).
58
-
59
- ## [1.1.1] - 2026-08-25
60
-
61
- ### Fixed
62
-
63
- - Published tarballs now carry real dependency versions instead of Bun's `catalog:` protocol (1.1.0 installs failed with "Unsupported URL Type catalog:").
64
-
65
- ## [1.1.0] - 2026-08-25
66
-
67
- ### Changed
68
-
69
- - 同步上游 OMP v18.0.3 / v18.0.4(内部运行时与构建改进,无独立用户可见变更)。
70
-
71
- ## [1.0.1] - 2026-08-14
72
-
73
- - Reset the version to 1.0.0 and republished under the `@linxiraos/*` scope, breaking from the `@linxiraos` version lineage.
74
-
75
- ## [1.0.0] - 2026-08-13
76
-
77
- ### Changed
78
-
79
- - Reset the version to 1.0.0 and republished under the `@linxiraos/*` scope, breaking from the `@linxiraos` version lineage.
80
- - Fixed Gemini thought summaries occasionally leaking a raw `` ```thinking `` / `` ``````thinking `` fence delimiter into the reasoning block, so it no longer shows up as fence spam in the thinking display or persisted transcripts ([#8719](https://github.com/can1357/oh-my-pi/issues/8719)).
@@ -28,6 +28,12 @@ export declare const Flag: {
28
28
  };
29
29
  export type Flag = (typeof Flag)[keyof typeof Flag];
30
30
  export declare const STREAM_READ_ERROR_PATTERN: RegExp;
31
+ /** Python h2/httpx diagnostics forwarded through provider or proxy error events. */
32
+ export declare const PYTHON_HTTP2_STREAM_RESET_PATTERN: RegExp;
33
+ /** Python h11/httpx EOF while reading an HTTP/1.1 chunked response body. */
34
+ export declare const PYTHON_HTTP_INCOMPLETE_CHUNK_PATTERN: RegExp;
35
+ /** reqwest body-frame failures forwarded by the Codex HTTP proxy. */
36
+ export declare const CODEX_HTTP_BODY_READ_ERROR_PATTERN: RegExp;
31
37
  export declare const TRANSIENT_TRANSPORT_PATTERN: RegExp;
32
38
  /**
33
39
  * Local llama.cpp / Ollama deterministic tool-call argument JSON parse failure.
@@ -135,7 +135,21 @@ export type FallbackBlockParam = {
135
135
  model: string;
136
136
  };
137
137
  };
138
- export type ContentBlockParam = TextBlockParam | ImageBlockParam | ToolUseBlockParam | ToolResultBlockParam | ServerToolUseBlockParam | WebSearchToolResultBlockParam | ToolSearchToolResultBlockParam | ToolAdditionBlockParam | ToolRemovalBlockParam | ThinkingBlockParam | RedactedThinkingBlockParam | FallbackBlockParam;
138
+ /** Beta enabling server-side compaction (`compact_20260112` edit, `compaction` blocks). */
139
+ export declare const COMPACTION_BETA = "compact-2026-01-12";
140
+ /**
141
+ * Server-side compaction summary (compact-2026-01-12). Returned at the start
142
+ * of the assistant response that crossed the trigger; on replay the API drops
143
+ * every block that precedes it, so it may open the messages array. The
144
+ * `encrypted_content` is opaque provider state, round-tripped verbatim.
145
+ */
146
+ export type CompactionBlockParam = {
147
+ type: "compaction";
148
+ content: string;
149
+ encrypted_content?: string | null;
150
+ cache_control?: CacheControlEphemeral | null;
151
+ };
152
+ export type ContentBlockParam = TextBlockParam | ImageBlockParam | ToolUseBlockParam | ToolResultBlockParam | ServerToolUseBlockParam | WebSearchToolResultBlockParam | ToolSearchToolResultBlockParam | ToolAdditionBlockParam | ToolRemovalBlockParam | ThinkingBlockParam | RedactedThinkingBlockParam | FallbackBlockParam | CompactionBlockParam;
139
153
  /**
140
154
  * A single conversation turn.
141
155
  *
@@ -229,12 +243,24 @@ export type FallbackParam = {
229
243
  output_config?: OutputConfig;
230
244
  speed?: "fast";
231
245
  };
246
+ /** Server-side compaction edit (compact-2026-01-12). */
247
+ export type CompactionEdit = {
248
+ type: "compact_20260112";
249
+ /** `input_tokens` is the only trigger; `value` must be at least 50,000. */
250
+ trigger?: {
251
+ type: "input_tokens";
252
+ value: number;
253
+ };
254
+ pause_after_compaction?: boolean;
255
+ /** Replaces the API's default summarization prompt entirely. */
256
+ instructions?: string;
257
+ };
232
258
  /** Claude Code context-management beta payload. */
233
259
  export type ContextManagement = {
234
260
  edits: Array<{
235
261
  type: "clear_thinking_20251015";
236
262
  keep: "all";
237
- }>;
263
+ } | CompactionEdit>;
238
264
  };
239
265
  export type MessageCreateParams = {
240
266
  model: string;
@@ -267,7 +293,7 @@ export type MessageCreateParams = {
267
293
  export type MessageCreateParamsStreaming = MessageCreateParams & {
268
294
  stream: true;
269
295
  };
270
- export type StopReason = "end_turn" | "max_tokens" | "stop_sequence" | "tool_use" | "pause_turn" | "refusal" | "sensitive" | "model_context_window_exceeded";
296
+ export type StopReason = "end_turn" | "max_tokens" | "stop_sequence" | "tool_use" | "pause_turn" | "refusal" | "sensitive" | "model_context_window_exceeded" | "compaction";
271
297
  export type CacheCreation = {
272
298
  ephemeral_5m_input_tokens?: number | null;
273
299
  ephemeral_1h_input_tokens?: number | null;
@@ -278,12 +304,15 @@ export type ServerToolUsage = {
278
304
  };
279
305
  /**
280
306
  * Per-attempt token accounting inside a multi-run turn
281
- * (server-side-fallback-2026-06-01). Populated whenever a fallback chain
282
- * ran, including sticky-served turns with no `fallback` content block.
283
- * A `fallback_message` entry is the definitive "served by fallback" signal.
307
+ * (server-side-fallback-2026-06-01, compact-2026-01-12). Populated whenever
308
+ * a fallback chain ran, including sticky-served turns with no `fallback`
309
+ * content block, and whenever the compaction beta is active. A
310
+ * `fallback_message` entry is the definitive "served by fallback" signal; a
311
+ * `compaction` entry is the summarization sampling the top-level usage
312
+ * excludes.
284
313
  */
285
314
  export type UsageIteration = {
286
- type?: "message" | "fallback_message" | string;
315
+ type?: "message" | "fallback_message" | "compaction" | string;
287
316
  model?: string | null;
288
317
  input_tokens?: number | null;
289
318
  output_tokens?: number | null;
@@ -343,6 +372,10 @@ export type ResponseContentBlock = {
343
372
  to: {
344
373
  model: string;
345
374
  };
375
+ } | {
376
+ type: "compaction";
377
+ content?: string | null;
378
+ encrypted_content?: string | null;
346
379
  };
347
380
  export type ContentBlockDelta = {
348
381
  type: "text_delta";
@@ -356,6 +389,10 @@ export type ContentBlockDelta = {
356
389
  } | {
357
390
  type: "signature_delta";
358
391
  signature: string;
392
+ } | {
393
+ type: "compaction_delta";
394
+ content?: string | null;
395
+ encrypted_content?: string | null;
359
396
  };
360
397
  export type StopDetails = {
361
398
  type: string;
@@ -165,6 +165,14 @@ export type AnthropicClientOptionsArgs = {
165
165
  fetch?: FetchImpl;
166
166
  maxRetryDelayMs?: number;
167
167
  sessionId?: string;
168
+ /** Working-identity cache key for this credential+host; undefined off the Copilot path. */
169
+ copilotCacheKey?: string;
170
+ /**
171
+ * Build-time cache provenance for the wrapper: the cached value the
172
+ * outgoing headers were built from, or `null` when the cache was empty at
173
+ * build. `undefined` rereads the cache at dispatch.
174
+ */
175
+ copilotCacheSnapshot?: string | null;
168
176
  };
169
177
  export type AnthropicClientOptionsResult = {
170
178
  isOAuthToken: boolean;
@@ -204,6 +212,32 @@ export type AnthropicUsageLike = {
204
212
  * zero-valued objects clear prior extras from earlier stream usage snapshots.
205
213
  */
206
214
  export declare function applyAnthropicUsageExtras(usage: Usage, source: AnthropicUsageLike): void;
215
+ /**
216
+ * Whether this model's requests reach the official Anthropic API, resolved the
217
+ * way the transport resolves it — including the Foundry and
218
+ * `ANTHROPIC_BASE_URL` reroutes that leave `compat.officialEndpoint` stale.
219
+ */
220
+ export declare function resolvesToOfficialAnthropicEndpoint(model: Model<"anthropic-messages">): boolean;
221
+ /**
222
+ * Whether server-side compaction (`compact-2026-01-12`) may be spoken for
223
+ * this model to the endpoint a request actually reaches: a model line the
224
+ * beta supports (`compat.supportsServerCompaction`, rule-owned in the
225
+ * catalog), on the official API for the first-party provider or on any
226
+ * endpoint that opted in through `remoteCompaction.enabled`, and never on one
227
+ * whose deployment contract excludes context management. The same predicate
228
+ * gates emitting the edit, attaching the beta, and replaying a persisted
229
+ * block, so a route or model change can never leave a session sending a block
230
+ * its endpoint rejects.
231
+ */
232
+ export declare function supportsAnthropicCompaction(model: Model<"anthropic-messages">, effectiveBaseUrl?: string): boolean;
233
+ /**
234
+ * {@link supportsAnthropicCompaction} for a request on a caller-owned client:
235
+ * the endpoint is whatever the client targets (an `AnthropicVertex` client
236
+ * carries an Anthropic model to Vertex), never the model's own routing. SDK
237
+ * clients expose it as `baseURL`; a client that exposes no endpoint only
238
+ * compacts through an explicit `remoteCompaction.enabled` opt-in.
239
+ */
240
+ export declare function supportsAnthropicCompactionOnClient(model: Model<"anthropic-messages">, client: AnthropicMessagesClientLike): boolean;
207
241
  /** Detects the preserved-thinking error caused by rewriting a signed block's conversation prefix. */
208
242
  export declare function isThinkingPrefixBindingError(message: string): boolean;
209
243
  export declare function isInvalidThinkingSignatureError(message: string): boolean;
@@ -250,9 +284,17 @@ export type AnthropicMessageParam = MessageParam;
250
284
  * `fallback` content block from a prior turn be replayed on the wire;
251
285
  * otherwise the block is dropped to avoid a 400 on non-fallback requests
252
286
  * that don't send the beta.
287
+ *
288
+ * `opts.replayCompaction` — replay a user-role compaction summary that
289
+ * carries an {@link AnthropicCompactionPayload} from this provider as a
290
+ * native `compaction` block instead of its text. The API drops every block
291
+ * before the compaction block, so the assistant turn carrying it may open
292
+ * the conversation; the request must send the compaction beta (the stream
293
+ * entry point adds it whenever such a payload is present).
253
294
  */
254
295
  export declare function convertAnthropicMessages(messages: Message[], model: Model<"anthropic-messages">, isOAuthToken: boolean, opts?: {
255
296
  serverSideFallbackEnabled?: boolean;
297
+ replayCompaction?: boolean;
256
298
  dropAllThinking?: boolean;
257
299
  droppedThinkingBlocks?: ReadonlySet<string>;
258
300
  }): AnthropicMessageParam[];
@@ -1,4 +1,4 @@
1
- import type { Message } from "../types.js";
1
+ import type { FetchImpl, Message } from "../types.js";
2
2
  /**
3
3
  * Infer whether the current request to Copilot is user-initiated or agent-initiated.
4
4
  * Accepts `unknown[]` because providers may pass pre-converted message shapes.
@@ -11,6 +11,70 @@ export type CopilotDynamicHeaders = {
11
11
  premiumRequests: CopilotPremiumRequests;
12
12
  };
13
13
  export declare function resolveGitHubCopilotBaseUrl(baseUrl: string | undefined, apiKey: string | undefined): string | undefined;
14
+ /**
15
+ * Opt-in `Copilot-Integration-Id` override for chat and model-policy requests.
16
+ * Reads `COPILOT_INTEGRATION_ID`; unset/invalid keeps the chat-surface default
17
+ * (`COPILOT_CHAT_INTEGRATION_ID`). Model discovery keeps the CLI identity: it
18
+ * unlocks enterprise/experimental models and listing models is not
19
+ * policy-gated the way chat completions are (#11372).
20
+ */
21
+ export declare function resolveCopilotIntegrationIdOverride(env?: Record<string, string | undefined>): string | undefined;
22
+ /**
23
+ * Effective identity before the chat-surface default: explicit value, then
24
+ * request headers, then `COPILOT_INTEGRATION_ID`. Pure given its inputs, so
25
+ * tests inject literals instead of mutating process state.
26
+ */
27
+ export declare function resolveCopilotRequestIdentity(headers?: Record<string, string>, explicit?: unknown, env?: Record<string, string | undefined>): string | undefined;
28
+ /**
29
+ * Stable cache key for a raw Copilot API key envelope on one effective host.
30
+ * Hashes the bearer with `Bun.hash` (repo-approved hashing API; same
31
+ * credential-scoped pattern as the GitLab Duo and Codex account keys) so token
32
+ * bytes never sit in the map as keys; enterprise/business routing inputs and
33
+ * the normalized effective base URL participate so the same token on two hosts
34
+ * does not share an entry.
35
+ */
36
+ export declare function getCopilotIntegrationCacheKey(apiKeyRaw: string | undefined, baseUrl?: string): string | undefined;
37
+ /** Cached working identity for a cache key, if one was learned. */
38
+ export declare function getCachedCopilotIntegrationId(cacheKey: string | undefined): string | undefined;
39
+ /**
40
+ * Remember the identity that cleared the identity gate for a credential.
41
+ * Every store refreshes recency, so hot credentials survive eviction.
42
+ */
43
+ export declare function rememberCopilotWorkingIntegrationId(cacheKey: string | undefined, integrationId: unknown): void;
44
+ /** Clear one cached identity, or the whole cache when no key is given. */
45
+ export declare function clearCopilotIntegrationCache(cacheKey?: string): void;
46
+ /**
47
+ * Reissue Copilot client-identity denials once with the other surface.
48
+ *
49
+ * Chat is the default surface (`COPILOT_CHAT_INTEGRATION_ID`) because Business
50
+ * organizations that gate premium models per client surface commonly allow
51
+ * chat while blocking CLI/agentic clients (issue #11372). Other Business and
52
+ * Enterprise orgs do the opposite and reject the chat identity — as an HTTP 403
53
+ * or, on `api.business.githubcopilot.com`, an HTTP 400 `model_not_supported`
54
+ * (issue #11669). Both denials retry once as the CLI. The retry fires only for
55
+ * requests carrying the chat default and only when the caller resolved no
56
+ * explicit identity — an explicit choice is never second-guessed. The denied
57
+ * body is drained before reissuing, and the retry carries the CLI identity so
58
+ * the guard passes it through: at most two requests, never a loop.
59
+ *
60
+ * When `cacheKey` is set, a 2xx retry remembers its identity via
61
+ * `rememberCopilotWorkingIntegrationId`, so later streams for the same
62
+ * credential start at the working shape. A cached CLI start that is itself
63
+ * denied (stale after an org-policy flip) retries once as chat and relearns.
64
+ * Only a 2xx retry proves its identity — 401s deny every identity equally and
65
+ * 408/429/5xx are transport-retryable (the transport resends the *original*
66
+ * headers), so those must never be recorded as working. Any non-2xx retry
67
+ * clears the entry instead: the next stream rediscovers rather than pinning a
68
+ * shape that just failed.
69
+ */
70
+ export declare function wrapFetchForCopilotFallback(base: FetchImpl | undefined, enabled: boolean, integrationId?: unknown, cacheKey?: string,
71
+ /**
72
+ * Build-time cache provenance: the exact cached value the outgoing headers
73
+ * were built from. `undefined` rereads the cache at dispatch (direct
74
+ * callers); `null` pins "cache was empty at build" so a sibling learning
75
+ * mid-flight cannot change this request's retry decision.
76
+ */
77
+ cacheSnapshot?: string | null): FetchImpl;
14
78
  export declare function inferCopilotInitiator(messages: unknown[]): CopilotInitiator;
15
79
  /** Check whether any message in the conversation contains image content. */
16
80
  export declare function hasCopilotVisionInput(messages: Message[]): boolean;
@@ -37,4 +101,10 @@ export declare function buildCopilotDynamicHeaders(params: {
37
101
  headers?: Record<string, string>;
38
102
  initiatorOverride?: CopilotInitiator;
39
103
  planTier?: string;
104
+ /** Enterprise login domain; Enterprise keeps the CLI identity that its private endpoint accepts. */
105
+ enterpriseUrl?: string;
106
+ /** Raw explicit identity; validated here, chat default when absent/invalid. */
107
+ integrationId?: unknown;
108
+ /** Learned working identity for this credential; explicit still wins, then this, then the defaults. */
109
+ cachedIntegrationId?: unknown;
40
110
  }): CopilotDynamicHeaders;
@@ -42,5 +42,5 @@ export interface OpenAICompletionsOptions extends StreamOptions {
42
42
  * assistant output commits the attempt.
43
43
  */
44
44
  export declare const streamOpenAICompletions: StreamFunction<"openai-completions">;
45
- export declare function parseChunkUsage(rawUsage: object, model: Model<"openai-completions">, premiumRequests: number | undefined): AssistantMessage["usage"];
45
+ export declare function parseChunkUsage(rawUsage: object, model: Model<"openai-completions">, premiumRequests: number | undefined, timestamp?: number): AssistantMessage["usage"];
46
46
  export declare function convertMessages(model: Model<"openai-completions">, context: Context, compat: ResolvedOpenAICompat): ChatCompletionMessageParam[];
@@ -65,6 +65,14 @@ export interface OpenAIRequestSetup {
65
65
  headers: Record<string, string>;
66
66
  query: Record<string, string> | undefined;
67
67
  requestHeaders: Record<string, string>;
68
+ /** Working-identity cache key for this credential+host; undefined off the Copilot path. */
69
+ copilotCacheKey: string | undefined;
70
+ /**
71
+ * Build-time cache provenance for the wrapper: the cached value the
72
+ * outgoing headers were built from, or `null` when the cache was empty at
73
+ * build. `undefined` off the Copilot path (wrapper rereads at dispatch).
74
+ */
75
+ copilotCacheSnapshot: string | null | undefined;
68
76
  }
69
77
  export declare function resolveOpenAIRequestSetup(model: OpenAIRequestSetupModel, options: OpenAIRequestSetupOptions): OpenAIRequestSetup;
70
78
  export declare function applyOpenAIServiceTier(params: {
@@ -1,4 +1,4 @@
1
- import type { ImageContent, Model, TextContent } from "../types.js";
1
+ import type { Api, ImageContent, Model, TextContent } from "../types.js";
2
2
  export declare const NON_VISION_IMAGE_PLACEHOLDER = "[image omitted: model does not support vision]";
3
3
  export declare function partitionVisionContent(content: ReadonlyArray<TextContent | ImageContent>, supportsImages: boolean): {
4
4
  textBlocks: TextContent[];
@@ -12,4 +12,18 @@ export declare function joinTextWithImagePlaceholder(text: string, omittedImages
12
12
  * misconfigured provider descriptors or user model entries (e.g. text-only
13
13
  * DashScope Qwen SKUs, DeepSeek models) whose endpoints reject `image_url`.
14
14
  */
15
- export declare function isOpenAICompletionsVisionSupported(model: Model<"openai-completions">): boolean;
15
+ export declare function isOpenAICompletionsVisionSupported(model: Model<"openai-completions" | "openrouter">): boolean;
16
+ /**
17
+ * Whether the transport that will carry `model` sends image content on the wire.
18
+ *
19
+ * The `pi-native` transport forwards the original context (images included) to
20
+ * the gateway, which resolves its own model server-side, so the Chat
21
+ * Completions guard below never runs client-side and the declared input
22
+ * applies. Otherwise the OpenAI Chat Completions path applies the text-only
23
+ * guard, as does the OpenRouter chat fallback (`PI_OPENROUTER_RESPONSES=0`,
24
+ * which dispatches `openrouter` models through `streamOpenAICompletions`);
25
+ * every other API ships the modalities the model declares. Callers that report
26
+ * or gate on the wire (for example the `omp models` table) read this
27
+ * predicate; declared capability reads `model.input`.
28
+ */
29
+ export declare function sendsImageInputOnWire(model: Model<Api>): boolean;
@@ -8,6 +8,7 @@ type GitHubCopilotLoginOptions = {
8
8
  allowEmpty?: boolean;
9
9
  }) => Promise<string>;
10
10
  onProgress?: (message: string) => void;
11
+ copilotIntegrationId?: unknown;
11
12
  signal?: AbortSignal;
12
13
  pollIntervalFloorMs?: number;
13
14
  pollIntervalScaleMs?: number;
@@ -6,8 +6,8 @@ declare const museCodeKeyResponseSchema: import("@linxiraos/pi-omptype").FluentT
6
6
  is_subs_active?: boolean | undefined;
7
7
  require_payment?: boolean | undefined;
8
8
  require_payment_action_url?: string | undefined;
9
- subs_tier_id?: string | undefined;
10
- subs_tier_name?: string | undefined;
9
+ subs_tier_id?: string | null | undefined;
10
+ subs_tier_name?: string | null | undefined;
11
11
  subs_usage?: {
12
12
  weekly?: {
13
13
  resets_at?: string | number | undefined;
@@ -28,8 +28,8 @@ declare const museCodeKeyResponseSchema: import("@linxiraos/pi-omptype").FluentT
28
28
  is_subs_active?: boolean | undefined;
29
29
  require_payment?: boolean | undefined;
30
30
  require_payment_action_url?: string | undefined;
31
- subs_tier_id?: string | undefined;
32
- subs_tier_name?: string | undefined;
31
+ subs_tier_id?: string | null | undefined;
32
+ subs_tier_name?: string | null | undefined;
33
33
  subs_usage?: {
34
34
  weekly?: {
35
35
  resets_at?: string | number | undefined;
@@ -195,6 +195,18 @@ export interface CodexCompactionMetadata {
195
195
  export interface CodexCompactionRequestContext extends CodexCompactionMetadata {
196
196
  operationId: string;
197
197
  }
198
+ /** Anthropic `compact_20260112` context-management edit (`compact-2026-01-12` beta). */
199
+ export interface AnthropicCompactionRequest {
200
+ /**
201
+ * Prompt input-token count at which the API compacts. The API enforces a
202
+ * 50,000-token floor and defaults to 150,000 when omitted.
203
+ */
204
+ triggerInputTokens?: number;
205
+ /** Stop after the compaction block instead of continuing the response. */
206
+ pauseAfterCompaction?: boolean;
207
+ /** Custom summarization prompt; replaces the API default entirely when set. */
208
+ instructions?: string;
209
+ }
198
210
  /** OpenAI's GPT-5.6+ explicit prompt-cache controls. */
199
211
  export interface OpenAIPromptCacheOptions {
200
212
  /** `explicit` disables OpenAI's automatic latest-message breakpoint. */
@@ -243,6 +255,15 @@ export interface StreamOptions {
243
255
  anthropicPrefixMismatchBehavior?: "drop_block" | "error";
244
256
  /** @internal Marks a replay-only Anthropic request that must use non-streaming `max_tokens: 0`. */
245
257
  anthropicCacheRefreshRequest?: boolean;
258
+ /**
259
+ * Anthropic server-side compaction (`compact-2026-01-12` beta). Sends the
260
+ * `compact_20260112` context-management edit so the API summarizes the
261
+ * prompt in-band once its input reaches the trigger; the resulting summary
262
+ * arrives as an {@link AnthropicCompactionPayload} on the assistant message.
263
+ * Ignored by every other provider and by Anthropic-compatible endpoints
264
+ * without context-management support.
265
+ */
266
+ anthropicCompaction?: AnthropicCompactionRequest;
246
267
  /**
247
268
  * Additional headers to include in provider requests.
248
269
  * These are merged on top of model-defined headers.
@@ -703,7 +724,32 @@ export interface AnthropicMessagePayload {
703
724
  name: string;
704
725
  }>;
705
726
  }
706
- export type ProviderPayload = OpenAIResponsesHistoryPayload | AnthropicMessagePayload;
727
+ /**
728
+ * Anthropic server-side compaction summary (`compact-2026-01-12` beta).
729
+ *
730
+ * Produced by the Anthropic provider on the assistant message of a request
731
+ * that streamed a `compaction` content block, and attached to the user-role
732
+ * compaction summary message that replaces the compacted history so the
733
+ * provider can replay the block verbatim: the API drops every block that
734
+ * precedes it. `content` is the plain-text summary, so every other provider
735
+ * reads the message text and ignores the payload.
736
+ */
737
+ export interface AnthropicCompactionPayload {
738
+ type: "anthropicCompaction";
739
+ /** Provider that produced the summary; only that provider replays it natively. */
740
+ provider: string;
741
+ content: string;
742
+ /** Opaque provider state the API attached to the block; replayed verbatim when present. */
743
+ encryptedContent?: string;
744
+ /**
745
+ * Harness-appended file metadata (`<files>` section) kept out of the
746
+ * byte-identical block. Replayed as a user message after the native block:
747
+ * the converter replaces the summary message with the block and skips its
748
+ * text, so without this the metadata would be invisible to this provider.
749
+ */
750
+ filesText?: string;
751
+ }
752
+ export type ProviderPayload = OpenAIResponsesHistoryPayload | AnthropicMessagePayload | AnthropicCompactionPayload;
707
753
  /** Provider-reported rewrite applied to request content before inference. */
708
754
  export interface ProviderInputTransformation {
709
755
  type: string;
@@ -0,0 +1,2 @@
1
+ import type { UsageProvider } from "../usage.js";
2
+ export declare const charmHyperUsageProvider: UsageProvider;
@@ -72,3 +72,44 @@ export type DemotedThinkingCarrier = object & {
72
72
  };
73
73
  /** True for text blocks synthesized by cross-model thinking demotion. */
74
74
  export declare function isDemotedThinking(block: DemotedThinkingCarrier | null | undefined): boolean;
75
+ /**
76
+ * Marks an Anthropic wire message that was serialized from a source
77
+ * `role: "user"` message.
78
+ *
79
+ * The wire role alone cannot answer this. `convertAnthropicMessages` emits
80
+ * `role: "user"` for several things that are not a conversational turn:
81
+ * `developer` messages on models without mid-conversation `system` support,
82
+ * `tool_result` runs, and the synthetic `Continue.` pads inserted between
83
+ * adjacent assistants. Prompt-cache decimation counts conversational turns,
84
+ * so it reads this marker instead of guessing from wire content.
85
+ *
86
+ * Symbol-keyed so the marker never persists across the JSONL round-trip and
87
+ * never reaches the wire.
88
+ */
89
+ export declare const kConversationalUser: unique symbol;
90
+ /** Carries the conversational-user marker without exposing a string-keyed property. */
91
+ export type ConversationalUserCarrier = object & {
92
+ [kConversationalUser]?: boolean;
93
+ };
94
+ /** True for wire messages serialized from a source `role: "user"` message. */
95
+ export declare function isConversationalUser(message: ConversationalUserCarrier | null | undefined): boolean;
96
+ /**
97
+ * Marks a `role: "user"` message that `transformMessages` synthesized rather
98
+ * than one the user sent.
99
+ *
100
+ * The stale-tool-result note is deliberately emitted as `user` so no provider
101
+ * elevates untrusted tool output to instruction priority, which leaves it
102
+ * indistinguishable from a real turn by role alone. Prompt-cache decimation
103
+ * must not count it, or an orphan result appearing or disappearing in a
104
+ * compacted history shifts every later checkpoint.
105
+ *
106
+ * Symbol-keyed so the marker never persists across the JSONL round-trip and
107
+ * never reaches the wire.
108
+ */
109
+ export declare const kSyntheticUser: unique symbol;
110
+ /** Carries the synthetic-user marker without exposing a string-keyed property. */
111
+ export type SyntheticUserCarrier = object & {
112
+ [kSyntheticUser]?: boolean;
113
+ };
114
+ /** True for `user` messages synthesized by message transformation. */
115
+ export declare function isSyntheticUser(message: SyntheticUserCarrier | null | undefined): boolean;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@linxiraos/pi-ai",
3
- "version": "1.1.14",
3
+ "version": "1.1.15",
4
4
  "description": "Unified LLM API with automatic model discovery and provider configuration",
5
5
  "keywords": [
6
6
  "ai",
@@ -124,11 +124,11 @@
124
124
  "fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
125
125
  },
126
126
  "dependencies": {
127
- "@linxiraos/pi-omptype": "1.1.14",
128
- "@linxiraos/pi-catalog": "1.1.14",
129
- "@linxiraos/pi-natives": "1.1.14",
130
- "@linxiraos/pi-utils": "1.1.14",
131
- "@linxiraos/pi-wire": "1.1.14"
127
+ "@linxiraos/pi-omptype": "1.1.15",
128
+ "@linxiraos/pi-catalog": "1.1.15",
129
+ "@linxiraos/pi-natives": "1.1.15",
130
+ "@linxiraos/pi-utils": "1.1.15",
131
+ "@linxiraos/pi-wire": "1.1.15"
132
132
  },
133
133
  "devDependencies": {
134
134
  "@types/bun": "^1.3.14"
@@ -51,6 +51,7 @@ import type {
51
51
  } from "./usage";
52
52
  import { resolveUsedFraction } from "./usage";
53
53
  import { alibabaTokenPlanRankingStrategy, alibabaTokenPlanUsageProvider } from "./usage/alibaba-token-plan";
54
+ import { charmHyperUsageProvider } from "./usage/charm-hyper";
54
55
  import { claudeRankingStrategy, claudeUsageProvider } from "./usage/claude";
55
56
  import { clinePassUsageProvider } from "./usage/cline-pass";
56
57
  import { cursorUsageProvider } from "./usage/cursor";
@@ -677,6 +678,7 @@ const DEFAULT_USAGE_PROVIDERS: UsageProvider[] = [
677
678
  syntheticUsageProvider,
678
679
  xaiOauthUsageProvider,
679
680
  devinUsageProvider,
681
+ charmHyperUsageProvider,
680
682
  ];
681
683
 
682
684
  const DEFAULT_USAGE_PROVIDER_MAP = new Map<Provider, UsageProvider>(
@@ -45,7 +45,8 @@ export interface FinalizeResult {
45
45
  */
46
46
  export async function finalize(error: unknown, opts: FinalizeOptions = {}): Promise<FinalizeResult> {
47
47
  const aborted = opts.abortTracker ? opts.abortTracker.wasCallerAbort() : opts.signal?.aborted === true;
48
- const currentStatus = status(error) ?? opts.capturedErrorResponse?.status;
48
+ const errorStatus = status(error);
49
+ const currentStatus = errorStatus ?? opts.capturedErrorResponse?.status;
49
50
 
50
51
  let message: string;
51
52
  try {
@@ -55,11 +56,17 @@ export async function finalize(error: unknown, opts: FinalizeOptions = {}): Prom
55
56
  message = error instanceof Error ? error.message : String(error);
56
57
  }
57
58
 
59
+ // A captured status is transport context for the original error. Put it at
60
+ // the root of the classification cause chain so terminal 4xx policy governs
61
+ // nested diagnostics before they can contribute transient flags.
62
+ const classificationError =
63
+ errorStatus === undefined && currentStatus !== undefined ? { status: currentStatus, cause: error } : error;
64
+
58
65
  const id = classifyMessage({
59
66
  api: opts.api,
60
67
  provider: opts.provider,
61
68
  model: opts.model,
62
- errorId: classify(error, opts.api),
69
+ errorId: classify(classificationError, opts.api),
63
70
  errorMessage: message,
64
71
  errorStatus: currentStatus,
65
72
  });