@linxiraos/pi-ai 1.1.14 → 1.1.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +2 -74
- package/dist/types/error/flags.d.ts +6 -0
- package/dist/types/providers/anthropic-wire.d.ts +44 -7
- package/dist/types/providers/anthropic.d.ts +42 -0
- package/dist/types/providers/github-copilot-headers.d.ts +71 -1
- package/dist/types/providers/openai-completions.d.ts +1 -1
- package/dist/types/providers/openai-shared.d.ts +8 -0
- package/dist/types/providers/vision-guard.d.ts +16 -2
- package/dist/types/registry/oauth/github-copilot.d.ts +1 -0
- package/dist/types/registry/oauth/muse-code.d.ts +4 -4
- package/dist/types/types.d.ts +47 -1
- package/dist/types/usage/charm-hyper.d.ts +2 -0
- package/dist/types/utils/block-symbols.d.ts +41 -0
- package/package.json +6 -6
- package/src/auth-storage.ts +2 -0
- package/src/error/finalize.ts +9 -2
- package/src/error/flags.ts +46 -1
- package/src/error/rate-limit.ts +14 -1
- package/src/error/retryable.ts +2 -0
- package/src/providers/amazon-bedrock.ts +35 -25
- package/src/providers/anthropic-wire.ts +42 -9
- package/src/providers/anthropic.ts +648 -82
- package/src/providers/cursor.ts +1 -1
- package/src/providers/devin.ts +1 -1
- package/src/providers/github-copilot-headers.ts +242 -1
- package/src/providers/google-gemini-cli.ts +1 -1
- package/src/providers/google-shared.ts +1 -1
- package/src/providers/ollama.ts +11 -2
- package/src/providers/openai-codex-responses.ts +7 -2
- package/src/providers/openai-completions.ts +23 -4
- package/src/providers/openai-responses.ts +17 -9
- package/src/providers/openai-shared.ts +25 -4
- package/src/providers/pi-native-server.ts +3 -0
- package/src/providers/transform-messages.ts +7 -3
- package/src/providers/vision-guard.ts +28 -2
- package/src/registry/oauth/github-copilot.ts +28 -4
- package/src/registry/oauth/muse-code.ts +2 -2
- package/src/registry/oauth/openai-codex.ts +7 -8
- package/src/stream.ts +1 -0
- package/src/types.ts +49 -1
- package/src/usage/charm-hyper.ts +95 -0
- package/src/usage/claude.ts +106 -12
- package/src/usage/kimi.ts +11 -1
- package/src/utils/block-symbols.ts +47 -0
- package/src/utils/empty-completion-retry.ts +7 -4
- package/src/utils/http-inspector.ts +1 -1
- package/src/utils/proxy.ts +4 -3
package/CHANGELOG.md
CHANGED
|
@@ -2,79 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
-
## [1.1.
|
|
5
|
+
## [1.1.15] - 2026-09-16
|
|
6
6
|
|
|
7
|
-
-
|
|
7
|
+
- 上游 v18.1.17–v18.1.21 同步:provider 修复与补全回退链强化(保序、预算、同模型区分)随同步带过。
|
|
8
8
|
|
|
9
|
-
## [1.1.13] - 2026-09-10
|
|
10
|
-
|
|
11
|
-
- 上游 v18.1.16 同步:AuthStorage 合并封锁契约——凭据级 `blockedUntilMs` 与 `providerTimed` 时间线合并,先到的更长封锁在短提示到来时保持有效;GitHub Copilot OAuth 拆分公共 GitHub / GHE 双 client-id;Codex WebSocket 传输 abort 携带 cause 链。
|
|
12
|
-
|
|
13
|
-
## [1.1.12] - 2026-09-10
|
|
14
|
-
|
|
15
|
-
- 品牌与合并工具链维护版本;无本包用户可见变更。
|
|
16
|
-
|
|
17
|
-
## [1.1.11] - 2026-09-08
|
|
18
|
-
|
|
19
|
-
### Added
|
|
20
|
-
|
|
21
|
-
- Muse Code subscription sign-in, credential refresh, inference, and quota reporting in `/usage`, with durable rate-limit backoff so quota refresh recovers instead of repeatedly retrying.
|
|
22
|
-
|
|
23
|
-
## [1.1.10-omp18.1.12] - 2026-09-06
|
|
24
|
-
|
|
25
|
-
### Added
|
|
26
|
-
|
|
27
|
-
- Added Muse Code subscription sign-in, credential refresh, inference, and quota reporting in `/usage`, with durable rate-limit backoff so quota refresh recovers instead of repeatedly retrying.
|
|
28
|
-
|
|
29
|
-
## [1.1.10-omp18.1.11] - 2026-09-05
|
|
30
|
-
|
|
31
|
-
### Fixed
|
|
32
|
-
|
|
33
|
-
- Fixed OpenCode Go usage polls (`GET /zen/go/v1/usage`) missing `x-opencode-session` and omp's `User-Agent`: background polls now attribute with the stable install id so the requests OpenCode flags as `Bun fetch` carry the required session header.
|
|
34
|
-
- GitHub Copilot sign-in now requests only basic profile access, restoring login for Enterprise organizations that reject repository, gist, and Codespaces permissions ([#10656](https://github.com/can1357/oh-my-pi/issues/10656)).
|
|
35
|
-
|
|
36
|
-
## [1.1.10] - 2026-09-07
|
|
37
|
-
|
|
38
|
-
- GitHub Copilot sign-in now requests only basic profile access, restoring login for Enterprise organizations that reject repository, gist, and Codespaces permissions.
|
|
39
|
-
- Transient gateway stream failures are now retried instead of surfacing as session errors.
|
|
40
|
-
|
|
41
|
-
## [1.1.9] - 2026-09-05
|
|
42
|
-
|
|
43
|
-
- Z.ai OAuth key name sends zeta (merge restored the upstream oh-my-pi literal in tests); xAI/OpenAI-compatible requests send the zeta User-Agent again.
|
|
44
|
-
|
|
45
|
-
## [1.1.6] - 2026-08-30
|
|
46
|
-
|
|
47
|
-
- 同步上游 OMP v18.0.9(`cc14e04f075d`)。
|
|
48
|
-
|
|
49
|
-
## [1.1.5] - 2026-08-26
|
|
50
|
-
|
|
51
|
-
- 同步上游 OMP v18.0.5 / v18.0.6:新增 Yolo-Auto / OpenRouter 浏览器登录与 DeepInfra 支持,空补全重试重构(withReplaySafeStreamRetry)。
|
|
52
|
-
|
|
53
|
-
## [1.1.2] - 2026-08-25
|
|
54
|
-
|
|
55
|
-
### Fixed
|
|
56
|
-
|
|
57
|
-
- Republished as 1.1.2 to reset the `latest` tag after the broken 1.1.0 (no functional change over 1.1.1).
|
|
58
|
-
|
|
59
|
-
## [1.1.1] - 2026-08-25
|
|
60
|
-
|
|
61
|
-
### Fixed
|
|
62
|
-
|
|
63
|
-
- Published tarballs now carry real dependency versions instead of Bun's `catalog:` protocol (1.1.0 installs failed with "Unsupported URL Type catalog:").
|
|
64
|
-
|
|
65
|
-
## [1.1.0] - 2026-08-25
|
|
66
|
-
|
|
67
|
-
### Changed
|
|
68
|
-
|
|
69
|
-
- 同步上游 OMP v18.0.3 / v18.0.4(内部运行时与构建改进,无独立用户可见变更)。
|
|
70
|
-
|
|
71
|
-
## [1.0.1] - 2026-08-14
|
|
72
|
-
|
|
73
|
-
- Reset the version to 1.0.0 and republished under the `@linxiraos/*` scope, breaking from the `@linxiraos` version lineage.
|
|
74
|
-
|
|
75
|
-
## [1.0.0] - 2026-08-13
|
|
76
|
-
|
|
77
|
-
### Changed
|
|
78
|
-
|
|
79
|
-
- Reset the version to 1.0.0 and republished under the `@linxiraos/*` scope, breaking from the `@linxiraos` version lineage.
|
|
80
|
-
- Fixed Gemini thought summaries occasionally leaking a raw `` ```thinking `` / `` ``````thinking `` fence delimiter into the reasoning block, so it no longer shows up as fence spam in the thinking display or persisted transcripts ([#8719](https://github.com/can1357/oh-my-pi/issues/8719)).
|
|
@@ -28,6 +28,12 @@ export declare const Flag: {
|
|
|
28
28
|
};
|
|
29
29
|
export type Flag = (typeof Flag)[keyof typeof Flag];
|
|
30
30
|
export declare const STREAM_READ_ERROR_PATTERN: RegExp;
|
|
31
|
+
/** Python h2/httpx diagnostics forwarded through provider or proxy error events. */
|
|
32
|
+
export declare const PYTHON_HTTP2_STREAM_RESET_PATTERN: RegExp;
|
|
33
|
+
/** Python h11/httpx EOF while reading an HTTP/1.1 chunked response body. */
|
|
34
|
+
export declare const PYTHON_HTTP_INCOMPLETE_CHUNK_PATTERN: RegExp;
|
|
35
|
+
/** reqwest body-frame failures forwarded by the Codex HTTP proxy. */
|
|
36
|
+
export declare const CODEX_HTTP_BODY_READ_ERROR_PATTERN: RegExp;
|
|
31
37
|
export declare const TRANSIENT_TRANSPORT_PATTERN: RegExp;
|
|
32
38
|
/**
|
|
33
39
|
* Local llama.cpp / Ollama deterministic tool-call argument JSON parse failure.
|
|
@@ -135,7 +135,21 @@ export type FallbackBlockParam = {
|
|
|
135
135
|
model: string;
|
|
136
136
|
};
|
|
137
137
|
};
|
|
138
|
-
|
|
138
|
+
/** Beta enabling server-side compaction (`compact_20260112` edit, `compaction` blocks). */
|
|
139
|
+
export declare const COMPACTION_BETA = "compact-2026-01-12";
|
|
140
|
+
/**
|
|
141
|
+
* Server-side compaction summary (compact-2026-01-12). Returned at the start
|
|
142
|
+
* of the assistant response that crossed the trigger; on replay the API drops
|
|
143
|
+
* every block that precedes it, so it may open the messages array. The
|
|
144
|
+
* `encrypted_content` is opaque provider state, round-tripped verbatim.
|
|
145
|
+
*/
|
|
146
|
+
export type CompactionBlockParam = {
|
|
147
|
+
type: "compaction";
|
|
148
|
+
content: string;
|
|
149
|
+
encrypted_content?: string | null;
|
|
150
|
+
cache_control?: CacheControlEphemeral | null;
|
|
151
|
+
};
|
|
152
|
+
export type ContentBlockParam = TextBlockParam | ImageBlockParam | ToolUseBlockParam | ToolResultBlockParam | ServerToolUseBlockParam | WebSearchToolResultBlockParam | ToolSearchToolResultBlockParam | ToolAdditionBlockParam | ToolRemovalBlockParam | ThinkingBlockParam | RedactedThinkingBlockParam | FallbackBlockParam | CompactionBlockParam;
|
|
139
153
|
/**
|
|
140
154
|
* A single conversation turn.
|
|
141
155
|
*
|
|
@@ -229,12 +243,24 @@ export type FallbackParam = {
|
|
|
229
243
|
output_config?: OutputConfig;
|
|
230
244
|
speed?: "fast";
|
|
231
245
|
};
|
|
246
|
+
/** Server-side compaction edit (compact-2026-01-12). */
|
|
247
|
+
export type CompactionEdit = {
|
|
248
|
+
type: "compact_20260112";
|
|
249
|
+
/** `input_tokens` is the only trigger; `value` must be at least 50,000. */
|
|
250
|
+
trigger?: {
|
|
251
|
+
type: "input_tokens";
|
|
252
|
+
value: number;
|
|
253
|
+
};
|
|
254
|
+
pause_after_compaction?: boolean;
|
|
255
|
+
/** Replaces the API's default summarization prompt entirely. */
|
|
256
|
+
instructions?: string;
|
|
257
|
+
};
|
|
232
258
|
/** Claude Code context-management beta payload. */
|
|
233
259
|
export type ContextManagement = {
|
|
234
260
|
edits: Array<{
|
|
235
261
|
type: "clear_thinking_20251015";
|
|
236
262
|
keep: "all";
|
|
237
|
-
}>;
|
|
263
|
+
} | CompactionEdit>;
|
|
238
264
|
};
|
|
239
265
|
export type MessageCreateParams = {
|
|
240
266
|
model: string;
|
|
@@ -267,7 +293,7 @@ export type MessageCreateParams = {
|
|
|
267
293
|
export type MessageCreateParamsStreaming = MessageCreateParams & {
|
|
268
294
|
stream: true;
|
|
269
295
|
};
|
|
270
|
-
export type StopReason = "end_turn" | "max_tokens" | "stop_sequence" | "tool_use" | "pause_turn" | "refusal" | "sensitive" | "model_context_window_exceeded";
|
|
296
|
+
export type StopReason = "end_turn" | "max_tokens" | "stop_sequence" | "tool_use" | "pause_turn" | "refusal" | "sensitive" | "model_context_window_exceeded" | "compaction";
|
|
271
297
|
export type CacheCreation = {
|
|
272
298
|
ephemeral_5m_input_tokens?: number | null;
|
|
273
299
|
ephemeral_1h_input_tokens?: number | null;
|
|
@@ -278,12 +304,15 @@ export type ServerToolUsage = {
|
|
|
278
304
|
};
|
|
279
305
|
/**
|
|
280
306
|
* Per-attempt token accounting inside a multi-run turn
|
|
281
|
-
* (server-side-fallback-2026-06-01). Populated whenever
|
|
282
|
-
* ran, including sticky-served turns with no `fallback`
|
|
283
|
-
*
|
|
307
|
+
* (server-side-fallback-2026-06-01, compact-2026-01-12). Populated whenever
|
|
308
|
+
* a fallback chain ran, including sticky-served turns with no `fallback`
|
|
309
|
+
* content block, and whenever the compaction beta is active. A
|
|
310
|
+
* `fallback_message` entry is the definitive "served by fallback" signal; a
|
|
311
|
+
* `compaction` entry is the summarization sampling the top-level usage
|
|
312
|
+
* excludes.
|
|
284
313
|
*/
|
|
285
314
|
export type UsageIteration = {
|
|
286
|
-
type?: "message" | "fallback_message" | string;
|
|
315
|
+
type?: "message" | "fallback_message" | "compaction" | string;
|
|
287
316
|
model?: string | null;
|
|
288
317
|
input_tokens?: number | null;
|
|
289
318
|
output_tokens?: number | null;
|
|
@@ -343,6 +372,10 @@ export type ResponseContentBlock = {
|
|
|
343
372
|
to: {
|
|
344
373
|
model: string;
|
|
345
374
|
};
|
|
375
|
+
} | {
|
|
376
|
+
type: "compaction";
|
|
377
|
+
content?: string | null;
|
|
378
|
+
encrypted_content?: string | null;
|
|
346
379
|
};
|
|
347
380
|
export type ContentBlockDelta = {
|
|
348
381
|
type: "text_delta";
|
|
@@ -356,6 +389,10 @@ export type ContentBlockDelta = {
|
|
|
356
389
|
} | {
|
|
357
390
|
type: "signature_delta";
|
|
358
391
|
signature: string;
|
|
392
|
+
} | {
|
|
393
|
+
type: "compaction_delta";
|
|
394
|
+
content?: string | null;
|
|
395
|
+
encrypted_content?: string | null;
|
|
359
396
|
};
|
|
360
397
|
export type StopDetails = {
|
|
361
398
|
type: string;
|
|
@@ -165,6 +165,14 @@ export type AnthropicClientOptionsArgs = {
|
|
|
165
165
|
fetch?: FetchImpl;
|
|
166
166
|
maxRetryDelayMs?: number;
|
|
167
167
|
sessionId?: string;
|
|
168
|
+
/** Working-identity cache key for this credential+host; undefined off the Copilot path. */
|
|
169
|
+
copilotCacheKey?: string;
|
|
170
|
+
/**
|
|
171
|
+
* Build-time cache provenance for the wrapper: the cached value the
|
|
172
|
+
* outgoing headers were built from, or `null` when the cache was empty at
|
|
173
|
+
* build. `undefined` rereads the cache at dispatch.
|
|
174
|
+
*/
|
|
175
|
+
copilotCacheSnapshot?: string | null;
|
|
168
176
|
};
|
|
169
177
|
export type AnthropicClientOptionsResult = {
|
|
170
178
|
isOAuthToken: boolean;
|
|
@@ -204,6 +212,32 @@ export type AnthropicUsageLike = {
|
|
|
204
212
|
* zero-valued objects clear prior extras from earlier stream usage snapshots.
|
|
205
213
|
*/
|
|
206
214
|
export declare function applyAnthropicUsageExtras(usage: Usage, source: AnthropicUsageLike): void;
|
|
215
|
+
/**
|
|
216
|
+
* Whether this model's requests reach the official Anthropic API, resolved the
|
|
217
|
+
* way the transport resolves it — including the Foundry and
|
|
218
|
+
* `ANTHROPIC_BASE_URL` reroutes that leave `compat.officialEndpoint` stale.
|
|
219
|
+
*/
|
|
220
|
+
export declare function resolvesToOfficialAnthropicEndpoint(model: Model<"anthropic-messages">): boolean;
|
|
221
|
+
/**
|
|
222
|
+
* Whether server-side compaction (`compact-2026-01-12`) may be spoken for
|
|
223
|
+
* this model to the endpoint a request actually reaches: a model line the
|
|
224
|
+
* beta supports (`compat.supportsServerCompaction`, rule-owned in the
|
|
225
|
+
* catalog), on the official API for the first-party provider or on any
|
|
226
|
+
* endpoint that opted in through `remoteCompaction.enabled`, and never on one
|
|
227
|
+
* whose deployment contract excludes context management. The same predicate
|
|
228
|
+
* gates emitting the edit, attaching the beta, and replaying a persisted
|
|
229
|
+
* block, so a route or model change can never leave a session sending a block
|
|
230
|
+
* its endpoint rejects.
|
|
231
|
+
*/
|
|
232
|
+
export declare function supportsAnthropicCompaction(model: Model<"anthropic-messages">, effectiveBaseUrl?: string): boolean;
|
|
233
|
+
/**
|
|
234
|
+
* {@link supportsAnthropicCompaction} for a request on a caller-owned client:
|
|
235
|
+
* the endpoint is whatever the client targets (an `AnthropicVertex` client
|
|
236
|
+
* carries an Anthropic model to Vertex), never the model's own routing. SDK
|
|
237
|
+
* clients expose it as `baseURL`; a client that exposes no endpoint only
|
|
238
|
+
* compacts through an explicit `remoteCompaction.enabled` opt-in.
|
|
239
|
+
*/
|
|
240
|
+
export declare function supportsAnthropicCompactionOnClient(model: Model<"anthropic-messages">, client: AnthropicMessagesClientLike): boolean;
|
|
207
241
|
/** Detects the preserved-thinking error caused by rewriting a signed block's conversation prefix. */
|
|
208
242
|
export declare function isThinkingPrefixBindingError(message: string): boolean;
|
|
209
243
|
export declare function isInvalidThinkingSignatureError(message: string): boolean;
|
|
@@ -250,9 +284,17 @@ export type AnthropicMessageParam = MessageParam;
|
|
|
250
284
|
* `fallback` content block from a prior turn be replayed on the wire;
|
|
251
285
|
* otherwise the block is dropped to avoid a 400 on non-fallback requests
|
|
252
286
|
* that don't send the beta.
|
|
287
|
+
*
|
|
288
|
+
* `opts.replayCompaction` — replay a user-role compaction summary that
|
|
289
|
+
* carries an {@link AnthropicCompactionPayload} from this provider as a
|
|
290
|
+
* native `compaction` block instead of its text. The API drops every block
|
|
291
|
+
* before the compaction block, so the assistant turn carrying it may open
|
|
292
|
+
* the conversation; the request must send the compaction beta (the stream
|
|
293
|
+
* entry point adds it whenever such a payload is present).
|
|
253
294
|
*/
|
|
254
295
|
export declare function convertAnthropicMessages(messages: Message[], model: Model<"anthropic-messages">, isOAuthToken: boolean, opts?: {
|
|
255
296
|
serverSideFallbackEnabled?: boolean;
|
|
297
|
+
replayCompaction?: boolean;
|
|
256
298
|
dropAllThinking?: boolean;
|
|
257
299
|
droppedThinkingBlocks?: ReadonlySet<string>;
|
|
258
300
|
}): AnthropicMessageParam[];
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type { Message } from "../types.js";
|
|
1
|
+
import type { FetchImpl, Message } from "../types.js";
|
|
2
2
|
/**
|
|
3
3
|
* Infer whether the current request to Copilot is user-initiated or agent-initiated.
|
|
4
4
|
* Accepts `unknown[]` because providers may pass pre-converted message shapes.
|
|
@@ -11,6 +11,70 @@ export type CopilotDynamicHeaders = {
|
|
|
11
11
|
premiumRequests: CopilotPremiumRequests;
|
|
12
12
|
};
|
|
13
13
|
export declare function resolveGitHubCopilotBaseUrl(baseUrl: string | undefined, apiKey: string | undefined): string | undefined;
|
|
14
|
+
/**
|
|
15
|
+
* Opt-in `Copilot-Integration-Id` override for chat and model-policy requests.
|
|
16
|
+
* Reads `COPILOT_INTEGRATION_ID`; unset/invalid keeps the chat-surface default
|
|
17
|
+
* (`COPILOT_CHAT_INTEGRATION_ID`). Model discovery keeps the CLI identity: it
|
|
18
|
+
* unlocks enterprise/experimental models and listing models is not
|
|
19
|
+
* policy-gated the way chat completions are (#11372).
|
|
20
|
+
*/
|
|
21
|
+
export declare function resolveCopilotIntegrationIdOverride(env?: Record<string, string | undefined>): string | undefined;
|
|
22
|
+
/**
|
|
23
|
+
* Effective identity before the chat-surface default: explicit value, then
|
|
24
|
+
* request headers, then `COPILOT_INTEGRATION_ID`. Pure given its inputs, so
|
|
25
|
+
* tests inject literals instead of mutating process state.
|
|
26
|
+
*/
|
|
27
|
+
export declare function resolveCopilotRequestIdentity(headers?: Record<string, string>, explicit?: unknown, env?: Record<string, string | undefined>): string | undefined;
|
|
28
|
+
/**
|
|
29
|
+
* Stable cache key for a raw Copilot API key envelope on one effective host.
|
|
30
|
+
* Hashes the bearer with `Bun.hash` (repo-approved hashing API; same
|
|
31
|
+
* credential-scoped pattern as the GitLab Duo and Codex account keys) so token
|
|
32
|
+
* bytes never sit in the map as keys; enterprise/business routing inputs and
|
|
33
|
+
* the normalized effective base URL participate so the same token on two hosts
|
|
34
|
+
* does not share an entry.
|
|
35
|
+
*/
|
|
36
|
+
export declare function getCopilotIntegrationCacheKey(apiKeyRaw: string | undefined, baseUrl?: string): string | undefined;
|
|
37
|
+
/** Cached working identity for a cache key, if one was learned. */
|
|
38
|
+
export declare function getCachedCopilotIntegrationId(cacheKey: string | undefined): string | undefined;
|
|
39
|
+
/**
|
|
40
|
+
* Remember the identity that cleared the identity gate for a credential.
|
|
41
|
+
* Every store refreshes recency, so hot credentials survive eviction.
|
|
42
|
+
*/
|
|
43
|
+
export declare function rememberCopilotWorkingIntegrationId(cacheKey: string | undefined, integrationId: unknown): void;
|
|
44
|
+
/** Clear one cached identity, or the whole cache when no key is given. */
|
|
45
|
+
export declare function clearCopilotIntegrationCache(cacheKey?: string): void;
|
|
46
|
+
/**
|
|
47
|
+
* Reissue Copilot client-identity denials once with the other surface.
|
|
48
|
+
*
|
|
49
|
+
* Chat is the default surface (`COPILOT_CHAT_INTEGRATION_ID`) because Business
|
|
50
|
+
* organizations that gate premium models per client surface commonly allow
|
|
51
|
+
* chat while blocking CLI/agentic clients (issue #11372). Other Business and
|
|
52
|
+
* Enterprise orgs do the opposite and reject the chat identity — as an HTTP 403
|
|
53
|
+
* or, on `api.business.githubcopilot.com`, an HTTP 400 `model_not_supported`
|
|
54
|
+
* (issue #11669). Both denials retry once as the CLI. The retry fires only for
|
|
55
|
+
* requests carrying the chat default and only when the caller resolved no
|
|
56
|
+
* explicit identity — an explicit choice is never second-guessed. The denied
|
|
57
|
+
* body is drained before reissuing, and the retry carries the CLI identity so
|
|
58
|
+
* the guard passes it through: at most two requests, never a loop.
|
|
59
|
+
*
|
|
60
|
+
* When `cacheKey` is set, a 2xx retry remembers its identity via
|
|
61
|
+
* `rememberCopilotWorkingIntegrationId`, so later streams for the same
|
|
62
|
+
* credential start at the working shape. A cached CLI start that is itself
|
|
63
|
+
* denied (stale after an org-policy flip) retries once as chat and relearns.
|
|
64
|
+
* Only a 2xx retry proves its identity — 401s deny every identity equally and
|
|
65
|
+
* 408/429/5xx are transport-retryable (the transport resends the *original*
|
|
66
|
+
* headers), so those must never be recorded as working. Any non-2xx retry
|
|
67
|
+
* clears the entry instead: the next stream rediscovers rather than pinning a
|
|
68
|
+
* shape that just failed.
|
|
69
|
+
*/
|
|
70
|
+
export declare function wrapFetchForCopilotFallback(base: FetchImpl | undefined, enabled: boolean, integrationId?: unknown, cacheKey?: string,
|
|
71
|
+
/**
|
|
72
|
+
* Build-time cache provenance: the exact cached value the outgoing headers
|
|
73
|
+
* were built from. `undefined` rereads the cache at dispatch (direct
|
|
74
|
+
* callers); `null` pins "cache was empty at build" so a sibling learning
|
|
75
|
+
* mid-flight cannot change this request's retry decision.
|
|
76
|
+
*/
|
|
77
|
+
cacheSnapshot?: string | null): FetchImpl;
|
|
14
78
|
export declare function inferCopilotInitiator(messages: unknown[]): CopilotInitiator;
|
|
15
79
|
/** Check whether any message in the conversation contains image content. */
|
|
16
80
|
export declare function hasCopilotVisionInput(messages: Message[]): boolean;
|
|
@@ -37,4 +101,10 @@ export declare function buildCopilotDynamicHeaders(params: {
|
|
|
37
101
|
headers?: Record<string, string>;
|
|
38
102
|
initiatorOverride?: CopilotInitiator;
|
|
39
103
|
planTier?: string;
|
|
104
|
+
/** Enterprise login domain; Enterprise keeps the CLI identity that its private endpoint accepts. */
|
|
105
|
+
enterpriseUrl?: string;
|
|
106
|
+
/** Raw explicit identity; validated here, chat default when absent/invalid. */
|
|
107
|
+
integrationId?: unknown;
|
|
108
|
+
/** Learned working identity for this credential; explicit still wins, then this, then the defaults. */
|
|
109
|
+
cachedIntegrationId?: unknown;
|
|
40
110
|
}): CopilotDynamicHeaders;
|
|
@@ -42,5 +42,5 @@ export interface OpenAICompletionsOptions extends StreamOptions {
|
|
|
42
42
|
* assistant output commits the attempt.
|
|
43
43
|
*/
|
|
44
44
|
export declare const streamOpenAICompletions: StreamFunction<"openai-completions">;
|
|
45
|
-
export declare function parseChunkUsage(rawUsage: object, model: Model<"openai-completions">, premiumRequests: number | undefined): AssistantMessage["usage"];
|
|
45
|
+
export declare function parseChunkUsage(rawUsage: object, model: Model<"openai-completions">, premiumRequests: number | undefined, timestamp?: number): AssistantMessage["usage"];
|
|
46
46
|
export declare function convertMessages(model: Model<"openai-completions">, context: Context, compat: ResolvedOpenAICompat): ChatCompletionMessageParam[];
|
|
@@ -65,6 +65,14 @@ export interface OpenAIRequestSetup {
|
|
|
65
65
|
headers: Record<string, string>;
|
|
66
66
|
query: Record<string, string> | undefined;
|
|
67
67
|
requestHeaders: Record<string, string>;
|
|
68
|
+
/** Working-identity cache key for this credential+host; undefined off the Copilot path. */
|
|
69
|
+
copilotCacheKey: string | undefined;
|
|
70
|
+
/**
|
|
71
|
+
* Build-time cache provenance for the wrapper: the cached value the
|
|
72
|
+
* outgoing headers were built from, or `null` when the cache was empty at
|
|
73
|
+
* build. `undefined` off the Copilot path (wrapper rereads at dispatch).
|
|
74
|
+
*/
|
|
75
|
+
copilotCacheSnapshot: string | null | undefined;
|
|
68
76
|
}
|
|
69
77
|
export declare function resolveOpenAIRequestSetup(model: OpenAIRequestSetupModel, options: OpenAIRequestSetupOptions): OpenAIRequestSetup;
|
|
70
78
|
export declare function applyOpenAIServiceTier(params: {
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type { ImageContent, Model, TextContent } from "../types.js";
|
|
1
|
+
import type { Api, ImageContent, Model, TextContent } from "../types.js";
|
|
2
2
|
export declare const NON_VISION_IMAGE_PLACEHOLDER = "[image omitted: model does not support vision]";
|
|
3
3
|
export declare function partitionVisionContent(content: ReadonlyArray<TextContent | ImageContent>, supportsImages: boolean): {
|
|
4
4
|
textBlocks: TextContent[];
|
|
@@ -12,4 +12,18 @@ export declare function joinTextWithImagePlaceholder(text: string, omittedImages
|
|
|
12
12
|
* misconfigured provider descriptors or user model entries (e.g. text-only
|
|
13
13
|
* DashScope Qwen SKUs, DeepSeek models) whose endpoints reject `image_url`.
|
|
14
14
|
*/
|
|
15
|
-
export declare function isOpenAICompletionsVisionSupported(model: Model<"openai-completions">): boolean;
|
|
15
|
+
export declare function isOpenAICompletionsVisionSupported(model: Model<"openai-completions" | "openrouter">): boolean;
|
|
16
|
+
/**
|
|
17
|
+
* Whether the transport that will carry `model` sends image content on the wire.
|
|
18
|
+
*
|
|
19
|
+
* The `pi-native` transport forwards the original context (images included) to
|
|
20
|
+
* the gateway, which resolves its own model server-side, so the Chat
|
|
21
|
+
* Completions guard below never runs client-side and the declared input
|
|
22
|
+
* applies. Otherwise the OpenAI Chat Completions path applies the text-only
|
|
23
|
+
* guard, as does the OpenRouter chat fallback (`PI_OPENROUTER_RESPONSES=0`,
|
|
24
|
+
* which dispatches `openrouter` models through `streamOpenAICompletions`);
|
|
25
|
+
* every other API ships the modalities the model declares. Callers that report
|
|
26
|
+
* or gate on the wire (for example the `omp models` table) read this
|
|
27
|
+
* predicate; declared capability reads `model.input`.
|
|
28
|
+
*/
|
|
29
|
+
export declare function sendsImageInputOnWire(model: Model<Api>): boolean;
|
|
@@ -6,8 +6,8 @@ declare const museCodeKeyResponseSchema: import("@linxiraos/pi-omptype").FluentT
|
|
|
6
6
|
is_subs_active?: boolean | undefined;
|
|
7
7
|
require_payment?: boolean | undefined;
|
|
8
8
|
require_payment_action_url?: string | undefined;
|
|
9
|
-
subs_tier_id?: string | undefined;
|
|
10
|
-
subs_tier_name?: string | undefined;
|
|
9
|
+
subs_tier_id?: string | null | undefined;
|
|
10
|
+
subs_tier_name?: string | null | undefined;
|
|
11
11
|
subs_usage?: {
|
|
12
12
|
weekly?: {
|
|
13
13
|
resets_at?: string | number | undefined;
|
|
@@ -28,8 +28,8 @@ declare const museCodeKeyResponseSchema: import("@linxiraos/pi-omptype").FluentT
|
|
|
28
28
|
is_subs_active?: boolean | undefined;
|
|
29
29
|
require_payment?: boolean | undefined;
|
|
30
30
|
require_payment_action_url?: string | undefined;
|
|
31
|
-
subs_tier_id?: string | undefined;
|
|
32
|
-
subs_tier_name?: string | undefined;
|
|
31
|
+
subs_tier_id?: string | null | undefined;
|
|
32
|
+
subs_tier_name?: string | null | undefined;
|
|
33
33
|
subs_usage?: {
|
|
34
34
|
weekly?: {
|
|
35
35
|
resets_at?: string | number | undefined;
|
package/dist/types/types.d.ts
CHANGED
|
@@ -195,6 +195,18 @@ export interface CodexCompactionMetadata {
|
|
|
195
195
|
export interface CodexCompactionRequestContext extends CodexCompactionMetadata {
|
|
196
196
|
operationId: string;
|
|
197
197
|
}
|
|
198
|
+
/** Anthropic `compact_20260112` context-management edit (`compact-2026-01-12` beta). */
|
|
199
|
+
export interface AnthropicCompactionRequest {
|
|
200
|
+
/**
|
|
201
|
+
* Prompt input-token count at which the API compacts. The API enforces a
|
|
202
|
+
* 50,000-token floor and defaults to 150,000 when omitted.
|
|
203
|
+
*/
|
|
204
|
+
triggerInputTokens?: number;
|
|
205
|
+
/** Stop after the compaction block instead of continuing the response. */
|
|
206
|
+
pauseAfterCompaction?: boolean;
|
|
207
|
+
/** Custom summarization prompt; replaces the API default entirely when set. */
|
|
208
|
+
instructions?: string;
|
|
209
|
+
}
|
|
198
210
|
/** OpenAI's GPT-5.6+ explicit prompt-cache controls. */
|
|
199
211
|
export interface OpenAIPromptCacheOptions {
|
|
200
212
|
/** `explicit` disables OpenAI's automatic latest-message breakpoint. */
|
|
@@ -243,6 +255,15 @@ export interface StreamOptions {
|
|
|
243
255
|
anthropicPrefixMismatchBehavior?: "drop_block" | "error";
|
|
244
256
|
/** @internal Marks a replay-only Anthropic request that must use non-streaming `max_tokens: 0`. */
|
|
245
257
|
anthropicCacheRefreshRequest?: boolean;
|
|
258
|
+
/**
|
|
259
|
+
* Anthropic server-side compaction (`compact-2026-01-12` beta). Sends the
|
|
260
|
+
* `compact_20260112` context-management edit so the API summarizes the
|
|
261
|
+
* prompt in-band once its input reaches the trigger; the resulting summary
|
|
262
|
+
* arrives as an {@link AnthropicCompactionPayload} on the assistant message.
|
|
263
|
+
* Ignored by every other provider and by Anthropic-compatible endpoints
|
|
264
|
+
* without context-management support.
|
|
265
|
+
*/
|
|
266
|
+
anthropicCompaction?: AnthropicCompactionRequest;
|
|
246
267
|
/**
|
|
247
268
|
* Additional headers to include in provider requests.
|
|
248
269
|
* These are merged on top of model-defined headers.
|
|
@@ -703,7 +724,32 @@ export interface AnthropicMessagePayload {
|
|
|
703
724
|
name: string;
|
|
704
725
|
}>;
|
|
705
726
|
}
|
|
706
|
-
|
|
727
|
+
/**
|
|
728
|
+
* Anthropic server-side compaction summary (`compact-2026-01-12` beta).
|
|
729
|
+
*
|
|
730
|
+
* Produced by the Anthropic provider on the assistant message of a request
|
|
731
|
+
* that streamed a `compaction` content block, and attached to the user-role
|
|
732
|
+
* compaction summary message that replaces the compacted history so the
|
|
733
|
+
* provider can replay the block verbatim: the API drops every block that
|
|
734
|
+
* precedes it. `content` is the plain-text summary, so every other provider
|
|
735
|
+
* reads the message text and ignores the payload.
|
|
736
|
+
*/
|
|
737
|
+
export interface AnthropicCompactionPayload {
|
|
738
|
+
type: "anthropicCompaction";
|
|
739
|
+
/** Provider that produced the summary; only that provider replays it natively. */
|
|
740
|
+
provider: string;
|
|
741
|
+
content: string;
|
|
742
|
+
/** Opaque provider state the API attached to the block; replayed verbatim when present. */
|
|
743
|
+
encryptedContent?: string;
|
|
744
|
+
/**
|
|
745
|
+
* Harness-appended file metadata (`<files>` section) kept out of the
|
|
746
|
+
* byte-identical block. Replayed as a user message after the native block:
|
|
747
|
+
* the converter replaces the summary message with the block and skips its
|
|
748
|
+
* text, so without this the metadata would be invisible to this provider.
|
|
749
|
+
*/
|
|
750
|
+
filesText?: string;
|
|
751
|
+
}
|
|
752
|
+
export type ProviderPayload = OpenAIResponsesHistoryPayload | AnthropicMessagePayload | AnthropicCompactionPayload;
|
|
707
753
|
/** Provider-reported rewrite applied to request content before inference. */
|
|
708
754
|
export interface ProviderInputTransformation {
|
|
709
755
|
type: string;
|
|
@@ -72,3 +72,44 @@ export type DemotedThinkingCarrier = object & {
|
|
|
72
72
|
};
|
|
73
73
|
/** True for text blocks synthesized by cross-model thinking demotion. */
|
|
74
74
|
export declare function isDemotedThinking(block: DemotedThinkingCarrier | null | undefined): boolean;
|
|
75
|
+
/**
|
|
76
|
+
* Marks an Anthropic wire message that was serialized from a source
|
|
77
|
+
* `role: "user"` message.
|
|
78
|
+
*
|
|
79
|
+
* The wire role alone cannot answer this. `convertAnthropicMessages` emits
|
|
80
|
+
* `role: "user"` for several things that are not a conversational turn:
|
|
81
|
+
* `developer` messages on models without mid-conversation `system` support,
|
|
82
|
+
* `tool_result` runs, and the synthetic `Continue.` pads inserted between
|
|
83
|
+
* adjacent assistants. Prompt-cache decimation counts conversational turns,
|
|
84
|
+
* so it reads this marker instead of guessing from wire content.
|
|
85
|
+
*
|
|
86
|
+
* Symbol-keyed so the marker never persists across the JSONL round-trip and
|
|
87
|
+
* never reaches the wire.
|
|
88
|
+
*/
|
|
89
|
+
export declare const kConversationalUser: unique symbol;
|
|
90
|
+
/** Carries the conversational-user marker without exposing a string-keyed property. */
|
|
91
|
+
export type ConversationalUserCarrier = object & {
|
|
92
|
+
[kConversationalUser]?: boolean;
|
|
93
|
+
};
|
|
94
|
+
/** True for wire messages serialized from a source `role: "user"` message. */
|
|
95
|
+
export declare function isConversationalUser(message: ConversationalUserCarrier | null | undefined): boolean;
|
|
96
|
+
/**
|
|
97
|
+
* Marks a `role: "user"` message that `transformMessages` synthesized rather
|
|
98
|
+
* than one the user sent.
|
|
99
|
+
*
|
|
100
|
+
* The stale-tool-result note is deliberately emitted as `user` so no provider
|
|
101
|
+
* elevates untrusted tool output to instruction priority, which leaves it
|
|
102
|
+
* indistinguishable from a real turn by role alone. Prompt-cache decimation
|
|
103
|
+
* must not count it, or an orphan result appearing or disappearing in a
|
|
104
|
+
* compacted history shifts every later checkpoint.
|
|
105
|
+
*
|
|
106
|
+
* Symbol-keyed so the marker never persists across the JSONL round-trip and
|
|
107
|
+
* never reaches the wire.
|
|
108
|
+
*/
|
|
109
|
+
export declare const kSyntheticUser: unique symbol;
|
|
110
|
+
/** Carries the synthetic-user marker without exposing a string-keyed property. */
|
|
111
|
+
export type SyntheticUserCarrier = object & {
|
|
112
|
+
[kSyntheticUser]?: boolean;
|
|
113
|
+
};
|
|
114
|
+
/** True for `user` messages synthesized by message transformation. */
|
|
115
|
+
export declare function isSyntheticUser(message: SyntheticUserCarrier | null | undefined): boolean;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@linxiraos/pi-ai",
|
|
3
|
-
"version": "1.1.
|
|
3
|
+
"version": "1.1.15",
|
|
4
4
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"ai",
|
|
@@ -124,11 +124,11 @@
|
|
|
124
124
|
"fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
|
|
125
125
|
},
|
|
126
126
|
"dependencies": {
|
|
127
|
-
"@linxiraos/pi-omptype": "1.1.
|
|
128
|
-
"@linxiraos/pi-catalog": "1.1.
|
|
129
|
-
"@linxiraos/pi-natives": "1.1.
|
|
130
|
-
"@linxiraos/pi-utils": "1.1.
|
|
131
|
-
"@linxiraos/pi-wire": "1.1.
|
|
127
|
+
"@linxiraos/pi-omptype": "1.1.15",
|
|
128
|
+
"@linxiraos/pi-catalog": "1.1.15",
|
|
129
|
+
"@linxiraos/pi-natives": "1.1.15",
|
|
130
|
+
"@linxiraos/pi-utils": "1.1.15",
|
|
131
|
+
"@linxiraos/pi-wire": "1.1.15"
|
|
132
132
|
},
|
|
133
133
|
"devDependencies": {
|
|
134
134
|
"@types/bun": "^1.3.14"
|
package/src/auth-storage.ts
CHANGED
|
@@ -51,6 +51,7 @@ import type {
|
|
|
51
51
|
} from "./usage";
|
|
52
52
|
import { resolveUsedFraction } from "./usage";
|
|
53
53
|
import { alibabaTokenPlanRankingStrategy, alibabaTokenPlanUsageProvider } from "./usage/alibaba-token-plan";
|
|
54
|
+
import { charmHyperUsageProvider } from "./usage/charm-hyper";
|
|
54
55
|
import { claudeRankingStrategy, claudeUsageProvider } from "./usage/claude";
|
|
55
56
|
import { clinePassUsageProvider } from "./usage/cline-pass";
|
|
56
57
|
import { cursorUsageProvider } from "./usage/cursor";
|
|
@@ -677,6 +678,7 @@ const DEFAULT_USAGE_PROVIDERS: UsageProvider[] = [
|
|
|
677
678
|
syntheticUsageProvider,
|
|
678
679
|
xaiOauthUsageProvider,
|
|
679
680
|
devinUsageProvider,
|
|
681
|
+
charmHyperUsageProvider,
|
|
680
682
|
];
|
|
681
683
|
|
|
682
684
|
const DEFAULT_USAGE_PROVIDER_MAP = new Map<Provider, UsageProvider>(
|
package/src/error/finalize.ts
CHANGED
|
@@ -45,7 +45,8 @@ export interface FinalizeResult {
|
|
|
45
45
|
*/
|
|
46
46
|
export async function finalize(error: unknown, opts: FinalizeOptions = {}): Promise<FinalizeResult> {
|
|
47
47
|
const aborted = opts.abortTracker ? opts.abortTracker.wasCallerAbort() : opts.signal?.aborted === true;
|
|
48
|
-
const
|
|
48
|
+
const errorStatus = status(error);
|
|
49
|
+
const currentStatus = errorStatus ?? opts.capturedErrorResponse?.status;
|
|
49
50
|
|
|
50
51
|
let message: string;
|
|
51
52
|
try {
|
|
@@ -55,11 +56,17 @@ export async function finalize(error: unknown, opts: FinalizeOptions = {}): Prom
|
|
|
55
56
|
message = error instanceof Error ? error.message : String(error);
|
|
56
57
|
}
|
|
57
58
|
|
|
59
|
+
// A captured status is transport context for the original error. Put it at
|
|
60
|
+
// the root of the classification cause chain so terminal 4xx policy governs
|
|
61
|
+
// nested diagnostics before they can contribute transient flags.
|
|
62
|
+
const classificationError =
|
|
63
|
+
errorStatus === undefined && currentStatus !== undefined ? { status: currentStatus, cause: error } : error;
|
|
64
|
+
|
|
58
65
|
const id = classifyMessage({
|
|
59
66
|
api: opts.api,
|
|
60
67
|
provider: opts.provider,
|
|
61
68
|
model: opts.model,
|
|
62
|
-
errorId: classify(
|
|
69
|
+
errorId: classify(classificationError, opts.api),
|
|
63
70
|
errorMessage: message,
|
|
64
71
|
errorStatus: currentStatus,
|
|
65
72
|
});
|