@oh-my-pi/pi-ai 18.0.11 → 18.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (96) hide show
  1. package/CHANGELOG.md +23 -1
  2. package/dist/types/auth-storage.d.ts +1 -1
  3. package/dist/types/error/flags.d.ts +7 -0
  4. package/dist/types/providers/gitlab-duo.d.ts +2 -22
  5. package/dist/types/providers/kimi.d.ts +1 -1
  6. package/dist/types/providers/mock.d.ts +1 -0
  7. package/dist/types/providers/openai-shared.d.ts +21 -11
  8. package/dist/types/providers/vision-guard.d.ts +0 -26
  9. package/dist/types/registry/api-key-login.d.ts +2 -0
  10. package/dist/types/registry/api-key-validation.d.ts +2 -0
  11. package/dist/types/registry/cline-pass.d.ts +7 -0
  12. package/dist/types/registry/registry.d.ts +5 -1
  13. package/dist/types/registry/zai.d.ts +1 -1
  14. package/dist/types/stream.d.ts +4 -1
  15. package/dist/types/types.d.ts +6 -4
  16. package/dist/types/usage/cline-pass.d.ts +2 -0
  17. package/dist/types/usage/devin.d.ts +12 -0
  18. package/dist/types/utils/harmony-leak.d.ts +13 -4
  19. package/dist/types/utils/http-inspector.d.ts +19 -0
  20. package/dist/types/utils/schema/normalize.d.ts +10 -0
  21. package/dist/types/utils/schema/strict-tool-validation.d.ts +3 -3
  22. package/dist/types/utils/stream-markup-healing.d.ts +2 -5
  23. package/dist/types/utils/thinking-loop.d.ts +3 -6
  24. package/package.json +9 -9
  25. package/src/auth-broker/client.ts +1 -1
  26. package/src/auth-broker/remote-store.ts +1 -1
  27. package/src/auth-broker/server.ts +4 -4
  28. package/src/auth-gateway/server.ts +5 -5
  29. package/src/auth-storage.ts +12 -16
  30. package/src/dialect/deepseek.ts +1 -1
  31. package/src/dialect/gemini.ts +1 -1
  32. package/src/dialect/gemma.ts +1 -1
  33. package/src/dialect/glm.ts +1 -1
  34. package/src/dialect/harmony.ts +1 -1
  35. package/src/dialect/kimi.ts +1 -1
  36. package/src/dialect/rendering.ts +2 -2
  37. package/src/dialect/thinking.ts +1 -1
  38. package/src/error/auth-classify.ts +4 -1
  39. package/src/error/classes.ts +1 -2
  40. package/src/error/flags.ts +12 -0
  41. package/src/error/format.ts +4 -0
  42. package/src/error/rate-limit.ts +9 -1
  43. package/src/providers/amazon-bedrock.ts +2 -0
  44. package/src/providers/anthropic-client.ts +1 -1
  45. package/src/providers/anthropic-messages-server.ts +4 -5
  46. package/src/providers/anthropic.ts +41 -42
  47. package/src/providers/aws-credentials.ts +2 -2
  48. package/src/providers/azure-openai-responses.ts +2 -1
  49. package/src/providers/cowork-fetch.ts +2 -3
  50. package/src/providers/cursor.ts +14 -9
  51. package/src/providers/devin.ts +139 -68
  52. package/src/providers/gitlab-duo-workflow.ts +10 -12
  53. package/src/providers/gitlab-duo.ts +18 -160
  54. package/src/providers/google-gemini-cli.ts +16 -18
  55. package/src/providers/google-shared.ts +6 -37
  56. package/src/providers/google-vertex.ts +2 -2
  57. package/src/providers/google.ts +2 -2
  58. package/src/providers/kimi.ts +6 -2
  59. package/src/providers/mock.ts +3 -0
  60. package/src/providers/ollama.ts +8 -22
  61. package/src/providers/openai-anthropic-shim.ts +1 -1
  62. package/src/providers/openai-codex/request-transformer.ts +5 -4
  63. package/src/providers/openai-codex-responses.ts +7 -13
  64. package/src/providers/openai-completions.ts +26 -28
  65. package/src/providers/openai-responses.ts +15 -46
  66. package/src/providers/openai-shared.ts +71 -36
  67. package/src/providers/pi-native-client.ts +1 -1
  68. package/src/providers/transform-messages.ts +186 -38
  69. package/src/providers/vision-guard.ts +1 -68
  70. package/src/registry/api-key-login.ts +4 -0
  71. package/src/registry/api-key-validation.ts +4 -2
  72. package/src/registry/cline-pass.ts +27 -0
  73. package/src/registry/cloudflare-ai-gateway.ts +18 -19
  74. package/src/registry/oauth/device-code.ts +1 -2
  75. package/src/registry/oauth/zai.ts +12 -2
  76. package/src/registry/registry.ts +2 -0
  77. package/src/registry/zai.ts +4 -4
  78. package/src/stream.ts +30 -22
  79. package/src/types.ts +18 -10
  80. package/src/usage/claude.ts +3 -2
  81. package/src/usage/cline-pass.ts +167 -0
  82. package/src/usage/devin.ts +310 -0
  83. package/src/usage/gemini.ts +2 -32
  84. package/src/usage/google-antigravity.ts +2 -12
  85. package/src/usage/openai-codex.ts +2 -1
  86. package/src/utils/aws-profile.ts +1 -1
  87. package/src/utils/harmony-leak.ts +8 -8
  88. package/src/utils/http-inspector.ts +40 -0
  89. package/src/utils/proxy.ts +2 -3
  90. package/src/utils/request-debug.ts +1 -1
  91. package/src/utils/schema/fields.ts +1 -1
  92. package/src/utils/schema/normalize.ts +173 -16
  93. package/src/utils/schema/strict-tool-validation.ts +4 -4
  94. package/src/utils/stream-markup-healing.ts +4 -26
  95. package/src/utils/thinking-loop.ts +12 -17
  96. package/src/utils.ts +13 -1
package/CHANGELOG.md CHANGED
@@ -2,6 +2,28 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [18.1.0] - 2026-09-01
6
+
7
+ ### Added
8
+
9
+ - Added an optional `completeSimple` callback that observes every result, including results from internal thinking-loop retries.
10
+ - Added compatibility options for Anthropic-compatible proxies that reject `context_management` and OpenAI Responses proxies that provide incomplete reasoning-summary streams.
11
+ - Added ClinePass API-key authentication via the official `CLINE_API_KEY` environment variable, with account validation, actionable subscription and quota errors, support for eligible ClinePass model rosters, and rolling quota-window reporting in `omp usage`.
12
+ - Added Devin router-model support, including assignment of the concrete model before each request, routed-model metadata, credit usage reporting, and plan, quota-window, and account details through `omp usage`.
13
+
14
+ ### Changed
15
+
16
+ - Provider behavior is now driven by each model's resolved compatibility, identity, thinking, and behavior policies rather than model-name matching, improving support for model-specific request formatting, vision, reasoning, routing, pricing, and quota handling.
17
+ - Devin integrations now use the current released CLI identity and support parallel tool calls when the model declares that capability.
18
+
19
+ ### Fixed
20
+
21
+ - Fixed OpenAI remote-compaction replay for persisted sessions, allowing sessions with previously stored compaction items to resume successfully.
22
+ - Fixed Cursor Fable requests failing when advertised tools used JSON Schema composition keywords.
23
+ - Fixed Z.AI (GLM Coding Plan) browser sign-in by using the registered CLI callback address.
24
+ - Fixed OpenAI Codex/Responses tool results being lost when composite call identifiers could not be paired with the corresponding assistant call.
25
+ - Fixed native OpenAI Responses history replay becoming stuck on malformed or truncated function-call arguments; invalid history items are now discarded so the session can recover.
26
+
5
27
  ## [18.0.11] - 2026-08-29
6
28
 
7
29
  ### Fixed
@@ -2015,7 +2037,7 @@
2015
2037
  - Fixed Anthropic-compatible proxies that omit `usage`/`delta` objects from `message_start`/`message_delta`/`content_block_*` envelopes crashing the turn with an unretryable `TypeError`; the missing payloads now degrade to logged envelope anomalies like every other malformed-frame case.
2016
2038
  - Fixed `applyPromptCaching` placing `cache_control` on `thinking`/`redacted_thinking` blocks — Anthropic rejects that with a 400. A thinking-only assistant turn inside the trailing cache window (e.g. followed by the synthetic `Continue.` pad) no longer receives a breakpoint.
2017
2039
  - Fixed consecutive `assistant` params reaching the wire when an empty user/developer turn between two assistant turns was dropped by the converter (e.g. an empty "nudge" submission after a length-truncated reply); Anthropic 400s on non-alternating assistant turns, and the broken triple replayed on every subsequent request. A `user: "Continue."` separator is now inserted, mirroring the trailing-prefill fallback.
2018
- - Fixed `supportsAdaptiveThinkingDisplay` misparsing bare dated Opus ids: `claude-opus-4-20250514` (Opus 4.0) parsed as minor `20250514` ≥ 4.7, which silently dropped the `interleaved-thinking-2025-05-14` beta for API-key Opus 4.0 requests.
2040
+ - Fixed adaptive-display classification misparsing bare dated Opus ids: `claude-opus-4-20250514` (Opus 4.0) parsed as minor `20250514` ≥ 4.7, which silently dropped the `interleaved-thinking-2025-05-14` beta for API-key Opus 4.0 requests.
2019
2041
  - Fixed `output_config.effort` shipping without the `effort-2025-11-24` beta on thinking-off requests against adaptive-only Claude models (the effort:"low" pin), and the mid-conversation `system` role shipping without `mid-conversation-system-2026-04-07` on API-key and OAuth-utility requests; both betas are now added whenever the request can carry the corresponding field.
2020
2042
  - Fixed GitHub Copilot anthropic-messages requests going out with no `Content-Type` and no `anthropic-version` header — the copilot branch builds its headers from scratch and Bun's fetch does not default `Content-Type` for string bodies. Both headers are now pinned to match every other branch.
2021
2043
  - Fixed Anthropic client/provider retry multiplication: with the first-event watchdog disabled (`PI_STREAM_FIRST_EVENT_TIMEOUT_MS=0`), the client's internal `maxRetries: 5` reactivated and stacked with the provider loop's 3 retries — up to 24 wire attempts with double backoff. The provider now pins per-request `maxRetries: 0` unconditionally.
@@ -3,7 +3,7 @@ import type { OAuthAuthInfo, OAuthController, OAuthCredentials, OAuthProviderId
3
3
  import type { Provider } from "./types.js";
4
4
  import type { ClientUsageIdentity, ClientUsageReport, ClientUsageSummary, CredentialRankingStrategy, ObservedUsageEntry, UsageHistoryEntry, UsageHistoryQuery, UsageLogger, UsageProvider, UsageReport } from "./usage.js";
5
5
  import { type CodexResetConsumeCode, type CodexResetCredit } from "./usage/openai-codex-reset.js";
6
- export { isSqliteBusyError, isSqliteCorruptionError, SqliteAuthCredentialStore, } from "./auth/sqlite-credential-store.js";
6
+ export { isSqliteBusyError, isSqliteCorruptionError, SqliteAuthCredentialStore } from "./auth/sqlite-credential-store.js";
7
7
  export type ApiKeyCredential = {
8
8
  type: "api_key";
9
9
  key: string;
@@ -80,6 +80,13 @@ export declare function isGrammarError(error: unknown): boolean;
80
80
  * Accessor for {@link Flag.FastModeUnsupported}.
81
81
  */
82
82
  export declare function isFastModeUnsupported(error: unknown): boolean;
83
+ /**
84
+ * Cline's gateway gates some roster entries (certain free-tier models) to its
85
+ * own product surfaces with a 403. The API key is valid — the restriction is
86
+ * per-model client policy — so it must neither rotate sibling credentials (they
87
+ * fail identically) nor surface as an auth failure.
88
+ */
89
+ export declare function isClinePassSurfaceGateMessage(errorMessage: string | undefined): boolean;
83
90
  /**
84
91
  * GitHub Copilot 400 `model_not_supported` response for a model advertised by
85
92
  * `/models` — transient fleet skew, not a malformed request. Reads the
@@ -1,27 +1,7 @@
1
+ import { getGitLabDuoModels } from "@oh-my-pi/pi-catalog/provider-models";
1
2
  import type { Api, Context, Model, SimpleStreamOptions } from "../types.js";
2
3
  import { AssistantMessageEventStream } from "../utils/event-stream.js";
3
- type GitLabProvider = "anthropic" | "openai";
4
- type GitLabOpenAIApiType = "chat" | "responses";
5
- export type GitLabModelMapping = {
6
- provider: GitLabProvider;
7
- model: string;
8
- openaiApiType?: GitLabOpenAIApiType;
9
- name: string;
10
- reasoning: boolean;
11
- input: ("text" | "image")[];
12
- cost: {
13
- input: number;
14
- output: number;
15
- cacheRead: number;
16
- cacheWrite: number;
17
- };
18
- contextWindow: number;
19
- maxTokens: number;
20
- };
21
- export declare const MODEL_MAPPINGS: Record<string, GitLabModelMapping>;
22
- export declare function getModelMapping(modelId: string): GitLabModelMapping | undefined;
23
- export declare function getGitLabDuoModels(): Model<Api>[];
4
+ export { getGitLabDuoModels };
24
5
  export declare function clearGitLabDuoDirectAccessCache(): void;
25
6
  export declare function isGitLabDuoModel(model: Model<Api>): boolean;
26
7
  export declare function streamGitLabDuo(model: Model<Api>, context: Context, options?: SimpleStreamOptions): AssistantMessageEventStream;
27
- export {};
@@ -13,7 +13,7 @@ import type { AssistantMessageEventStream } from "../utils/event-stream.js";
13
13
  import { type OpenAIAnthropicApiFormat, type OpenAIAnthropicShimOptions } from "./openai-anthropic-shim.js";
14
14
  export type KimiApiFormat = OpenAIAnthropicApiFormat;
15
15
  export interface KimiOptions extends OpenAIAnthropicShimOptions {
16
- /** Explicit API format override. Defaults to the model's discovered protocol. */
16
+ /** Explicit API format override. Defaults to the model's resolved protocol policy. */
17
17
  format?: KimiApiFormat;
18
18
  }
19
19
  /**
@@ -150,6 +150,7 @@ export declare class MockModel implements Model<MockApi> {
150
150
  readonly contextWindow: number;
151
151
  readonly maxTokens: number;
152
152
  readonly compat: undefined;
153
+ readonly identity: Model["identity"];
153
154
  /** Recorded calls in invocation order. */
154
155
  readonly calls: MockCall[];
155
156
  iterator?: Iterator<MockHandler> | AsyncIterator<MockHandler>;
@@ -20,6 +20,7 @@ export declare const NO_AUTH_SENTINEL = "N/A";
20
20
  export interface OpenAIModelIdentity {
21
21
  provider: string;
22
22
  id: string;
23
+ identity?: Model["identity"];
23
24
  baseUrl?: string;
24
25
  }
25
26
  export interface OpenAIStrictToolsScope {
@@ -68,7 +69,7 @@ export interface OpenAIRequestSetup {
68
69
  export declare function resolveOpenAIRequestSetup(model: OpenAIRequestSetupModel, options: OpenAIRequestSetupOptions): OpenAIRequestSetup;
69
70
  export declare function applyOpenAIServiceTier(params: {
70
71
  service_tier?: ServiceTier | null | undefined;
71
- }, serviceTier: ServiceTier | null | undefined, model: Pick<Model, "provider" | "api" | "id">): void;
72
+ }, serviceTier: ServiceTier | null | undefined, model: Pick<Model, "provider" | "api" | "identity">): void;
72
73
  /**
73
74
  * Adjust resolved cost by the service tier OpenAI actually billed — parity with
74
75
  * Codex (`applyCodexServiceTierPricing`), but with the standard (non-Codex)
@@ -77,9 +78,9 @@ export declare function applyOpenAIServiceTier(params: {
77
78
  * Responses biller) so an echoed `service_tier` from an Azure/OpenRouter/Copilot
78
79
  * proxy can never skew those costs.
79
80
  */
80
- export declare function applyOpenAIResponsesServiceTierCost(model: Pick<Model, "provider">, usage: AssistantMessage["usage"], responseServiceTier: unknown, requestServiceTier: ServiceTier | null | undefined): void;
81
- /** Reconcile token-price estimates with OpenRouter's authoritative account charge. */
82
- export declare function applyOpenRouterReportedCost(model: Pick<Model, "provider">, usage: Usage, rawUsage: unknown): void;
81
+ export declare function applyOpenAIResponsesServiceTierCost(model: Pick<Model, "provider" | "serviceTierCost">, usage: AssistantMessage["usage"], responseServiceTier: unknown, requestServiceTier: ServiceTier | null | undefined): void;
82
+ /** Reconcile token-price estimates with a gateway's authoritative account charge. */
83
+ export declare function applyProviderReportedCost(model: Pick<Model, "provider">, usage: Usage, rawUsage: unknown): void;
83
84
  export interface OpenAIUsageAccountingInput {
84
85
  promptTokens: number;
85
86
  outputTokens: number;
@@ -112,7 +113,6 @@ export declare function clearOpenAIStrictToolsState(state: OpenAIStrictToolsStat
112
113
  export declare function getOpenAIStrictToolsScope(model: OpenAIModelIdentity, resolvedBaseUrl: string | undefined): OpenAIStrictToolsScope;
113
114
  export declare function isStrictToolsDisabledForScope(state: OpenAIStrictToolsState | undefined, scope: OpenAIStrictToolsScope | undefined): boolean;
114
115
  export declare function disableStrictToolsForScope(state: OpenAIStrictToolsState | undefined, scope: OpenAIStrictToolsScope | undefined): void;
115
- export declare function isOpenRouterAnthropicModel(model: OpenAIModelIdentity): boolean;
116
116
  /**
117
117
  * Append an OpenRouter routing-variant suffix (e.g. `:nitro`, `:floor`, `:online`, `:exacto`)
118
118
  * to a model id when no explicit variant is already present. A variant is considered
@@ -240,8 +240,8 @@ export type OpenAICompletionsParams = Omit<ChatCompletionCreateParamsStreaming,
240
240
  };
241
241
  reasoning?: {
242
242
  effort?: string;
243
- } | {
244
- enabled: false;
243
+ enabled?: boolean;
244
+ max_tokens?: number;
245
245
  };
246
246
  venice_parameters?: {
247
247
  disable_thinking?: boolean;
@@ -262,6 +262,7 @@ export type OpenAICompletionsParams = Omit<ChatCompletionCreateParamsStreaming,
262
262
  export interface ChatCompletionsReasoningOptions {
263
263
  reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
264
264
  disableReasoning?: boolean;
265
+ thinkingBudgets?: Partial<Record<Effort, number>>;
265
266
  }
266
267
  export type OpenAICompatEndpoint = "chat-completions" | "responses";
267
268
  export type OpenAIReasoningDisableReason = "caller" | "forced-tool-choice" | "tool-choice" | "not-requested";
@@ -348,9 +349,9 @@ export declare function disableChatCompletionsReasoningForDialect(params: OpenAI
348
349
  * Provider-specific Chat Completions output clamp.
349
350
  *
350
351
  * Most OpenAI-compatible endpoints retain the conservative 64k ceiling from
351
- * {@link resolveOpenAIOutputTokenParam}. Z.AI/GLM-5.2 reasoning and native
352
- * Moonshot K3 explicitly accept their full advertised model caps, so those
353
- * routes clamp to `model.maxTokens` instead.
352
+ * {@link resolveOpenAIOutputTokenParam}. ClinePass, Z.AI/GLM-5.2 reasoning,
353
+ * and native Moonshot K3 explicitly accept their full advertised model caps,
354
+ * so those routes clamp to `model.maxTokens` instead.
354
355
  */
355
356
  export declare function resolveOpenAICompletionsOutputClamp(model: Model<"openai-completions">, compat: ResolvedOpenAICompat): number | undefined;
356
357
  /**
@@ -645,7 +646,7 @@ type CommonSamplingOptions = Pick<StreamOptions, "temperature" | "topP" | "topK"
645
646
  * can let the upstream apply its own default instead of 400-ing on `maxTokens` values that
646
647
  * reflect the model's context window rather than the upstream output limit.
647
648
  */
648
- export declare function applyCommonResponsesSamplingParams<P extends CommonResponsesParams>(params: P, options: CommonSamplingOptions | undefined, model: Pick<Model, "provider" | "api" | "id" | "omitMaxOutputTokens" | "maxTokens"> & {
649
+ export declare function applyCommonResponsesSamplingParams<P extends CommonResponsesParams>(params: P, options: CommonSamplingOptions | undefined, model: Pick<Model, "provider" | "api" | "id" | "omitMaxOutputTokens" | "maxTokens" | "identity"> & {
649
650
  compat: Pick<ResolvedOpenAISharedCompat, "supportsSamplingParams" | "supportsPenaltyAndStopParams">;
650
651
  }): void;
651
652
  type ReasoningOptions = {
@@ -654,6 +655,15 @@ type ReasoningOptions = {
654
655
  disableReasoning?: boolean;
655
656
  toolChoice?: unknown;
656
657
  };
658
+ /**
659
+ * Resolve the caller's reasoning-summary request against catalog compat.
660
+ * Hosts that reject `reasoning.summary` get an explicit `null` (wire omission)
661
+ * whenever reasoning is engaged, so the policy never fills the `"auto"` default.
662
+ */
663
+ export declare function resolveReasoningSummaryOption(model: Model<"openai-responses" | "azure-openai-responses" | "openai-codex-responses">, options: {
664
+ reasoning?: string;
665
+ reasoningSummary?: "auto" | "detailed" | "concise" | null;
666
+ } | undefined): "auto" | "detailed" | "concise" | null | undefined;
657
667
  export interface ApplyResponsesCompatPolicyOptions {
658
668
  reasoningSummary?: "auto" | "detailed" | "concise" | null;
659
669
  mapEffort?: (effort: string) => string;
@@ -6,32 +6,6 @@ export declare function partitionVisionContent(content: ReadonlyArray<TextConten
6
6
  omittedImages: boolean;
7
7
  };
8
8
  export declare function joinTextWithImagePlaceholder(text: string, omittedImages: boolean): string;
9
- /**
10
- * Detect known text-only Qwen models served via Alibaba DashScope's consumer
11
- * `compatible-mode` endpoint that the upstream chat-completions API rejects
12
- * multimodal content arrays for. The compatible-mode endpoint also serves
13
- * multimodal Qwen SKUs without `vl` in the id (e.g. `qwen3.7-plus`), so this
14
- * guard only covers families verified to be text-only for issue #1859:
15
- * `qwen*-coder*` and `qwen*-max` up to and including `qwen3.7-max`.
16
- *
17
- * Qwen-Max became multimodal at `qwen3.8-max` (image input, issue #8019), so
18
- * `-max` SKUs at version 3.8 or newer are excluded — otherwise the override
19
- * would strip images from a genuinely vision-capable flagship (issue #8305).
20
- *
21
- * Used as a defensive override in `convertMessages` so a misconfigured custom
22
- * provider (issue #1859) can't drive the request into an unrecoverable 400.
23
- */
24
- export declare function isDashscopeCompatibleModeTextOnlyQwen(model: Model<"openai-completions">): boolean;
25
- /**
26
- * Detect known text-only DeepSeek models served via OpenAI-compatible Chat
27
- * Completions endpoints whose server-side deserializers reject `image_url`
28
- * content parts with HTTP 400 (`unknown variant \`image_url\`, expected \`text\``).
29
- *
30
- * Used as a defensive override in `convertMessages` so misconfigured model
31
- * definitions or user overrides (e.g. `models.yml` claiming `input: [text, image]`)
32
- * do not crash the session with an unrecoverable 400.
33
- */
34
- export declare function isTextOnlyDeepSeek(model: Model<"openai-completions">): boolean;
35
9
  /**
36
10
  * Evaluates whether an OpenAI-compatible Chat Completions model genuinely
37
11
  * supports multimodal image inputs on the wire. Defensive guards override
@@ -13,6 +13,8 @@ type ChatCompletionsValidation = {
13
13
  model: string;
14
14
  /** Treat an authenticated 401 (`invalid_model`) as a valid key. */
15
15
  tolerateModelDenied?: boolean;
16
+ maxTokensField?: "max_tokens" | "max_completion_tokens";
17
+ maxTokens?: number;
16
18
  };
17
19
  type AnthropicMessagesValidation = {
18
20
  kind: "anthropic-messages";
@@ -4,6 +4,8 @@ type OpenAICompatibleValidationOptions = {
4
4
  apiKey: string;
5
5
  baseUrl: string;
6
6
  model: string;
7
+ maxTokensField?: "max_tokens" | "max_completion_tokens";
8
+ maxTokens?: number;
7
9
  signal?: AbortSignal;
8
10
  fetch?: FetchImpl;
9
11
  tolerateModelDenied?: boolean;
@@ -0,0 +1,7 @@
1
+ import type { OAuthLoginCallbacks } from "./oauth/types.js";
2
+ export declare const loginClinePass: (options: import("./oauth/index.js").OAuthController) => Promise<string>;
3
+ export declare const clinePassProvider: {
4
+ readonly id: "cline-pass";
5
+ readonly name: "ClinePass";
6
+ readonly login: (callbacks: OAuthLoginCallbacks) => Promise<string>;
7
+ };
@@ -69,6 +69,10 @@ declare const ALL: ({
69
69
  readonly id: "cerebras";
70
70
  readonly name: "Cerebras";
71
71
  readonly login: (cb: import("./oauth/index.js").OAuthLoginCallbacks) => Promise<string>;
72
+ } | {
73
+ readonly id: "cline-pass";
74
+ readonly name: "ClinePass";
75
+ readonly login: (callbacks: import("./oauth/index.js").OAuthLoginCallbacks) => Promise<string>;
72
76
  } | {
73
77
  readonly id: "cloudflare-ai-gateway";
74
78
  readonly name: "Cloudflare AI Gateway";
@@ -365,7 +369,7 @@ declare const ALL: ({
365
369
  readonly id: "zai-coding-plan";
366
370
  readonly name: "Z.AI (GLM Coding Plan · Sign in)";
367
371
  readonly storeCredentialsAs: "zai";
368
- readonly callbackPort: 54548;
372
+ readonly callbackPort: 9999;
369
373
  readonly pasteCodeFlow: true;
370
374
  readonly login: (cb: import("./oauth/index.js").OAuthLoginCallbacks) => Promise<string | import("./oauth/index.js").OAuthCredentials>;
371
375
  } | {
@@ -9,7 +9,7 @@ export declare const zaiCodingPlanProvider: {
9
9
  readonly id: "zai-coding-plan";
10
10
  readonly name: "Z.AI (GLM Coding Plan · Sign in)";
11
11
  readonly storeCredentialsAs: "zai";
12
- readonly callbackPort: 54548;
12
+ readonly callbackPort: 9999;
13
13
  readonly pasteCodeFlow: true;
14
14
  readonly login: (cb: OAuthLoginCallbacks) => Promise<string | import("./oauth/index.js").OAuthCredentials>;
15
15
  };
@@ -46,7 +46,10 @@ export declare function listProvidersWithEnvKey(): string[];
46
46
  export declare function stream<TApi extends Api>(model: Model<TApi>, context: Context, options?: OptionsForApi<TApi>): AssistantMessageEventStream;
47
47
  export declare function complete<TApi extends Api>(model: Model<TApi>, context: Context, options?: OptionsForApi<TApi>): Promise<AssistantMessage>;
48
48
  export declare function streamSimple<TApi extends Api>(model: Model<TApi>, context: Context, options?: SimpleStreamOptions): AssistantMessageEventStream;
49
- export declare function completeSimple<TApi extends Api>(model: Model<TApi>, context: Context, options?: SimpleStreamOptions): Promise<AssistantMessage>;
49
+ export declare function completeSimple<TApi extends Api>(model: Model<TApi>, context: Context, options?: SimpleStreamOptions & {
50
+ /** Receives every completed result, including results retried by the thinking-loop guard. */
51
+ onAttempt?: (message: AssistantMessage) => void;
52
+ }): Promise<AssistantMessage>;
50
53
  export declare const OUTPUT_FALLBACK_BUFFER = 4000;
51
54
  export declare const ANTHROPIC_THINKING: Record<Effort, number>;
52
55
  export declare function mapAnthropicToolChoice(choice?: ToolChoice): AnthropicOptions["toolChoice"];
@@ -103,7 +103,7 @@ export type ServiceTierFamily = "openai" | "anthropic" | "google";
103
103
  * models mid-session.
104
104
  */
105
105
  export type ServiceTierByFamily = Partial<Record<ServiceTierFamily, ServiceTier>>;
106
- type ServiceTierModel = Pick<Model, "provider" | "api" | "id">;
106
+ type ServiceTierModel = Pick<Model, "provider" | "api" | "identity">;
107
107
  /**
108
108
  * Classify a model into the service-tier family whose knob governs it, or
109
109
  * `undefined` when the model exposes no serving-priority control.
@@ -120,7 +120,7 @@ export declare function serviceTierFamily(model: ServiceTierModel): ServiceTierF
120
120
  * Reduce a per-family tier map to the single wire tier for `model` — the entry
121
121
  * for the model's family, or `undefined` when the model has no family.
122
122
  */
123
- export declare function resolveModelServiceTier(tiers: ServiceTierByFamily | null | undefined, model: Pick<Model, "provider" | "api" | "id">): ServiceTier | undefined;
123
+ export declare function resolveModelServiceTier(tiers: ServiceTierByFamily | null | undefined, model: ServiceTierModel): ServiceTier | undefined;
124
124
  /**
125
125
  * True when the tier should be sent on the wire as the provider's service-tier
126
126
  * request field. `auto` is never forwarded — it is OpenAI's implicit default, so
@@ -139,7 +139,7 @@ export declare function shouldSendServiceTier(serviceTier: ServiceTier | null |
139
139
  * Google-family upstreams. Bedrock/Vertex Claude and OpenRouter Anthropic
140
140
  * models do not realize priority and return `false`.
141
141
  */
142
- export declare function realizesPriorityServiceTier(serviceTier: ServiceTier | null | undefined, model: Pick<Model, "provider" | "api" | "id">): boolean;
142
+ export declare function realizesPriorityServiceTier(serviceTier: ServiceTier | null | undefined, model: ServiceTierModel): boolean;
143
143
  /**
144
144
  * Premium-request weight contributed by a priority request to a provider that
145
145
  * realizes it and bills extra. Mirrors GitHub Copilot's `premiumRequests`
@@ -152,7 +152,7 @@ export declare function realizesPriorityServiceTier(serviceTier: ServiceTier | n
152
152
  * Copilot-premium semantics — as are Bedrock/Vertex Claude, where priority is
153
153
  * silently dropped.
154
154
  */
155
- export declare function getPriorityPremiumRequests(serviceTier: ServiceTier | null | undefined, model: Pick<Model, "provider" | "api" | "id">): number;
155
+ export declare function getPriorityPremiumRequests(serviceTier: ServiceTier | null | undefined, model: ServiceTierModel): number;
156
156
  /**
157
157
  * Coerce a persisted service-tier value to a {@link ServiceTierByFamily}. Newer
158
158
  * sessions store the family map directly; legacy sessions stored a single
@@ -760,6 +760,8 @@ export interface AssistantMessage {
760
760
  * providers that expose no such field.
761
761
  */
762
762
  upstreamProvider?: string;
763
+ /** Provider-reported concrete model when a router selected one for this turn. */
764
+ upstreamModel?: string;
763
765
  usage: Usage;
764
766
  stopReason: StopReason;
765
767
  stopDetails?: StopDetails | null;
@@ -0,0 +1,2 @@
1
+ import type { UsageProvider } from "../usage.js";
2
+ export declare const clinePassUsageProvider: UsageProvider;
@@ -0,0 +1,12 @@
1
+ /**
2
+ * Devin (Codeium Cascade) account plan + credit usage provider.
3
+ *
4
+ * Devin ships no REST usage endpoint: plan tier, credit balances, the
5
+ * daily/weekly quota windows and the account identity all come back from the
6
+ * single `SeatManagementService/GetUserStatus` unary Connect RPC the native CLI
7
+ * issues at startup. The request body is raw (unframed) protobuf carrying the
8
+ * CLI identity metadata plus the session token; the backend gates the CLI
9
+ * surface on that identity tuple.
10
+ */
11
+ import type { UsageProvider } from "../usage.js";
12
+ export declare const devinUsageProvider: UsageProvider;
@@ -1,3 +1,12 @@
1
+ /**
2
+ * GPT-5 Harmony-header leakage detection and recovery.
3
+ *
4
+ * Background and policy: see `docs/ERRATA-GPT5-HARMONY.md`. This module
5
+ * implements §3 of that document: detection by signal fusion, plus a
6
+ * truncate-and-resume primitive for the `edit` tool when its input is in
7
+ * hashline DSL form. Other tools and surfaces fall through to
8
+ * abort-and-retry handled by the agent loop.
9
+ */
1
10
  import type { AssistantMessage, Model, ToolCall } from "../types.js";
2
11
  /**
3
12
  * Escape reserved Harmony control tokens in arbitrary text so it can be
@@ -59,10 +68,10 @@ export interface HarmonyRecoveredToolCall {
59
68
  removed: string;
60
69
  }
61
70
  /**
62
- * Whether to run leak detection on responses from this model. We default-on
63
- * for every openai-codex model rather than enumerating ids, so a future
64
- * gpt-5.6 (or whatever) doesn't silently bypass the mitigation. Detection
65
- * itself is cheap; the cost of missing a leak on a new model is not.
71
+ * Whether to run leak detection on responses from this model. The default-on policy
72
+ * lives on the harmony-leak-mitigation axis in providers/openai-codex.kdl.
73
+ * It targets the provider rather than enumerating model ids so future models
74
+ * do not silently bypass this cheap mitigation.
66
75
  */
67
76
  export declare function isHarmonyLeakMitigationTarget(model: Model): boolean;
68
77
  export declare function signalListLabel(signals: readonly HarmonySignal[]): string;
@@ -47,3 +47,22 @@ export declare function finalizeErrorMessage(error: unknown, rawRequestDump: Raw
47
47
  * do NOT reuse the auth-failed string (which triggers credential removal).
48
48
  */
49
49
  export declare function rewriteCopilotError(errorMessage: string, error: unknown, provider: string): string;
50
+ /**
51
+ * Rewrite error messages for ClinePass request failures. Gated to the
52
+ * cline-pass provider: the "model not found" marker is too generic to match
53
+ * for other hosts.
54
+ *
55
+ * not-subscribed (400) = the key is valid but the account has no ClinePass
56
+ * subscription; free-tier models remain usable on the same key.
57
+ * org restriction (400) = organization accounts cannot use individual
58
+ * inference subscriptions; a personal-account key is required.
59
+ * model-not-found (400) = roster rotation removed the model since selection;
60
+ * the fix is reselection, not retry. (Quota windows — "clinepass
61
+ * limit", "free limit reached on model" — are classified upstream in
62
+ * error/rate-limit and need no rewrite.)
63
+ * surface-gate (403) = the model is restricted to Cline's official clients.
64
+ * Requests carry the mirrored CLI identity headers, so reaching this
65
+ * means Cline's gate policy changed; the classifier exempts it from
66
+ * credential rotation (sibling keys fail identically).
67
+ */
68
+ export declare function rewriteClinePassError(errorMessage: string, provider: string): string;
@@ -40,6 +40,16 @@ export declare function copySchemaWithout(schema: JsonObject, combiner: string):
40
40
  * create new anyOf in merged subtrees after child normalization already ran.
41
41
  */
42
42
  export declare function stripResidualCombiners(value: unknown, epoch?: number): unknown;
43
+ /**
44
+ * Project a tool's wire schema onto the subset Cursor's MCP tool catalog
45
+ * accepts. Cursor rejects the entire request with a provider 400 when any
46
+ * advertised schema carries a composition keyword (issue #10432); this removes
47
+ * `anyOf`/`oneOf`/`allOf` everywhere while preserving representable guidance and
48
+ * only ever widening acceptance, so every input the canonical schema accepts is
49
+ * still accepted by the advertised projection. The canonical schema (used for
50
+ * execution-time argument validation) is never mutated.
51
+ */
52
+ export declare function sanitizeSchemaForCursor(schema: JsonObject): JsonObject;
43
53
  export declare function normalizeSchema(value: unknown, options: NormalizeSchemaOptions): unknown;
44
54
  export declare function normalizeSchemaForGoogle(value: unknown): unknown;
45
55
  export declare function normalizeSchemaForCCA(value: unknown): unknown;
@@ -10,12 +10,12 @@
10
10
  *
11
11
  * xAI additionally rejects a leftover *root* `anyOf`/`oneOf` whose branches
12
12
  * are not objects ("tool parameter root must be an object type"). That class
13
- * is opt-in via {@link FindStrictToolSchemaViolationOptions.rejectXaiRootObjectUnion}
13
+ * is opt-in via {@link FindStrictToolSchemaViolationOptions.rejectRootObjectUnion}
14
14
  * so OpenAI/Azure/Codex keep valid object-root unions.
15
15
  */
16
16
  export interface FindStrictToolSchemaViolationOptions {
17
- /** xAI (paid + OAuth) only: leftover object-root unions 400 the whole turn. */
18
- rejectXaiRootObjectUnion?: boolean;
17
+ /** Reject leftover object-root unions; xAI (paid + OAuth) 400s the whole turn when they remain. */
18
+ rejectRootObjectUnion?: boolean;
19
19
  }
20
20
  /**
21
21
  * Walk a tool parameter schema for OpenAI-strict `enum`/`const`-vs-`type`
@@ -6,6 +6,7 @@
6
6
  * dialect scanners used by owned in-band tool calling; this file keeps the
7
7
  * provider-facing compatibility wrapper and model/provider gating.
8
8
  */
9
+ import type { Model } from "../types.js";
9
10
  export interface HealedToolCall {
10
11
  readonly id: string;
11
12
  readonly name: string;
@@ -73,10 +74,6 @@ export declare class StreamMarkupHealing {
73
74
  /** True once any configured tool-call section/envelope has fully closed. */
74
75
  get sectionClosed(): boolean;
75
76
  }
76
- /** Cheap model/provider gate for Kimi-K2 chat-template token leaks. */
77
- export declare function modelMayLeakKimiToolCalls(provider: string, modelId: string): boolean;
78
- /** Cheap model/provider gate for DeepSeek DSML envelope leaks. */
79
- export declare function modelMayLeakDsmlToolCalls(provider: string, modelId: string): boolean;
80
77
  /**
81
78
  * Pick the leaked-markup healer for an OpenAI-compatible / Ollama visible-text
82
79
  * stream. Kimi chat-template tokens and DeepSeek DSML envelopes need their
@@ -84,4 +81,4 @@ export declare function modelMayLeakDsmlToolCalls(provider: string, modelId: str
84
81
  * patterns run the generic {@link ThinkingInbandScanner}, so leaked reasoning
85
82
  * idioms (e.g. a Gemini ` ```thinking ` fence on OpenRouter) are always healed.
86
83
  */
87
- export declare function getStreamMarkupHealingPattern(provider: string, modelId: string): StreamMarkupHealingPattern;
84
+ export declare function getStreamMarkupHealingPattern(model: Model<"ollama-chat">): StreamMarkupHealingPattern;
@@ -5,12 +5,9 @@ import { AssistantMessageEventStream } from "./event-stream.js";
5
5
  * classifiers treat it as a transient (retryable) stop without bespoke rules. */
6
6
  export declare const THINKING_LOOP_ERROR_MARKER = "Thinking loop detected";
7
7
  /**
8
- * True when `model.id` belongs to a family guarded by the semantic loop
9
- * heuristics: Gemini, DeepSeek, or Grok. Exact suffix-cycle detection applies to
10
- * every enabled model independently of this predicate.
11
- *
12
- * Model identity is derived only from its id; provider and compatibility metadata
13
- * do not opt opaque aliases into semantic detection.
8
+ * True when resolved compatibility policy enables semantic loop heuristics for
9
+ * this model. Exact suffix-cycle detection applies to every enabled model
10
+ * independently of this predicate.
14
11
  */
15
12
  export declare function isLoopGuardedModel(model: Model<Api>, options?: StreamOptions): boolean;
16
13
  /**
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@oh-my-pi/pi-ai",
4
- "version": "18.0.11",
4
+ "version": "18.1.1",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://omp.sh",
7
7
  "author": "Stencil Labs, Inc.",
@@ -29,18 +29,18 @@
29
29
  "main": "./src/index.ts",
30
30
  "types": "./dist/types/index.d.ts",
31
31
  "scripts": {
32
- "check": "biome check . && bun run check:types",
32
+ "check": "oxlint . && oxfmt --check --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts' && bun run check:types",
33
33
  "check:types": "tsgo -p tsconfig.json --noEmit",
34
- "lint": "biome lint .",
34
+ "lint": "oxlint .",
35
35
  "test": "bun test --parallel",
36
- "fix": "biome check --write --unsafe .",
37
- "fmt": "biome format --write ."
36
+ "fix": "oxlint --fix --fix-suggestions . && bun run fmt",
37
+ "fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
38
38
  },
39
39
  "dependencies": {
40
- "@oh-my-pi/omptype": "18.0.11",
41
- "@oh-my-pi/pi-catalog": "18.0.11",
42
- "@oh-my-pi/pi-utils": "18.0.11",
43
- "@oh-my-pi/pi-wire": "18.0.11"
40
+ "@oh-my-pi/omptype": "18.1.1",
41
+ "@oh-my-pi/pi-catalog": "18.1.1",
42
+ "@oh-my-pi/pi-utils": "18.1.1",
43
+ "@oh-my-pi/pi-wire": "18.1.1"
44
44
  },
45
45
  "devDependencies": {
46
46
  "@types/bun": "^1.3.14"
@@ -445,7 +445,7 @@ export class AuthBrokerClient {
445
445
  ): Promise<Response> {
446
446
  const auth = opts.auth ?? true;
447
447
  const url = `${this.#baseUrl}${path}`;
448
- const headers: Record<string, string> = { Accept: "application/json", ...(opts.headers ?? {}) };
448
+ const headers: Record<string, string> = { Accept: "application/json", ...opts.headers };
449
449
  if (auth) headers.Authorization = `Bearer ${this.#token}`;
450
450
  let payload: string | undefined;
451
451
  if (opts.body !== undefined) {
@@ -204,7 +204,7 @@ function mergeUsageReports(base: UsageReport, overlay: UsageReport): UsageReport
204
204
  limits,
205
205
  metadata: {
206
206
  ...overlayMetadata,
207
- ...(base.metadata ?? {}),
207
+ ...base.metadata,
208
208
  ...(overlayMetadata.headersUpdatedAt !== undefined
209
209
  ? { headersUpdatedAt: overlayMetadata.headersUpdatedAt }
210
210
  : {}),
@@ -89,7 +89,7 @@ export interface AuthBrokerServerHandle {
89
89
  function json(status: number, body: unknown, headers?: Record<string, string>): Response {
90
90
  return new Response(JSON.stringify(body), {
91
91
  status,
92
- headers: { "Content-Type": "application/json", ...(headers ?? {}) },
92
+ headers: { "Content-Type": "application/json", ...headers },
93
93
  });
94
94
  }
95
95
 
@@ -260,9 +260,9 @@ class GenerationGate {
260
260
  }
261
261
 
262
262
  #wake(generation: number): void {
263
- for (const [waitingFor, waiters] of [...this.#waiters]) {
263
+ for (const [waitingFor, waiters] of Array.from(this.#waiters)) {
264
264
  if (generation <= waitingFor) continue;
265
- for (const resolve of [...waiters]) resolve();
265
+ for (const resolve of Array.from(waiters)) resolve();
266
266
  }
267
267
  }
268
268
  }
@@ -588,7 +588,7 @@ function serveSnapshotStream(
588
588
  generation: snapshot.generation,
589
589
  });
590
590
  }
591
- for (const id of [...lastByCredId.keys()]) {
591
+ for (const id of Array.from(lastByCredId.keys())) {
592
592
  if (seenIds.has(id)) continue;
593
593
  lastByCredId.delete(id);
594
594
  const payload: SnapshotStreamRemovedEvent = {