@gajae-code/ai 0.12.0 → 0.12.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +20 -0
- package/dist/types/model-thinking.d.ts +15 -0
- package/dist/types/providers/azure-openai-responses.d.ts +2 -0
- package/dist/types/providers/google-vertex.d.ts +2 -0
- package/dist/types/utils/fallback-transport.d.ts +2 -0
- package/dist/types/utils/http-inspector.d.ts +10 -0
- package/dist/types/utils/idle-iterator.d.ts +4 -0
- package/dist/types/utils/oauth/kimi.d.ts +2 -0
- package/dist/types/utils/oauth/perplexity.d.ts +2 -7
- package/package.json +2 -2
- package/src/model-thinking.ts +19 -0
- package/src/providers/amazon-bedrock.ts +5 -20
- package/src/providers/anthropic.ts +95 -44
- package/src/providers/azure-openai-responses.ts +28 -16
- package/src/providers/google-vertex.ts +35 -4
- package/src/providers/ollama.ts +32 -4
- package/src/providers/openai-codex-responses.ts +48 -31
- package/src/providers/openai-completions.ts +13 -12
- package/src/providers/openai-responses.ts +9 -11
- package/src/providers/register-builtins.ts +12 -2
- package/src/utils/fallback-transport.ts +17 -10
- package/src/utils/http-inspector.ts +47 -1
- package/src/utils/idle-iterator.ts +12 -3
- package/src/utils/oauth/kimi.ts +17 -2
- package/src/utils/oauth/perplexity.ts +21 -2
- package/src/utils/schema/adapt.ts +2 -2
- package/src/utils/tool-choice-capability.ts +2 -1
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,24 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [0.12.1] - 2026-07-29
|
|
6
|
+
|
|
7
|
+
### Fixed
|
|
8
|
+
|
|
9
|
+
- Lazy-stream first-event timeouts now abort with `FirstEventTimeoutError` so `transportFailure.providerCode` is `stream_first_event_timeout` on the outer watchdog path shared by all bundled providers via `createLazyStream`. Idle stalls remain bare `Error`s (distinct class intentionally) (#3496).
|
|
10
|
+
|
|
11
|
+
- Provider streams now surface first-event watchdog expiry as a typed timeout so callers can apply bounded retry policy without parsing error prose.
|
|
12
|
+
- Codex websocket first-event timeouts now discard the timed-out connection before the outer retry/fallback layer handles the typed failure, preventing late frames from the abandoned request from being consumed by the replayed turn.
|
|
13
|
+
- Codex named-tool requests now recognize provider `Tool choice '<name>' not found in 'tools' parameter` errors as runtime capability failures and retry once without forcing the choice.
|
|
14
|
+
- The Kimi OAuth host (`KIMI_CODE_OAUTH_HOST` / `KIMI_OAUTH_HOST`) is now resolved from trusted environment sources only. That host receives the device-authorization request, the authorization-code exchange, and the refresh call that carries the existing refresh token, so reading it through the merged view that includes the caller's `cwd/.env` let a repository redirect the login flow and collect the user's Kimi credentials. Resolution now uses the non-project resolver; shell and user-level configuration is unchanged.
|
|
15
|
+
- The documented `GJC_NO_STRICT` environment variable now takes effect. `adaptSchemaForStrict` read only the legacy `PI_NO_STRICT`, so an operator hitting a provider that rejects strict function schemas set the documented name and strict mode stayed on. Both names are honoured, canonical name first, and `GJC_NO_STRICT` is now listed in the environment-variable reference rather than only in the schema-normalisation note.
|
|
16
|
+
- The documented `GJC_AUTH_NO_BORROW` environment variable now takes effect. Only the legacy `PI_AUTH_NO_BORROW` was read, so an operator who followed the documentation to disable macOS native-app token borrowing still had a JWT read out of the Perplexity desktop application during login. Both names are now honoured, and the contract stays presence-based as documented so that setting it to `0` cannot silently re-enable borrowing.
|
|
17
|
+
- The Azure client's `AZURE_OPENAI_API_KEY` fallback is now resolved from trusted environment sources only. It read the merged view that includes the caller's `cwd/.env`, so a repository could supply the credential the client authenticates with; provider credential resolution is documented as excluding the project `.env`, and this fallback now matches. An explicit caller-supplied key still takes precedence, and shell / user-level configuration is unchanged.
|
|
18
|
+
- Anthropic and Ollama tool calls cut off by an output-token limit are now marked incomplete before dispatch, so repaired partial JSON is rejected instead of executing with truncated arguments.
|
|
19
|
+
- The Anthropic "thinking blocks in the latest assistant message cannot be modified" 400 now escalates its one-shot replay repair. The error names the latest assistant message but its cited `messages.N.content.M` path can point at an earlier replayed turn, so the latest-only repair was rejected identically and killed the turn; recovery now retries once more with thinking dropped from every replayed assistant message.
|
|
20
|
+
- Anthropic adaptive-thinking `display` support is now decided by the canonical model-version parser instead of a provider-local `claude-opus-(\d+)-(\d+)` regex. The regex only matched two-component ids, so a single-component alias such as `claude-opus-5` was classified as pre-4.7 while its dated snapshot `claude-opus-5-20260101` was not: the alias sent `thinking: { type: "adaptive" }` without `display: "summarized"`, additionally requested the `interleaved-thinking-2025-05-14` beta, and had its returned thinking blocks recorded as raw rather than summarized. Both Anthropic and Bedrock providers now share `supportsAnthropicAdaptiveThinkingDisplay`, so alias and dated ids of the same model send an identical request shape.
|
|
21
|
+
- Anthropic requests that force a tool choice no longer replay signed thinking blocks. Forcing `tool_choice` strips `thinking` from the request (the API rejects the combination), but the converted history still carried native `thinking`/`redacted_thinking` blocks from thinking-enabled turns, so eager tool-forcing turns (e.g. the todo bootstrap) sent a request whose history contradicted its own thinking setting and drew a 400. The replay now degrades in the same rebuild; the forced request trades its prompt-cache prefix for a shape the API accepts.
|
|
22
|
+
|
|
5
23
|
## [0.12.0] - 2026-07-28
|
|
6
24
|
|
|
7
25
|
### Added
|
|
@@ -26,6 +44,8 @@
|
|
|
26
44
|
- The OpenAI and Azure endpoint decisions are now resolved from trusted environment sources only: `OPENAI_BASE_URL` (streaming responses, completions, and the model manager), `AZURE_OPENAI_BASE_URL`, and `AZURE_OPENAI_RESOURCE_NAME` (the alternate constructor for the same Azure host). `Bun.env` is `process.env` and the env module merges the caller's `cwd/.env` into it, so a repository could previously plant a `.env` that redirected authenticated requests; the two provider paths already reached for `$inheritedEnv` but re-admitted the project `.env` through a fallback. Resolution now goes through the non-project resolver (launching shell plus GJC/user-owned `.env` files); shell and user-level configuration is unchanged.
|
|
27
45
|
- Google credential material is now resolved from trusted environment sources only: the `GOOGLE_APPLICATION_CREDENTIALS` service-account / authorized-user file path used by the ADC loader, and `GOOGLE_CLOUD_API_KEY` used as the Vertex API key. Both were read through the merged view that includes the caller's `cwd/.env`, so a repository could ship a key file and point the agent at it, making it authenticate to Google as an identity the repository chose. `stream.ts` already resolved the same ADC variable through the non-project resolver; the two now agree. An explicit caller-supplied API key still takes precedence.
|
|
28
46
|
- The Grok usage token fallback (`GROK_CLI_OAUTH_TOKEN`) is now resolved from trusted environment sources only. It authenticates the billing/usage call, and reading it through the merged view that includes the caller's `cwd/.env` let a repository decide which account that call ran against. Stored credentials keep precedence, and shell / user-level configuration is unchanged.
|
|
47
|
+
- The Vertex AI location (`GOOGLE_CLOUD_LOCATION`) can no longer redirect authenticated requests off Google. It is interpolated into the request host (`${location}-aiplatform.googleapis.com`), so a value containing `/` terminated the authority component and sent the Google access token to an arbitrary origin — and it was read through the merged view that includes the caller's `cwd/.env`. It now resolves from trusted sources only and must be a region label; `GOOGLE_CLOUD_PROJECT` / `GCLOUD_PROJECT` moved to the same trusted resolver.
|
|
48
|
+
- HTTP 400 request dumps are now bounded. Every 400 wrote a file containing the full sanitized request body and nothing ever removed one, so the directory grew without limit — a developer machine reached 27,249 files totalling 7.0 GB, which was 96% of everything under `~/.gjc`. The newest 50 are retained, matching the bounded retention the rotating application log already uses, and pruning stays best-effort so diagnostics never turn a request failure into a second failure.
|
|
29
49
|
- Anthropic `ping` keepalives no longer reset stream progress, so responses that stop producing content now reach the idle timeout instead of hanging indefinitely.
|
|
30
50
|
- The Anthropic endpoint decision is now resolved from trusted environment sources only: `ANTHROPIC_BASE_URL`, `FOUNDRY_BASE_URL`, `ZCODE_PLAN_ANTHROPIC_BASE_URL`, and the `CLAUDE_CODE_USE_FOUNDRY` mode switch. `Bun.env` is `process.env` and the env module merges the caller's `cwd/.env` into it, so a repository could previously plant a `.env` that redirected authenticated Anthropic requests — the resolved base URL becomes `${baseUrl}/v1/messages` while the headers carry the API key or OAuth token. Resolution now goes through the non-project resolver (launching shell plus GJC/user-owned `.env` files); shell and user-level configuration is unchanged.
|
|
31
51
|
- The documented `GJC_OPENAI_STREAM_IDLE_TIMEOUT_MS` environment variable now takes effect: the stream-watchdog idle-timeout helpers resolve it GJC-first before the legacy `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` / `PI_STREAM_IDLE_TIMEOUT_MS` aliases (previously only the `PI_`-prefixed names were read, so setting the documented GJC name was a silent no-op).
|
|
@@ -72,3 +72,18 @@ export declare function mapEffortToAnthropicAdaptiveEffort<TApi extends Api>(mod
|
|
|
72
72
|
* - Thinking content is omitted by default (needs display: "summarized")
|
|
73
73
|
*/
|
|
74
74
|
export declare function hasOpus47ApiRestrictions(modelId: string): boolean;
|
|
75
|
+
/**
|
|
76
|
+
* Adaptive thinking `display` is supported starting with Anthropic Opus 4.7.
|
|
77
|
+
* Older adaptive-thinking models (Opus 4.6, Sonnet 4.6+) reject the field.
|
|
78
|
+
* Fable (5+) postdates Opus 4.7, accepts `display`, and defaults it to
|
|
79
|
+
* "omitted" — thinking tokens are billed but no content streams back — so it
|
|
80
|
+
* must opt in like Opus 4.7+ (issue #2791).
|
|
81
|
+
*
|
|
82
|
+
* Shares `hasOpus47ApiRestrictions` version parsing on purpose: the two
|
|
83
|
+
* predicates describe the same API generation, and a private `claude-opus-(\d+)-(\d+)`
|
|
84
|
+
* regex silently disagreed with it for single-component aliases (`claude-opus-5`
|
|
85
|
+
* matched nothing while `claude-opus-5-20260101` matched), so the same model sent
|
|
86
|
+
* a different thinking shape and beta set depending on which id string was used.
|
|
87
|
+
* Bedrock region/inference-profile prefixes are handled by the canonical parser.
|
|
88
|
+
*/
|
|
89
|
+
export declare function supportsAnthropicAdaptiveThinkingDisplay(modelId: string): boolean;
|
|
@@ -18,3 +18,5 @@ export declare function resolveAzureConfigForTest(model: Model<"azure-openai-res
|
|
|
18
18
|
baseUrl: string;
|
|
19
19
|
apiVersion: string;
|
|
20
20
|
};
|
|
21
|
+
/** Test seam: the client API key as resolved from a caller value plus trusted env. */
|
|
22
|
+
export declare function resolveAzureClientApiKeyForTest(apiKey: string): string | undefined;
|
|
@@ -7,3 +7,5 @@ export interface GoogleVertexOptions extends GoogleSharedStreamOptions {
|
|
|
7
7
|
export declare const streamGoogleVertex: StreamFunction<"google-vertex">;
|
|
8
8
|
/** Test seam: the Vertex API key as resolved from options plus trusted env. */
|
|
9
9
|
export declare function resolveVertexApiKeyForTest(options?: GoogleVertexOptions): string | undefined;
|
|
10
|
+
/** Test seam: the Vertex location as resolved from options plus trusted env. */
|
|
11
|
+
export declare function resolveVertexLocationForTest(options?: GoogleVertexOptions): string;
|
|
@@ -3,6 +3,8 @@ export interface FallbackTrigger {
|
|
|
3
3
|
class: FallbackTriggerClass;
|
|
4
4
|
retryAfterMs?: number;
|
|
5
5
|
}
|
|
6
|
+
/** Stable code for streams that time out before producing semantic progress. */
|
|
7
|
+
export declare const STREAM_FIRST_EVENT_TIMEOUT_PROVIDER_CODE = "stream_first_event_timeout";
|
|
6
8
|
export type TransportHeaders = Headers | Record<string, string | undefined>;
|
|
7
9
|
/**
|
|
8
10
|
* Structured facts from an upstream HTTP or transport failure. Retry decisions
|
|
@@ -17,6 +17,16 @@ export type CapturedHttpErrorResponse = {
|
|
|
17
17
|
export declare function isModelUnavailableError(message: string, error: unknown): boolean;
|
|
18
18
|
/** Actionable guidance for selecting an available model/provider. */
|
|
19
19
|
export declare function formatModelUnavailableGuidance(dump: RawHttpRequestDump | undefined): string;
|
|
20
|
+
/** Directory holding the retained HTTP 400 dumps. */
|
|
21
|
+
export declare function httpRequestDumpDir(): string;
|
|
22
|
+
/**
|
|
23
|
+
* Drop the oldest dumps beyond the cap. Best-effort: diagnostics must never turn
|
|
24
|
+
* a request failure into a second failure, so every step swallows its error.
|
|
25
|
+
*
|
|
26
|
+
* File names are `${Date.now()}-${hash}.json`, so a lexical sort is chronological
|
|
27
|
+
* for the millisecond timestamps this writer produces.
|
|
28
|
+
*/
|
|
29
|
+
export declare function pruneHttpRequestDumps(dir?: string): Promise<number>;
|
|
20
30
|
export declare function appendRawHttpRequestDumpFor400(message: string, error: unknown, dump: RawHttpRequestDump | undefined): Promise<string>;
|
|
21
31
|
/**
|
|
22
32
|
* Name the failed connection when the request never produced an HTTP status.
|
|
@@ -30,6 +30,10 @@ export declare function getOpenAIStreamIdleTimeoutMs(): number | undefined;
|
|
|
30
30
|
*/
|
|
31
31
|
export declare function getStreamFirstEventTimeoutMs(idleTimeoutMs?: number, fallbackMs?: number): number | undefined;
|
|
32
32
|
export type Watchdog = NodeJS.Timeout | undefined;
|
|
33
|
+
export declare class FirstEventTimeoutError extends Error {
|
|
34
|
+
readonly providerCode = "stream_first_event_timeout";
|
|
35
|
+
constructor(message: string);
|
|
36
|
+
}
|
|
33
37
|
/**
|
|
34
38
|
* Starts a watchdog that aborts a request if no first stream event arrives in time.
|
|
35
39
|
* Call `markFirstEventReceived()` as soon as the first event is observed.
|
|
@@ -2,6 +2,8 @@
|
|
|
2
2
|
* Kimi Code OAuth flow (device authorization grant)
|
|
3
3
|
*/
|
|
4
4
|
import type { OAuthController, OAuthCredentials } from "./types";
|
|
5
|
+
/** Test seam: the OAuth host as resolved from trusted env. */
|
|
6
|
+
export declare function resolveKimiOAuthHostForTest(): string;
|
|
5
7
|
export declare let getKimiCommonHeaders: () => Readonly<{
|
|
6
8
|
"User-Agent": `KimiCLI/${string}`;
|
|
7
9
|
"X-Msh-Platform": "kimi_cli";
|
|
@@ -1,9 +1,4 @@
|
|
|
1
1
|
import type { OAuthController, OAuthCredentials } from "./types";
|
|
2
|
-
/**
|
|
3
|
-
|
|
4
|
-
*
|
|
5
|
-
* Tries auto-extraction from the desktop app, then runs HTTP email OTP login.
|
|
6
|
-
*
|
|
7
|
-
* No browser/manual token paste fallback is used.
|
|
8
|
-
*/
|
|
2
|
+
/** Test seam: the resolved native-app borrowing opt-out. */
|
|
3
|
+
export declare function authBorrowDisabledForTest(): boolean;
|
|
9
4
|
export declare function loginPerplexity(ctrl: OAuthController): Promise<OAuthCredentials>;
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@gajae-code/ai",
|
|
4
|
-
"version": "0.12.
|
|
4
|
+
"version": "0.12.1",
|
|
5
5
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
6
6
|
"homepage": "https://gajae-code.com",
|
|
7
7
|
"author": "Yeachan-Heo and Gajae Code Contributors",
|
|
@@ -40,7 +40,7 @@
|
|
|
40
40
|
"dependencies": {
|
|
41
41
|
"@anthropic-ai/sdk": "^0.94.0",
|
|
42
42
|
"@bufbuild/protobuf": "^2.12.0",
|
|
43
|
-
"@gajae-code/utils": "0.12.
|
|
43
|
+
"@gajae-code/utils": "0.12.1",
|
|
44
44
|
"openai": "^6.36.0",
|
|
45
45
|
"partial-json": "^0.1.7",
|
|
46
46
|
"zod": "4.4.3"
|
package/src/model-thinking.ts
CHANGED
|
@@ -350,6 +350,25 @@ export function hasOpus47ApiRestrictions(modelId: string): boolean {
|
|
|
350
350
|
return semverGte(parsed.version, "4.7") && parsed.kind === "opus";
|
|
351
351
|
}
|
|
352
352
|
|
|
353
|
+
/**
|
|
354
|
+
* Adaptive thinking `display` is supported starting with Anthropic Opus 4.7.
|
|
355
|
+
* Older adaptive-thinking models (Opus 4.6, Sonnet 4.6+) reject the field.
|
|
356
|
+
* Fable (5+) postdates Opus 4.7, accepts `display`, and defaults it to
|
|
357
|
+
* "omitted" — thinking tokens are billed but no content streams back — so it
|
|
358
|
+
* must opt in like Opus 4.7+ (issue #2791).
|
|
359
|
+
*
|
|
360
|
+
* Shares `hasOpus47ApiRestrictions` version parsing on purpose: the two
|
|
361
|
+
* predicates describe the same API generation, and a private `claude-opus-(\d+)-(\d+)`
|
|
362
|
+
* regex silently disagreed with it for single-component aliases (`claude-opus-5`
|
|
363
|
+
* matched nothing while `claude-opus-5-20260101` matched), so the same model sent
|
|
364
|
+
* a different thinking shape and beta set depending on which id string was used.
|
|
365
|
+
* Bedrock region/inference-profile prefixes are handled by the canonical parser.
|
|
366
|
+
*/
|
|
367
|
+
export function supportsAnthropicAdaptiveThinkingDisplay(modelId: string): boolean {
|
|
368
|
+
if (/claude-fable-\d/.test(modelId)) return true;
|
|
369
|
+
return hasOpus47ApiRestrictions(modelId);
|
|
370
|
+
}
|
|
371
|
+
|
|
353
372
|
function anthropicModelHasRealXHighEffort<TApi extends Api>(model: ApiModel<TApi>): boolean {
|
|
354
373
|
if (model.api !== "anthropic-messages") return false;
|
|
355
374
|
const parsedModel = parseKnownModel(model.id);
|
|
@@ -9,7 +9,11 @@
|
|
|
9
9
|
|
|
10
10
|
import { $credentialEnv, $env, $flag, extractHttpStatusFromError, fetchWithRetry } from "@gajae-code/utils";
|
|
11
11
|
import type { Effort } from "../model-thinking";
|
|
12
|
-
import {
|
|
12
|
+
import {
|
|
13
|
+
mapEffortToAnthropicAdaptiveEffort,
|
|
14
|
+
requireSupportedEffort,
|
|
15
|
+
supportsAnthropicAdaptiveThinkingDisplay as supportsAdaptiveThinkingDisplay,
|
|
16
|
+
} from "../model-thinking";
|
|
13
17
|
import { calculateCost } from "../models";
|
|
14
18
|
import type {
|
|
15
19
|
Api,
|
|
@@ -907,25 +911,6 @@ function buildAdditionalModelRequestFields(
|
|
|
907
911
|
return result;
|
|
908
912
|
}
|
|
909
913
|
|
|
910
|
-
/**
|
|
911
|
-
* Adaptive thinking `display` is supported starting with Anthropic model Opus 4.7.
|
|
912
|
-
* Older adaptive-thinking models (Opus 4.6, Sonnet 4.6+) reject the field.
|
|
913
|
-
* Fable (5+) postdates Opus 4.7, accepts `display`, and defaults it to
|
|
914
|
-
* "omitted" — thinking tokens are billed but no content streams back — so it
|
|
915
|
-
* must opt in like Opus 4.7+ (issue #2791).
|
|
916
|
-
* Bedrock model ids are prefixed with region/inference-profile slugs (e.g.
|
|
917
|
-
* `eu.anthropic.Anthropic model-opus-4-7-...`); the regex matches the `Anthropic model-opus-X-Y`
|
|
918
|
-
* fragment regardless of prefix.
|
|
919
|
-
*/
|
|
920
|
-
function supportsAdaptiveThinkingDisplay(modelId: string): boolean {
|
|
921
|
-
if (/claude-fable-\d/.test(modelId)) return true;
|
|
922
|
-
const match = /claude-opus-(\d+)-(\d+)/.exec(modelId);
|
|
923
|
-
if (!match) return false;
|
|
924
|
-
const major = Number(match[1]);
|
|
925
|
-
const minor = Number(match[2]);
|
|
926
|
-
return major > 4 || (major === 4 && minor >= 7);
|
|
927
|
-
}
|
|
928
|
-
|
|
929
914
|
/**
|
|
930
915
|
* Bedrock's wire format expects the image as `{ source: { bytes: <base64-string> }, format }`.
|
|
931
916
|
* The caller already passes base64-encoded data, so no decode/re-encode round-trip is needed.
|
|
@@ -20,7 +20,11 @@ import {
|
|
|
20
20
|
logger,
|
|
21
21
|
readSseEvents,
|
|
22
22
|
} from "@gajae-code/utils";
|
|
23
|
-
import {
|
|
23
|
+
import {
|
|
24
|
+
hasOpus47ApiRestrictions,
|
|
25
|
+
mapEffortToAnthropicAdaptiveEffort,
|
|
26
|
+
supportsAnthropicAdaptiveThinkingDisplay as supportsAdaptiveThinkingDisplay,
|
|
27
|
+
} from "../model-thinking";
|
|
24
28
|
import { calculateCost } from "../models";
|
|
25
29
|
import { isUsageLimitError } from "../rate-limit-utils";
|
|
26
30
|
import { getEnvApiKey, OUTPUT_FALLBACK_BUFFER } from "../stream";
|
|
@@ -62,12 +66,13 @@ import { transportFailureFacts } from "../utils/fallback-transport";
|
|
|
62
66
|
import { isFoundryEnabled } from "../utils/foundry";
|
|
63
67
|
import { finalizeErrorMessage, type RawHttpRequestDump, rewriteCopilotError } from "../utils/http-inspector";
|
|
64
68
|
import {
|
|
69
|
+
FirstEventTimeoutError,
|
|
65
70
|
getProviderFirstEventTimeoutFallbackMs,
|
|
66
71
|
getStreamFirstEventTimeoutMs,
|
|
67
72
|
getStreamIdleTimeoutMs,
|
|
68
73
|
iterateWithIdleTimeout,
|
|
69
74
|
} from "../utils/idle-iterator";
|
|
70
|
-
import { parseJsonWithRepair, parseStreamingJson } from "../utils/json-parse";
|
|
75
|
+
import { isCompleteJson, parseJsonWithRepair, parseStreamingJson } from "../utils/json-parse";
|
|
71
76
|
import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
|
|
72
77
|
import { notifyProviderResponse } from "../utils/provider-response";
|
|
73
78
|
import { isCopilotTransientModelError } from "../utils/retry";
|
|
@@ -309,22 +314,6 @@ type AnthropicSamplingParams = MessageCreateParamsStreaming & {
|
|
|
309
314
|
const ANTHROPIC_STOP_SEQUENCES_MAX = 4;
|
|
310
315
|
let warnedStopSequencesTrim = false;
|
|
311
316
|
|
|
312
|
-
/**
|
|
313
|
-
* Adaptive thinking `display` is supported starting with Anthropic model Opus 4.7.
|
|
314
|
-
* Older adaptive-thinking models (Opus 4.6, Sonnet 4.6+) reject the field.
|
|
315
|
-
* Fable (5+) postdates Opus 4.7, accepts `display`, and defaults it to
|
|
316
|
-
* "omitted" — thinking tokens are billed but no content streams back — so it
|
|
317
|
-
* must opt in like Opus 4.7+ (issue #2791).
|
|
318
|
-
*/
|
|
319
|
-
function supportsAdaptiveThinkingDisplay(modelId: string): boolean {
|
|
320
|
-
if (/claude-fable-\d/.test(modelId)) return true;
|
|
321
|
-
const match = /claude-opus-(\d+)-(\d+)/.exec(modelId);
|
|
322
|
-
if (!match) return false;
|
|
323
|
-
const major = Number(match[1]);
|
|
324
|
-
const minor = Number(match[2]);
|
|
325
|
-
return major > 4 || (major === 4 && minor >= 7);
|
|
326
|
-
}
|
|
327
|
-
|
|
328
317
|
const ANTHROPIC_PROVIDER_SESSION_STATE_KEY = "anthropic-messages";
|
|
329
318
|
|
|
330
319
|
type AnthropicProviderSessionState = ProviderSessionState & {
|
|
@@ -1422,6 +1411,8 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1422
1411
|
) & { index: number };
|
|
1423
1412
|
const blocks = output.content as Block[];
|
|
1424
1413
|
const blocksByAnthropicIndex = new Map<number, Block>();
|
|
1414
|
+
const truncatedToolCalls = new Set<ToolCall>();
|
|
1415
|
+
let sawTerminalStopReason = false;
|
|
1425
1416
|
// Derive from the ACTUAL request shape, not the option default: the request
|
|
1426
1417
|
// only sends `display: "summarized"` on specific paths (adaptive display is
|
|
1427
1418
|
// omitted for models where supportsAdaptiveThinkingDisplay is false). Defaulting
|
|
@@ -1439,8 +1430,14 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1439
1430
|
// finalize the orphaned block so no internal stream fields leak into output.
|
|
1440
1431
|
const orphaned = blocksByAnthropicIndex.get(anthropicIndex);
|
|
1441
1432
|
if (orphaned) {
|
|
1442
|
-
if (orphaned.type === "toolCall"
|
|
1443
|
-
|
|
1433
|
+
if (orphaned.type === "toolCall") {
|
|
1434
|
+
if (!isCompleteJson(orphaned.partialJson)) {
|
|
1435
|
+
orphaned.incompleteArguments = true;
|
|
1436
|
+
truncatedToolCalls.add(orphaned);
|
|
1437
|
+
}
|
|
1438
|
+
if (orphaned.partialJson.trim()) {
|
|
1439
|
+
orphaned.arguments = parseStreamingJson(orphaned.partialJson);
|
|
1440
|
+
}
|
|
1444
1441
|
}
|
|
1445
1442
|
delete (orphaned as { index?: number }).index;
|
|
1446
1443
|
delete (orphaned as { partialJson?: string }).partialJson;
|
|
@@ -1457,6 +1454,8 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1457
1454
|
output.usage = createEmptyUsage(copilotDynamicHeaders?.premiumRequests);
|
|
1458
1455
|
output.stopReason = "stop";
|
|
1459
1456
|
firstTokenTime = undefined;
|
|
1457
|
+
truncatedToolCalls.clear();
|
|
1458
|
+
sawTerminalStopReason = false;
|
|
1460
1459
|
};
|
|
1461
1460
|
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs();
|
|
1462
1461
|
const firstEventFallbackMs = getProviderFirstEventTimeoutFallbackMs(model.provider);
|
|
@@ -1467,12 +1466,13 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1467
1466
|
// Provider-level transport/rate-limit failures: only before any streamed content starts.
|
|
1468
1467
|
// Malformed envelopes/JSON: only before replay-unsafe text/tool events are visible on this stream.
|
|
1469
1468
|
let providerRetryAttempt = 0;
|
|
1470
|
-
let thinkingRepairAttempted = false;
|
|
1471
1469
|
while (true) {
|
|
1472
1470
|
// Retries reset output.content; drop stale block correlations from the aborted attempt.
|
|
1473
1471
|
blocksByAnthropicIndex.clear();
|
|
1472
|
+
truncatedToolCalls.clear();
|
|
1473
|
+
sawTerminalStopReason = false;
|
|
1474
1474
|
activeAbortTracker = createAbortSourceTracker(options?.signal);
|
|
1475
|
-
const firstEventTimeoutAbortError = new
|
|
1475
|
+
const firstEventTimeoutAbortError = new FirstEventTimeoutError(
|
|
1476
1476
|
"Anthropic stream timed out while waiting for the first event",
|
|
1477
1477
|
);
|
|
1478
1478
|
const idleTimeoutAbortError = new Error("Anthropic stream stalled while waiting for the next event");
|
|
@@ -1495,6 +1495,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1495
1495
|
let sawEvent = false;
|
|
1496
1496
|
let sawMessageStart = false;
|
|
1497
1497
|
let sawTerminalEnvelope = false;
|
|
1498
|
+
let sawMessageStop = false;
|
|
1498
1499
|
const isProgressEvent = createAnthropicStreamProgressPredicate();
|
|
1499
1500
|
|
|
1500
1501
|
for await (const event of iterateWithIdleTimeout(anthropicStream, {
|
|
@@ -1508,9 +1509,13 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1508
1509
|
isProgressItem: isProgressEvent,
|
|
1509
1510
|
})) {
|
|
1510
1511
|
sawEvent = true;
|
|
1512
|
+
if (sawMessageStop) {
|
|
1513
|
+
throw createAnthropicStreamEnvelopeError("received event after message_stop");
|
|
1514
|
+
}
|
|
1511
1515
|
if (sawProviderSafetyStop) {
|
|
1512
1516
|
if (event.type === "message_stop") {
|
|
1513
1517
|
sawTerminalEnvelope = true;
|
|
1518
|
+
sawMessageStop = true;
|
|
1514
1519
|
}
|
|
1515
1520
|
continue;
|
|
1516
1521
|
}
|
|
@@ -1698,6 +1703,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1698
1703
|
partial: output,
|
|
1699
1704
|
});
|
|
1700
1705
|
} else if (block.type === "toolCall") {
|
|
1706
|
+
if (!isCompleteJson(block.partialJson)) truncatedToolCalls.add(block);
|
|
1701
1707
|
if (block.partialJson.trim()) {
|
|
1702
1708
|
block.arguments = parseStreamingJson(block.partialJson);
|
|
1703
1709
|
}
|
|
@@ -1718,6 +1724,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1718
1724
|
if (rawStopReason) {
|
|
1719
1725
|
output.stopReason = isProviderSafetyStop ? "error" : mapStopReason(rawStopReason);
|
|
1720
1726
|
sawTerminalEnvelope = true;
|
|
1727
|
+
sawTerminalStopReason = true;
|
|
1721
1728
|
}
|
|
1722
1729
|
if (isProviderSafetyStop) {
|
|
1723
1730
|
sawProviderSafetyStop = true;
|
|
@@ -1759,6 +1766,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1759
1766
|
calculateCost(model, output.usage);
|
|
1760
1767
|
} else if (event.type === "message_stop") {
|
|
1761
1768
|
sawTerminalEnvelope = true;
|
|
1769
|
+
sawMessageStop = true;
|
|
1762
1770
|
}
|
|
1763
1771
|
}
|
|
1764
1772
|
|
|
@@ -1781,8 +1789,9 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1781
1789
|
}
|
|
1782
1790
|
break;
|
|
1783
1791
|
} catch (streamError) {
|
|
1784
|
-
const
|
|
1785
|
-
|
|
1792
|
+
const localAbortReason = activeAbortTracker.getLocalAbortReason();
|
|
1793
|
+
const streamFailure = localAbortReason ?? streamError;
|
|
1794
|
+
if (localAbortReason || sawProviderSafetyStop) {
|
|
1786
1795
|
throw streamFailure;
|
|
1787
1796
|
}
|
|
1788
1797
|
if (
|
|
@@ -1834,21 +1843,22 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1834
1843
|
const thinkingSignatureInvalid = isAnthropicThinkingSignatureInvalidError(streamFailure);
|
|
1835
1844
|
if (
|
|
1836
1845
|
!options?.fallbackManaged &&
|
|
1837
|
-
!
|
|
1846
|
+
!repairAllAssistantThinking &&
|
|
1838
1847
|
firstTokenTime === undefined &&
|
|
1839
1848
|
(thinkingSignatureInvalid || isAnthropicThinkingBlockMutationError(streamFailure))
|
|
1840
1849
|
) {
|
|
1850
|
+
// The mutation 400 blames the "latest assistant message", but its cited
|
|
1851
|
+
// `messages.N.content.M` path can point at an EARLIER replayed turn, so the
|
|
1852
|
+
// latest-only repair gets rejected identically. Escalate to the full-history
|
|
1853
|
+
// repair instead of burning the single retry on one scope.
|
|
1854
|
+
const escalateToAll: boolean = thinkingSignatureInvalid || repairLatestAssistantThinking;
|
|
1841
1855
|
logger.debug("anthropic: repairing assistant thinking replay after provider rejection", {
|
|
1842
1856
|
model: model.id,
|
|
1843
|
-
scope:
|
|
1857
|
+
scope: escalateToAll ? "all" : "latest",
|
|
1844
1858
|
error: streamFailure instanceof Error ? streamFailure.message : String(streamFailure),
|
|
1845
1859
|
});
|
|
1846
|
-
|
|
1847
|
-
|
|
1848
|
-
repairAllAssistantThinking = true;
|
|
1849
|
-
} else {
|
|
1850
|
-
repairLatestAssistantThinking = true;
|
|
1851
|
-
}
|
|
1860
|
+
repairLatestAssistantThinking = !escalateToAll;
|
|
1861
|
+
repairAllAssistantThinking = escalateToAll;
|
|
1852
1862
|
params = await prepareParams();
|
|
1853
1863
|
providerRetryAttempt = 0;
|
|
1854
1864
|
resetOutputForRetry();
|
|
@@ -1897,6 +1907,24 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1897
1907
|
}
|
|
1898
1908
|
}
|
|
1899
1909
|
|
|
1910
|
+
for (const block of blocksByAnthropicIndex.values()) {
|
|
1911
|
+
delete (block as { index?: number }).index;
|
|
1912
|
+
if (block.type === "toolCall") {
|
|
1913
|
+
truncatedToolCalls.add(block);
|
|
1914
|
+
if (block.partialJson.trim()) {
|
|
1915
|
+
block.arguments = parseStreamingJson(block.partialJson);
|
|
1916
|
+
}
|
|
1917
|
+
delete (block as { partialJson?: string }).partialJson;
|
|
1918
|
+
}
|
|
1919
|
+
}
|
|
1920
|
+
blocksByAnthropicIndex.clear();
|
|
1921
|
+
if (output.stopReason === "length" || !sawTerminalStopReason) {
|
|
1922
|
+
for (const block of output.content) {
|
|
1923
|
+
if (block.type === "toolCall" && truncatedToolCalls.has(block)) {
|
|
1924
|
+
block.incompleteArguments = true;
|
|
1925
|
+
}
|
|
1926
|
+
}
|
|
1927
|
+
}
|
|
1900
1928
|
output.duration = Date.now() - startTime;
|
|
1901
1929
|
if (firstTokenTime) output.ttft = firstTokenTime - startTime;
|
|
1902
1930
|
if (dropFastMode && resolveServiceTier(options?.serviceTier, model.provider) === "priority") {
|
|
@@ -1909,13 +1937,12 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1909
1937
|
delete (block as { index?: number }).index;
|
|
1910
1938
|
delete (block as { partialJson?: string }).partialJson;
|
|
1911
1939
|
}
|
|
1912
|
-
const
|
|
1940
|
+
const localAbortReason = activeAbortTracker.getLocalAbortReason();
|
|
1913
1941
|
output.stopReason = activeAbortTracker.wasCallerAbort() ? "aborted" : "error";
|
|
1914
|
-
output.errorStatus = extractHttpStatusFromError(error);
|
|
1915
|
-
output.transportFailure = transportFailureFacts(error);
|
|
1942
|
+
output.errorStatus = extractHttpStatusFromError(localAbortReason ?? error);
|
|
1943
|
+
output.transportFailure = transportFailureFacts(localAbortReason ?? error);
|
|
1916
1944
|
if (output.errorKind !== "provider_safety_stop" || !output.errorMessage) {
|
|
1917
|
-
output.errorMessage =
|
|
1918
|
-
firstEventTimeoutError?.message ?? (await finalizeErrorMessage(error, rawRequestDump));
|
|
1945
|
+
output.errorMessage = localAbortReason?.message ?? (await finalizeErrorMessage(error, rawRequestDump));
|
|
1919
1946
|
}
|
|
1920
1947
|
output.errorMessage = rewriteCopilotError(output.errorMessage, error, model.provider);
|
|
1921
1948
|
output.duration = Date.now() - startTime;
|
|
@@ -2100,13 +2127,26 @@ function createClient(
|
|
|
2100
2127
|
return { client, isOAuthToken: oauthToken };
|
|
2101
2128
|
}
|
|
2102
2129
|
|
|
2103
|
-
|
|
2130
|
+
/**
|
|
2131
|
+
* Anthropic rejects extended thinking combined with a forced tool choice, so such a
|
|
2132
|
+
* request drops `thinking`/`output_config`. Reports whether the forced-choice branch
|
|
2133
|
+
* applied so the caller can keep the replayed history consistent with it.
|
|
2134
|
+
*/
|
|
2135
|
+
function disableThinkingIfToolChoiceForced(params: MessageCreateParamsStreaming): boolean {
|
|
2104
2136
|
const toolChoice = params.tool_choice;
|
|
2105
|
-
if (!toolChoice) return;
|
|
2106
|
-
if (toolChoice.type
|
|
2107
|
-
|
|
2108
|
-
|
|
2109
|
-
|
|
2137
|
+
if (!toolChoice) return false;
|
|
2138
|
+
if (toolChoice.type !== "any" && toolChoice.type !== "tool") return false;
|
|
2139
|
+
delete params.thinking;
|
|
2140
|
+
delete params.output_config;
|
|
2141
|
+
return true;
|
|
2142
|
+
}
|
|
2143
|
+
|
|
2144
|
+
function hasNativeThinkingBlocks(messages: MessageParam[]): boolean {
|
|
2145
|
+
return messages.some(
|
|
2146
|
+
message =>
|
|
2147
|
+
Array.isArray(message.content) &&
|
|
2148
|
+
message.content.some(block => block.type === "thinking" || block.type === "redacted_thinking"),
|
|
2149
|
+
);
|
|
2110
2150
|
}
|
|
2111
2151
|
|
|
2112
2152
|
function mapAnthropicToolChoice(
|
|
@@ -2430,6 +2470,18 @@ function buildParams(
|
|
|
2430
2470
|
}
|
|
2431
2471
|
}
|
|
2432
2472
|
|
|
2473
|
+
// A forced tool choice strips `thinking` from the request. Signed thinking blocks
|
|
2474
|
+
// replayed from history belong to a thinking-enabled request, and Anthropic rejects
|
|
2475
|
+
// that pair with `thinking`/`redacted_thinking` blocks "cannot be modified", so the
|
|
2476
|
+
// replay has to degrade in the same rebuild. Runs before the billing/system payload
|
|
2477
|
+
// snapshot so the attribution hash covers the messages actually sent.
|
|
2478
|
+
if (disableThinkingIfToolChoiceForced(params) && hasNativeThinkingBlocks(params.messages)) {
|
|
2479
|
+
params.messages = convertAnthropicMessages(context.messages, model, isOAuthToken, {
|
|
2480
|
+
...thinkingRepair,
|
|
2481
|
+
repairAllAssistantThinking: true,
|
|
2482
|
+
});
|
|
2483
|
+
}
|
|
2484
|
+
|
|
2433
2485
|
const shouldInjectClaudeCodeInstruction = isOAuthToken && !model.id.startsWith("claude-3-5-haiku");
|
|
2434
2486
|
const billingSystemPrompts = normalizeSystemPrompts(context.systemPrompt);
|
|
2435
2487
|
const billingPayload = shouldInjectClaudeCodeInstruction
|
|
@@ -2445,7 +2497,6 @@ function buildParams(
|
|
|
2445
2497
|
if (systemBlocks) {
|
|
2446
2498
|
params.system = systemBlocks;
|
|
2447
2499
|
}
|
|
2448
|
-
disableThinkingIfToolChoiceForced(params);
|
|
2449
2500
|
ensureMaxTokensForThinking(params, model);
|
|
2450
2501
|
applyPromptCaching(params as AnthropicCacheParams, cacheMode, cacheControl);
|
|
2451
2502
|
enforceCacheControlLimit(params, 4);
|
|
@@ -22,7 +22,6 @@ import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
|
22
22
|
import { transportFailureFacts } from "../utils/fallback-transport";
|
|
23
23
|
import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
|
|
24
24
|
import {
|
|
25
|
-
createWatchdog,
|
|
26
25
|
getOpenAIStreamIdleTimeoutMs,
|
|
27
26
|
getStreamFirstEventTimeoutMs,
|
|
28
27
|
iterateWithIdleTimeout,
|
|
@@ -118,7 +117,6 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
|
|
118
117
|
);
|
|
119
118
|
let rawRequestDump: RawHttpRequestDump | undefined;
|
|
120
119
|
const abortTracker = createAbortSourceTracker(options?.signal);
|
|
121
|
-
const firstEventTimeoutAbortError = new Error(AZURE_OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE);
|
|
122
120
|
const { requestAbortController, requestSignal } = abortTracker;
|
|
123
121
|
|
|
124
122
|
try {
|
|
@@ -127,7 +125,7 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
|
|
127
125
|
const client = createClient(model, apiKey, options);
|
|
128
126
|
const { baseUrl } = resolveAzureConfig(model, options);
|
|
129
127
|
const params = buildParams(model, context, options, deploymentName, baseUrl);
|
|
130
|
-
const idleTimeoutMs = getOpenAIStreamIdleTimeoutMs();
|
|
128
|
+
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs();
|
|
131
129
|
options?.onPayload?.(params);
|
|
132
130
|
rawRequestDump = {
|
|
133
131
|
provider: model.provider,
|
|
@@ -164,18 +162,17 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
|
|
164
162
|
rawRequestDump = { ...rawRequestDump, body: params };
|
|
165
163
|
openaiStream = await client.responses.create(params, { signal: requestSignal });
|
|
166
164
|
}
|
|
167
|
-
const
|
|
168
|
-
options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs),
|
|
169
|
-
() => abortTracker.abortLocally(firstEventTimeoutAbortError),
|
|
170
|
-
);
|
|
165
|
+
const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs);
|
|
171
166
|
stream.push({ type: "start", partial: output });
|
|
172
167
|
|
|
173
168
|
await processResponsesStream(
|
|
174
169
|
iterateWithIdleTimeout(openaiStream, {
|
|
175
|
-
|
|
170
|
+
firstItemTimeoutMs: firstEventTimeoutMs,
|
|
171
|
+
firstItemErrorMessage: AZURE_OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE,
|
|
176
172
|
idleTimeoutMs,
|
|
177
173
|
errorMessage: "Azure OpenAI responses stream stalled while waiting for the next event",
|
|
178
174
|
onIdle: () => requestAbortController.abort(),
|
|
175
|
+
onFirstItemTimeout: () => requestAbortController.abort(),
|
|
179
176
|
}),
|
|
180
177
|
output,
|
|
181
178
|
stream,
|
|
@@ -272,16 +269,31 @@ export function resolveAzureConfigForTest(
|
|
|
272
269
|
return resolveAzureConfig(model, options);
|
|
273
270
|
}
|
|
274
271
|
|
|
272
|
+
/**
|
|
273
|
+
* Azure API key for the client, from trusted environment sources only.
|
|
274
|
+
*
|
|
275
|
+
* `$env` merges the caller's `cwd/.env`, so reading the key there would let
|
|
276
|
+
* repository content supply the credential this client authenticates with.
|
|
277
|
+
* Provider credentials are resolved from the launching shell plus GJC/user-owned
|
|
278
|
+
* `.env` files, never the project `.env` — this fallback now matches that rule.
|
|
279
|
+
*/
|
|
280
|
+
function resolveAzureClientApiKey(apiKey: string): string | undefined {
|
|
281
|
+
if (apiKey) return apiKey;
|
|
282
|
+
return $credentialEnv("AZURE_OPENAI_API_KEY");
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
/** Test seam: the client API key as resolved from a caller value plus trusted env. */
|
|
286
|
+
export function resolveAzureClientApiKeyForTest(apiKey: string): string | undefined {
|
|
287
|
+
return resolveAzureClientApiKey(apiKey);
|
|
288
|
+
}
|
|
275
289
|
function createClient(model: Model<"azure-openai-responses">, apiKey: string, options?: AzureOpenAIResponsesOptions) {
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
);
|
|
282
|
-
}
|
|
283
|
-
apiKey = envKey;
|
|
290
|
+
const resolvedApiKey = resolveAzureClientApiKey(apiKey);
|
|
291
|
+
if (!resolvedApiKey) {
|
|
292
|
+
throw new Error(
|
|
293
|
+
"Azure OpenAI API key is required. Set AZURE_OPENAI_API_KEY environment variable or pass it as an argument.",
|
|
294
|
+
);
|
|
284
295
|
}
|
|
296
|
+
apiKey = resolvedApiKey;
|
|
285
297
|
|
|
286
298
|
const headers = { ...(model.headers ?? {}) };
|
|
287
299
|
|